diff --git a/pageindex/agent_tools.py b/pageindex/agent_tools.py
index afac6b9d5..40b76cf23 100644
--- a/pageindex/agent_tools.py
+++ b/pageindex/agent_tools.py
@@ -1596,7 +1596,7 @@ def build_agent_tools(client, include_management: bool = False,
_DISCOVERY = """\
DOCUMENT DISCOVERY:
-- browse_documents() — DEFAULT discovery tool, first choice for any document-related question. It lists your documents newest first with names and descriptions; match them against the user's intent, and page through with `offset: next_offset` while has_more is true."""
+- browse_documents() — DEFAULT discovery tool, first choice for any document-related question. The bare call returns your documents newest first with names and descriptions; match them against the user's intent."""
_DECISION = """\
DECISION:
@@ -1612,9 +1612,9 @@ def build_agent_tools(client, include_management: bool = False,
PERSISTENCE (before concluding the target document is not in the library):
This protocol applies both when results are empty AND when results are returned but none match the user's intent. Do NOT give up after a single discovery attempt. Follow these steps in order:
1. browse_documents() and compare every returned name/description against the user's intent
-2. Page through the ENTIRE library with `limit: 50` and `offset: next_offset` until has_more is false — MANDATORY, must be completed before concluding "not found"
-3. Re-scan for loose matches: synonyms, abbreviations, and partial titles in names/descriptions can identify the target
-Only after ALL three steps have been tried may you conclude the document is not in the library. Do NOT fall back to general knowledge — if the user's question references their own documents, exhaust every discovery path first."""
+2. Rephrase the query with synonyms or alternative terms and browse again
+3. Page through the ENTIRE library with `limit: 50` and `offset: next_offset` until has_more is false — MANDATORY, must be completed before concluding "not found"
+Only after ALL steps have been tried may you conclude the document is not in the library. Do NOT fall back to general knowledge — if the user's question references their own documents, exhaust every discovery path first."""
AGENT_INSTRUCTIONS = "\n\n".join([
_INSTRUCTIONS_HEADER,
@@ -1651,8 +1651,7 @@ def build_agent_tools(client, include_management: bool = False,
- Cite only statements supported by tool outputs: or . Place immediately after the claim.
- When page content includes block_id values, citations MUST be block-level: copy the exact block_id of the supporting block. Page-only cites are allowed ONLY when the tool output carries no block_id (legacy documents, structure outlines). NEVER invent or alter block_id values.
- For a claim drawn from multiple blocks on one page, add one tag per supporting block (at most 3); beyond that, cite the single strongest block.
-- Each tag must reference a SINGLE page integer. For multi-page citations, use separate tags.
-- Close every answer with a "Sources" section: one plain-text line per cited block, formatted `- , page (block )`. Keep it even when the inline tags are present — a client that strips unknown HTML tags would otherwise leave the answer with no visible citations at all.""",
+- Each tag must reference a SINGLE page integer. For multi-page citations, use separate tags.""",
"footnote": """\
GROUNDING
- Answer only from the user's PageIndex documents. Call get_page_content() and state only what was actually read there.
diff --git a/pageindex/client.py b/pageindex/client.py
index ffac91d02..aaa93a594 100644
--- a/pageindex/client.py
+++ b/pageindex/client.py
@@ -1653,6 +1653,19 @@ def get_document(self, doc_id: str) -> dict[str, Any]:
"""
return self._api.get_document(doc_id=doc_id)
+ def get_document_id(self, name: str) -> str:
+ """
+ Look up a document's ID by its name. Useful for resolving
+ citation doc names (from ````) to IDs.
+
+ Raises PageIndexAPIError if no document with that name exists.
+ """
+ result = self._api.list_documents(limit=1, name=name)
+ docs = result.get("documents", [])
+ if docs:
+ return docs[0]["id"]
+ raise PageIndexAPIError(f"No document named {name!r} found.")
+
def delete_document(self, doc_id: str) -> dict[str, Any]:
"""
Delete a PageIndex document and all its associated data.
diff --git a/pageindex/cloud_api.py b/pageindex/cloud_api.py
index 94e2cfd39..a8fc7a7e6 100644
--- a/pageindex/cloud_api.py
+++ b/pageindex/cloud_api.py
@@ -369,7 +369,7 @@ def delete_document(self, doc_id: str) -> Dict[str, Any]:
status_code=response.status_code)
return response.json() if response.content else {}
- def list_documents(self, limit: int = 50, offset: int = 0, folder_id: Optional[str] = None) -> Dict[str, Any]:
+ def list_documents(self, limit: int = 50, offset: int = 0, folder_id: Optional[str] = None, name: Optional[str] = None) -> Dict[str, Any]:
"""
List all documents for the authenticated user with pagination.
@@ -394,6 +394,8 @@ def list_documents(self, limit: int = 50, offset: int = 0, folder_id: Optional[s
params: Dict[str, Any] = {"limit": limit, "offset": offset}
if folder_id is not None:
params["folder_id"] = folder_id
+ if name is not None:
+ params["name"] = name
response = requests.get(
f"{self.BASE_URL}/docs/",
diff --git a/pageindex/flash/README.md b/pageindex/flash/README.md
index 0d23a3c35..7ecff60f6 100644
--- a/pageindex/flash/README.md
+++ b/pageindex/flash/README.md
@@ -40,14 +40,26 @@ Writes the tree to `results/_structure.json`.
"node_id": str, # 4-digit, zero-padded
"start_index": int, # 1-based, inclusive
"end_index": int,
- "summary": str,
+ "summary": str, # summary=True only
"key_items": [str], # optimize only: titles of subsections merged away
- "nodes": [...],
+ "nodes": [...], # entries with children only
}
],
+ "toc_source": str, # "detected" | "bookmarks" | "hybrid" | "pages" | "unreadable"
}
```
+`toc_source` says where the structure came from: `"detected"` from the layout,
+`"bookmarks"` from the embedded outline, `"hybrid"` when bookmarks frame the
+detected sections. `"pages"` means the layout yielded no hierarchy, so every
+page became one node titled `Page N`; past `FLAT_TREE_MAX_NODES` (10) pages that
+flat tree comes back without summaries or optimization, and the local client and
+CLI refuse it. `"unreadable"` means no page carries text and `structure` is
+empty.
+
+Every page is in some node: a hierarchy that starts after page 1 is preceded by
+a `Preface` node covering the pages before it, as in standard mode.
+
## Benchmark
Nine PDFs, each run end to end with tree optimization: PDF parse, layout
diff --git a/pageindex/flash/api.py b/pageindex/flash/api.py
index 6f7361e27..02a6eebf1 100644
--- a/pageindex/flash/api.py
+++ b/pageindex/flash/api.py
@@ -1,4 +1,4 @@
-"""Public API for PageIndex Flash. The only supported entry point is :func:`page_index_flash`. Everything else in this package is internal pipeline machinery."""
+"""Public API for PageIndex Flash: :func:`page_index_flash` builds the tree, :func:`flash_rejection_reason` is the refusal policy the local client and CLI share. Everything else in this package is internal pipeline machinery."""
from __future__ import annotations
@@ -10,6 +10,9 @@
from .main import extract_toc
+# Largest page-node fallback the managed pipelines accept as an index.
+FLAT_TREE_MAX_NODES = 10
+
def _is_pdfium_password_error(exc: Exception) -> bool:
msg = str(exc).lower()
@@ -95,11 +98,50 @@ def _optimize(structure, page_texts, do_expand, model):
"before": outcome["before"], "after": outcome["after"]}
+def _page_nodes(page_texts: list[str]) -> list[dict]:
+ """One node per page, so every page is reachable."""
+ from ..utils import write_node_id
+ if not any(text.strip() for text in page_texts):
+ return []
+ nodes = [{"title": f"Page {index}", "node_id": "", "start_index": index,
+ "end_index": index} for index in range(1, len(page_texts) + 1)]
+ write_node_id(nodes)
+ return nodes
+
+
+def _add_preface(structure: list[dict]) -> None:
+ """The pages before a hierarchy that starts late become a Preface node, as in standard mode."""
+ from ..utils import write_node_id
+ structure.insert(0, {"title": "Preface", "start_index": 1,
+ "end_index": structure[0]["start_index"] - 1})
+ write_node_id(structure)
+
+
+def flash_rejection_reason(result: dict, standard_hint: str = "mode='standard'") -> str | None:
+ """Why a managed pipeline should refuse this flash result, or None to accept it.
+
+ The local client and the CLI share this policy so they refuse the same
+ documents; ``standard_hint`` is how each spells the standard-mode switch.
+ """
+ structure = result.get("structure") or []
+ if result.get("toc_source") == "unreadable":
+ return ("PageIndex Flash found no text layer in this PDF (scanned or "
+ "image-only); run OCR before indexing it.")
+ if result.get("toc_source") == "pages" and len(structure) > FLAT_TREE_MAX_NODES:
+ return (f"PageIndex Flash found no layout structure in this document "
+ f"({len(structure)} pages); try {standard_hint}, which builds "
+ "the structure with the model.")
+ if not structure:
+ return ("PageIndex Flash could not extract a structure from this PDF; "
+ f"try {standard_hint}, which builds the structure with the model.")
+ return None
+
+
def page_index_flash(pdf, summary=True, summary_model=None,
optimize: str | bool | None = None, optimize_expand=None,
optimize_model=None, summary_concurrency=None,
use_embedded_toc=True) -> dict:
- """Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored; adds a ``toc_source`` key to the result. On by default; pass False for the pure detected structure. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics. """
+ """Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """
if optimize_expand is not None:
import warnings
warnings.warn(
@@ -118,6 +160,17 @@ def page_index_flash(pdf, summary=True, summary_model=None,
f"optimize must be 'full', 'merge', or False, got {optimize!r}")
result = extract_toc(_validate_pdf(pdf), use_embedded_toc=use_embedded_toc)
structure = result.get("structure", [])
+ if not structure:
+ # the layout yields no hierarchy; the pages themselves are the tree
+ structure = _page_nodes(result.get("page_texts") or [])
+ result["structure"] = structure
+ result["toc_source"] = "pages" if structure else "unreadable"
+ elif structure[0]["start_index"] > 1:
+ _add_preface(structure)
+ if result.get("toc_source") == "pages" and len(structure) > FLAT_TREE_MAX_NODES:
+ # the managed pipelines refuse a flat tree this size; skip the model passes
+ result.pop("page_texts", None)
+ return result
if optimize and structure:
# bookmark-only extractions carry no page_texts and scanned ones
# only empty strings; expand needs text
diff --git a/pageindex/flash/heading_detection/page_scan.py b/pageindex/flash/heading_detection/page_scan.py
index d6ebe6d8e..fc52c4927 100644
--- a/pageindex/flash/heading_detection/page_scan.py
+++ b/pageindex/flash/heading_detection/page_scan.py
@@ -295,7 +295,6 @@ def __init__(self, doc, labeled):
def filter_page_candidates(doc_collector: DocCandidateCollector, page, page_candidates: list[HeadingCandidate]) -> None:
"""Per-page candidate filter for noisy pages, title overlap, page headers, and numbering continuity."""
- from ..outline_assembly import is_script_compatible
from ..model import intervals_overlap, is_caps_heavy
from ..stats import column_index_of
@@ -350,9 +349,6 @@ def _ih_key(heading_candidate: HeadingCandidate):
active_candidate = page_candidates[index]
next_item = page_candidates[index + 1] if index + 1 < count else None
- if is_script_compatible(doc_collector.previous_slot.secondary_slot.tertiary_slot, active_candidate):
- continue
-
if not doc_collector.auxiliary_slot and active_candidate.type != 11:
bottom = active_candidate.group_slot.top_edge()
if (title is not None and bottom > title.top_edge()
@@ -428,7 +424,7 @@ def build_doc_heading_candidates(doc, labeled: Optional[list] = None) -> list[He
def find_section_openers(doc, start_page_idx: int) -> list:
"""Find the first valid heading on each page, then clique-filter the result."""
- from ..outline_assembly import is_script_compatible, has_conflict_in_context, OutlineContext, OutlineNode
+ from ..outline_assembly import has_conflict_in_context, OutlineContext, OutlineNode
item_list: list[HeadingCandidate] = []
index = start_page_idx
@@ -450,7 +446,7 @@ def find_section_openers(doc, start_page_idx: int) -> list:
break
if block.line_count() > 2:
break
- if current_candidate is not None and not is_script_compatible(doc.secondary_slot.tertiary_slot, current_candidate):
+ if current_candidate is not None:
item_list.append(current_candidate)
index += 1
diff --git a/pageindex/flash/main.py b/pageindex/flash/main.py
index 791874538..3f8b3d9c4 100644
--- a/pageindex/flash/main.py
+++ b/pageindex/flash/main.py
@@ -21,7 +21,7 @@
from .labels import detect_captions, build_caption_regions, CaptionContext
from .model import Rect, numbering_kind, block_text, deaccented_text, Block
from .outline_assembly import (
- build_heading_from_block, is_landscape_or_empty, is_outline_valid, is_chapter_outline_valid, mark_outline_block_types, assemble_outline, compute_max_heading_gap, has_table_or_prominent, OutlineNode, outline_to_dict_tree,
+ build_heading_from_block, is_outline_valid, is_chapter_outline_valid, mark_outline_block_types, assemble_outline, compute_max_heading_gap, has_table_or_prominent, OutlineNode, outline_to_dict_tree,
)
from .parser_pdfium_parallel import parse_charlevel_meta_parallel
from .phases import assign_reading_order, PageView, process_page
@@ -126,7 +126,7 @@ def extract_toc(
workers: Optional[int] = None,
use_embedded_toc: bool = True,
) -> dict:
- """Run the full pipeline. Returns a dict shaped like:: { "doc_name": "...", "doc_title": "...", "structure": [ {"title": "...", "start_index": 1, "end_index": 3, "nodes": [...]}, ... ], "has_abstract_or_references_section": False } ``has_abstract_or_references_section`` is True when any TOP-LEVEL outline entry is an abstract-keyword heading or carries the prominent-heading flag (a references-keyword heading, plain or numbered). The near-empty bail and the valid-outline branch both report False. ``workers`` sets the process count for the per-page parallel parser: None = auto (CPU count - 1), 1 forces the sequential path; output is identical either way. ``use_embedded_toc`` consumes the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame with the detected sections they lack grafted back in, coarse ones become the chapter frame with detected nodes re-hung under them, garbage ones are ignored; adds ``toc_source`` to the result. On by default; pass False for the pure detected structure. """
+ """Run the full pipeline. Returns a dict shaped like:: { "doc_name": "...", "doc_title": "...", "structure": [ {"title": "...", "start_index": 1, "end_index": 3, "nodes": [...]}, ... ], "has_abstract_or_references_section": False } ``has_abstract_or_references_section`` is True when any TOP-LEVEL outline entry is an abstract-keyword heading or carries the prominent-heading flag (a references-keyword heading, plain or numbered). The valid-outline branch reports False. ``workers`` sets the process count for the per-page parallel parser: None = auto (CPU count - 1), 1 forces the sequential path; output is identical either way. ``use_embedded_toc`` consumes the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame with the detected sections they lack grafted back in, coarse ones become the chapter frame with detected nodes re-hung under them, garbage ones are ignored. On by default; pass False for the pure detected structure. ``toc_source`` is always present: ``"detected"``, ``"bookmarks"``, or ``"hybrid"``. """
# ----- 1) Parse PDF -> flat spans per page --------------------------
# per-page (view box, /Rotate) comes from the same engine (PDFium) that
# produced the block coordinates, so the geometry frame is consistent.
@@ -157,34 +157,6 @@ def extract_toc(
page.blocks = cluster_lines_into_blocks(ctx)
assign_reading_order(page, page.blocks)
- # ----- Early empty-outline gate ------------------------------------
- # Short, near-empty, unsupported-script, or mostly-landscape documents
- # emit an empty outline rather than a fabricated structure.
- if (doc.secondary_slot.state_slot <= 300 or doc.secondary_slot.previous_slot <= 200
- or doc.secondary_slot.tertiary_slot in (0, 2, 10) or is_landscape_or_empty(doc)):
- if isinstance(doc_handle, (str, Path)):
- doc_name = Path(str(doc_handle)).name
- else:
- doc_name = "document.pdf"
- result = {
- "doc_name": doc_name,
- "doc_title": None,
- "structure": [],
- "has_abstract_or_references_section": False,
- # the summary/expand passes read these like on the normal path
- "page_texts": ["\n".join(block_text(block)
- for block in (page.secondary_slot or []))
- for page in pages],
- }
- # Bookmarks need no extracted text, so they can still structure a
- # document this gate wrote off as unreadable.
- if use_embedded_toc:
- from .embedded_toc import apply_embedded_toc
- result["structure"], result["toc_source"] = apply_embedded_toc(
- [], doc_handle, len(pages), page_texts=result["page_texts"],
- )
- return result
-
# ----- 5) Classification: header / footer / watermark / TOC pages ---
detect_header_footer(HeaderFooterContext(doc, 1)) # HEADER
detect_header_footer(HeaderFooterContext(doc, 2)) # FOOTER
@@ -280,7 +252,7 @@ def extract_toc(
# ----- 11) Outline assembly and validation gate ---------------------
outline_nodes = assemble_outline(doc, section_openers)
# Validate the assembled outline. Structured outlines must cover enough
- # chapters; unstructured outlines are filtered by script and density gap.
+ # chapters; unstructured outlines are filtered by density gap.
# The abstract/references signal rides along with this gate: it is False on
# the valid-outline branch, and on the other branch it is read off the
# possibly-emptied list once the density filter has run.
@@ -291,9 +263,9 @@ def extract_toc(
else:
mark_outline_block_types(outline_nodes)
page_count = len(doc.primary_slot)
- if doc.secondary_slot.tertiary_slot == 7 or (
+ if (
page_count >= 3
- and compute_max_heading_gap(outline_nodes, 1)["max_gap"] > (0.65 if doc.secondary_slot.tertiary_slot == 4 else 0.85) * page_count
+ and compute_max_heading_gap(outline_nodes, 1)["max_gap"] > 0.85 * page_count
):
outline_nodes = []
has_abstract_or_references = has_table_or_prominent(outline_nodes)
@@ -321,6 +293,7 @@ def extract_toc(
"structure": structure,
"has_abstract_or_references_section": has_abstract_or_references,
"page_texts": page_texts,
+ "toc_source": "detected",
}
if use_embedded_toc:
from .embedded_toc import apply_embedded_toc
diff --git a/pageindex/flash/outline_assembly/__init__.py b/pageindex/flash/outline_assembly/__init__.py
index 303c14883..2abc42ef8 100644
--- a/pageindex/flash/outline_assembly/__init__.py
+++ b/pageindex/flash/outline_assembly/__init__.py
@@ -34,7 +34,6 @@
compare_heading_order,
_compare_block_order,
heading_order_key,
- is_script_compatible,
heading_signature,
parent_signature,
cached_signature,
@@ -87,7 +86,6 @@
mark_outline_block_types,
compute_max_heading_gap,
has_table_or_prominent,
- is_landscape_or_empty,
build_heading_from_block,
assemble_outline,
_flatten_outline_nodes,
@@ -98,7 +96,7 @@
__all__ = [
"HeadingCandidate", "OutlineNode",
"compare_heading_order", "heading_order_key", "compare_heading_depth",
- "is_script_compatible", "heading_signature", "parent_signature", "cached_signature", "is_in_oo_range", "has_style_neighbor", "pick_style_bucket", "has_conflict_in_context", "is_compatible_with_context",
+ "heading_signature", "parent_signature", "cached_signature", "is_in_oo_range", "has_style_neighbor", "pick_style_bucket", "has_conflict_in_context", "is_compatible_with_context",
"StyleCluster", "OutlineContext", "NumberingTrie", "insert_numbering", "count_sibling_numberings",
"OutlineState",
"find_keyword_clique", "detect_body_headings", "CliqueFilterContext",
diff --git a/pageindex/flash/outline_assembly/assembly.py b/pageindex/flash/outline_assembly/assembly.py
index ea0ff4027..816c5411f 100644
--- a/pageindex/flash/outline_assembly/assembly.py
+++ b/pageindex/flash/outline_assembly/assembly.py
@@ -64,20 +64,6 @@ def has_table_or_prominent(outline_nodes: list[OutlineNode]) -> bool:
return any(secondary_item.heading.type == 5 or secondary_item.heading.is_prominent for secondary_item in outline_nodes)
-def is_landscape_or_empty(doc) -> bool:
- """Return True for mostly-landscape or near-empty documents with little outline text."""
- if doc.secondary_slot.secondary_slot >= 1e3:
- return False
- secondary_item = 0
- candidate_item = 0.0
- for page in doc.primary_slot:
- if page.bounds.bbox_width() > page.bounds.bbox_height() and page.primary_slot.secondary_slot < 1e3:
- secondary_item += 1
- candidate_item += page.primary_slot.secondary_slot
- count_item = len(doc.primary_slot)
- return secondary_item >= 0.9 * count_item or (secondary_item >= 0.7 * count_item and candidate_item >= 0.5 * doc.secondary_slot.state_slot)
-
-
# --------------------------------------------------------------------------- #
# Build a heading candidate from a block #
# --------------------------------------------------------------------------- #
diff --git a/pageindex/flash/outline_assembly/candidates.py b/pageindex/flash/outline_assembly/candidates.py
index 630f4ecc2..6d0ca2846 100644
--- a/pageindex/flash/outline_assembly/candidates.py
+++ b/pageindex/flash/outline_assembly/candidates.py
@@ -136,30 +136,6 @@ def heading_order_key(heading_candidate: HeadingCandidate) -> tuple:
# --------------------------------------------------------------------------- #
-def is_script_compatible(number: int, other_heading_candidate: HeadingCandidate) -> bool:
- """Return whether candidate script/type is compatible with prior context. Args: a: integer previous script/type context b: heading candidate """
- candidate_item = other_heading_candidate.state_slot
- if candidate_item == 0 or candidate_item == 2 or candidate_item == 10:
- return True
- if number == candidate_item:
- return False
- if other_heading_candidate.type == 5:
- return False
- if other_heading_candidate.is_prominent:
- return False
- if len(other_heading_candidate.numbering) > 0:
- return False
- if number == 3 and candidate_item == 9:
- return False
- if number == 9 and candidate_item == 3:
- return False
- if number == 7 and candidate_item == 5:
- return False
- if number == 6 and candidate_item == 3:
- return False
- return True
-
-
def heading_signature(heading_candidate: HeadingCandidate) -> str:
"""Return a full heading signature including numbering or text."""
if len(heading_candidate.numbering) > 0:
diff --git a/pageindex/flash/parser_pdfium_charlevel/text_normalize.py b/pageindex/flash/parser_pdfium_charlevel/text_normalize.py
index 9c477e3f5..21a6f7240 100644
--- a/pageindex/flash/parser_pdfium_charlevel/text_normalize.py
+++ b/pageindex/flash/parser_pdfium_charlevel/text_normalize.py
@@ -260,8 +260,8 @@ def _apply_bidi_reordering(text: str, start_level: int = -1, vertical: bool = Fa
def _rtl_sign(char: str) -> int:
- """+1 for LTR runs, -1 for a strong right-to-left char (bidi class R/AL, e.g. Hebrew/Arabic). PDFium reports RTL text in logical order with decreasing char origins, so the LTR ``advance = ox - prev_text_x`` model (prev_text_x = ox+glyph_w, a right edge) yields a large negative advance. For RTL chunks the x-axis is signed with ``sign*ox`` so the reading-direction advance is positive and the existing LTR merge logic applies unchanged."""
- return -1 if unicodedata.bidirectional(char) in ("R", "AL") else 1
+ """+1 for LTR runs, -1 for a strong right-to-left char (bidi class R/AL, e.g. Hebrew/Arabic). PDFium reports RTL text in logical order with decreasing char origins, so the LTR ``advance = ox - prev_text_x`` model (prev_text_x = ox+glyph_w, a right edge) yields a large negative advance. For RTL chunks the x-axis is signed with ``sign*ox`` so the reading-direction advance is positive and the existing LTR merge logic applies unchanged. A multi-code-point value (a ligature, a Devanagari conjunct, a Thai cluster) takes the class of its first code point, like ``_reverse_if_rtl``."""
+ return -1 if char and unicodedata.bidirectional(char[0]) in ("R", "AL") else 1
def _reverse_if_rtl(chars: str) -> str:
diff --git a/pageindex/flash/title/detect.py b/pageindex/flash/title/detect.py
index 49a022227..14f561e94 100644
--- a/pageindex/flash/title/detect.py
+++ b/pageindex/flash/title/detect.py
@@ -61,21 +61,6 @@ def detect_title(doc) -> Optional[TitleCandidate]:
has_seen_da = False # "broke into body" flag
for page in doc.primary_slot:
- # Special branch: landscape cover document
- if (
- doc.secondary_slot.style_slot > len(doc.primary_slot) / 2
- and page.page_index <= 1
- and page.bounds.bbox_width() > page.bounds.bbox_height()
- and page.primary_slot.secondary_slot < 500
- ):
- for idx, block in enumerate(page.secondary_slot):
- if (
- is_title_candidate_block(block) and id(block) not in state.primary_slot
- and heading_score(block) > page.primary_slot.primary_slot - 0.1
- ):
- score_title_candidate(state, page, idx)
- break
-
if (
is_cover_like_page(doc, page)
or (page.page_index <= 1 and len(doc.primary_slot) >= 10 and page.primary_slot.secondary_slot < 0.8 * doc.secondary_slot.secondary_slot)
diff --git a/pageindex/flash/title/scoring.py b/pageindex/flash/title/scoring.py
index 7c2c47308..89c382875 100644
--- a/pageindex/flash/title/scoring.py
+++ b/pageindex/flash/title/scoring.py
@@ -15,7 +15,6 @@
Line,
last_line_of,
first_span_of,
- block_text,
deaccented_text,
letter_count,
dominant_style_of,
@@ -24,7 +23,7 @@
alignment_code,
Block,
)
-from ..stats import DocStats, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
+from ..stats import DocStats, column_index_of
from ..tokens import is_superscript_adjacent, clamp_value, enumerate_tokens, jenkins_hash, trie_prefix_match, set_case_fold, TrieConfig, build_trie, tokenize_block, _de_norm, BuiltTrie, is_word_token
from .dicts import (
@@ -247,17 +246,10 @@ def score_title_candidate(zp_state, page, index: int) -> None:
# Email penalty
email = 1.0 / ((1 + email_count) ** 2)
- # Script-family match: build a script histogram over the candidate group's text and
- # compare the candidate script family against the document script family.
- script_acc = ScriptHistogram()
- for result_value in group:
- tally_scripts(script_acc, block_text(result_value))
- script = 1.0 if dominant_script_family(script_acc) == doc_state.secondary_slot.tertiary_slot else 0.5
-
score = (
max_heading_score * len_value * width_ratio_sq * right_pen * bracket_factor * page_pos
* density_factor * top * factor * recurrence_factor * institution * label
- * email * script
+ * email
)
if zp_state.secondary_slot is None or score > zp_state.secondary_slot.score:
diff --git a/pageindex/local_api.py b/pageindex/local_api.py
index e60433932..21a3abcd8 100644
--- a/pageindex/local_api.py
+++ b/pageindex/local_api.py
@@ -227,6 +227,7 @@ def _index_standard(self, file_path: str, page_texts: list[str]) -> tuple[list,
def _index_flash(self, file_path: str) -> tuple[list, str | None]:
from .flash import page_index_flash
+ from .flash.api import flash_rejection_reason
from .utils import (create_clean_structure_for_description,
generate_doc_description, write_node_id)
result = page_index_flash(file_path, summary=True,
@@ -234,12 +235,9 @@ def _index_flash(self, file_path: str) -> tuple[list, str | None]:
optimize="full",
optimize_model=self._summary_model)
structure = result.get("structure", [])
- if not structure:
- raise PageIndexAPIError(
- "Failed to submit document: PageIndex Flash could not extract "
- "a structure from this PDF. Try mode='standard', which builds "
- "the structure with the model."
- )
+ reason = flash_rejection_reason(result)
+ if reason:
+ raise PageIndexAPIError(f"Failed to submit document: {reason}")
write_node_id(structure)
description = generate_doc_description(
create_clean_structure_for_description(structure),
@@ -348,6 +346,7 @@ def list_documents(
limit: int = 50,
offset: int = 0,
folder_id: str | None = None,
+ name: str | None = None,
) -> dict[str, Any]:
if limit < 1 or limit > 100:
raise ValueError("limit must be between 1 and 100")
@@ -359,6 +358,8 @@ def list_documents(
)
metas = sorted(self._store.list_metas(), key=lambda m: m.get("id") or "")
metas.sort(key=lambda m: m.get("createdAt") or "", reverse=True)
+ if name is not None:
+ metas = [m for m in metas if m.get("name") == name]
documents = [{
"id": m.get("id"),
"name": m.get("name"),
diff --git a/pageindex/utils.py b/pageindex/utils.py
index f23995057..995d60ec1 100644
--- a/pageindex/utils.py
+++ b/pageindex/utils.py
@@ -317,7 +317,7 @@ def structure_to_list(structure):
def get_leaf_nodes(structure):
if isinstance(structure, dict):
- if not structure['nodes']:
+ if not structure.get('nodes'):
structure_node = copy.deepcopy(structure)
structure_node.pop('nodes', None)
return [structure_node]
diff --git a/run_pageindex.py b/run_pageindex.py
index 8e3e1e525..0d515c369 100644
--- a/run_pageindex.py
+++ b/run_pageindex.py
@@ -90,6 +90,7 @@
if args.mode == 'flash':
from pageindex.flash import page_index_flash
+ from pageindex.flash.api import flash_rejection_reason
summary_model = ConfigLoader().load({k: v for k, v in {
'summary_model': args.summary_model,
'index_model': args.index_model,
@@ -104,9 +105,10 @@
use_embedded_toc=args.embedded_toc if args.embedded_toc is not None else True,
summary=will_summarize,
)
- if not toc_with_page_number.get('structure'):
- raise ValueError("PageIndex Flash could not extract a structure from this PDF; "
- "try --mode standard, which builds the structure with the model")
+ reason = flash_rejection_reason(toc_with_page_number,
+ standard_hint="--mode standard")
+ if reason:
+ raise ValueError(reason)
if 'optimize' in toc_with_page_number:
o = toc_with_page_number['optimize']
print(f"Optimize: merges={o['merges']} expands={o['expands']}, "
diff --git a/tests/data/flash/ar_report.pdf b/tests/data/flash/ar_report.pdf
new file mode 100644
index 000000000..57dd6f068
Binary files /dev/null and b/tests/data/flash/ar_report.pdf differ
diff --git a/tests/data/flash/hi_report.pdf b/tests/data/flash/hi_report.pdf
new file mode 100644
index 000000000..f191e5c79
Binary files /dev/null and b/tests/data/flash/hi_report.pdf differ
diff --git a/tests/data/flash/ja_report.pdf b/tests/data/flash/ja_report.pdf
new file mode 100644
index 000000000..7eab57e7c
Binary files /dev/null and b/tests/data/flash/ja_report.pdf differ
diff --git a/tests/data/flash/make_fixtures.py b/tests/data/flash/make_fixtures.py
new file mode 100644
index 000000000..fb29cf2ae
--- /dev/null
+++ b/tests/data/flash/make_fixtures.py
@@ -0,0 +1,58 @@
+"""Regenerate the flash fixtures: python tests/data/flash/make_fixtures.py
+
+Needs pymupdf and pymupdf-fonts (FiraGO), neither a PageIndex dependency. Each
+PDF is a title page and three 20pt headings over 11pt body lines, with the font
+subset embedded. Output is byte-stable, so a regeneration leaves git clean.
+"""
+from pathlib import Path
+
+import pymupdf
+
+HERE = Path(__file__).parent
+
+ZH = ["本公司致力于为客户提供高质量的产品和服务,持续推动技术创新与业务增长。",
+ "报告期内,公司实现营业收入同比增长,主要得益于核心业务的稳步扩张。",
+ "管理层将继续优化资源配置,加强风险管理,提升整体运营效率。",
+ "未来公司将围绕战略目标,深化数字化转型,拓展新的市场机会。",
+ "董事会对全体员工的辛勤付出表示衷心感谢,并对未来发展充满信心。"]
+JA = ["当社は、お客様に高品質な製品とサービスを提供し、技術革新と事業成長を推進しています。",
+ "当期において、当社の売上高は主力事業の着実な拡大により前年同期比で増加しました。",
+ "経営陣は引き続き資源配分を最適化し、リスク管理を強化して業務効率を高めていきます。",
+ "今後は戦略目標を軸にデジタル変革を深め、新たな市場機会の開拓を進めてまいります。",
+ "取締役会は全従業員の努力に心より感謝し、今後の発展に自信を持っております。"]
+HI = ["कंपनी ग्राहकों को उच्च गुणवत्ता वाले उत्पाद और सेवाएँ प्रदान करने के लिए प्रतिबद्ध है।",
+ "रिपोर्टिंग अवधि में कंपनी की आय में मुख्य व्यवसाय के विस्तार के कारण वृद्धि हुई।",
+ "प्रबंधन संसाधनों का अनुकूलन और जोखिम प्रबंधन को मजबूत करना जारी रखेगा।",
+ "भविष्य में कंपनी रणनीतिक लक्ष्यों के अनुरूप डिजिटल परिवर्तन को गहरा करेगी।",
+ "निदेशक मंडल सभी कर्मचारियों की कड़ी मेहनत के लिए हृदय से आभार व्यक्त करता है।"]
+AR = ["تلتزم الشركة بتقديم منتجات وخدمات عالية الجودة لعملائها في جميع الأسواق.",
+ "خلال فترة التقرير ارتفعت إيرادات الشركة بفضل التوسع المستمر في الأعمال الأساسية.",
+ "ستواصل الإدارة تحسين توزيع الموارد وتعزيز إدارة المخاطر لرفع الكفاءة التشغيلية.",
+ "في المستقبل ستعمق الشركة التحول الرقمي وفق أهدافها الاستراتيجية وتستكشف أسواقا جديدة.",
+ "يتقدم مجلس الإدارة بخالص الشكر لجميع الموظفين على جهودهم ويثق بمستقبل الشركة."]
+
+FIXTURES = {
+ "zh_body_en_headings.pdf": ("china-s", ["公司年度报告", "Financial Review", "Risk Factors", "Business Outlook"], ZH),
+ "ja_report.pdf": ("japan", ["年次報告書", "財務ハイライト", "リスク要因", "今後の見通し"], JA),
+ "hi_report.pdf": ("figo", ["वार्षिक रिपोर्ट", "वित्तीय समीक्षा", "जोखिम कारक", "भविष्य की दिशा"], HI),
+ "ar_report.pdf": ("figo", ["التقرير السنوي", "المراجعة المالية", "عوامل المخاطر", "التوقعات المستقبلية"], AR),
+}
+
+
+def build(font, headings, body):
+ doc = pymupdf.Document()
+ buffer = pymupdf.Font(font).buffer
+ for heading in headings:
+ page = doc.new_page(width=595, height=842)
+ page.insert_font(fontname="f", fontbuffer=buffer)
+ page.insert_text((72, 90), heading, fontsize=20, fontname="f")
+ for row, line in enumerate(body):
+ page.insert_text((72, 130 + 16 * row), line, fontsize=11, fontname="f")
+ doc.subset_fonts()
+ return doc.tobytes(garbage=4, deflate=True, no_new_id=True)
+
+
+if __name__ == "__main__":
+ for name, (font, headings, body) in FIXTURES.items():
+ (HERE / name).write_bytes(build(font, headings, body))
+ print(name)
diff --git a/tests/data/flash/zh_body_en_headings.pdf b/tests/data/flash/zh_body_en_headings.pdf
new file mode 100644
index 000000000..3e72b2b04
Binary files /dev/null and b/tests/data/flash/zh_body_en_headings.pdf differ
diff --git a/tests/test_agent_tools.py b/tests/test_agent_tools.py
index 3de8384ab..01d91d664 100644
--- a/tests/test_agent_tools.py
+++ b/tests/test_agent_tools.py
@@ -2406,14 +2406,19 @@ def test_citation_prompt_local_frozen_copy(client):
@pytest.mark.skipif(not LIVE_KEY, reason="PAGEINDEX_API_KEY not set")
def test_live_local_citation_prompts_match_cloud():
- """The frozen local copies are the server's texts minus the one bullet
- naming get_document_image(); any other server edit fails here."""
+ """The frozen local copies are the server's texts minus the bullet
+ naming get_document_image() and the 'Sources' footer (cite format
+ only — SDK users who can't render tags use the markdown or
+ footnote format instead)."""
from pageindex.agent_tools import LOCAL_CITATION_PROMPTS
cloud = PageIndexCloudClient(api_key=LIVE_KEY)
for fmt, frozen in LOCAL_CITATION_PROMPTS.items():
live = cloud.citation_prompt(format=fmt).split("\n")
- dropped = [line for line in live if "get_document_image()" in line]
- assert len(dropped) == 1, fmt
+ dropped = [line for line in live
+ if "get_document_image()" in line
+ or (fmt == "cite" and "Sources" in line)]
+ expected_drops = 2 if fmt == "cite" else 1
+ assert len(dropped) == expected_drops, f"{fmt}: expected {expected_drops} dropped lines, got {len(dropped)}"
assert "\n".join(line for line in live if line not in dropped) == frozen
diff --git a/tests/test_client.py b/tests/test_client.py
index 4e63223bf..19df6c6e0 100644
--- a/tests/test_client.py
+++ b/tests/test_client.py
@@ -2186,3 +2186,45 @@ def test_instructions_must_be_a_string(tmp_path):
with pytest.raises(PageIndexAPIError, match="instructions must be a str"):
PageIndexClient(storage_path=str(tmp_path / "s"),
instructions=[{"type": "text", "text": "x"}])
+
+
+def test_submit_flash_rejects_unreadable_text_layer(local_client, sample_pdf,
+ monkeypatch):
+ monkeypatch.setattr(
+ pageindex.flash, "page_index_flash",
+ lambda pdf, **kwargs: {"doc_name": "sample.pdf", "structure": [],
+ "toc_source": "unreadable"})
+ with pytest.raises(PageIndexAPIError, match="no text layer"):
+ local_client.submit_document(sample_pdf, mode="flash")
+
+
+def test_submit_flash_accepts_page_fallback(local_client, sample_pdf, monkeypatch):
+ """A small flat tree is a valid index."""
+ monkeypatch.setattr(
+ pageindex.flash, "page_index_flash",
+ lambda pdf, **kwargs: {
+ "doc_name": "sample.pdf", "toc_source": "pages",
+ "structure": [{"title": "Hello", "node_id": "0000",
+ "start_index": 1, "end_index": 1},
+ {"title": "World", "node_id": "0001",
+ "start_index": 2, "end_index": 2}]})
+ monkeypatch.setattr(pageindex.utils, "llm_completion",
+ lambda model, prompt, **kw: "Flash description.")
+ doc_id = local_client.submit_document(sample_pdf, mode="flash")["doc_id"]
+ tree = local_client.get_tree(doc_id)["result"]
+ assert [node["title"] for node in tree] == ["Hello", "World"]
+
+
+def test_submit_flash_rejects_oversized_flat_tree(local_client, sample_pdf,
+ monkeypatch):
+ from pageindex.flash.api import FLAT_TREE_MAX_NODES
+
+ nodes = [{"title": f"Page {n}", "start_index": 1, "end_index": 1, "nodes": []}
+ for n in range(FLAT_TREE_MAX_NODES + 1)]
+ monkeypatch.setattr(
+ pageindex.flash, "page_index_flash",
+ lambda pdf, **kwargs: {"doc_name": "sample.pdf", "toc_source": "pages",
+ "structure": nodes})
+ with pytest.raises(PageIndexAPIError,
+ match="no layout structure.*mode='standard'"):
+ local_client.submit_document(sample_pdf, mode="flash")
diff --git a/tests/test_flash_extraction.py b/tests/test_flash_extraction.py
index 28c0a963e..f799db237 100644
--- a/tests/test_flash_extraction.py
+++ b/tests/test_flash_extraction.py
@@ -60,6 +60,17 @@ def test_page_mode_walk_uses_merged_surrogate_census():
assert unmapped["ch"] == "β"
+def test_rtl_sign_takes_a_multi_code_point_glyph():
+ """A ToUnicode value can be several code points (a Devanagari conjunct, a
+ Thai cluster, an Arabic ligature); the first one decides the direction."""
+ from pageindex.flash.parser_pdfium_charlevel.text_normalize import _rtl_sign
+
+ assert _rtl_sign("\u094d\u0924") == 1 # Devanagari conjunct
+ assert _rtl_sign("\u0e01\u0e34") == 1 # Thai cluster
+ assert _rtl_sign("\u0626\u062c") == -1 # Arabic ligature
+ assert _rtl_sign("") == 1
+
+
def test_optimize_full_keyless_reports_file_errors_first(tmp_path, monkeypatch):
"""No credential pre-check: a bad path is a FileNotFoundError even
keyless (validation runs first), and the LLM-free spellings still run
@@ -81,16 +92,16 @@ def test_optimize_full_keyless_reports_file_errors_first(tmp_path, monkeypatch):
assert "structure" in result
-def test_empty_outline_gate_carries_page_texts(tmp_path):
- """The gate's bookmark-built trees feed the same summary/expand passes
- as detected ones, so its result must carry the per-page text too."""
+def test_no_heading_result_carries_page_texts(tmp_path):
+ """A document that yields no headings still carries its per-page text,
+ so the page-node fallback and the summary/expand passes can read it."""
from conftest import build_pdf
from pageindex.flash.main import extract_toc
pdf = tmp_path / "doc.pdf"
pdf.write_bytes(build_pdf(["Alpha body", "Beta body"]))
result = extract_toc(str(pdf))
- assert result["structure"] == [] # the short-document gate fired
+ assert result["structure"] == [] # two body-only pages: nothing to detect
assert len(result["page_texts"]) == 2
assert "Alpha" in result["page_texts"][0]
@@ -420,7 +431,7 @@ def test_optimize_expand_warning_names_the_behavior_change(tmp_path,
SCRIPT = Path(__file__).resolve().parent.parent / "run_pageindex.py"
-def _run_flash_cli(monkeypatch, tmp_path, argv, structure):
+def _run_flash_cli(monkeypatch, tmp_path, argv, structure, toc_source=None):
"""Drive run_pageindex.py in-process with a stubbed flash indexer."""
import runpy
import sys
@@ -433,7 +444,10 @@ def _run_flash_cli(monkeypatch, tmp_path, argv, structure):
def fake_flash(path, **kw):
captured.update(kw)
- return {"structure": structure}
+ result = {"structure": structure}
+ if toc_source:
+ result["toc_source"] = toc_source
+ return result
monkeypatch.setattr(pageindex.flash, "page_index_flash", fake_flash)
monkeypatch.setattr(sys, "argv",
["run_pageindex.py", "--pdf_path", str(pdf), *argv])
@@ -467,3 +481,299 @@ def test_flash_cli_rejects_empty_structure(monkeypatch, tmp_path):
with pytest.raises(ValueError, match="try --mode standard"):
_run_flash_cli(monkeypatch, tmp_path, [], [])
assert not (tmp_path / "results").exists()
+
+
+def test_flash_cli_rejects_oversized_flat_tree(monkeypatch, tmp_path):
+ """The CLI applies the same refusal policy as the local client, worded
+ for its own flag."""
+ from pageindex.flash.api import FLAT_TREE_MAX_NODES
+
+ nodes = [{"title": f"Page {n}", "start_index": n, "end_index": n, "nodes": []}
+ for n in range(1, FLAT_TREE_MAX_NODES + 2)]
+ with pytest.raises(ValueError, match="no layout structure.*--mode standard"):
+ _run_flash_cli(monkeypatch, tmp_path, [], nodes, toc_source="pages")
+ assert not (tmp_path / "results").exists()
+
+
+def _layout_pdf(pages, landscape=False):
+ """Uncompressed PDF, per page a 20pt heading line then 11pt body lines:
+ ``pages`` is a list of ``(heading, [body line, ...])``."""
+ width, height = (792, 612) if landscape else (595, 842)
+ objs = {1: "<>", 3: "<>>>",
+ 5: "<>"}
+ kids, nxt = [], 6
+ for heading, lines in pages:
+ ops, y = [(20, height - 80, heading)], height - 120
+ for line in lines:
+ ops.append((11, y, line))
+ y -= 16
+ stream = "".join(f"BT /F1 {size} Tf 72 {top} Td ({text}) Tj ET\n"
+ for size, top, text in ops)
+ objs[nxt] = f"<>\nstream\n{stream}endstream"
+ objs[nxt + 1] = (f"<>")
+ kids.append(nxt + 1)
+ nxt += 2
+ objs[2] = (f"<>")
+ out, offsets = bytearray(b"%PDF-1.7\n"), {}
+ for num in sorted(objs):
+ offsets[num] = len(out)
+ out += f"{num} 0 obj\n{objs[num]}\nendobj\n".encode("latin-1")
+ xref = len(out)
+ out += f"xref\n0 {nxt}\n0000000000 65535 f \n".encode()
+ for num in range(1, nxt):
+ out += (f"{offsets[num]:010d} 00000 n \n" if num in offsets
+ else "0000000000 65535 f \n").encode()
+ out += (f"trailer\n<>\nstartxref\n{xref}\n"
+ "%%EOF\n").encode()
+ return bytes(out)
+
+
+DECK_TITLES = ["Revenue Overview", "Operating Expenses", "Customer Growth",
+ "Product Roadmap", "Regional Performance", "Engineering Metrics",
+ "Risk Factors", "Outlook and Guidance"]
+
+
+def _deck_pdf():
+ return _layout_pdf(
+ [(title, [f"Detail {n} for slide {slide} with a few more words of body text"
+ for n in range(1, 6)]) for slide, title in enumerate(DECK_TITLES, 1)],
+ landscape=True)
+
+
+def test_landscape_deck_is_structured(tmp_path):
+ """A text-light landscape deck is an ordinary document: every slide title
+ becomes a node."""
+ from pageindex.flash.main import extract_toc
+
+ pdf = tmp_path / "deck.pdf"
+ pdf.write_bytes(_deck_pdf())
+ result = extract_toc(str(pdf))
+ # slide 1 is spent on the document title, as on any title page
+ assert [node["title"] for node in result["structure"]] == DECK_TITLES[1:]
+ assert result["toc_source"] == "detected"
+
+
+def test_short_document_headings_are_detected(tmp_path):
+ """Three pages of heading-plus-a-line are enough for detection."""
+ from pageindex.flash.main import extract_toc
+
+ pdf = tmp_path / "memo.pdf"
+ pdf.write_bytes(_layout_pdf([("Mission", ["The launch is named Skylark."]),
+ ("Budget", ["The budget is 420 euros."]),
+ ("Team", ["The lead is Ada."])]))
+ result = extract_toc(str(pdf))
+ assert [node["title"] for node in result["structure"]] == ["Budget", "Team"]
+ assert result["doc_title"] == "Mission"
+
+
+def test_toc_source_is_always_present(tmp_path):
+ """The pure-detected path (no bookmark pass) labels its result too, so
+ callers can rely on the key."""
+ from pageindex.flash import page_index_flash
+
+ pdf = tmp_path / "memo.pdf"
+ pdf.write_bytes(_layout_pdf([("Mission", ["The launch is named Skylark."]),
+ ("Budget", ["The budget is 420 euros."]),
+ ("Team", ["The lead is Ada."])]))
+ result = page_index_flash(str(pdf), summary=False, optimize=False,
+ use_embedded_toc=False)
+ assert result["toc_source"] == "detected"
+ assert [node["title"] for node in result["structure"]] == [
+ "Preface", "Budget", "Team"]
+
+
+def test_no_hierarchy_falls_back_to_page_nodes(tmp_path):
+ """Two pages leave one heading after the title claims the other; with no
+ hierarchy to infer, the pages themselves are the tree, labelled as such."""
+ from pageindex.flash import page_index_flash
+
+ pdf = tmp_path / "two.pdf"
+ pdf.write_bytes(_layout_pdf([("Mission", ["The launch is named Skylark."]),
+ ("Budget", ["The budget is 420 euros."])]))
+ result = page_index_flash(str(pdf), summary=False, optimize=False)
+ assert result["toc_source"] == "pages"
+ assert result["structure"] == [
+ {"title": "Page 1", "node_id": "0000", "start_index": 1, "end_index": 1},
+ {"title": "Page 2", "node_id": "0001", "start_index": 2, "end_index": 2},
+ ]
+ assert "page_texts" not in result
+
+
+def test_flat_fallback_over_limit_skips_model_passes(tmp_path, monkeypatch):
+ """A flat tree larger than the managed pipelines accept is returned
+ unsummarized: neither optimize nor summary may spend model calls on it."""
+ from conftest import build_pdf
+ from pageindex.flash import api as flash_api
+
+ monkeypatch.setattr(flash_api, "FLAT_TREE_MAX_NODES", 2)
+ monkeypatch.setattr(flash_api, "_optimize", lambda *a, **k: pytest.fail(
+ "optimize ran on a refused flat tree"))
+ monkeypatch.setattr(flash_api, "_summarize", lambda *a, **k: pytest.fail(
+ "summary ran on a refused flat tree"))
+ pdf = tmp_path / "letter.pdf"
+ pdf.write_bytes(build_pdf(["Alpha body", "Beta body", "Gamma body"]))
+ result = flash_api.page_index_flash(str(pdf), summary=True, summary_model="m")
+ assert result["toc_source"] == "pages"
+ assert [node["title"] for node in result["structure"]] == [
+ "Page 1", "Page 2", "Page 3"]
+ assert "page_texts" not in result
+
+
+def test_textless_pdf_is_unreadable(tmp_path):
+ """A PDF with no text on any page has nothing to index: an empty
+ structure labelled unreadable, no page nodes."""
+ from conftest import build_pdf
+ from pageindex.flash import page_index_flash
+
+ pdf = tmp_path / "scan.pdf"
+ pdf.write_bytes(build_pdf(["", ""]))
+ result = page_index_flash(str(pdf), summary=False, optimize=False)
+ assert result["structure"] == []
+ assert result["toc_source"] == "unreadable"
+
+
+def test_flash_rejection_reason():
+ from pageindex.flash.api import FLAT_TREE_MAX_NODES, flash_rejection_reason
+
+ node = {"title": "T", "start_index": 1, "end_index": 1, "nodes": []}
+ assert flash_rejection_reason(
+ {"structure": [node], "toc_source": "detected"}) is None
+ assert flash_rejection_reason(
+ {"structure": [node] * 2, "toc_source": "pages"}) is None
+ unreadable = flash_rejection_reason({"structure": [], "toc_source": "unreadable"})
+ assert "no text layer" in unreadable and "standard" not in unreadable
+ oversized = {"structure": [node] * (FLAT_TREE_MAX_NODES + 1),
+ "toc_source": "pages"}
+ flat = flash_rejection_reason(oversized)
+ assert "no layout structure" in flat and "mode='standard'" in flat
+ assert "--mode standard" in flash_rejection_reason(
+ oversized, standard_hint="--mode standard")
+ assert "could not extract" in flash_rejection_reason({"structure": []})
+
+
+FLASH_DATA = Path(__file__).parent / "data" / "flash" # see its make_fixtures.py
+
+
+def test_page_fallback_covers_every_page(tmp_path):
+ """Pages without text still get a node: the flat tree covers the whole
+ document, so no page is unreachable."""
+ from conftest import build_pdf
+ from pageindex.flash import page_index_flash
+
+ pdf = tmp_path / "sparse.pdf"
+ pdf.write_bytes(build_pdf(["", "Alpha body", "", "Delta body"]))
+ result = page_index_flash(str(pdf), summary=False, optimize=False)
+ assert result["toc_source"] == "pages"
+ assert [(n["title"], n["start_index"], n["end_index"])
+ for n in result["structure"]] == [
+ ("Page 1", 1, 1), ("Page 2", 2, 2), ("Page 3", 3, 3), ("Page 4", 4, 4)]
+ assert all("nodes" not in n for n in result["structure"])
+
+
+def test_page_nodes_are_leaves(tmp_path):
+ """``get_leaf_nodes`` takes a flat page tree, whose nodes carry no ``nodes`` key."""
+ from conftest import build_pdf
+ from pageindex import get_leaf_nodes
+ from pageindex.flash import page_index_flash
+
+ pdf = tmp_path / "flat.pdf"
+ pdf.write_bytes(build_pdf(["Alpha body", "Beta body"]))
+ structure = page_index_flash(str(pdf), summary=False, optimize=False)["structure"]
+ assert [n["title"] for n in get_leaf_nodes(structure)] == ["Page 1", "Page 2"]
+
+
+def test_every_page_is_in_a_node(tmp_path):
+ """Top-level ranges cover the whole document: a hierarchy that starts after
+ page 1 is preceded by a Preface node, as in standard mode."""
+ import pypdfium2 as pdfium
+ from pageindex.flash import page_index_flash
+
+ memo = tmp_path / "memo.pdf"
+ memo.write_bytes(_layout_pdf([("Mission", ["The launch is named Skylark."]),
+ ("Budget", ["The budget is 420 euros."]),
+ ("Team", ["The lead is Ada."])]))
+ deck = tmp_path / "deck.pdf"
+ deck.write_bytes(_deck_pdf())
+ for pdf in [memo, deck, *sorted(FLASH_DATA.glob("*.pdf"))]:
+ document = pdfium.PdfDocument(str(pdf))
+ pages = len(document)
+ document.close()
+ structure = page_index_flash(str(pdf), summary=False, optimize=False)["structure"]
+ covered = {page for node in structure
+ for page in range(node["start_index"], node["end_index"] + 1)}
+ assert covered == set(range(1, pages + 1)), pdf.name
+ assert structure[0] == {"title": "Preface", "node_id": "0000",
+ "start_index": 1, "end_index": 1}, pdf.name
+ assert structure[1]["node_id"] == "0001", pdf.name
+
+
+def test_preface_page_is_retrievable(tmp_path, monkeypatch):
+ """The page a late-starting hierarchy skips reaches the client's tree text."""
+ import pageindex.utils
+ from pageindex import PageIndexClient
+ from pageindex.flash import api as flash_api
+
+ async def no_summary(*args, **kwargs):
+ return None
+ monkeypatch.setattr(flash_api, "_optimize", lambda *a, **k: {"merges": 0})
+ monkeypatch.setattr(flash_api, "_summarize", no_summary)
+ monkeypatch.setattr(pageindex.utils, "llm_completion",
+ lambda model, prompt, **kw: "A memo.")
+ pdf = tmp_path / "memo.pdf"
+ pdf.write_bytes(_layout_pdf([("Mission", ["The launch is named Skylark."]),
+ ("Budget", ["The budget is 420 euros."]),
+ ("Team", ["The lead is Ada."])]))
+ client = PageIndexClient(storage_path=str(tmp_path / "store"))
+ doc_id = client.submit_document(str(pdf), mode="flash")["doc_id"]
+ tree = client.get_tree(doc_id)["result"]
+ assert [(node["title"], node["page_index"]) for node in tree] == [
+ ("Preface", 1), ("Budget", 2), ("Team", 3)]
+ assert "Skylark" in tree[0]["text"]
+
+
+@pytest.mark.parametrize("name", ["hi_report.pdf", "ar_report.pdf"])
+def test_non_latin_document_is_indexed(name):
+ """Script never decides whether a document is indexable."""
+ from pageindex.flash import page_index_flash
+ from pageindex.flash.api import flash_rejection_reason
+
+ result = page_index_flash(str(FLASH_DATA / name), summary=False, optimize=False)
+ assert result["structure"]
+ assert flash_rejection_reason(result) is None
+
+
+def test_japanese_headings_are_kept():
+ from pageindex.flash.main import extract_toc
+
+ result = extract_toc(str(FLASH_DATA / "ja_report.pdf"), use_embedded_toc=False)
+ assert [n["title"] for n in result["structure"]] == [
+ "財務ハイライト", "リスク要因", "今後の見通し"]
+
+
+def test_cross_script_headings_are_kept():
+ from pageindex.flash.main import extract_toc
+
+ result = extract_toc(str(FLASH_DATA / "zh_body_en_headings.pdf"),
+ use_embedded_toc=False)
+ assert [n["title"] for n in result["structure"]] == [
+ "Financial Review", "Risk Factors", "Business Outlook"]
+
+
+@pytest.mark.parametrize("name", ["hi_report.pdf", "ar_report.pdf"])
+def test_non_latin_headings_are_detected(name):
+ from pageindex.flash.main import extract_toc
+
+ result = extract_toc(str(FLASH_DATA / name), use_embedded_toc=False)
+ assert result["toc_source"] == "detected"
+ assert len(result["structure"]) == 3
+
+
+def test_landscape_deck_title_is_the_slide_heading(tmp_path):
+ from pageindex.flash.main import extract_toc
+
+ pdf = tmp_path / "deck.pdf"
+ pdf.write_bytes(_deck_pdf())
+ assert extract_toc(str(pdf))["doc_title"] == DECK_TITLES[0]