Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 14 additions & 6 deletions pageindex/client.py
Original file line number Diff line number Diff line change
Expand Up @@ -949,14 +949,22 @@ def get_tree(self, doc_id: str, node_summary: bool = False,
Returns:
dict: {'doc_id', 'status', 'retrieval_ready', 'result', ...} where
result nodes are {'title', 'node_id', 'page_index', ('summary' /
'prefix_summary',) ('text',) 'nodes'}.
result nodes are {'title', 'node_id', 'start_index', 'end_index',
('summary',) ('text',) 'nodes'} in both modes. Each node's summary
describes its own pages start_index..end_index; its text is the
part no child holds, though a parent's may share the page its
first child starts on.
"""
tree = self._api.get_tree(doc_id=doc_id, node_summary=node_summary,
include_text=include_text)
if not include_text and tree.get("result"):
from .utils import remove_fields
tree["result"] = remove_fields(tree["result"], fields=["text"])
if tree.get("result"):
from .utils import _subtree, remove_fields, unify_tree
page_count = None
if any("end_index" not in node for node in _subtree(tree["result"])):
page_count = self._api.get_document(doc_id).get("pageNum")
tree["result"] = unify_tree(tree["result"], page_count)
if not include_text:
tree["result"] = remove_fields(tree["result"], fields=["text"])
return tree

def get_document_structure(self, doc_id: str) -> list[dict[str, Any]]:
Expand All @@ -975,7 +983,7 @@ def is_retrieval_ready(self, doc_id: str) -> bool:
failures, timeouts) propagate.
"""
try:
result = self.get_tree(doc_id)
result = self._api.get_tree(doc_id)
return result.get("retrieval_ready", False)
except PageIndexAPIError:
return False
Expand Down
35 changes: 26 additions & 9 deletions pageindex/flash/api.py
Original file line number Diff line number Diff line change
Expand Up @@ -89,9 +89,7 @@ async def _optimize_async(structure, page_texts, do_expand, model, on_final=None
the summaries use.
"""
from ..tree_optimize import optimize
lines = [[line_text.strip() for line_text in (page_text or "").splitlines()
if line_text.strip()]
for page_text in page_texts]
lines = _page_lines(page_texts)
outcome = await optimize(structure, page_texts, lines, model=model,
do_expand=do_expand, page_count=len(page_texts),
on_final=on_final, concurrency=concurrency)
Expand Down Expand Up @@ -134,14 +132,31 @@ def _page_nodes(page_texts: list[str]) -> list[dict]:
return nodes


def _add_preface(structure: list[dict]) -> None:
"""The pages before a hierarchy that starts late become a Preface node, as in standard mode."""
def _page_lines(page_texts: list[str]) -> list[list[str]]:
return [[line_text.strip() for line_text in (page_text or "").splitlines()
if line_text.strip()]
for page_text in page_texts]


def _add_intros(structure: list[dict], lines: list[list[str]]) -> None:
"""The pages a parent opens with before its first child become its intro node."""
from ..tree_optimize import add_intro_nodes
from ..utils import write_node_id
structure.insert(0, {"title": "Preface", "start_index": 1,
"end_index": structure[0]["start_index"] - 1})
add_intro_nodes(structure, lines)
write_node_id(structure)


def _add_preface(structure: list[dict], lines: list[list[str]]) -> None:
"""The pages before a hierarchy that starts late become a Preface node, as in
standard mode. It runs onto the first section's page unless that heading opens it."""
from ..tree_optimize import heading_at_page_start
first = structure[0]
opens = first["start_index"] <= len(lines) and heading_at_page_start(
lines, first["start_index"], first["title"])
structure.insert(0, {"title": "Preface", "start_index": 1,
"end_index": first["start_index"] - (1 if opens else 0)})


def flash_rejection_reason(result: dict, standard_hint: str = "mode='standard'") -> str | None:
"""Why a managed pipeline should refuse this flash result, or None to accept it.
Expand All @@ -166,7 +181,7 @@ def page_index_flash(pdf, summary=True, summary_model=None,
optimize: str | bool | None = None, optimize_expand=None,
optimize_model=None, summary_concurrency=None,
use_embedded_toc=True, summary_max_words=None) -> dict:
"""Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """
"""Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode, and the first section's page too unless that heading opens it; a parent whose first child starts on a later page opens with a child titled ``"<parent title> (intro)"`` holding the pages before it; a parent's range and summary cover its whole subtree) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """
for name, value in (("summary_concurrency", summary_concurrency),
("summary_max_words", summary_max_words)):
if value is not None and not (isinstance(value, numbers.Integral) and int(value) >= 1):
Expand Down Expand Up @@ -195,7 +210,9 @@ def page_index_flash(pdf, summary=True, summary_model=None,
result["structure"] = structure
result["toc_source"] = "pages" if structure else "unreadable"
elif structure[0]["start_index"] > 1:
_add_preface(structure)
_add_preface(structure, _page_lines(result.get("page_texts") or []))
if structure:
_add_intros(structure, _page_lines(result.get("page_texts") or []))
if result.get("toc_source") == "pages" and len(structure) > FLAT_TREE_MAX_NODES:
# the managed pipelines refuse a flat tree this size; skip the model passes
result.pop("page_texts", None)
Expand Down
12 changes: 5 additions & 7 deletions pageindex/local_api.py
Original file line number Diff line number Diff line change
Expand Up @@ -260,6 +260,7 @@ def _index_flash(self, file_path: str) -> tuple[list, str | None]:
# ── tree / ocr ──

def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list:
"""Each node's own text (see utils.own_pages)."""
from .utils import add_node_text
structure = self._require_data(
self._store.get_tree(doc_id), error_prefix)
Expand All @@ -269,8 +270,7 @@ def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list:
return structure

def raw_tree(self, doc_id: str) -> list | None:
"""Stored tree verbatim — keeps start_index/end_index, which
get_tree's cloud wire shape renames and drops."""
"""Stored tree verbatim, every key kept."""
return self._store.get_tree(doc_id)

def get_tree(self, doc_id: str, node_summary: bool = False,
Expand Down Expand Up @@ -398,17 +398,15 @@ def _format_tree_node(node: dict, node_summary: bool) -> dict:
out = {
"title": node.get("title", ""),
"node_id": node.get("node_id"),
"page_index": node.get("start_index"),
"start_index": node.get("start_index"),
"end_index": node.get("end_index"),
}
if node.get("key_items"):
out["key_items"] = node["key_items"]
if node_summary:
summary = node.get("summary")
if summary is not None:
if children:
out["prefix_summary"] = summary
else:
out["summary"] = summary
out["summary"] = summary
if "text" in node:
out["text"] = node["text"]
if children:
Expand Down
24 changes: 13 additions & 11 deletions pageindex/page_index_classic.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@
import re
from .utils import *
from .naming import sanitize_filename as sanitize_upload_filename
from .tree_optimize import merge_tree
from .tree_optimize import add_intro_nodes, merge_tree
import os

######################### Hardening for prompt injection patterns ####################################################
Expand Down Expand Up @@ -1169,7 +1169,8 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None)
node_page_list = page_list[node['start_index']-1:node['end_index']]
token_num = sum([page[1] for page in node_page_list])

if node['end_index'] - node['start_index'] > opt.max_page_num_each_node and token_num >= opt.max_token_num_each_node:
if (not node.get('nodes') and node['end_index'] - node['start_index'] > opt.max_page_num_each_node
and token_num >= opt.max_token_num_each_node):
print('large node:', node['title'], 'start_index:', node['start_index'], 'end_index:', node['end_index'], 'token_num:', token_num)

node_toc_tree = await meta_processor(node_page_list, mode='process_no_toc', start_index=node['start_index'], opt=opt, logger=logger)
Expand All @@ -1178,7 +1179,9 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None)
# Filter out items with None physical_index before post_processing
valid_node_toc_items = [item for item in node_toc_tree if item.get('physical_index') is not None]

if valid_node_toc_items and node['title'].strip() == valid_node_toc_items[0]['title'].strip():
# an intro node's pages open with its parent's heading
title = node['title'].strip()
if valid_node_toc_items and valid_node_toc_items[0]['title'].strip() in (title, title.removesuffix(INTRO_SUFFIX)):
node['nodes'] = post_processing(valid_node_toc_items[1:], node['end_index'])
node['end_index'] = valid_node_toc_items[1]['start_index'] if len(valid_node_toc_items) > 1 else node['end_index']
else:
Expand Down Expand Up @@ -1221,14 +1224,15 @@ async def tree_parser(page_list, opt, doc=None, logger=None):
# Filter out items with None physical_index before post_processings
valid_toc_items = [item for item in toc_with_page_number if item.get('physical_index') is not None]

toc_tree = post_processing(valid_toc_items, len(page_list))
toc_tree = add_intro_nodes(post_processing(valid_toc_items, len(page_list)))
tasks = [
process_large_node_recursively(node, page_list, opt, logger=logger)
for node in toc_tree
]
await asyncio.gather(*tasks)

return toc_tree

# a split leaf is now a parent that may open before its first child
return add_intro_nodes(toc_tree)


def page_index_main(doc, opt=None, logger=None, page_list=None):
Expand Down Expand Up @@ -1257,21 +1261,19 @@ async def page_index_builder():
if opt.if_add_node_text == 'yes':
add_node_text(structure, page_list)
if opt.if_add_node_summary == 'yes':
if opt.if_add_node_text == 'no':
add_node_text(structure, page_list)
await generate_summaries_for_structure(structure, model=getattr(opt, 'summary_model', None) or opt.model)
if opt.if_add_node_text == 'no':
remove_structure_text(structure)
await summarize_tree(structure, page_list, model=getattr(opt, 'summary_model', None) or opt.model)
if opt.if_add_doc_description == 'yes':
# Create a clean structure without unnecessary fields for description generation
clean_structure = create_clean_structure_for_description(structure)
doc_description = generate_doc_description(clean_structure, model=getattr(opt, 'summary_model', None) or opt.model)
cover_subtree_ranges(structure)
structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes'])
return {
'doc_name': get_pdf_name(doc),
'doc_description': doc_description,
'structure': structure,
}
cover_subtree_ranges(structure)
structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes'])
return {
'doc_name': get_pdf_name(doc),
Expand Down
Loading
Loading