diff --git a/pageindex/client.py b/pageindex/client.py index 4dea417c1..0d77269bf 100644 --- a/pageindex/client.py +++ b/pageindex/client.py @@ -949,14 +949,22 @@ def get_tree(self, doc_id: str, node_summary: bool = False, Returns: dict: {'doc_id', 'status', 'retrieval_ready', 'result', ...} where - result nodes are {'title', 'node_id', 'page_index', ('summary' / - 'prefix_summary',) ('text',) 'nodes'}. + result nodes are {'title', 'node_id', 'start_index', 'end_index', + ('summary',) ('text',) 'nodes'} in both modes. Each node's summary + describes its own pages start_index..end_index; its text is the + part no child holds, though a parent's may share the page its + first child starts on. """ tree = self._api.get_tree(doc_id=doc_id, node_summary=node_summary, include_text=include_text) - if not include_text and tree.get("result"): - from .utils import remove_fields - tree["result"] = remove_fields(tree["result"], fields=["text"]) + if tree.get("result"): + from .utils import _subtree, remove_fields, unify_tree + page_count = None + if any("end_index" not in node for node in _subtree(tree["result"])): + page_count = self._api.get_document(doc_id).get("pageNum") + tree["result"] = unify_tree(tree["result"], page_count) + if not include_text: + tree["result"] = remove_fields(tree["result"], fields=["text"]) return tree def get_document_structure(self, doc_id: str) -> list[dict[str, Any]]: @@ -975,7 +983,7 @@ def is_retrieval_ready(self, doc_id: str) -> bool: failures, timeouts) propagate. """ try: - result = self.get_tree(doc_id) + result = self._api.get_tree(doc_id) return result.get("retrieval_ready", False) except PageIndexAPIError: return False diff --git a/pageindex/flash/api.py b/pageindex/flash/api.py index bc8c71446..5e3e397e1 100644 --- a/pageindex/flash/api.py +++ b/pageindex/flash/api.py @@ -89,9 +89,7 @@ async def _optimize_async(structure, page_texts, do_expand, model, on_final=None the summaries use. """ from ..tree_optimize import optimize - lines = [[line_text.strip() for line_text in (page_text or "").splitlines() - if line_text.strip()] - for page_text in page_texts] + lines = _page_lines(page_texts) outcome = await optimize(structure, page_texts, lines, model=model, do_expand=do_expand, page_count=len(page_texts), on_final=on_final, concurrency=concurrency) @@ -134,14 +132,31 @@ def _page_nodes(page_texts: list[str]) -> list[dict]: return nodes -def _add_preface(structure: list[dict]) -> None: - """The pages before a hierarchy that starts late become a Preface node, as in standard mode.""" +def _page_lines(page_texts: list[str]) -> list[list[str]]: + return [[line_text.strip() for line_text in (page_text or "").splitlines() + if line_text.strip()] + for page_text in page_texts] + + +def _add_intros(structure: list[dict], lines: list[list[str]]) -> None: + """The pages a parent opens with before its first child become its intro node.""" + from ..tree_optimize import add_intro_nodes from ..utils import write_node_id - structure.insert(0, {"title": "Preface", "start_index": 1, - "end_index": structure[0]["start_index"] - 1}) + add_intro_nodes(structure, lines) write_node_id(structure) +def _add_preface(structure: list[dict], lines: list[list[str]]) -> None: + """The pages before a hierarchy that starts late become a Preface node, as in + standard mode. It runs onto the first section's page unless that heading opens it.""" + from ..tree_optimize import heading_at_page_start + first = structure[0] + opens = first["start_index"] <= len(lines) and heading_at_page_start( + lines, first["start_index"], first["title"]) + structure.insert(0, {"title": "Preface", "start_index": 1, + "end_index": first["start_index"] - (1 if opens else 0)}) + + def flash_rejection_reason(result: dict, standard_hint: str = "mode='standard'") -> str | None: """Why a managed pipeline should refuse this flash result, or None to accept it. @@ -166,7 +181,7 @@ def page_index_flash(pdf, summary=True, summary_model=None, optimize: str | bool | None = None, optimize_expand=None, optimize_model=None, summary_concurrency=None, use_embedded_toc=True, summary_max_words=None) -> dict: - """Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """ + """Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode, and the first section's page too unless that heading opens it; a parent whose first child starts on a later page opens with a child titled ``" (intro)"`` holding the pages before it; a parent's range and summary cover its whole subtree) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """ for name, value in (("summary_concurrency", summary_concurrency), ("summary_max_words", summary_max_words)): if value is not None and not (isinstance(value, numbers.Integral) and int(value) >= 1): @@ -195,7 +210,9 @@ def page_index_flash(pdf, summary=True, summary_model=None, result["structure"] = structure result["toc_source"] = "pages" if structure else "unreadable" elif structure[0]["start_index"] > 1: - _add_preface(structure) + _add_preface(structure, _page_lines(result.get("page_texts") or [])) + if structure: + _add_intros(structure, _page_lines(result.get("page_texts") or [])) if result.get("toc_source") == "pages" and len(structure) > FLAT_TREE_MAX_NODES: # the managed pipelines refuse a flat tree this size; skip the model passes result.pop("page_texts", None) diff --git a/pageindex/local_api.py b/pageindex/local_api.py index 025ab716d..4802634b8 100644 --- a/pageindex/local_api.py +++ b/pageindex/local_api.py @@ -260,6 +260,7 @@ def _index_flash(self, file_path: str) -> tuple[list, str | None]: # ── tree / ocr ── def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: + """Each node's own text (see utils.own_pages).""" from .utils import add_node_text structure = self._require_data( self._store.get_tree(doc_id), error_prefix) @@ -269,8 +270,7 @@ def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: return structure def raw_tree(self, doc_id: str) -> list | None: - """Stored tree verbatim — keeps start_index/end_index, which - get_tree's cloud wire shape renames and drops.""" + """Stored tree verbatim, every key kept.""" return self._store.get_tree(doc_id) def get_tree(self, doc_id: str, node_summary: bool = False, @@ -398,17 +398,15 @@ def _format_tree_node(node: dict, node_summary: bool) -> dict: out = { "title": node.get("title", ""), "node_id": node.get("node_id"), - "page_index": node.get("start_index"), + "start_index": node.get("start_index"), + "end_index": node.get("end_index"), } if node.get("key_items"): out["key_items"] = node["key_items"] if node_summary: summary = node.get("summary") if summary is not None: - if children: - out["prefix_summary"] = summary - else: - out["summary"] = summary + out["summary"] = summary if "text" in node: out["text"] = node["text"] if children: diff --git a/pageindex/page_index_classic.py b/pageindex/page_index_classic.py index d680154eb..7788e236b 100644 --- a/pageindex/page_index_classic.py +++ b/pageindex/page_index_classic.py @@ -6,7 +6,7 @@ import re from .utils import * from .naming import sanitize_filename as sanitize_upload_filename -from .tree_optimize import merge_tree +from .tree_optimize import add_intro_nodes, merge_tree import os ######################### Hardening for prompt injection patterns #################################################### @@ -1169,7 +1169,8 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None) node_page_list = page_list[node['start_index']-1:node['end_index']] token_num = sum([page[1] for page in node_page_list]) - if node['end_index'] - node['start_index'] > opt.max_page_num_each_node and token_num >= opt.max_token_num_each_node: + if (not node.get('nodes') and node['end_index'] - node['start_index'] > opt.max_page_num_each_node + and token_num >= opt.max_token_num_each_node): print('large node:', node['title'], 'start_index:', node['start_index'], 'end_index:', node['end_index'], 'token_num:', token_num) node_toc_tree = await meta_processor(node_page_list, mode='process_no_toc', start_index=node['start_index'], opt=opt, logger=logger) @@ -1178,7 +1179,9 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None) # Filter out items with None physical_index before post_processing valid_node_toc_items = [item for item in node_toc_tree if item.get('physical_index') is not None] - if valid_node_toc_items and node['title'].strip() == valid_node_toc_items[0]['title'].strip(): + # an intro node's pages open with its parent's heading + title = node['title'].strip() + if valid_node_toc_items and valid_node_toc_items[0]['title'].strip() in (title, title.removesuffix(INTRO_SUFFIX)): node['nodes'] = post_processing(valid_node_toc_items[1:], node['end_index']) node['end_index'] = valid_node_toc_items[1]['start_index'] if len(valid_node_toc_items) > 1 else node['end_index'] else: @@ -1221,14 +1224,15 @@ async def tree_parser(page_list, opt, doc=None, logger=None): # Filter out items with None physical_index before post_processings valid_toc_items = [item for item in toc_with_page_number if item.get('physical_index') is not None] - toc_tree = post_processing(valid_toc_items, len(page_list)) + toc_tree = add_intro_nodes(post_processing(valid_toc_items, len(page_list))) tasks = [ process_large_node_recursively(node, page_list, opt, logger=logger) for node in toc_tree ] await asyncio.gather(*tasks) - - return toc_tree + + # a split leaf is now a parent that may open before its first child + return add_intro_nodes(toc_tree) def page_index_main(doc, opt=None, logger=None, page_list=None): @@ -1257,21 +1261,19 @@ async def page_index_builder(): if opt.if_add_node_text == 'yes': add_node_text(structure, page_list) if opt.if_add_node_summary == 'yes': - if opt.if_add_node_text == 'no': - add_node_text(structure, page_list) - await generate_summaries_for_structure(structure, model=getattr(opt, 'summary_model', None) or opt.model) - if opt.if_add_node_text == 'no': - remove_structure_text(structure) + await summarize_tree(structure, page_list, model=getattr(opt, 'summary_model', None) or opt.model) if opt.if_add_doc_description == 'yes': # Create a clean structure without unnecessary fields for description generation clean_structure = create_clean_structure_for_description(structure) doc_description = generate_doc_description(clean_structure, model=getattr(opt, 'summary_model', None) or opt.model) + cover_subtree_ranges(structure) structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes']) return { 'doc_name': get_pdf_name(doc), 'doc_description': doc_description, 'structure': structure, } + cover_subtree_ranges(structure) structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes']) return { 'doc_name': get_pdf_name(doc), diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index ddf397c7d..e607d5f00 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -62,8 +62,8 @@ import sys from types import SimpleNamespace -from .utils import (ConfigLoader, _is_unrecoverable, llm_acompletion, - strip_internal_keys) +from .utils import (ConfigLoader, _is_unrecoverable, intro_title, is_intro, + llm_acompletion, strip_internal_keys) TRIGGER_PAGES = 5 # only look ahead on nodes larger than this ROUTING_COST = 1 # R(v), in pages @@ -171,11 +171,13 @@ def is_frontier(node): def heading_at_page_start(lines, page_no, heading): - """Is the heading the first line on its page?""" + """Is the heading the first line on its page? No when that cannot be told + (a heading with no Latin letter or digit to match), so the page is shared.""" page = lines[page_no - 1] - if not page: + key = normalize(heading) + if not page or not key: return False - return normalize(heading) in normalize(page[0]) + return key in normalize(page[0]) def assign_ends(node, children, lines): @@ -202,9 +204,30 @@ def attach_children(node, children, lines): # so gaining children leaves it unchanged sized = assign_ends(node, children, lines) node["nodes"] = sized + add_intro_nodes([node], lines) + if len(node["nodes"]) > len(sized): # an intro went in; numbered like the proposals + node["nodes"][0]["node_id"] = f"{node['node_id']}.0" return sized +def add_intro_nodes(structure, lines=None): + """Give every parent whose first child starts on a later page an intro child + for the pages before it, titled " (intro)". It ends where the + parent's own pages do, and no later than the first child's page, the page + before it when that child's heading opens its page.""" + for node, _ in list(flatten(structure)): + children = node.get("nodes") or [] + first = children[0]["start_index"] if children else None + if first is None or first <= node["start_index"]: + continue + opens = bool(lines) and first <= len(lines) and heading_at_page_start( + lines, first, children[0]["title"]) + end = max(node["start_index"], min(node["end_index"], first - 1 if opens else first)) + node["nodes"] = [{"title": intro_title(node.get("title")), + "start_index": node["start_index"], "end_index": end}] + children + return structure + + def relabel(structure, width=4): """Renumber every node_id in document order: 0000, 0001, 0002, ... @@ -558,8 +581,9 @@ def visit(node): # titles are routing information; keep them on the parent, in document # order, carrying forward anything an earlier merge already folded in titles = [] - for child, _ in flatten(node["nodes"]): - titles.append(child["title"]) + for child, parent in flatten(node["nodes"]): + if not is_intro(parent or node, child): # its parent's heading again + titles.append(child["title"]) titles.extend(child.get("key_items") or []) log.append({"op": "merge", "node_id": node.get("node_id"), "S": span, "tree_cost": cost, "frontier_cost": checked, diff --git a/pageindex/utils.py b/pageindex/utils.py index 269218117..775e1f68a 100644 --- a/pageindex/utils.py +++ b/pageindex/utils.py @@ -708,8 +708,7 @@ def convert_page_to_int(data): def add_node_text(node, pdf_pages): if isinstance(node, dict): - start_page = node.get('start_index') - end_page = node.get('end_index') + start_page, end_page = own_pages(node) node['text'] = get_text_of_pdf_pages(pdf_pages, start_page, end_page) if 'nodes' in node: add_node_text(node['nodes'], pdf_pages) @@ -721,8 +720,7 @@ def add_node_text(node, pdf_pages): def add_node_text_with_labels(node, pdf_pages): if isinstance(node, dict): - start_page = node.get('start_index') - end_page = node.get('end_index') + start_page, end_page = own_pages(node) node['text'] = get_text_of_pdf_pages_with_labels(pdf_pages, start_page, end_page) if 'nodes' in node: add_node_text_with_labels(node['nodes'], pdf_pages) @@ -743,6 +741,63 @@ async def generate_node_summary(node, model=None): return response +FALLBACK_SUMMARY_CHARS = 600 + + +def fallback_summary(node, text=""): + """The summary of a node the model left unsummarized, so no node goes + without one: a parent's subsection titles, a leaf's own text.""" + children = node.get('nodes') or [] + if children: + summary = "; ".join(child['title'] for child in children if child.get('title')) + else: + summary = " ".join(text.split()) + return summary[:FALLBACK_SUMMARY_CHARS] or node.get('title') or "" + + +INTRO_SUFFIX = " (intro)" + + +def intro_title(title): + """The title of a parent's intro node; a split intro keeps its own.""" + title = (title or "").strip() + if title and not title.endswith(INTRO_SUFFIX): + title += INTRO_SUFFIX + return title or "Intro" + + +def is_intro(parent, child): + """Whether child is the intro node holding parent's opening pages.""" + return (child.get('start_index') == parent.get('start_index') + and child.get('title') == intro_title(parent.get('title'))) + + +def own_pages(node): + """The first and last page of a node's own text. A parent's runs onto the + page its first child starts on, where that child's heading may sit mid-page; + it has none (end None) when its intro holds those pages.""" + start, end = node.get('start_index'), node.get('end_index') + children = node.get('nodes') or [] + if children: + first = children[0].get('start_index') + if is_intro(node, children[0]): + end = None + elif end is not None and first is not None: + end = min(end, first) + return start, end + + +def cover_subtree_ranges(structure): + """Widen every parent's range to cover its whole subtree.""" + for node in structure if isinstance(structure, list) else [structure]: + children = node.get('nodes') or [] + if children: + cover_subtree_ranges(children) + node['start_index'] = min([node['start_index']] + [c['start_index'] for c in children]) + node['end_index'] = max([node['end_index']] + [c['end_index'] for c in children]) + return structure + + async def generate_summaries_for_structure(structure, model=None): nodes = structure_to_list(structure) tasks = [generate_node_summary(node, model=model) for node in nodes] @@ -1040,6 +1095,9 @@ async def _visit(self, node, depth): node['summary'] = "" if _is_unrecoverable(e): raise + if not node['summary']: + node['summary'] = fallback_summary(node, "" if children else get_text_of_pdf_pages( + self._pdf_pages, node['start_index'], node['end_index'])) async def finish(self): """Wait for every summary; fails loud if the model never answered.""" @@ -1252,9 +1310,58 @@ def load(self, user_opt=None) -> config: _resolve_models(merged) return config(**merged) +_TREE_WIRE_KEYS = frozenset({"title", "node_id", "start_index", "end_index", "page_index", + "summary", "prefix_summary", "text", "nodes"}) + + +def unify_tree(nodes, page_count=None): + """get_tree's nodes in the field names both modes return. + + The structure is the index's own: nodes are neither added nor moved, so + every node keeps the summary written for its own pages start_index .. + end_index. A node's first page comes as start_index (local) or + page_index (cloud), and a parent's summary as summary or prefix_summary. + Without an end_index (an older server) a node ends on the page where the + next node in reading order starts, the last on page_count. + """ + flat = list(_subtree(nodes)) + starts = [node.get("start_index", node.get("page_index")) for node in flat] + ranges = {} + for i, node in enumerate(flat): + start, end = starts[i], node.get("end_index") + if end is None: + end = starts[i + 1] if i + 1 < len(flat) else page_count + if end is None or (start is not None and end < start): + end = start + ranges[id(node)] = (start, end) + + def build(siblings): + out = [] + for node in siblings: + unified = {"title": node.get("title", "")} + if "node_id" in node: + unified["node_id"] = node["node_id"] + unified["start_index"], unified["end_index"] = ranges[id(node)] + unified.update((key, value) for key, value in node.items() + if key not in _TREE_WIRE_KEYS) + for key in ("summary", "prefix_summary"): + if key in node: + unified["summary"] = node[key] + if "text" in node: + unified["text"] = node["text"] + if node.get("nodes"): + unified["nodes"] = build(node["nodes"]) + out.append(unified) + return out + + return build(nodes) + + def create_node_mapping(tree, include_page_ranges=False, max_page=None): """Map node_id to node; with include_page_ranges, to {"node", "start_index", - "end_index"} (end = next node's page_index, or max_page for the last node).""" + "end_index"}: the node's own range as get_tree returns it, or, for a tree + that carries page_index only, end = next node's page_index (max_page for + the last node).""" def get_all_nodes(tree): if isinstance(tree, dict): return [tree] + [node for child in tree.get('nodes', []) for node in get_all_nodes(child)] @@ -1268,10 +1375,14 @@ def get_all_nodes(tree): mapping = {} for i, node in enumerate(all_nodes): if node.get("node_id"): - end_page = all_nodes[i + 1].get("page_index") if i + 1 < len(all_nodes) else max_page + if "end_index" in node: + start_page, end_page = node["start_index"], node["end_index"] + else: + start_page = node["page_index"] + end_page = all_nodes[i + 1].get("page_index") if i + 1 < len(all_nodes) else max_page mapping[node["node_id"]] = { "node": node, - "start_index": node["page_index"], + "start_index": start_page, "end_index": end_page, } return mapping diff --git a/tests/test_client.py b/tests/test_client.py index d8eec89b9..8bffc36a4 100644 --- a/tests/test_client.py +++ b/tests/test_client.py @@ -549,24 +549,24 @@ def test_submit_and_get_tree(local_client, indexed_doc, tmp_path, monkeypatch): assert tree["status"] == "completed" assert tree["retrieval_ready"] is True root = tree["result"][0] - assert root["page_index"] == 1 - assert "start_index" not in root and "end_index" not in root - assert root["prefix_summary"] == "root summary" - assert "summary" not in root - child = root["nodes"][0] + assert (root["start_index"], root["end_index"]) == (1, 2) + assert "page_index" not in root and "prefix_summary" not in root + assert root["summary"] == "root summary" + assert "Hello page one" in root["text"] + child, = root["nodes"] + assert (child["start_index"], child["end_index"]) == (2, 2) assert child["summary"] == "child summary" assert child["text"] == "Second page about bananas" no_summary = local_client.get_tree(indexed_doc)["result"][0] - assert "summary" not in no_summary and "prefix_summary" not in no_summary + assert all("summary" not in node for node in [no_summary, *no_summary["nodes"]]) def test_get_tree_include_text_false(local_client, indexed_doc): tree = local_client.get_tree(indexed_doc, include_text=False) root = tree["result"][0] - assert "text" not in root - assert "text" not in root["nodes"][0] - assert root["page_index"] == 1 + assert all("text" not in node for node in [root, *root["nodes"]]) + assert (root["start_index"], root["end_index"]) == (1, 2) with_text = local_client.get_tree(indexed_doc)["result"][0] assert "text" in with_text @@ -576,10 +576,72 @@ def test_get_document_structure(local_client, indexed_doc): result = local_client.get_document_structure(indexed_doc) assert isinstance(result, list) root = result[0] - assert "text" not in root - assert "text" not in root["nodes"][0] - assert "prefix_summary" in root - assert root["nodes"][0]["summary"] == "child summary" + assert all("text" not in node for node in [root, *root["nodes"]]) + assert (root["start_index"], root["end_index"]) == (1, 2) + assert [node.get("summary") for node in [root, *root["nodes"]]] == [ + "root summary", "child summary"] + + +@pytest.mark.parametrize("mode", ["flash", "standard"]) +def test_get_tree_keeps_each_index_self_consistent(local_client, tmp_path, + monkeypatch, mode): + """get_tree adds and moves no node: every node keeps the summary written + for its own range, a flash parent's range covering its subtree and a + standard parent's ending where its first child starts.""" + from conftest import build_pdf + pdf = tmp_path / "report.pdf" + pdf.write_bytes(build_pdf(["Opening words", "Alpha body", "Beta body"])) + structure = [{"title": "Report", "node_id": "0000", "start_index": 1, + "end_index": 3 if mode == "flash" else 2, "summary": "report", + "nodes": [{"title": "Alpha", "node_id": "0001", "start_index": 2, + "end_index": 2, "summary": "alpha"}, + {"title": "Beta", "node_id": "0002", "start_index": 3, + "end_index": 3, "summary": "beta"}]}] + result = {"doc_name": "report.pdf", "doc_description": "d", "structure": structure} + monkeypatch.setattr(pageindex.flash, "page_index_flash", lambda pdf, **kw: result) + monkeypatch.setattr(page_index_module, "page_index_main", lambda *a, **kw: result) + monkeypatch.setattr(pageindex.utils, "llm_completion", lambda model, prompt, **kw: "d") + doc_id = local_client.submit_document(str(pdf), mode=mode)["doc_id"] + + root = local_client.get_tree(doc_id, node_summary=True)["result"][0] + assert [(node["title"], node["start_index"], node["end_index"], node["summary"]) + for node in [root, *root["nodes"]]] == [ + ("Report", 1, 3 if mode == "flash" else 2, "report"), + ("Alpha", 2, 2, "alpha"), ("Beta", 3, 3, "beta")] + # the parent's text runs onto its first child's page, where that heading may sit mid-page + assert "Alpha body" in root["text"] and "Beta body" not in root["text"] + + +def test_unify_tree_edges(): + from pageindex.utils import unify_tree + tree = [{"title": "A", "page_index": 3, "prefix_summary": "a", + "nodes": [{"title": "A.1", "page_index": 5, "summary": "a1"}, + {"title": "A.2", "page_index": 7, "prefix_summary": "a2", + "nodes": [{"title": "A.2.1", "page_index": 8}]}]}] + once = unify_tree(tree, 9) + # without end_index each node ends where the next one in reading order + # starts: a parent's range is its own opening, as its summary describes + assert once == [ + {"title": "A", "start_index": 3, "end_index": 5, "summary": "a", "nodes": [ + {"title": "A.1", "start_index": 5, "end_index": 7, "summary": "a1"}, + {"title": "A.2", "start_index": 7, "end_index": 8, "summary": "a2", "nodes": [ + {"title": "A.2.1", "start_index": 8, "end_index": 9}]}]}] + assert unify_tree(once, 9) == once + # a node with no page (a hand-edited store) passes through without raising + assert unify_tree([{"title": "B", "start_index": None}], None) == [ + {"title": "B", "start_index": None, "end_index": None}] + + +def test_create_node_mapping_reads_get_tree_ranges(): + from pageindex.utils import create_node_mapping + tree = [{"title": "A", "node_id": "0000", "start_index": 1, "end_index": 9, + "nodes": [{"title": "B", "node_id": "0001", "start_index": 4, "end_index": 9}]}] + mapping = create_node_mapping(tree, include_page_ranges=True, max_page=12) + assert [(m["start_index"], m["end_index"]) for m in mapping.values()] == [(1, 9), (4, 9)] + legacy = [{"title": "A", "node_id": "0000", "page_index": 1, + "nodes": [{"title": "B", "node_id": "0001", "page_index": 4}]}] + mapping = create_node_mapping(legacy, include_page_ranges=True, max_page=12) + assert [(m["start_index"], m["end_index"]) for m in mapping.values()] == [(1, 4), (4, 12)] def test_get_page_content(local_client, indexed_doc): @@ -1696,6 +1758,69 @@ def test_cloud_request_wiring(cloud, sample_pdf): assert calls[-1]["headers"] == {"api_key": "other"} +def test_cloud_get_tree_unified_shape(monkeypatch): + """An older server's tree wire (page_index only, a parent's + prefix_summary) leaves the SDK in local's field names: a node ends where + the next one in reading order starts, the last on the document's page + count, so a parent's range is the opening its summary describes.""" + client = PageIndexClient(api_key="secret") + wire = {"doc_id": "pi-1", "status": "completed", "retrieval_ready": True, + "metadata": None, "features": {}, "result": [ + {"title": "Ch 1", "node_id": "0000", "page_index": 2, + "prefix_summary": "opening", "text": "ch1 own", "nodes": [ + {"title": "1.1", "node_id": "0001", "page_index": 4, "summary": "s11", "text": "t11"}, + {"title": "1.2", "node_id": "0002", "page_index": 6, "summary": "s12", "text": "t12"}]}, + {"title": "Ch 2", "node_id": "0003", "page_index": 9, + "prefix_summary": "same page", "text": "ch2 own", "nodes": [ + {"title": "2.1", "node_id": "0004", "page_index": 9, "summary": "s21", "text": "t21"}]}]} + urls = [] + + def handler(method, url, kw): + urls.append(url) + return FakeResponse(wire if "type=tree" in url else {"pageNum": 12}) + _patch_requests(monkeypatch, handler) + + result = client.get_tree("pi-1", node_summary=True)["result"] + assert result == [ + {"title": "Ch 1", "node_id": "0000", "start_index": 2, "end_index": 4, + "summary": "opening", "text": "ch1 own", "nodes": [ + {"title": "1.1", "node_id": "0001", "start_index": 4, "end_index": 6, + "summary": "s11", "text": "t11"}, + {"title": "1.2", "node_id": "0002", "start_index": 6, "end_index": 9, + "summary": "s12", "text": "t12"}]}, + {"title": "Ch 2", "node_id": "0003", "start_index": 9, "end_index": 9, + "summary": "same page", "text": "ch2 own", "nodes": [ + {"title": "2.1", "node_id": "0004", "start_index": 9, "end_index": 12, + "summary": "s21", "text": "t21"}]}] + assert list(result[0]) == ["title", "node_id", "start_index", "end_index", + "summary", "text", "nodes"] + assert urls[-1] == "https://api.pageindex.ai/doc/pi-1/metadata/" + + +def test_cloud_get_tree_takes_served_ranges(monkeypatch): + """A server that sends start_index/end_index has its ranges taken as + given, with no metadata request.""" + client = PageIndexClient(api_key="secret") + wire = {"status": "completed", "retrieval_ready": True, "result": [ + {"title": "Ch", "node_id": "0000", "page_index": 2, "start_index": 2, + "end_index": 3, "prefix_summary": "opening", "nodes": [ + {"title": "S", "node_id": "0001", "page_index": 4, "start_index": 4, + "end_index": 8, "summary": "s"}]}]} + urls = [] + + def handler(method, url, kw): + urls.append(url) + return FakeResponse(wire) + _patch_requests(monkeypatch, handler) + + assert client.get_tree("pi-1", node_summary=True)["result"] == [ + {"title": "Ch", "node_id": "0000", "start_index": 2, "end_index": 3, + "summary": "opening", "nodes": [ + {"title": "S", "node_id": "0001", "start_index": 4, "end_index": 8, + "summary": "s"}]}] + assert len(urls) == 1 + + def test_cloud_error_and_empty_delete(cloud, monkeypatch): client, calls, fake = cloud _patch_requests(monkeypatch, @@ -2424,8 +2549,8 @@ async def blank(model, prompt): def test_summarize_tree_partial_empty_reply_absorbed(monkeypatch): - """One blank reply among good ones stays the documented per-node - absorption: blank summary, run survives.""" + """One blank reply among good ones is absorbed per node: the node falls + back to its own text, the run survives.""" async def flaky(model, prompt): if "alpha" in prompt: return "" @@ -2435,7 +2560,7 @@ async def flaky(model, prompt): structure = [{"title": "A", "start_index": 1, "end_index": 1}, {"title": "B", "start_index": 2, "end_index": 2}] out = asyncio.run(pageindex.utils.summarize_tree(structure, pdf_pages)) - assert [n["summary"] for n in out] == ["", "ok"] + assert [n["summary"] for n in out] == [" ".join(["alpha"] * 300)[:600], "ok"] def test_generate_doc_description_absorbs_context_overflow(monkeypatch): diff --git a/tests/test_flash_extraction.py b/tests/test_flash_extraction.py index c53d3aa6b..c2fc301df 100644 --- a/tests/test_flash_extraction.py +++ b/tests/test_flash_extraction.py @@ -707,9 +707,12 @@ def test_every_page_is_in_a_node(tmp_path): covered = {page for node in structure for page in range(node["start_index"], node["end_index"] + 1)} assert covered == set(range(1, pages + 1)), pdf.name - assert structure[0] == {"title": "Preface", "node_id": "0000", - "start_index": 1, "end_index": 1}, pdf.name - assert structure[1]["node_id"] == "0001", pdf.name + preface, first = structure[0], structure[1] + assert (preface["title"], preface["node_id"], preface["start_index"]) == ( + "Preface", "0000", 1), pdf.name + # it runs onto the first section's page unless that heading is known to open it + assert preface["end_index"] in (first["start_index"] - 1, first["start_index"]), pdf.name + assert first["node_id"] == "0001", pdf.name def test_preface_page_is_retrievable(tmp_path, monkeypatch): @@ -732,8 +735,8 @@ async def no_summary(*args, **kwargs): client = PageIndexClient(storage_path=str(tmp_path / "store")) doc_id = client.submit_document(str(pdf), mode="flash")["doc_id"] tree = client.get_tree(doc_id)["result"] - assert [(node["title"], node["page_index"]) for node in tree] == [ - ("Preface", 1), ("Budget", 2), ("Team", 3)] + assert [(node["title"], node["start_index"], node["end_index"]) for node in tree] == [ + ("Preface", 1, 1), ("Budget", 2, 2), ("Team", 3, 3)] assert "Skylark" in tree[0]["text"] diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py new file mode 100644 index 000000000..36066a2d5 --- /dev/null +++ b/tests/test_tree_format.py @@ -0,0 +1,229 @@ +"""The tree every local index writes: intro nodes, covering ranges, leaf-only +splitting, and summaries every node gets, a parent's built from its children's.""" + +import asyncio +import importlib +from types import SimpleNamespace + +import pageindex.flash +import pageindex.tree_optimize as tree_optimize +import pageindex.utils as utils +from conftest import build_pdf +from pageindex import PageIndexClient + +classic = importlib.import_module("pageindex.page_index_classic") + + +def shape(nodes): + return [(n["title"], n["start_index"], n["end_index"]) for n in utils._subtree(nodes)] + + +def test_intro_node_holds_the_pages_a_parent_opens_with(): + lines = [["body"] for _ in range(40)] + lines[11] = ["3.1 Scope", "body"] # 3.1 opens its page, 4.1 starts mid-page + tree = [{"title": "Ch 3", "start_index": 10, "end_index": 30, "nodes": [ + {"title": "3.1 Scope", "start_index": 12, "end_index": 20}, + {"title": "3.2", "start_index": 21, "end_index": 30, "nodes": [ + {"title": "3.2.1", "start_index": 21, "end_index": 30}]}]}, + {"title": "", "start_index": 31, "end_index": 40, "nodes": [ + {"title": "4.1", "start_index": 33, "end_index": 40}]}] + + tree_optimize.add_intro_nodes(tree, lines) + + assert shape(tree) == [ + ("Ch 3", 10, 30), ("Ch 3 (intro)", 10, 11), ("3.1 Scope", 12, 20), + ("3.2", 21, 30), ("3.2.1", 21, 30), # a first child on its parent's page: no intro + ("", 31, 40), ("Intro", 31, 33), ("4.1", 33, 40)] + assert tree_optimize.add_intro_nodes(tree, lines) == tree + # a standard parent ends where its first child starts; a split intro keeps its title + split = [{"title": "Ch 3 (intro)", "start_index": 10, "end_index": 11, "nodes": [ + {"title": "Background", "start_index": 12, "end_index": 14}]}] + assert shape(tree_optimize.add_intro_nodes(split))[1] == ("Ch 3 (intro)", 10, 11) + # a heading with nothing to match cannot be placed on its page, so the page is shared + assert not tree_optimize.heading_at_page_start([["第一章 总则"]], 1, "第一章 总则") + + +def test_expand_gives_a_split_node_its_intro(monkeypatch): + body = "body " * 250 + pages = [body] * 12 + pages[5] = "Sub One\n" + body + pages[8] = "Sub Two\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 12, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "X", "start_index": 4, "end_index": 12, "node_id": "0002"}]}] + + async def propose(model, prompt): + return {"subsections": [{"title": "Sub One", "page": 6}, {"title": "Sub Two", "page": 9}]} + + async def summarize(model, prompt): + return '{"summary": "ok"}' + monkeypatch.setattr(tree_optimize, "ask_model", propose) + monkeypatch.setattr(utils, "llm_acompletion", summarize) + + async def run(): + scheduler = utils.SummaryScheduler(tree, [(page, 0) for page in pages], model="m") + await tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, + on_final=scheduler.mark_final) + await scheduler.finish() + asyncio.run(run()) + + x = tree[0]["nodes"][1] + assert [(n["title"], n["start_index"], n["end_index"], n["node_id"]) for n in x["nodes"]] == [ + ("X (intro)", 4, 5, "0003"), ("Sub One", 6, 8, "0004"), ("Sub Two", 9, 12, "0005")] + assert all(n["summary"] == "ok" for n in utils._subtree(tree)) + + +def test_merge_folds_an_intro_without_listing_its_title(): + tree = [{"title": "P", "start_index": 1, "end_index": 4, "nodes": [ + {"title": "P (intro)", "start_index": 1, "end_index": 1}, + {"title": "C", "start_index": 2, "end_index": 4}]}] + tree_optimize.merge_tree(tree) + assert "nodes" not in tree[0] and tree[0]["key_items"] == ["C"] + + +LARGE = SimpleNamespace(max_page_num_each_node=10, max_token_num_each_node=20000, model="m") + + +def test_large_node_split_keeps_existing_children(monkeypatch): + async def meta_processor(*args, **kwargs): + raise AssertionError("a parent's subsections must not be replaced") + monkeypatch.setattr(classic, "meta_processor", meta_processor) + node = {"title": "Ch", "start_index": 1, "end_index": 20, + "nodes": [{"title": "S", "start_index": 20, "end_index": 25}]} + + asyncio.run(classic.process_large_node_recursively(node, [("page", 5000)] * 25, LARGE)) + + assert [child["title"] for child in node["nodes"]] == ["S"] + + +def test_split_intro_skips_its_parents_heading(monkeypatch): + async def meta_processor(*args, **kwargs): + return [{"title": "Ch", "physical_index": 1}, + {"title": "Background", "physical_index": 5}, + {"title": "Scope", "physical_index": 9}] + + async def appear(items, *args, **kwargs): + for item in items: + item["appear_start"] = "yes" + return items + monkeypatch.setattr(classic, "meta_processor", meta_processor) + monkeypatch.setattr(classic, "check_title_appearance_in_start_concurrent", appear) + intro = {"title": "Ch (intro)", "start_index": 1, "end_index": 14} + + asyncio.run(classic.process_large_node_recursively(intro, [("page", 5000)] * 14, LARGE)) + + assert [child["title"] for child in intro["nodes"]] == ["Background", "Scope"] + + +def test_standard_index_stores_intros_covering_ranges_and_section_summaries(tmp_path, monkeypatch): + texts = ["Opening words"] + [f"page {n}" for n in range(2, 41)] + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(texts)) + + async def tree_parser(page_list, opt, doc=None, logger=None): + return [{"title": "P", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "P (intro)", "start_index": 1, "end_index": 2}, + {"title": "C1", "start_index": 3, "end_index": 20}, + {"title": "C2", "start_index": 21, "end_index": 21, "nodes": [ + {"title": "C2.1", "start_index": 21, "end_index": 30}, + {"title": "C2.2", "start_index": 31, "end_index": 40}]}]}] + prompts = [] + + async def reply(model, prompt): + prompts.append(prompt) + return "whole " + prompt.split("Section Title: ")[1].split("\n")[0] + monkeypatch.setattr(classic, "tree_parser", tree_parser) + monkeypatch.setattr(utils, "llm_acompletion", reply) + monkeypatch.setattr(utils, "llm_completion", lambda model, prompt, **kw: "d") + opt = utils.ConfigLoader().load({"model": "m", "if_add_node_id": "yes", + "if_add_node_summary": "yes", "if_add_node_text": "yes", + "if_add_doc_description": "yes"}) + + result = classic.page_index_main(str(pdf), opt, logger=SimpleNamespace(info=print), + page_list=[(text, 1) for text in texts]) + + p = result["structure"][0] + c2 = p["nodes"][2] + assert shape([p]) == [("P", 1, 40), ("P (intro)", 1, 2), ("C1", 3, 20), + ("C2", 21, 40), ("C2.1", 21, 30), ("C2.2", 31, 40)] + assert (p["text"], p["summary"]) == ("", "whole P") # its intro holds pages 1-2 + assert "Opening words" in p["nodes"][0]["text"] + # C2's first child starts on C2's page; C2 keeps that page as its opening + assert (c2["text"], c2["summary"]) == ("page 21", "whole C2") + # short leaves keep their own text; only the parents are asked, deepest first + assert [n["summary"] for n in p["nodes"][:2]] == [ + "Opening wordspage 2", "".join(f"page {n}" for n in range(3, 21))] + assert [q.split("Section Title: ")[1].split("\n")[0] for q in prompts] == ["C2", "P"] + assert '"summary": "whole C2"' in prompts[-1] + + +def test_flash_gives_parents_their_intro_nodes(tmp_path, monkeypatch): + import pageindex.flash.api as flash_api + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["x"])) + monkeypatch.setattr(flash_api, "extract_toc", lambda pdf, **kw: { + "structure": [{"title": "R", "node_id": "0000", "start_index": 1, "end_index": 6, "nodes": [ + {"title": "A", "node_id": "0001", "start_index": 3, "end_index": 4}, + {"title": "X", "node_id": "0002", "start_index": 5, "end_index": 6}]}], + "page_texts": ["Cover", "Foreword", "A\nalpha", "alpha", "X\nbody", "body"]}) + + result = flash_api.page_index_flash(str(pdf), summary=False, optimize=False) + + assert [(n["title"], n["node_id"], n["start_index"], n["end_index"]) + for n in utils._subtree(result["structure"])] == [ + ("R", "0000", 1, 6), ("R (intro)", "0001", 1, 2), ("A", "0002", 3, 4), + ("X", "0003", 5, 6)] + + +def test_flash_preface_runs_onto_a_page_its_first_heading_does_not_open(tmp_path, monkeypatch): + import pageindex.flash.api as flash_api + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["x"])) + ends = [] + for third in ["A\nalpha", "Contents, continued\nA\nalpha"]: + monkeypatch.setattr(flash_api, "extract_toc", lambda pdf, third=third, **kw: { + "structure": [{"title": "A", "node_id": "0000", "start_index": 3, "end_index": 4}], + "page_texts": ["Cover", "Contents", third, "alpha"]}) + preface = flash_api.page_index_flash(str(pdf), summary=False, optimize=False)["structure"][0] + ends.append((preface["title"], preface["node_id"], preface["end_index"])) + assert ends == [("Preface", "0000", 2), ("Preface", "0000", 3)] + + +def test_summarize_tree_parent_falls_back_to_its_subsection_titles(monkeypatch): + async def reply(model, prompt): + return "" if "Section Title" in prompt else '{"summary": "ok"}' + monkeypatch.setattr(utils, "llm_acompletion", reply) + structure = [{"title": "R", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "A", "start_index": 1, "end_index": 1}, + {"title": "B", "start_index": 2, "end_index": 2}]}] + + out = asyncio.run(utils.summarize_tree(structure, [("alpha " * 300, 0), ("beta " * 300, 0)])) + + assert [n["summary"] for n in utils._subtree(out)] == ["A; B", "ok", "ok"] + + +def test_get_tree_parent_text_shares_its_first_childs_page_unless_an_intro_holds_it( + tmp_path, monkeypatch): + pdf = tmp_path / "report.pdf" + pdf.write_bytes(build_pdf(["Opening words", "Alpha body", "Beta body"])) + structure = [{"title": "Report", "node_id": "0000", "start_index": 1, "end_index": 3, + "summary": "report", "nodes": [ + {"title": "Report (intro)", "node_id": "0001", "start_index": 1, + "end_index": 1, "summary": "opening"}, + {"title": "Alpha", "node_id": "0002", "start_index": 2, + "end_index": 3, "summary": "alpha", "nodes": [ + {"title": "Alpha one", "node_id": "0003", "start_index": 2, + "end_index": 2, "summary": "one"}, + {"title": "Alpha two", "node_id": "0004", "start_index": 3, + "end_index": 3, "summary": "two"}]}]}] + monkeypatch.setattr(pageindex.flash, "page_index_flash", lambda pdf, **kw: { + "doc_name": "report.pdf", "structure": structure}) + monkeypatch.setattr(utils, "llm_completion", lambda model, prompt, **kw: "d") + client = PageIndexClient(storage_path=str(tmp_path / "store")) + doc_id = client.submit_document(str(pdf), mode="flash")["doc_id"] + + root = client.get_tree(doc_id)["result"][0] + + assert [n["text"] for n in utils._subtree([root])] == [ + "", "Opening words", "Alpha body", "Alpha body", "Beta body"]