From 79d88e8346468995c111b04c0274dd8f6ba956a1 Mon Sep 17 00:00:00 2001 From: Ray Date: Thu, 1 Oct 2026 19:35:59 +0800 Subject: [PATCH 1/3] Unify the document tree across local and cloud get_tree returns the same node shape in both modes: start_index and end_index (no page_index) and summary (no prefix_summary). The SDK only maps field names on the way out. A stored tree is never restructured, so older documents keep their own ranges and summaries. New local indexes, standard and flash, are built in the unified shape: - a parent whose first child starts on a later page gets a first child " (intro)" holding those pages - a parent's range covers its whole subtree, and its summary is written from its children's summaries, deepest first - a node the model leaves unsummarized falls back to its subsection titles or its opening text; a run with no answer at all still fails - the standard large-node split acts on leaves only, so it no longer replaces a parent's existing subsections A page cut that cannot tell where a heading sits gives the page to both sides. A parent's text runs onto its first child's page, and is empty when its intro holds those pages. The flash Preface takes the first section's page unless that heading opens it. A heading with nothing to match counts as not at the top of its page. --- pageindex/client.py | 20 ++- pageindex/flash/api.py | 35 +++-- pageindex/local_api.py | 31 +++-- pageindex/page_index_classic.py | 20 ++- pageindex/tree_optimize.py | 38 +++++- pageindex/utils.py | 176 ++++++++++++++++++++++-- tests/test_client.py | 160 +++++++++++++++++++--- tests/test_flash_extraction.py | 13 +- tests/test_tree_format.py | 229 ++++++++++++++++++++++++++++++++ 9 files changed, 652 insertions(+), 70 deletions(-) create mode 100644 tests/test_tree_format.py diff --git a/pageindex/client.py b/pageindex/client.py index 4dea417c1..0d77269bf 100644 --- a/pageindex/client.py +++ b/pageindex/client.py @@ -949,14 +949,22 @@ def get_tree(self, doc_id: str, node_summary: bool = False, Returns: dict: {'doc_id', 'status', 'retrieval_ready', 'result', ...} where - result nodes are {'title', 'node_id', 'page_index', ('summary' / - 'prefix_summary',) ('text',) 'nodes'}. + result nodes are {'title', 'node_id', 'start_index', 'end_index', + ('summary',) ('text',) 'nodes'} in both modes. Each node's summary + describes its own pages start_index..end_index; its text is the + part no child holds, though a parent's may share the page its + first child starts on. """ tree = self._api.get_tree(doc_id=doc_id, node_summary=node_summary, include_text=include_text) - if not include_text and tree.get("result"): - from .utils import remove_fields - tree["result"] = remove_fields(tree["result"], fields=["text"]) + if tree.get("result"): + from .utils import _subtree, remove_fields, unify_tree + page_count = None + if any("end_index" not in node for node in _subtree(tree["result"])): + page_count = self._api.get_document(doc_id).get("pageNum") + tree["result"] = unify_tree(tree["result"], page_count) + if not include_text: + tree["result"] = remove_fields(tree["result"], fields=["text"]) return tree def get_document_structure(self, doc_id: str) -> list[dict[str, Any]]: @@ -975,7 +983,7 @@ def is_retrieval_ready(self, doc_id: str) -> bool: failures, timeouts) propagate. """ try: - result = self.get_tree(doc_id) + result = self._api.get_tree(doc_id) return result.get("retrieval_ready", False) except PageIndexAPIError: return False diff --git a/pageindex/flash/api.py b/pageindex/flash/api.py index bc8c71446..5e3e397e1 100644 --- a/pageindex/flash/api.py +++ b/pageindex/flash/api.py @@ -89,9 +89,7 @@ async def _optimize_async(structure, page_texts, do_expand, model, on_final=None the summaries use. """ from ..tree_optimize import optimize - lines = [[line_text.strip() for line_text in (page_text or "").splitlines() - if line_text.strip()] - for page_text in page_texts] + lines = _page_lines(page_texts) outcome = await optimize(structure, page_texts, lines, model=model, do_expand=do_expand, page_count=len(page_texts), on_final=on_final, concurrency=concurrency) @@ -134,14 +132,31 @@ def _page_nodes(page_texts: list[str]) -> list[dict]: return nodes -def _add_preface(structure: list[dict]) -> None: - """The pages before a hierarchy that starts late become a Preface node, as in standard mode.""" +def _page_lines(page_texts: list[str]) -> list[list[str]]: + return [[line_text.strip() for line_text in (page_text or "").splitlines() + if line_text.strip()] + for page_text in page_texts] + + +def _add_intros(structure: list[dict], lines: list[list[str]]) -> None: + """The pages a parent opens with before its first child become its intro node.""" + from ..tree_optimize import add_intro_nodes from ..utils import write_node_id - structure.insert(0, {"title": "Preface", "start_index": 1, - "end_index": structure[0]["start_index"] - 1}) + add_intro_nodes(structure, lines) write_node_id(structure) +def _add_preface(structure: list[dict], lines: list[list[str]]) -> None: + """The pages before a hierarchy that starts late become a Preface node, as in + standard mode. It runs onto the first section's page unless that heading opens it.""" + from ..tree_optimize import heading_at_page_start + first = structure[0] + opens = first["start_index"] <= len(lines) and heading_at_page_start( + lines, first["start_index"], first["title"]) + structure.insert(0, {"title": "Preface", "start_index": 1, + "end_index": first["start_index"] - (1 if opens else 0)}) + + def flash_rejection_reason(result: dict, standard_hint: str = "mode='standard'") -> str | None: """Why a managed pipeline should refuse this flash result, or None to accept it. @@ -166,7 +181,7 @@ def page_index_flash(pdf, summary=True, summary_model=None, optimize: str | bool | None = None, optimize_expand=None, optimize_model=None, summary_concurrency=None, use_embedded_toc=True, summary_max_words=None) -> dict: - """Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """ + """Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode, and the first section's page too unless that heading opens it; a parent whose first child starts on a later page opens with a child titled ``" (intro)"`` holding the pages before it; a parent's range and summary cover its whole subtree) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """ for name, value in (("summary_concurrency", summary_concurrency), ("summary_max_words", summary_max_words)): if value is not None and not (isinstance(value, numbers.Integral) and int(value) >= 1): @@ -195,7 +210,9 @@ def page_index_flash(pdf, summary=True, summary_model=None, result["structure"] = structure result["toc_source"] = "pages" if structure else "unreadable" elif structure[0]["start_index"] > 1: - _add_preface(structure) + _add_preface(structure, _page_lines(result.get("page_texts") or [])) + if structure: + _add_intros(structure, _page_lines(result.get("page_texts") or [])) if result.get("toc_source") == "pages" and len(structure) > FLAT_TREE_MAX_NODES: # the managed pipelines refuse a flat tree this size; skip the model passes result.pop("page_texts", None) diff --git a/pageindex/local_api.py b/pageindex/local_api.py index 025ab716d..63f431ac6 100644 --- a/pageindex/local_api.py +++ b/pageindex/local_api.py @@ -260,17 +260,32 @@ def _index_flash(self, file_path: str) -> tuple[list, str | None]: # ── tree / ocr ── def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: - from .utils import add_node_text + """Each node's own text. A parent's runs onto the page its first child + starts on, where that child's heading may sit mid-page; it is empty when + an intro child holds the parent's opening pages.""" + from .utils import get_text_of_pdf_pages, is_intro structure = self._require_data( self._store.get_tree(doc_id), error_prefix) pages = self._require_pages(doc_id, error_prefix) pdf_pages = [(p.get("markdown", ""), 0) for p in pages] - add_node_text(structure, pdf_pages) + + def add_own_text(nodes): + for node in nodes: + children = node.get("nodes") or [] + end = node.get("end_index") + first = children[0].get("start_index") if children else None + if children and is_intro(node, children[0]): + end = None + elif end is not None and first is not None: + end = min(end, first) + node["text"] = get_text_of_pdf_pages(pdf_pages, node.get("start_index"), end) + add_own_text(children) + + add_own_text(structure) return structure def raw_tree(self, doc_id: str) -> list | None: - """Stored tree verbatim — keeps start_index/end_index, which - get_tree's cloud wire shape renames and drops.""" + """Stored tree verbatim, every key kept.""" return self._store.get_tree(doc_id) def get_tree(self, doc_id: str, node_summary: bool = False, @@ -398,17 +413,15 @@ def _format_tree_node(node: dict, node_summary: bool) -> dict: out = { "title": node.get("title", ""), "node_id": node.get("node_id"), - "page_index": node.get("start_index"), + "start_index": node.get("start_index"), + "end_index": node.get("end_index"), } if node.get("key_items"): out["key_items"] = node["key_items"] if node_summary: summary = node.get("summary") if summary is not None: - if children: - out["prefix_summary"] = summary - else: - out["summary"] = summary + out["summary"] = summary if "text" in node: out["text"] = node["text"] if children: diff --git a/pageindex/page_index_classic.py b/pageindex/page_index_classic.py index d680154eb..015660249 100644 --- a/pageindex/page_index_classic.py +++ b/pageindex/page_index_classic.py @@ -6,7 +6,7 @@ import re from .utils import * from .naming import sanitize_filename as sanitize_upload_filename -from .tree_optimize import merge_tree +from .tree_optimize import add_intro_nodes, merge_tree import os ######################### Hardening for prompt injection patterns #################################################### @@ -1169,7 +1169,8 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None) node_page_list = page_list[node['start_index']-1:node['end_index']] token_num = sum([page[1] for page in node_page_list]) - if node['end_index'] - node['start_index'] > opt.max_page_num_each_node and token_num >= opt.max_token_num_each_node: + if (not node.get('nodes') and node['end_index'] - node['start_index'] > opt.max_page_num_each_node + and token_num >= opt.max_token_num_each_node): print('large node:', node['title'], 'start_index:', node['start_index'], 'end_index:', node['end_index'], 'token_num:', token_num) node_toc_tree = await meta_processor(node_page_list, mode='process_no_toc', start_index=node['start_index'], opt=opt, logger=logger) @@ -1178,7 +1179,9 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None) # Filter out items with None physical_index before post_processing valid_node_toc_items = [item for item in node_toc_tree if item.get('physical_index') is not None] - if valid_node_toc_items and node['title'].strip() == valid_node_toc_items[0]['title'].strip(): + # an intro node's pages open with its parent's heading + title = node['title'].strip() + if valid_node_toc_items and valid_node_toc_items[0]['title'].strip() in (title, title.removesuffix(INTRO_SUFFIX)): node['nodes'] = post_processing(valid_node_toc_items[1:], node['end_index']) node['end_index'] = valid_node_toc_items[1]['start_index'] if len(valid_node_toc_items) > 1 else node['end_index'] else: @@ -1221,14 +1224,15 @@ async def tree_parser(page_list, opt, doc=None, logger=None): # Filter out items with None physical_index before post_processings valid_toc_items = [item for item in toc_with_page_number if item.get('physical_index') is not None] - toc_tree = post_processing(valid_toc_items, len(page_list)) + toc_tree = add_intro_nodes(post_processing(valid_toc_items, len(page_list))) tasks = [ process_large_node_recursively(node, page_list, opt, logger=logger) for node in toc_tree ] await asyncio.gather(*tasks) - - return toc_tree + + # a split leaf is now a parent that may open before its first child + return add_intro_nodes(toc_tree) def page_index_main(doc, opt=None, logger=None, page_list=None): @@ -1256,9 +1260,11 @@ async def page_index_builder(): write_node_id(structure) if opt.if_add_node_text == 'yes': add_node_text(structure, page_list) + remove_intro_parent_text(structure) if opt.if_add_node_summary == 'yes': if opt.if_add_node_text == 'no': add_node_text(structure, page_list) + remove_intro_parent_text(structure) await generate_summaries_for_structure(structure, model=getattr(opt, 'summary_model', None) or opt.model) if opt.if_add_node_text == 'no': remove_structure_text(structure) @@ -1266,12 +1272,14 @@ async def page_index_builder(): # Create a clean structure without unnecessary fields for description generation clean_structure = create_clean_structure_for_description(structure) doc_description = generate_doc_description(clean_structure, model=getattr(opt, 'summary_model', None) or opt.model) + cover_subtree_ranges(structure) structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes']) return { 'doc_name': get_pdf_name(doc), 'doc_description': doc_description, 'structure': structure, } + cover_subtree_ranges(structure) structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes']) return { 'doc_name': get_pdf_name(doc), diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index ddf397c7d..e607d5f00 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -62,8 +62,8 @@ import sys from types import SimpleNamespace -from .utils import (ConfigLoader, _is_unrecoverable, llm_acompletion, - strip_internal_keys) +from .utils import (ConfigLoader, _is_unrecoverable, intro_title, is_intro, + llm_acompletion, strip_internal_keys) TRIGGER_PAGES = 5 # only look ahead on nodes larger than this ROUTING_COST = 1 # R(v), in pages @@ -171,11 +171,13 @@ def is_frontier(node): def heading_at_page_start(lines, page_no, heading): - """Is the heading the first line on its page?""" + """Is the heading the first line on its page? No when that cannot be told + (a heading with no Latin letter or digit to match), so the page is shared.""" page = lines[page_no - 1] - if not page: + key = normalize(heading) + if not page or not key: return False - return normalize(heading) in normalize(page[0]) + return key in normalize(page[0]) def assign_ends(node, children, lines): @@ -202,9 +204,30 @@ def attach_children(node, children, lines): # so gaining children leaves it unchanged sized = assign_ends(node, children, lines) node["nodes"] = sized + add_intro_nodes([node], lines) + if len(node["nodes"]) > len(sized): # an intro went in; numbered like the proposals + node["nodes"][0]["node_id"] = f"{node['node_id']}.0" return sized +def add_intro_nodes(structure, lines=None): + """Give every parent whose first child starts on a later page an intro child + for the pages before it, titled " (intro)". It ends where the + parent's own pages do, and no later than the first child's page, the page + before it when that child's heading opens its page.""" + for node, _ in list(flatten(structure)): + children = node.get("nodes") or [] + first = children[0]["start_index"] if children else None + if first is None or first <= node["start_index"]: + continue + opens = bool(lines) and first <= len(lines) and heading_at_page_start( + lines, first, children[0]["title"]) + end = max(node["start_index"], min(node["end_index"], first - 1 if opens else first)) + node["nodes"] = [{"title": intro_title(node.get("title")), + "start_index": node["start_index"], "end_index": end}] + children + return structure + + def relabel(structure, width=4): """Renumber every node_id in document order: 0000, 0001, 0002, ... @@ -558,8 +581,9 @@ def visit(node): # titles are routing information; keep them on the parent, in document # order, carrying forward anything an earlier merge already folded in titles = [] - for child, _ in flatten(node["nodes"]): - titles.append(child["title"]) + for child, parent in flatten(node["nodes"]): + if not is_intro(parent or node, child): # its parent's heading again + titles.append(child["title"]) titles.extend(child.get("key_items") or []) log.append({"op": "merge", "node_id": node.get("node_id"), "S": span, "tree_cost": cost, "frontier_cost": checked, diff --git a/pageindex/utils.py b/pageindex/utils.py index 269218117..fe6e1c376 100644 --- a/pageindex/utils.py +++ b/pageindex/utils.py @@ -743,16 +743,114 @@ async def generate_node_summary(node, model=None): return response +async def generate_section_summary(node, model=None): + listing = json.dumps( + [{'title': child.get('title', ''), 'summary': child.get('summary', '')} + for child in node['nodes']], + ensure_ascii=False) + prompt = f"""You are given a section of a document: the text that opens the section (possibly empty) and the titles and summaries of its subsections. + Your task is to generate a concise description of everything that is covered in the whole section, summarizing all its points without omitting any type of content. + Keep the description concise and to the point, avoiding unnecessary details. + + Section Title: {node.get('title', '')} + + Opening Text: {node.get('text') or ''} + + Subsection Titles and Summaries: {listing} + + Directly return the description, do not include any other text. + """ + return await llm_acompletion(model, prompt) + + +FALLBACK_SUMMARY_CHARS = 600 + + +def fallback_summary(node, text=None): + """The summary of a node the model left unsummarized, so no node goes + without one: a parent's subsection titles, a leaf's opening text.""" + children = node.get('nodes') or [] + if children: + summary = "; ".join(child['title'] for child in children if child.get('title')) + else: + source = node.get('text') if text is None else text + summary = " ".join((source or "").split()) + return summary[:FALLBACK_SUMMARY_CHARS] or node.get('title') or "" + + +INTRO_SUFFIX = " (intro)" + + +def intro_title(title): + """The title of a parent's intro node; a split intro keeps its own.""" + title = (title or "").strip() + if title and not title.endswith(INTRO_SUFFIX): + title += INTRO_SUFFIX + return title or "Intro" + + +def is_intro(parent, child): + """Whether child is the intro node holding parent's opening pages.""" + return (child.get('start_index') == parent.get('start_index') + and child.get('title') == intro_title(parent.get('title'))) + + +def remove_intro_parent_text(structure): + """Clear the text of every parent that opens with its intro: the intro + holds the same pages.""" + for node in structure_to_list(structure): + children = node.get('nodes') or [] + if children and is_intro(node, children[0]) and 'text' in node: + node['text'] = "" + return structure + + +def cover_subtree_ranges(structure): + """Widen every parent's range to cover its whole subtree.""" + for node in structure if isinstance(structure, list) else [structure]: + children = node.get('nodes') or [] + if children: + cover_subtree_ranges(children) + node['start_index'] = min([node['start_index']] + [c['start_index'] for c in children]) + node['end_index'] = max([node['end_index']] + [c['end_index'] for c in children]) + return structure + + async def generate_summaries_for_structure(structure, model=None): + """Leaves are summarized from their own text, parents from their children's + summaries, deepest first; a node the model leaves unsummarized gets + fallback_summary. Fails loud when no call got an answer.""" nodes = structure_to_list(structure) - tasks = [generate_node_summary(node, model=model) for node in nodes] - summaries = await asyncio.gather(*tasks, return_exceptions=True) - - for node, summary in zip(nodes, summaries): - if isinstance(summary, Exception) and _is_unrecoverable(summary): - raise summary - node['summary'] = "" if isinstance(summary, BaseException) else summary - if nodes and not any(node['summary'] for node in nodes): + levels = [] + + def collect(siblings, depth): + for node in siblings: + if node.get('nodes'): + if len(levels) <= depth: + levels.append([]) + levels[depth].append(node) + collect(node['nodes'], depth + 1) + + collect(structure if isinstance(structure, list) else [structure], 0) + answered = False + + async def summarize(group, ask): + nonlocal answered + replies = await asyncio.gather(*(ask(node, model=model) for node in group), + return_exceptions=True) + for node, reply in zip(group, replies): + if isinstance(reply, Exception) and _is_unrecoverable(reply): + raise reply + if isinstance(reply, str) and reply: + answered = True + node['summary'] = reply + else: + node['summary'] = fallback_summary(node) + + await summarize([node for node in nodes if not node.get('nodes')], generate_node_summary) + for parents in reversed(levels): + await summarize(parents, generate_section_summary) + if nodes and not answered: raise RuntimeError( "Summary generation failed for all nodes " "(every summary call failed or returned empty; " @@ -1040,6 +1138,9 @@ async def _visit(self, node, depth): node['summary'] = "" if _is_unrecoverable(e): raise + if not node['summary']: + node['summary'] = fallback_summary(node, None if children else get_text_of_pdf_pages( + self._pdf_pages, node['start_index'], node['end_index'])) async def finish(self): """Wait for every summary; fails loud if the model never answered.""" @@ -1252,9 +1353,58 @@ def load(self, user_opt=None) -> config: _resolve_models(merged) return config(**merged) +_TREE_WIRE_KEYS = frozenset({"title", "node_id", "start_index", "end_index", "page_index", + "summary", "prefix_summary", "text", "nodes"}) + + +def unify_tree(nodes, page_count=None): + """get_tree's nodes in the field names both modes return. + + The structure is the index's own: nodes are neither added nor moved, so + every node keeps the summary written for its own pages start_index .. + end_index. A node's first page comes as start_index (local) or + page_index (cloud), and a parent's summary as summary or prefix_summary. + Without an end_index (an older server) a node ends on the page where the + next node in reading order starts, the last on page_count. + """ + flat = list(_subtree(nodes)) + starts = [node.get("start_index", node.get("page_index")) for node in flat] + ranges = {} + for i, node in enumerate(flat): + start, end = starts[i], node.get("end_index") + if end is None: + end = starts[i + 1] if i + 1 < len(flat) else page_count + if end is None or (start is not None and end < start): + end = start + ranges[id(node)] = (start, end) + + def build(siblings): + out = [] + for node in siblings: + unified = {"title": node.get("title", "")} + if "node_id" in node: + unified["node_id"] = node["node_id"] + unified["start_index"], unified["end_index"] = ranges[id(node)] + unified.update((key, value) for key, value in node.items() + if key not in _TREE_WIRE_KEYS) + for key in ("summary", "prefix_summary"): + if key in node: + unified["summary"] = node[key] + if "text" in node: + unified["text"] = node["text"] + if node.get("nodes"): + unified["nodes"] = build(node["nodes"]) + out.append(unified) + return out + + return build(nodes) + + def create_node_mapping(tree, include_page_ranges=False, max_page=None): """Map node_id to node; with include_page_ranges, to {"node", "start_index", - "end_index"} (end = next node's page_index, or max_page for the last node).""" + "end_index"}: the node's own range as get_tree returns it, or, for a tree + that carries page_index only, end = next node's page_index (max_page for + the last node).""" def get_all_nodes(tree): if isinstance(tree, dict): return [tree] + [node for child in tree.get('nodes', []) for node in get_all_nodes(child)] @@ -1268,10 +1418,14 @@ def get_all_nodes(tree): mapping = {} for i, node in enumerate(all_nodes): if node.get("node_id"): - end_page = all_nodes[i + 1].get("page_index") if i + 1 < len(all_nodes) else max_page + if "end_index" in node: + start_page, end_page = node["start_index"], node["end_index"] + else: + start_page = node["page_index"] + end_page = all_nodes[i + 1].get("page_index") if i + 1 < len(all_nodes) else max_page mapping[node["node_id"]] = { "node": node, - "start_index": node["page_index"], + "start_index": start_page, "end_index": end_page, } return mapping diff --git a/tests/test_client.py b/tests/test_client.py index d8eec89b9..dc1180054 100644 --- a/tests/test_client.py +++ b/tests/test_client.py @@ -549,24 +549,24 @@ def test_submit_and_get_tree(local_client, indexed_doc, tmp_path, monkeypatch): assert tree["status"] == "completed" assert tree["retrieval_ready"] is True root = tree["result"][0] - assert root["page_index"] == 1 - assert "start_index" not in root and "end_index" not in root - assert root["prefix_summary"] == "root summary" - assert "summary" not in root - child = root["nodes"][0] + assert (root["start_index"], root["end_index"]) == (1, 2) + assert "page_index" not in root and "prefix_summary" not in root + assert root["summary"] == "root summary" + assert "Hello page one" in root["text"] + child, = root["nodes"] + assert (child["start_index"], child["end_index"]) == (2, 2) assert child["summary"] == "child summary" assert child["text"] == "Second page about bananas" no_summary = local_client.get_tree(indexed_doc)["result"][0] - assert "summary" not in no_summary and "prefix_summary" not in no_summary + assert all("summary" not in node for node in [no_summary, *no_summary["nodes"]]) def test_get_tree_include_text_false(local_client, indexed_doc): tree = local_client.get_tree(indexed_doc, include_text=False) root = tree["result"][0] - assert "text" not in root - assert "text" not in root["nodes"][0] - assert root["page_index"] == 1 + assert all("text" not in node for node in [root, *root["nodes"]]) + assert (root["start_index"], root["end_index"]) == (1, 2) with_text = local_client.get_tree(indexed_doc)["result"][0] assert "text" in with_text @@ -576,10 +576,72 @@ def test_get_document_structure(local_client, indexed_doc): result = local_client.get_document_structure(indexed_doc) assert isinstance(result, list) root = result[0] - assert "text" not in root - assert "text" not in root["nodes"][0] - assert "prefix_summary" in root - assert root["nodes"][0]["summary"] == "child summary" + assert all("text" not in node for node in [root, *root["nodes"]]) + assert (root["start_index"], root["end_index"]) == (1, 2) + assert [node.get("summary") for node in [root, *root["nodes"]]] == [ + "root summary", "child summary"] + + +@pytest.mark.parametrize("mode", ["flash", "standard"]) +def test_get_tree_keeps_each_index_self_consistent(local_client, tmp_path, + monkeypatch, mode): + """get_tree adds and moves no node: every node keeps the summary written + for its own range, a flash parent's range covering its subtree and a + standard parent's ending where its first child starts.""" + from conftest import build_pdf + pdf = tmp_path / "report.pdf" + pdf.write_bytes(build_pdf(["Opening words", "Alpha body", "Beta body"])) + structure = [{"title": "Report", "node_id": "0000", "start_index": 1, + "end_index": 3 if mode == "flash" else 2, "summary": "report", + "nodes": [{"title": "Alpha", "node_id": "0001", "start_index": 2, + "end_index": 2, "summary": "alpha"}, + {"title": "Beta", "node_id": "0002", "start_index": 3, + "end_index": 3, "summary": "beta"}]}] + result = {"doc_name": "report.pdf", "doc_description": "d", "structure": structure} + monkeypatch.setattr(pageindex.flash, "page_index_flash", lambda pdf, **kw: result) + monkeypatch.setattr(page_index_module, "page_index_main", lambda *a, **kw: result) + monkeypatch.setattr(pageindex.utils, "llm_completion", lambda model, prompt, **kw: "d") + doc_id = local_client.submit_document(str(pdf), mode=mode)["doc_id"] + + root = local_client.get_tree(doc_id, node_summary=True)["result"][0] + assert [(node["title"], node["start_index"], node["end_index"], node["summary"]) + for node in [root, *root["nodes"]]] == [ + ("Report", 1, 3 if mode == "flash" else 2, "report"), + ("Alpha", 2, 2, "alpha"), ("Beta", 3, 3, "beta")] + # the parent's text runs onto its first child's page, where that heading may sit mid-page + assert "Alpha body" in root["text"] and "Beta body" not in root["text"] + + +def test_unify_tree_edges(): + from pageindex.utils import unify_tree + tree = [{"title": "A", "page_index": 3, "prefix_summary": "a", + "nodes": [{"title": "A.1", "page_index": 5, "summary": "a1"}, + {"title": "A.2", "page_index": 7, "prefix_summary": "a2", + "nodes": [{"title": "A.2.1", "page_index": 8}]}]}] + once = unify_tree(tree, 9) + # without end_index each node ends where the next one in reading order + # starts: a parent's range is its own opening, as its summary describes + assert once == [ + {"title": "A", "start_index": 3, "end_index": 5, "summary": "a", "nodes": [ + {"title": "A.1", "start_index": 5, "end_index": 7, "summary": "a1"}, + {"title": "A.2", "start_index": 7, "end_index": 8, "summary": "a2", "nodes": [ + {"title": "A.2.1", "start_index": 8, "end_index": 9}]}]}] + assert unify_tree(once, 9) == once + # a node with no page (a hand-edited store) passes through without raising + assert unify_tree([{"title": "B", "start_index": None}], None) == [ + {"title": "B", "start_index": None, "end_index": None}] + + +def test_create_node_mapping_reads_get_tree_ranges(): + from pageindex.utils import create_node_mapping + tree = [{"title": "A", "node_id": "0000", "start_index": 1, "end_index": 9, + "nodes": [{"title": "B", "node_id": "0001", "start_index": 4, "end_index": 9}]}] + mapping = create_node_mapping(tree, include_page_ranges=True, max_page=12) + assert [(m["start_index"], m["end_index"]) for m in mapping.values()] == [(1, 9), (4, 9)] + legacy = [{"title": "A", "node_id": "0000", "page_index": 1, + "nodes": [{"title": "B", "node_id": "0001", "page_index": 4}]}] + mapping = create_node_mapping(legacy, include_page_ranges=True, max_page=12) + assert [(m["start_index"], m["end_index"]) for m in mapping.values()] == [(1, 4), (4, 12)] def test_get_page_content(local_client, indexed_doc): @@ -1166,7 +1228,8 @@ async def flaky(model, prompt): result = asyncio.run(pageindex.utils.generate_summaries_for_structure(structure)) summaries = {n["title"]: n["summary"] for n in pageindex.utils.structure_to_list(result)} - assert summaries == {"A": "", "B": "ok"} + # the failed parent falls back to its subsection titles + assert summaries == {"A": "B", "B": "ok"} def test_generate_summaries_unrecoverable_raises(monkeypatch): @@ -1696,6 +1759,69 @@ def test_cloud_request_wiring(cloud, sample_pdf): assert calls[-1]["headers"] == {"api_key": "other"} +def test_cloud_get_tree_unified_shape(monkeypatch): + """An older server's tree wire (page_index only, a parent's + prefix_summary) leaves the SDK in local's field names: a node ends where + the next one in reading order starts, the last on the document's page + count, so a parent's range is the opening its summary describes.""" + client = PageIndexClient(api_key="secret") + wire = {"doc_id": "pi-1", "status": "completed", "retrieval_ready": True, + "metadata": None, "features": {}, "result": [ + {"title": "Ch 1", "node_id": "0000", "page_index": 2, + "prefix_summary": "opening", "text": "ch1 own", "nodes": [ + {"title": "1.1", "node_id": "0001", "page_index": 4, "summary": "s11", "text": "t11"}, + {"title": "1.2", "node_id": "0002", "page_index": 6, "summary": "s12", "text": "t12"}]}, + {"title": "Ch 2", "node_id": "0003", "page_index": 9, + "prefix_summary": "same page", "text": "ch2 own", "nodes": [ + {"title": "2.1", "node_id": "0004", "page_index": 9, "summary": "s21", "text": "t21"}]}]} + urls = [] + + def handler(method, url, kw): + urls.append(url) + return FakeResponse(wire if "type=tree" in url else {"pageNum": 12}) + _patch_requests(monkeypatch, handler) + + result = client.get_tree("pi-1", node_summary=True)["result"] + assert result == [ + {"title": "Ch 1", "node_id": "0000", "start_index": 2, "end_index": 4, + "summary": "opening", "text": "ch1 own", "nodes": [ + {"title": "1.1", "node_id": "0001", "start_index": 4, "end_index": 6, + "summary": "s11", "text": "t11"}, + {"title": "1.2", "node_id": "0002", "start_index": 6, "end_index": 9, + "summary": "s12", "text": "t12"}]}, + {"title": "Ch 2", "node_id": "0003", "start_index": 9, "end_index": 9, + "summary": "same page", "text": "ch2 own", "nodes": [ + {"title": "2.1", "node_id": "0004", "start_index": 9, "end_index": 12, + "summary": "s21", "text": "t21"}]}] + assert list(result[0]) == ["title", "node_id", "start_index", "end_index", + "summary", "text", "nodes"] + assert urls[-1] == "https://api.pageindex.ai/doc/pi-1/metadata/" + + +def test_cloud_get_tree_takes_served_ranges(monkeypatch): + """A server that sends start_index/end_index has its ranges taken as + given, with no metadata request.""" + client = PageIndexClient(api_key="secret") + wire = {"status": "completed", "retrieval_ready": True, "result": [ + {"title": "Ch", "node_id": "0000", "page_index": 2, "start_index": 2, + "end_index": 3, "prefix_summary": "opening", "nodes": [ + {"title": "S", "node_id": "0001", "page_index": 4, "start_index": 4, + "end_index": 8, "summary": "s"}]}]} + urls = [] + + def handler(method, url, kw): + urls.append(url) + return FakeResponse(wire) + _patch_requests(monkeypatch, handler) + + assert client.get_tree("pi-1", node_summary=True)["result"] == [ + {"title": "Ch", "node_id": "0000", "start_index": 2, "end_index": 3, + "summary": "opening", "nodes": [ + {"title": "S", "node_id": "0001", "start_index": 4, "end_index": 8, + "summary": "s"}]}] + assert len(urls) == 1 + + def test_cloud_error_and_empty_delete(cloud, monkeypatch): client, calls, fake = cloud _patch_requests(monkeypatch, @@ -2424,8 +2550,8 @@ async def blank(model, prompt): def test_summarize_tree_partial_empty_reply_absorbed(monkeypatch): - """One blank reply among good ones stays the documented per-node - absorption: blank summary, run survives.""" + """One blank reply among good ones is absorbed per node: the node falls + back to its own text, the run survives.""" async def flaky(model, prompt): if "alpha" in prompt: return "" @@ -2435,7 +2561,7 @@ async def flaky(model, prompt): structure = [{"title": "A", "start_index": 1, "end_index": 1}, {"title": "B", "start_index": 2, "end_index": 2}] out = asyncio.run(pageindex.utils.summarize_tree(structure, pdf_pages)) - assert [n["summary"] for n in out] == ["", "ok"] + assert [n["summary"] for n in out] == [" ".join(["alpha"] * 300)[:600], "ok"] def test_generate_doc_description_absorbs_context_overflow(monkeypatch): diff --git a/tests/test_flash_extraction.py b/tests/test_flash_extraction.py index c53d3aa6b..c2fc301df 100644 --- a/tests/test_flash_extraction.py +++ b/tests/test_flash_extraction.py @@ -707,9 +707,12 @@ def test_every_page_is_in_a_node(tmp_path): covered = {page for node in structure for page in range(node["start_index"], node["end_index"] + 1)} assert covered == set(range(1, pages + 1)), pdf.name - assert structure[0] == {"title": "Preface", "node_id": "0000", - "start_index": 1, "end_index": 1}, pdf.name - assert structure[1]["node_id"] == "0001", pdf.name + preface, first = structure[0], structure[1] + assert (preface["title"], preface["node_id"], preface["start_index"]) == ( + "Preface", "0000", 1), pdf.name + # it runs onto the first section's page unless that heading is known to open it + assert preface["end_index"] in (first["start_index"] - 1, first["start_index"]), pdf.name + assert first["node_id"] == "0001", pdf.name def test_preface_page_is_retrievable(tmp_path, monkeypatch): @@ -732,8 +735,8 @@ async def no_summary(*args, **kwargs): client = PageIndexClient(storage_path=str(tmp_path / "store")) doc_id = client.submit_document(str(pdf), mode="flash")["doc_id"] tree = client.get_tree(doc_id)["result"] - assert [(node["title"], node["page_index"]) for node in tree] == [ - ("Preface", 1), ("Budget", 2), ("Team", 3)] + assert [(node["title"], node["start_index"], node["end_index"]) for node in tree] == [ + ("Preface", 1, 1), ("Budget", 2, 2), ("Team", 3, 3)] assert "Skylark" in tree[0]["text"] diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py new file mode 100644 index 000000000..b5bf3d733 --- /dev/null +++ b/tests/test_tree_format.py @@ -0,0 +1,229 @@ +"""The tree every local index writes: intro nodes, covering ranges, leaf-only +splitting, and summaries every node gets, a parent's built from its children's.""" + +import asyncio +import importlib +from types import SimpleNamespace + +import pageindex.flash +import pageindex.tree_optimize as tree_optimize +import pageindex.utils as utils +from conftest import build_pdf +from pageindex import PageIndexClient + +classic = importlib.import_module("pageindex.page_index_classic") + + +def shape(nodes): + return [(n["title"], n["start_index"], n["end_index"]) for n in utils._subtree(nodes)] + + +def test_intro_node_holds_the_pages_a_parent_opens_with(): + lines = [["body"] for _ in range(40)] + lines[11] = ["3.1 Scope", "body"] # 3.1 opens its page, 4.1 starts mid-page + tree = [{"title": "Ch 3", "start_index": 10, "end_index": 30, "nodes": [ + {"title": "3.1 Scope", "start_index": 12, "end_index": 20}, + {"title": "3.2", "start_index": 21, "end_index": 30, "nodes": [ + {"title": "3.2.1", "start_index": 21, "end_index": 30}]}]}, + {"title": "", "start_index": 31, "end_index": 40, "nodes": [ + {"title": "4.1", "start_index": 33, "end_index": 40}]}] + + tree_optimize.add_intro_nodes(tree, lines) + + assert shape(tree) == [ + ("Ch 3", 10, 30), ("Ch 3 (intro)", 10, 11), ("3.1 Scope", 12, 20), + ("3.2", 21, 30), ("3.2.1", 21, 30), # a first child on its parent's page: no intro + ("", 31, 40), ("Intro", 31, 33), ("4.1", 33, 40)] + assert tree_optimize.add_intro_nodes(tree, lines) == tree + # a standard parent ends where its first child starts; a split intro keeps its title + split = [{"title": "Ch 3 (intro)", "start_index": 10, "end_index": 11, "nodes": [ + {"title": "Background", "start_index": 12, "end_index": 14}]}] + assert shape(tree_optimize.add_intro_nodes(split))[1] == ("Ch 3 (intro)", 10, 11) + # a heading with nothing to match cannot be placed on its page, so the page is shared + assert not tree_optimize.heading_at_page_start([["第一章 总则"]], 1, "第一章 总则") + + +def test_expand_gives_a_split_node_its_intro(monkeypatch): + body = "body " * 250 + pages = [body] * 12 + pages[5] = "Sub One\n" + body + pages[8] = "Sub Two\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 12, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "X", "start_index": 4, "end_index": 12, "node_id": "0002"}]}] + + async def propose(model, prompt): + return {"subsections": [{"title": "Sub One", "page": 6}, {"title": "Sub Two", "page": 9}]} + + async def summarize(model, prompt): + return '{"summary": "ok"}' + monkeypatch.setattr(tree_optimize, "ask_model", propose) + monkeypatch.setattr(utils, "llm_acompletion", summarize) + + async def run(): + scheduler = utils.SummaryScheduler(tree, [(page, 0) for page in pages], model="m") + await tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, + on_final=scheduler.mark_final) + await scheduler.finish() + asyncio.run(run()) + + x = tree[0]["nodes"][1] + assert [(n["title"], n["start_index"], n["end_index"], n["node_id"]) for n in x["nodes"]] == [ + ("X (intro)", 4, 5, "0003"), ("Sub One", 6, 8, "0004"), ("Sub Two", 9, 12, "0005")] + assert all(n["summary"] == "ok" for n in utils._subtree(tree)) + + +def test_merge_folds_an_intro_without_listing_its_title(): + tree = [{"title": "P", "start_index": 1, "end_index": 4, "nodes": [ + {"title": "P (intro)", "start_index": 1, "end_index": 1}, + {"title": "C", "start_index": 2, "end_index": 4}]}] + tree_optimize.merge_tree(tree) + assert "nodes" not in tree[0] and tree[0]["key_items"] == ["C"] + + +LARGE = SimpleNamespace(max_page_num_each_node=10, max_token_num_each_node=20000, model="m") + + +def test_large_node_split_keeps_existing_children(monkeypatch): + async def meta_processor(*args, **kwargs): + raise AssertionError("a parent's subsections must not be replaced") + monkeypatch.setattr(classic, "meta_processor", meta_processor) + node = {"title": "Ch", "start_index": 1, "end_index": 20, + "nodes": [{"title": "S", "start_index": 20, "end_index": 25}]} + + asyncio.run(classic.process_large_node_recursively(node, [("page", 5000)] * 25, LARGE)) + + assert [child["title"] for child in node["nodes"]] == ["S"] + + +def test_split_intro_skips_its_parents_heading(monkeypatch): + async def meta_processor(*args, **kwargs): + return [{"title": "Ch", "physical_index": 1}, + {"title": "Background", "physical_index": 5}, + {"title": "Scope", "physical_index": 9}] + + async def appear(items, *args, **kwargs): + for item in items: + item["appear_start"] = "yes" + return items + monkeypatch.setattr(classic, "meta_processor", meta_processor) + monkeypatch.setattr(classic, "check_title_appearance_in_start_concurrent", appear) + intro = {"title": "Ch (intro)", "start_index": 1, "end_index": 14} + + asyncio.run(classic.process_large_node_recursively(intro, [("page", 5000)] * 14, LARGE)) + + assert [child["title"] for child in intro["nodes"]] == ["Background", "Scope"] + + +def test_standard_index_stores_intros_covering_ranges_and_section_summaries(tmp_path, monkeypatch): + texts = ["Opening words"] + [f"page {n}" for n in range(2, 41)] + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(texts)) + + async def tree_parser(page_list, opt, doc=None, logger=None): + return [{"title": "P", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "P (intro)", "start_index": 1, "end_index": 2}, + {"title": "C1", "start_index": 3, "end_index": 20}, + {"title": "C2", "start_index": 21, "end_index": 21, "nodes": [ + {"title": "C2.1", "start_index": 21, "end_index": 30}, + {"title": "C2.2", "start_index": 31, "end_index": 40}]}]}] + prompts = [] + + async def reply(model, prompt): + prompts.append(prompt) + if "Section Title: " in prompt: + return "whole " + prompt.split("Section Title: ")[1].split("\n")[0] + return "leaf" + monkeypatch.setattr(classic, "tree_parser", tree_parser) + monkeypatch.setattr(utils, "llm_acompletion", reply) + monkeypatch.setattr(utils, "llm_completion", lambda model, prompt, **kw: "d") + opt = utils.ConfigLoader().load({"model": "m", "if_add_node_id": "yes", + "if_add_node_summary": "yes", "if_add_node_text": "yes", + "if_add_doc_description": "yes"}) + + result = classic.page_index_main(str(pdf), opt, logger=SimpleNamespace(info=print), + page_list=[(text, 1) for text in texts]) + + p = result["structure"][0] + c2 = p["nodes"][2] + assert shape([p]) == [("P", 1, 40), ("P (intro)", 1, 2), ("C1", 3, 20), + ("C2", 21, 40), ("C2.1", 21, 30), ("C2.2", 31, 40)] + assert (p["text"], p["summary"]) == ("", "whole P") # its intro holds pages 1-2 + assert "Opening words" in p["nodes"][0]["text"] + # C2's first child starts on C2's page; C2 keeps that page as its opening + assert (c2["text"], c2["summary"]) == ("page 21", "whole C2") + assert any("Section Title: C2" in q and "Opening Text: page 21" in q for q in prompts) + assert [n["summary"] for n in p["nodes"]] == ["leaf", "leaf", "whole C2"] + assert '"summary": "whole C2"' in prompts[-1] # deepest first: P is asked last + + +def test_flash_gives_parents_their_intro_nodes(tmp_path, monkeypatch): + import pageindex.flash.api as flash_api + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["x"])) + monkeypatch.setattr(flash_api, "extract_toc", lambda pdf, **kw: { + "structure": [{"title": "R", "node_id": "0000", "start_index": 1, "end_index": 6, "nodes": [ + {"title": "A", "node_id": "0001", "start_index": 3, "end_index": 4}, + {"title": "X", "node_id": "0002", "start_index": 5, "end_index": 6}]}], + "page_texts": ["Cover", "Foreword", "A\nalpha", "alpha", "X\nbody", "body"]}) + + result = flash_api.page_index_flash(str(pdf), summary=False, optimize=False) + + assert [(n["title"], n["node_id"], n["start_index"], n["end_index"]) + for n in utils._subtree(result["structure"])] == [ + ("R", "0000", 1, 6), ("R (intro)", "0001", 1, 2), ("A", "0002", 3, 4), + ("X", "0003", 5, 6)] + + +def test_flash_preface_runs_onto_a_page_its_first_heading_does_not_open(tmp_path, monkeypatch): + import pageindex.flash.api as flash_api + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["x"])) + ends = [] + for third in ["A\nalpha", "Contents, continued\nA\nalpha"]: + monkeypatch.setattr(flash_api, "extract_toc", lambda pdf, third=third, **kw: { + "structure": [{"title": "A", "node_id": "0000", "start_index": 3, "end_index": 4}], + "page_texts": ["Cover", "Contents", third, "alpha"]}) + preface = flash_api.page_index_flash(str(pdf), summary=False, optimize=False)["structure"][0] + ends.append((preface["title"], preface["node_id"], preface["end_index"])) + assert ends == [("Preface", "0000", 2), ("Preface", "0000", 3)] + + +def test_summarize_tree_parent_falls_back_to_its_subsection_titles(monkeypatch): + async def reply(model, prompt): + return "" if "Section Title" in prompt else '{"summary": "ok"}' + monkeypatch.setattr(utils, "llm_acompletion", reply) + structure = [{"title": "R", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "A", "start_index": 1, "end_index": 1}, + {"title": "B", "start_index": 2, "end_index": 2}]}] + + out = asyncio.run(utils.summarize_tree(structure, [("alpha " * 300, 0), ("beta " * 300, 0)])) + + assert [n["summary"] for n in utils._subtree(out)] == ["A; B", "ok", "ok"] + + +def test_get_tree_parent_text_shares_its_first_childs_page_unless_an_intro_holds_it( + tmp_path, monkeypatch): + pdf = tmp_path / "report.pdf" + pdf.write_bytes(build_pdf(["Opening words", "Alpha body", "Beta body"])) + structure = [{"title": "Report", "node_id": "0000", "start_index": 1, "end_index": 3, + "summary": "report", "nodes": [ + {"title": "Report (intro)", "node_id": "0001", "start_index": 1, + "end_index": 1, "summary": "opening"}, + {"title": "Alpha", "node_id": "0002", "start_index": 2, + "end_index": 3, "summary": "alpha", "nodes": [ + {"title": "Alpha one", "node_id": "0003", "start_index": 2, + "end_index": 2, "summary": "one"}, + {"title": "Alpha two", "node_id": "0004", "start_index": 3, + "end_index": 3, "summary": "two"}]}]}] + monkeypatch.setattr(pageindex.flash, "page_index_flash", lambda pdf, **kw: { + "doc_name": "report.pdf", "structure": structure}) + monkeypatch.setattr(utils, "llm_completion", lambda model, prompt, **kw: "d") + client = PageIndexClient(storage_path=str(tmp_path / "store")) + doc_id = client.submit_document(str(pdf), mode="flash")["doc_id"] + + root = client.get_tree(doc_id)["result"][0] + + assert [n["text"] for n in utils._subtree([root])] == [ + "", "Opening words", "Alpha body", "Alpha body", "Beta body"] From 6a64364118dfa4e72586d333e5a9a624bd363540 Mon Sep 17 00:00:00 2001 From: Ray Date: Thu, 1 Oct 2026 21:12:41 +0800 Subject: [PATCH 2/3] Cut each node's own text in one place A parent's text runs onto the page its first child starts on, and is empty when its intro holds those pages. LocalAPI and the standard builder each applied this rule themselves. utils.own_pages now holds it, and add_node_text and add_node_text_with_labels use it, so both get the same text as before. --- pageindex/local_api.py | 21 +++------------------ pageindex/page_index_classic.py | 2 -- pageindex/utils.py | 27 +++++++++++++++------------ 3 files changed, 18 insertions(+), 32 deletions(-) diff --git a/pageindex/local_api.py b/pageindex/local_api.py index 63f431ac6..4802634b8 100644 --- a/pageindex/local_api.py +++ b/pageindex/local_api.py @@ -260,28 +260,13 @@ def _index_flash(self, file_path: str) -> tuple[list, str | None]: # ── tree / ocr ── def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: - """Each node's own text. A parent's runs onto the page its first child - starts on, where that child's heading may sit mid-page; it is empty when - an intro child holds the parent's opening pages.""" - from .utils import get_text_of_pdf_pages, is_intro + """Each node's own text (see utils.own_pages).""" + from .utils import add_node_text structure = self._require_data( self._store.get_tree(doc_id), error_prefix) pages = self._require_pages(doc_id, error_prefix) pdf_pages = [(p.get("markdown", ""), 0) for p in pages] - - def add_own_text(nodes): - for node in nodes: - children = node.get("nodes") or [] - end = node.get("end_index") - first = children[0].get("start_index") if children else None - if children and is_intro(node, children[0]): - end = None - elif end is not None and first is not None: - end = min(end, first) - node["text"] = get_text_of_pdf_pages(pdf_pages, node.get("start_index"), end) - add_own_text(children) - - add_own_text(structure) + add_node_text(structure, pdf_pages) return structure def raw_tree(self, doc_id: str) -> list | None: diff --git a/pageindex/page_index_classic.py b/pageindex/page_index_classic.py index 015660249..ff656f9dc 100644 --- a/pageindex/page_index_classic.py +++ b/pageindex/page_index_classic.py @@ -1260,11 +1260,9 @@ async def page_index_builder(): write_node_id(structure) if opt.if_add_node_text == 'yes': add_node_text(structure, page_list) - remove_intro_parent_text(structure) if opt.if_add_node_summary == 'yes': if opt.if_add_node_text == 'no': add_node_text(structure, page_list) - remove_intro_parent_text(structure) await generate_summaries_for_structure(structure, model=getattr(opt, 'summary_model', None) or opt.model) if opt.if_add_node_text == 'no': remove_structure_text(structure) diff --git a/pageindex/utils.py b/pageindex/utils.py index fe6e1c376..2055bdc47 100644 --- a/pageindex/utils.py +++ b/pageindex/utils.py @@ -708,8 +708,7 @@ def convert_page_to_int(data): def add_node_text(node, pdf_pages): if isinstance(node, dict): - start_page = node.get('start_index') - end_page = node.get('end_index') + start_page, end_page = own_pages(node) node['text'] = get_text_of_pdf_pages(pdf_pages, start_page, end_page) if 'nodes' in node: add_node_text(node['nodes'], pdf_pages) @@ -721,8 +720,7 @@ def add_node_text(node, pdf_pages): def add_node_text_with_labels(node, pdf_pages): if isinstance(node, dict): - start_page = node.get('start_index') - end_page = node.get('end_index') + start_page, end_page = own_pages(node) node['text'] = get_text_of_pdf_pages_with_labels(pdf_pages, start_page, end_page) if 'nodes' in node: add_node_text_with_labels(node['nodes'], pdf_pages) @@ -795,14 +793,19 @@ def is_intro(parent, child): and child.get('title') == intro_title(parent.get('title'))) -def remove_intro_parent_text(structure): - """Clear the text of every parent that opens with its intro: the intro - holds the same pages.""" - for node in structure_to_list(structure): - children = node.get('nodes') or [] - if children and is_intro(node, children[0]) and 'text' in node: - node['text'] = "" - return structure +def own_pages(node): + """The first and last page of a node's own text. A parent's runs onto the + page its first child starts on, where that child's heading may sit mid-page; + it has none (end None) when its intro holds those pages.""" + start, end = node.get('start_index'), node.get('end_index') + children = node.get('nodes') or [] + if children: + first = children[0].get('start_index') + if is_intro(node, children[0]): + end = None + elif end is not None and first is not None: + end = min(end, first) + return start, end def cover_subtree_ranges(structure): From 281d8779740d87e15955c5aadca054232f52e816 Mon Sep 17 00:00:00 2001 From: Ray Date: Thu, 1 Oct 2026 22:52:20 +0800 Subject: [PATCH 3/3] Standard mode summarizes with summarize_tree Standard mode ran its section summaries level by level through a second, slower copy of what summarize_tree already does for flash. It now calls summarize_tree: a parent waits only for its own children, calls run deepest first under the concurrency cap, short leaves keep their raw text, and the prompts are flash's. generate_summaries_for_structure is back to its main version. --- pageindex/page_index_classic.py | 6 +-- pageindex/utils.py | 70 ++++++--------------------------- tests/test_client.py | 3 +- tests/test_tree_format.py | 12 +++--- 4 files changed, 20 insertions(+), 71 deletions(-) diff --git a/pageindex/page_index_classic.py b/pageindex/page_index_classic.py index ff656f9dc..7788e236b 100644 --- a/pageindex/page_index_classic.py +++ b/pageindex/page_index_classic.py @@ -1261,11 +1261,7 @@ async def page_index_builder(): if opt.if_add_node_text == 'yes': add_node_text(structure, page_list) if opt.if_add_node_summary == 'yes': - if opt.if_add_node_text == 'no': - add_node_text(structure, page_list) - await generate_summaries_for_structure(structure, model=getattr(opt, 'summary_model', None) or opt.model) - if opt.if_add_node_text == 'no': - remove_structure_text(structure) + await summarize_tree(structure, page_list, model=getattr(opt, 'summary_model', None) or opt.model) if opt.if_add_doc_description == 'yes': # Create a clean structure without unnecessary fields for description generation clean_structure = create_clean_structure_for_description(structure) diff --git a/pageindex/utils.py b/pageindex/utils.py index 2055bdc47..775e1f68a 100644 --- a/pageindex/utils.py +++ b/pageindex/utils.py @@ -741,38 +741,17 @@ async def generate_node_summary(node, model=None): return response -async def generate_section_summary(node, model=None): - listing = json.dumps( - [{'title': child.get('title', ''), 'summary': child.get('summary', '')} - for child in node['nodes']], - ensure_ascii=False) - prompt = f"""You are given a section of a document: the text that opens the section (possibly empty) and the titles and summaries of its subsections. - Your task is to generate a concise description of everything that is covered in the whole section, summarizing all its points without omitting any type of content. - Keep the description concise and to the point, avoiding unnecessary details. - - Section Title: {node.get('title', '')} - - Opening Text: {node.get('text') or ''} - - Subsection Titles and Summaries: {listing} - - Directly return the description, do not include any other text. - """ - return await llm_acompletion(model, prompt) - - FALLBACK_SUMMARY_CHARS = 600 -def fallback_summary(node, text=None): +def fallback_summary(node, text=""): """The summary of a node the model left unsummarized, so no node goes - without one: a parent's subsection titles, a leaf's opening text.""" + without one: a parent's subsection titles, a leaf's own text.""" children = node.get('nodes') or [] if children: summary = "; ".join(child['title'] for child in children if child.get('title')) else: - source = node.get('text') if text is None else text - summary = " ".join((source or "").split()) + summary = " ".join(text.split()) return summary[:FALLBACK_SUMMARY_CHARS] or node.get('title') or "" @@ -820,40 +799,15 @@ def cover_subtree_ranges(structure): async def generate_summaries_for_structure(structure, model=None): - """Leaves are summarized from their own text, parents from their children's - summaries, deepest first; a node the model leaves unsummarized gets - fallback_summary. Fails loud when no call got an answer.""" nodes = structure_to_list(structure) - levels = [] - - def collect(siblings, depth): - for node in siblings: - if node.get('nodes'): - if len(levels) <= depth: - levels.append([]) - levels[depth].append(node) - collect(node['nodes'], depth + 1) - - collect(structure if isinstance(structure, list) else [structure], 0) - answered = False - - async def summarize(group, ask): - nonlocal answered - replies = await asyncio.gather(*(ask(node, model=model) for node in group), - return_exceptions=True) - for node, reply in zip(group, replies): - if isinstance(reply, Exception) and _is_unrecoverable(reply): - raise reply - if isinstance(reply, str) and reply: - answered = True - node['summary'] = reply - else: - node['summary'] = fallback_summary(node) - - await summarize([node for node in nodes if not node.get('nodes')], generate_node_summary) - for parents in reversed(levels): - await summarize(parents, generate_section_summary) - if nodes and not answered: + tasks = [generate_node_summary(node, model=model) for node in nodes] + summaries = await asyncio.gather(*tasks, return_exceptions=True) + + for node, summary in zip(nodes, summaries): + if isinstance(summary, Exception) and _is_unrecoverable(summary): + raise summary + node['summary'] = "" if isinstance(summary, BaseException) else summary + if nodes and not any(node['summary'] for node in nodes): raise RuntimeError( "Summary generation failed for all nodes " "(every summary call failed or returned empty; " @@ -1142,7 +1096,7 @@ async def _visit(self, node, depth): if _is_unrecoverable(e): raise if not node['summary']: - node['summary'] = fallback_summary(node, None if children else get_text_of_pdf_pages( + node['summary'] = fallback_summary(node, "" if children else get_text_of_pdf_pages( self._pdf_pages, node['start_index'], node['end_index'])) async def finish(self): diff --git a/tests/test_client.py b/tests/test_client.py index dc1180054..8bffc36a4 100644 --- a/tests/test_client.py +++ b/tests/test_client.py @@ -1228,8 +1228,7 @@ async def flaky(model, prompt): result = asyncio.run(pageindex.utils.generate_summaries_for_structure(structure)) summaries = {n["title"]: n["summary"] for n in pageindex.utils.structure_to_list(result)} - # the failed parent falls back to its subsection titles - assert summaries == {"A": "B", "B": "ok"} + assert summaries == {"A": "", "B": "ok"} def test_generate_summaries_unrecoverable_raises(monkeypatch): diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index b5bf3d733..36066a2d5 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -132,9 +132,7 @@ async def tree_parser(page_list, opt, doc=None, logger=None): async def reply(model, prompt): prompts.append(prompt) - if "Section Title: " in prompt: - return "whole " + prompt.split("Section Title: ")[1].split("\n")[0] - return "leaf" + return "whole " + prompt.split("Section Title: ")[1].split("\n")[0] monkeypatch.setattr(classic, "tree_parser", tree_parser) monkeypatch.setattr(utils, "llm_acompletion", reply) monkeypatch.setattr(utils, "llm_completion", lambda model, prompt, **kw: "d") @@ -153,9 +151,11 @@ async def reply(model, prompt): assert "Opening words" in p["nodes"][0]["text"] # C2's first child starts on C2's page; C2 keeps that page as its opening assert (c2["text"], c2["summary"]) == ("page 21", "whole C2") - assert any("Section Title: C2" in q and "Opening Text: page 21" in q for q in prompts) - assert [n["summary"] for n in p["nodes"]] == ["leaf", "leaf", "whole C2"] - assert '"summary": "whole C2"' in prompts[-1] # deepest first: P is asked last + # short leaves keep their own text; only the parents are asked, deepest first + assert [n["summary"] for n in p["nodes"][:2]] == [ + "Opening wordspage 2", "".join(f"page {n}" for n in range(3, 21))] + assert [q.split("Section Title: ")[1].split("\n")[0] for q in prompts] == ["C2", "P"] + assert '"summary": "whole C2"' in prompts[-1] def test_flash_gives_parents_their_intro_nodes(tmp_path, monkeypatch):