From 6d23caf416858f2ca136840305d1f479a86f6ef7 Mon Sep 17 00:00:00 2001 From: Ray Date: Thu, 1 Oct 2026 23:20:09 +0800 Subject: [PATCH 1/4] Unify the document tree across local and cloud (#541) get_tree returns one node shape in local and cloud mode: {title, node_id, start_index, end_index, summary, text, nodes}. page_index and prefix_summary no longer appear. The SDK only renames fields on the way out, so a document indexed before keeps its own ranges and summaries. New local indexes, standard and flash: - A parent whose first child starts on a later page gets a first child " (intro)" that holds those pages. - A parent's range covers its whole subtree, and its summary is written from its children's summaries. Standard mode now summarizes with summarize_tree, as flash does. - A node the model leaves unsummarized falls back to its subsection titles or its own text. - The standard large-node split acts on leaves only. A node's text is its own pages. A parent's runs onto the page its first child starts on, and is empty when its intro holds those pages. --- pageindex/client.py | 20 ++- pageindex/flash/api.py | 35 +++-- pageindex/local_api.py | 12 +- pageindex/page_index_classic.py | 24 ++-- pageindex/tree_optimize.py | 38 +++++- pageindex/utils.py | 125 ++++++++++++++++- tests/test_client.py | 157 +++++++++++++++++++--- tests/test_flash_extraction.py | 13 +- tests/test_tree_format.py | 229 ++++++++++++++++++++++++++++++++ 9 files changed, 585 insertions(+), 68 deletions(-) create mode 100644 tests/test_tree_format.py diff --git a/pageindex/client.py b/pageindex/client.py index 4dea417c1..0d77269bf 100644 --- a/pageindex/client.py +++ b/pageindex/client.py @@ -949,14 +949,22 @@ def get_tree(self, doc_id: str, node_summary: bool = False, Returns: dict: {'doc_id', 'status', 'retrieval_ready', 'result', ...} where - result nodes are {'title', 'node_id', 'page_index', ('summary' / - 'prefix_summary',) ('text',) 'nodes'}. + result nodes are {'title', 'node_id', 'start_index', 'end_index', + ('summary',) ('text',) 'nodes'} in both modes. Each node's summary + describes its own pages start_index..end_index; its text is the + part no child holds, though a parent's may share the page its + first child starts on. """ tree = self._api.get_tree(doc_id=doc_id, node_summary=node_summary, include_text=include_text) - if not include_text and tree.get("result"): - from .utils import remove_fields - tree["result"] = remove_fields(tree["result"], fields=["text"]) + if tree.get("result"): + from .utils import _subtree, remove_fields, unify_tree + page_count = None + if any("end_index" not in node for node in _subtree(tree["result"])): + page_count = self._api.get_document(doc_id).get("pageNum") + tree["result"] = unify_tree(tree["result"], page_count) + if not include_text: + tree["result"] = remove_fields(tree["result"], fields=["text"]) return tree def get_document_structure(self, doc_id: str) -> list[dict[str, Any]]: @@ -975,7 +983,7 @@ def is_retrieval_ready(self, doc_id: str) -> bool: failures, timeouts) propagate. """ try: - result = self.get_tree(doc_id) + result = self._api.get_tree(doc_id) return result.get("retrieval_ready", False) except PageIndexAPIError: return False diff --git a/pageindex/flash/api.py b/pageindex/flash/api.py index bc8c71446..5e3e397e1 100644 --- a/pageindex/flash/api.py +++ b/pageindex/flash/api.py @@ -89,9 +89,7 @@ async def _optimize_async(structure, page_texts, do_expand, model, on_final=None the summaries use. """ from ..tree_optimize import optimize - lines = [[line_text.strip() for line_text in (page_text or "").splitlines() - if line_text.strip()] - for page_text in page_texts] + lines = _page_lines(page_texts) outcome = await optimize(structure, page_texts, lines, model=model, do_expand=do_expand, page_count=len(page_texts), on_final=on_final, concurrency=concurrency) @@ -134,14 +132,31 @@ def _page_nodes(page_texts: list[str]) -> list[dict]: return nodes -def _add_preface(structure: list[dict]) -> None: - """The pages before a hierarchy that starts late become a Preface node, as in standard mode.""" +def _page_lines(page_texts: list[str]) -> list[list[str]]: + return [[line_text.strip() for line_text in (page_text or "").splitlines() + if line_text.strip()] + for page_text in page_texts] + + +def _add_intros(structure: list[dict], lines: list[list[str]]) -> None: + """The pages a parent opens with before its first child become its intro node.""" + from ..tree_optimize import add_intro_nodes from ..utils import write_node_id - structure.insert(0, {"title": "Preface", "start_index": 1, - "end_index": structure[0]["start_index"] - 1}) + add_intro_nodes(structure, lines) write_node_id(structure) +def _add_preface(structure: list[dict], lines: list[list[str]]) -> None: + """The pages before a hierarchy that starts late become a Preface node, as in + standard mode. It runs onto the first section's page unless that heading opens it.""" + from ..tree_optimize import heading_at_page_start + first = structure[0] + opens = first["start_index"] <= len(lines) and heading_at_page_start( + lines, first["start_index"], first["title"]) + structure.insert(0, {"title": "Preface", "start_index": 1, + "end_index": first["start_index"] - (1 if opens else 0)}) + + def flash_rejection_reason(result: dict, standard_hint: str = "mode='standard'") -> str | None: """Why a managed pipeline should refuse this flash result, or None to accept it. @@ -166,7 +181,7 @@ def page_index_flash(pdf, summary=True, summary_model=None, optimize: str | bool | None = None, optimize_expand=None, optimize_model=None, summary_concurrency=None, use_embedded_toc=True, summary_max_words=None) -> dict: - """Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """ + """Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode, and the first section's page too unless that heading opens it; a parent whose first child starts on a later page opens with a child titled ``" (intro)"`` holding the pages before it; a parent's range and summary cover its whole subtree) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """ for name, value in (("summary_concurrency", summary_concurrency), ("summary_max_words", summary_max_words)): if value is not None and not (isinstance(value, numbers.Integral) and int(value) >= 1): @@ -195,7 +210,9 @@ def page_index_flash(pdf, summary=True, summary_model=None, result["structure"] = structure result["toc_source"] = "pages" if structure else "unreadable" elif structure[0]["start_index"] > 1: - _add_preface(structure) + _add_preface(structure, _page_lines(result.get("page_texts") or [])) + if structure: + _add_intros(structure, _page_lines(result.get("page_texts") or [])) if result.get("toc_source") == "pages" and len(structure) > FLAT_TREE_MAX_NODES: # the managed pipelines refuse a flat tree this size; skip the model passes result.pop("page_texts", None) diff --git a/pageindex/local_api.py b/pageindex/local_api.py index 025ab716d..4802634b8 100644 --- a/pageindex/local_api.py +++ b/pageindex/local_api.py @@ -260,6 +260,7 @@ def _index_flash(self, file_path: str) -> tuple[list, str | None]: # ── tree / ocr ── def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: + """Each node's own text (see utils.own_pages).""" from .utils import add_node_text structure = self._require_data( self._store.get_tree(doc_id), error_prefix) @@ -269,8 +270,7 @@ def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: return structure def raw_tree(self, doc_id: str) -> list | None: - """Stored tree verbatim — keeps start_index/end_index, which - get_tree's cloud wire shape renames and drops.""" + """Stored tree verbatim, every key kept.""" return self._store.get_tree(doc_id) def get_tree(self, doc_id: str, node_summary: bool = False, @@ -398,17 +398,15 @@ def _format_tree_node(node: dict, node_summary: bool) -> dict: out = { "title": node.get("title", ""), "node_id": node.get("node_id"), - "page_index": node.get("start_index"), + "start_index": node.get("start_index"), + "end_index": node.get("end_index"), } if node.get("key_items"): out["key_items"] = node["key_items"] if node_summary: summary = node.get("summary") if summary is not None: - if children: - out["prefix_summary"] = summary - else: - out["summary"] = summary + out["summary"] = summary if "text" in node: out["text"] = node["text"] if children: diff --git a/pageindex/page_index_classic.py b/pageindex/page_index_classic.py index d680154eb..7788e236b 100644 --- a/pageindex/page_index_classic.py +++ b/pageindex/page_index_classic.py @@ -6,7 +6,7 @@ import re from .utils import * from .naming import sanitize_filename as sanitize_upload_filename -from .tree_optimize import merge_tree +from .tree_optimize import add_intro_nodes, merge_tree import os ######################### Hardening for prompt injection patterns #################################################### @@ -1169,7 +1169,8 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None) node_page_list = page_list[node['start_index']-1:node['end_index']] token_num = sum([page[1] for page in node_page_list]) - if node['end_index'] - node['start_index'] > opt.max_page_num_each_node and token_num >= opt.max_token_num_each_node: + if (not node.get('nodes') and node['end_index'] - node['start_index'] > opt.max_page_num_each_node + and token_num >= opt.max_token_num_each_node): print('large node:', node['title'], 'start_index:', node['start_index'], 'end_index:', node['end_index'], 'token_num:', token_num) node_toc_tree = await meta_processor(node_page_list, mode='process_no_toc', start_index=node['start_index'], opt=opt, logger=logger) @@ -1178,7 +1179,9 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None) # Filter out items with None physical_index before post_processing valid_node_toc_items = [item for item in node_toc_tree if item.get('physical_index') is not None] - if valid_node_toc_items and node['title'].strip() == valid_node_toc_items[0]['title'].strip(): + # an intro node's pages open with its parent's heading + title = node['title'].strip() + if valid_node_toc_items and valid_node_toc_items[0]['title'].strip() in (title, title.removesuffix(INTRO_SUFFIX)): node['nodes'] = post_processing(valid_node_toc_items[1:], node['end_index']) node['end_index'] = valid_node_toc_items[1]['start_index'] if len(valid_node_toc_items) > 1 else node['end_index'] else: @@ -1221,14 +1224,15 @@ async def tree_parser(page_list, opt, doc=None, logger=None): # Filter out items with None physical_index before post_processings valid_toc_items = [item for item in toc_with_page_number if item.get('physical_index') is not None] - toc_tree = post_processing(valid_toc_items, len(page_list)) + toc_tree = add_intro_nodes(post_processing(valid_toc_items, len(page_list))) tasks = [ process_large_node_recursively(node, page_list, opt, logger=logger) for node in toc_tree ] await asyncio.gather(*tasks) - - return toc_tree + + # a split leaf is now a parent that may open before its first child + return add_intro_nodes(toc_tree) def page_index_main(doc, opt=None, logger=None, page_list=None): @@ -1257,21 +1261,19 @@ async def page_index_builder(): if opt.if_add_node_text == 'yes': add_node_text(structure, page_list) if opt.if_add_node_summary == 'yes': - if opt.if_add_node_text == 'no': - add_node_text(structure, page_list) - await generate_summaries_for_structure(structure, model=getattr(opt, 'summary_model', None) or opt.model) - if opt.if_add_node_text == 'no': - remove_structure_text(structure) + await summarize_tree(structure, page_list, model=getattr(opt, 'summary_model', None) or opt.model) if opt.if_add_doc_description == 'yes': # Create a clean structure without unnecessary fields for description generation clean_structure = create_clean_structure_for_description(structure) doc_description = generate_doc_description(clean_structure, model=getattr(opt, 'summary_model', None) or opt.model) + cover_subtree_ranges(structure) structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes']) return { 'doc_name': get_pdf_name(doc), 'doc_description': doc_description, 'structure': structure, } + cover_subtree_ranges(structure) structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes']) return { 'doc_name': get_pdf_name(doc), diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index ddf397c7d..e607d5f00 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -62,8 +62,8 @@ import sys from types import SimpleNamespace -from .utils import (ConfigLoader, _is_unrecoverable, llm_acompletion, - strip_internal_keys) +from .utils import (ConfigLoader, _is_unrecoverable, intro_title, is_intro, + llm_acompletion, strip_internal_keys) TRIGGER_PAGES = 5 # only look ahead on nodes larger than this ROUTING_COST = 1 # R(v), in pages @@ -171,11 +171,13 @@ def is_frontier(node): def heading_at_page_start(lines, page_no, heading): - """Is the heading the first line on its page?""" + """Is the heading the first line on its page? No when that cannot be told + (a heading with no Latin letter or digit to match), so the page is shared.""" page = lines[page_no - 1] - if not page: + key = normalize(heading) + if not page or not key: return False - return normalize(heading) in normalize(page[0]) + return key in normalize(page[0]) def assign_ends(node, children, lines): @@ -202,9 +204,30 @@ def attach_children(node, children, lines): # so gaining children leaves it unchanged sized = assign_ends(node, children, lines) node["nodes"] = sized + add_intro_nodes([node], lines) + if len(node["nodes"]) > len(sized): # an intro went in; numbered like the proposals + node["nodes"][0]["node_id"] = f"{node['node_id']}.0" return sized +def add_intro_nodes(structure, lines=None): + """Give every parent whose first child starts on a later page an intro child + for the pages before it, titled " (intro)". It ends where the + parent's own pages do, and no later than the first child's page, the page + before it when that child's heading opens its page.""" + for node, _ in list(flatten(structure)): + children = node.get("nodes") or [] + first = children[0]["start_index"] if children else None + if first is None or first <= node["start_index"]: + continue + opens = bool(lines) and first <= len(lines) and heading_at_page_start( + lines, first, children[0]["title"]) + end = max(node["start_index"], min(node["end_index"], first - 1 if opens else first)) + node["nodes"] = [{"title": intro_title(node.get("title")), + "start_index": node["start_index"], "end_index": end}] + children + return structure + + def relabel(structure, width=4): """Renumber every node_id in document order: 0000, 0001, 0002, ... @@ -558,8 +581,9 @@ def visit(node): # titles are routing information; keep them on the parent, in document # order, carrying forward anything an earlier merge already folded in titles = [] - for child, _ in flatten(node["nodes"]): - titles.append(child["title"]) + for child, parent in flatten(node["nodes"]): + if not is_intro(parent or node, child): # its parent's heading again + titles.append(child["title"]) titles.extend(child.get("key_items") or []) log.append({"op": "merge", "node_id": node.get("node_id"), "S": span, "tree_cost": cost, "frontier_cost": checked, diff --git a/pageindex/utils.py b/pageindex/utils.py index 31ee768b1..4981a63c5 100644 --- a/pageindex/utils.py +++ b/pageindex/utils.py @@ -708,8 +708,7 @@ def convert_page_to_int(data): def add_node_text(node, pdf_pages): if isinstance(node, dict): - start_page = node.get('start_index') - end_page = node.get('end_index') + start_page, end_page = own_pages(node) node['text'] = get_text_of_pdf_pages(pdf_pages, start_page, end_page) if 'nodes' in node: add_node_text(node['nodes'], pdf_pages) @@ -721,8 +720,7 @@ def add_node_text(node, pdf_pages): def add_node_text_with_labels(node, pdf_pages): if isinstance(node, dict): - start_page = node.get('start_index') - end_page = node.get('end_index') + start_page, end_page = own_pages(node) node['text'] = get_text_of_pdf_pages_with_labels(pdf_pages, start_page, end_page) if 'nodes' in node: add_node_text_with_labels(node['nodes'], pdf_pages) @@ -743,6 +741,63 @@ async def generate_node_summary(node, model=None): return response +FALLBACK_SUMMARY_CHARS = 600 + + +def fallback_summary(node, text=""): + """The summary of a node the model left unsummarized, so no node goes + without one: a parent's subsection titles, a leaf's own text.""" + children = node.get('nodes') or [] + if children: + summary = "; ".join(child['title'] for child in children if child.get('title')) + else: + summary = " ".join(text.split()) + return summary[:FALLBACK_SUMMARY_CHARS] or node.get('title') or "" + + +INTRO_SUFFIX = " (intro)" + + +def intro_title(title): + """The title of a parent's intro node; a split intro keeps its own.""" + title = (title or "").strip() + if title and not title.endswith(INTRO_SUFFIX): + title += INTRO_SUFFIX + return title or "Intro" + + +def is_intro(parent, child): + """Whether child is the intro node holding parent's opening pages.""" + return (child.get('start_index') == parent.get('start_index') + and child.get('title') == intro_title(parent.get('title'))) + + +def own_pages(node): + """The first and last page of a node's own text. A parent's runs onto the + page its first child starts on, where that child's heading may sit mid-page; + it has none (end None) when its intro holds those pages.""" + start, end = node.get('start_index'), node.get('end_index') + children = node.get('nodes') or [] + if children: + first = children[0].get('start_index') + if is_intro(node, children[0]): + end = None + elif end is not None and first is not None: + end = min(end, first) + return start, end + + +def cover_subtree_ranges(structure): + """Widen every parent's range to cover its whole subtree.""" + for node in structure if isinstance(structure, list) else [structure]: + children = node.get('nodes') or [] + if children: + cover_subtree_ranges(children) + node['start_index'] = min([node['start_index']] + [c['start_index'] for c in children]) + node['end_index'] = max([node['end_index']] + [c['end_index'] for c in children]) + return structure + + async def generate_summaries_for_structure(structure, model=None): nodes = structure_to_list(structure) tasks = [generate_node_summary(node, model=model) for node in nodes] @@ -1040,6 +1095,9 @@ async def _visit(self, node, depth): node['summary'] = "" if _is_unrecoverable(e): raise + if not node['summary']: + node['summary'] = fallback_summary(node, "" if children else get_text_of_pdf_pages( + self._pdf_pages, node['start_index'], node['end_index'])) async def finish(self): """Wait for every summary; fails loud if the model never answered.""" @@ -1252,9 +1310,58 @@ def load(self, user_opt=None) -> config: _resolve_models(merged) return config(**merged) +_TREE_WIRE_KEYS = frozenset({"title", "node_id", "start_index", "end_index", "page_index", + "summary", "prefix_summary", "text", "nodes"}) + + +def unify_tree(nodes, page_count=None): + """get_tree's nodes in the field names both modes return. + + The structure is the index's own: nodes are neither added nor moved, so + every node keeps the summary written for its own pages start_index .. + end_index. A node's first page comes as start_index (local) or + page_index (cloud), and a parent's summary as summary or prefix_summary. + Without an end_index (an older server) a node ends on the page where the + next node in reading order starts, the last on page_count. + """ + flat = list(_subtree(nodes)) + starts = [node.get("start_index", node.get("page_index")) for node in flat] + ranges = {} + for i, node in enumerate(flat): + start, end = starts[i], node.get("end_index") + if end is None: + end = starts[i + 1] if i + 1 < len(flat) else page_count + if end is None or (start is not None and end < start): + end = start + ranges[id(node)] = (start, end) + + def build(siblings): + out = [] + for node in siblings: + unified = {"title": node.get("title", "")} + if "node_id" in node: + unified["node_id"] = node["node_id"] + unified["start_index"], unified["end_index"] = ranges[id(node)] + unified.update((key, value) for key, value in node.items() + if key not in _TREE_WIRE_KEYS) + for key in ("summary", "prefix_summary"): + if key in node: + unified["summary"] = node[key] + if "text" in node: + unified["text"] = node["text"] + if node.get("nodes"): + unified["nodes"] = build(node["nodes"]) + out.append(unified) + return out + + return build(nodes) + + def create_node_mapping(tree, include_page_ranges=False, max_page=None): """Map node_id to node; with include_page_ranges, to {"node", "start_index", - "end_index"} (end = next node's page_index, or max_page for the last node).""" + "end_index"}: the node's own range as get_tree returns it, or, for a tree + that carries page_index only, end = next node's page_index (max_page for + the last node).""" def get_all_nodes(tree): if isinstance(tree, dict): return [tree] + [node for child in tree.get('nodes') or [] for node in get_all_nodes(child)] @@ -1268,10 +1375,14 @@ def get_all_nodes(tree): mapping = {} for i, node in enumerate(all_nodes): if node.get("node_id"): - end_page = all_nodes[i + 1].get("page_index") if i + 1 < len(all_nodes) else max_page + if "end_index" in node: + start_page, end_page = node["start_index"], node["end_index"] + else: + start_page = node["page_index"] + end_page = all_nodes[i + 1].get("page_index") if i + 1 < len(all_nodes) else max_page mapping[node["node_id"]] = { "node": node, - "start_index": node["page_index"], + "start_index": start_page, "end_index": end_page, } return mapping diff --git a/tests/test_client.py b/tests/test_client.py index 66b069cd3..814415ca6 100644 --- a/tests/test_client.py +++ b/tests/test_client.py @@ -549,24 +549,24 @@ def test_submit_and_get_tree(local_client, indexed_doc, tmp_path, monkeypatch): assert tree["status"] == "completed" assert tree["retrieval_ready"] is True root = tree["result"][0] - assert root["page_index"] == 1 - assert "start_index" not in root and "end_index" not in root - assert root["prefix_summary"] == "root summary" - assert "summary" not in root - child = root["nodes"][0] + assert (root["start_index"], root["end_index"]) == (1, 2) + assert "page_index" not in root and "prefix_summary" not in root + assert root["summary"] == "root summary" + assert "Hello page one" in root["text"] + child, = root["nodes"] + assert (child["start_index"], child["end_index"]) == (2, 2) assert child["summary"] == "child summary" assert child["text"] == "Second page about bananas" no_summary = local_client.get_tree(indexed_doc)["result"][0] - assert "summary" not in no_summary and "prefix_summary" not in no_summary + assert all("summary" not in node for node in [no_summary, *no_summary["nodes"]]) def test_get_tree_include_text_false(local_client, indexed_doc): tree = local_client.get_tree(indexed_doc, include_text=False) root = tree["result"][0] - assert "text" not in root - assert "text" not in root["nodes"][0] - assert root["page_index"] == 1 + assert all("text" not in node for node in [root, *root["nodes"]]) + assert (root["start_index"], root["end_index"]) == (1, 2) with_text = local_client.get_tree(indexed_doc)["result"][0] assert "text" in with_text @@ -576,10 +576,72 @@ def test_get_document_structure(local_client, indexed_doc): result = local_client.get_document_structure(indexed_doc) assert isinstance(result, list) root = result[0] - assert "text" not in root - assert "text" not in root["nodes"][0] - assert "prefix_summary" in root - assert root["nodes"][0]["summary"] == "child summary" + assert all("text" not in node for node in [root, *root["nodes"]]) + assert (root["start_index"], root["end_index"]) == (1, 2) + assert [node.get("summary") for node in [root, *root["nodes"]]] == [ + "root summary", "child summary"] + + +@pytest.mark.parametrize("mode", ["flash", "standard"]) +def test_get_tree_keeps_each_index_self_consistent(local_client, tmp_path, + monkeypatch, mode): + """get_tree adds and moves no node: every node keeps the summary written + for its own range, a flash parent's range covering its subtree and a + standard parent's ending where its first child starts.""" + from conftest import build_pdf + pdf = tmp_path / "report.pdf" + pdf.write_bytes(build_pdf(["Opening words", "Alpha body", "Beta body"])) + structure = [{"title": "Report", "node_id": "0000", "start_index": 1, + "end_index": 3 if mode == "flash" else 2, "summary": "report", + "nodes": [{"title": "Alpha", "node_id": "0001", "start_index": 2, + "end_index": 2, "summary": "alpha"}, + {"title": "Beta", "node_id": "0002", "start_index": 3, + "end_index": 3, "summary": "beta"}]}] + result = {"doc_name": "report.pdf", "doc_description": "d", "structure": structure} + monkeypatch.setattr(pageindex.flash, "page_index_flash", lambda pdf, **kw: result) + monkeypatch.setattr(page_index_module, "page_index_main", lambda *a, **kw: result) + monkeypatch.setattr(pageindex.utils, "llm_completion", lambda model, prompt, **kw: "d") + doc_id = local_client.submit_document(str(pdf), mode=mode)["doc_id"] + + root = local_client.get_tree(doc_id, node_summary=True)["result"][0] + assert [(node["title"], node["start_index"], node["end_index"], node["summary"]) + for node in [root, *root["nodes"]]] == [ + ("Report", 1, 3 if mode == "flash" else 2, "report"), + ("Alpha", 2, 2, "alpha"), ("Beta", 3, 3, "beta")] + # the parent's text runs onto its first child's page, where that heading may sit mid-page + assert "Alpha body" in root["text"] and "Beta body" not in root["text"] + + +def test_unify_tree_edges(): + from pageindex.utils import unify_tree + tree = [{"title": "A", "page_index": 3, "prefix_summary": "a", + "nodes": [{"title": "A.1", "page_index": 5, "summary": "a1"}, + {"title": "A.2", "page_index": 7, "prefix_summary": "a2", + "nodes": [{"title": "A.2.1", "page_index": 8}]}]}] + once = unify_tree(tree, 9) + # without end_index each node ends where the next one in reading order + # starts: a parent's range is its own opening, as its summary describes + assert once == [ + {"title": "A", "start_index": 3, "end_index": 5, "summary": "a", "nodes": [ + {"title": "A.1", "start_index": 5, "end_index": 7, "summary": "a1"}, + {"title": "A.2", "start_index": 7, "end_index": 8, "summary": "a2", "nodes": [ + {"title": "A.2.1", "start_index": 8, "end_index": 9}]}]}] + assert unify_tree(once, 9) == once + # a node with no page (a hand-edited store) passes through without raising + assert unify_tree([{"title": "B", "start_index": None}], None) == [ + {"title": "B", "start_index": None, "end_index": None}] + + +def test_create_node_mapping_reads_get_tree_ranges(): + from pageindex.utils import create_node_mapping + tree = [{"title": "A", "node_id": "0000", "start_index": 1, "end_index": 9, + "nodes": [{"title": "B", "node_id": "0001", "start_index": 4, "end_index": 9}]}] + mapping = create_node_mapping(tree, include_page_ranges=True, max_page=12) + assert [(m["start_index"], m["end_index"]) for m in mapping.values()] == [(1, 9), (4, 9)] + legacy = [{"title": "A", "node_id": "0000", "page_index": 1, + "nodes": [{"title": "B", "node_id": "0001", "page_index": 4}]}] + mapping = create_node_mapping(legacy, include_page_ranges=True, max_page=12) + assert [(m["start_index"], m["end_index"]) for m in mapping.values()] == [(1, 4), (4, 12)] def test_get_page_content(local_client, indexed_doc): @@ -1696,6 +1758,69 @@ def test_cloud_request_wiring(cloud, sample_pdf): assert calls[-1]["headers"] == {"api_key": "other"} +def test_cloud_get_tree_unified_shape(monkeypatch): + """An older server's tree wire (page_index only, a parent's + prefix_summary) leaves the SDK in local's field names: a node ends where + the next one in reading order starts, the last on the document's page + count, so a parent's range is the opening its summary describes.""" + client = PageIndexClient(api_key="secret") + wire = {"doc_id": "pi-1", "status": "completed", "retrieval_ready": True, + "metadata": None, "features": {}, "result": [ + {"title": "Ch 1", "node_id": "0000", "page_index": 2, + "prefix_summary": "opening", "text": "ch1 own", "nodes": [ + {"title": "1.1", "node_id": "0001", "page_index": 4, "summary": "s11", "text": "t11"}, + {"title": "1.2", "node_id": "0002", "page_index": 6, "summary": "s12", "text": "t12"}]}, + {"title": "Ch 2", "node_id": "0003", "page_index": 9, + "prefix_summary": "same page", "text": "ch2 own", "nodes": [ + {"title": "2.1", "node_id": "0004", "page_index": 9, "summary": "s21", "text": "t21"}]}]} + urls = [] + + def handler(method, url, kw): + urls.append(url) + return FakeResponse(wire if "type=tree" in url else {"pageNum": 12}) + _patch_requests(monkeypatch, handler) + + result = client.get_tree("pi-1", node_summary=True)["result"] + assert result == [ + {"title": "Ch 1", "node_id": "0000", "start_index": 2, "end_index": 4, + "summary": "opening", "text": "ch1 own", "nodes": [ + {"title": "1.1", "node_id": "0001", "start_index": 4, "end_index": 6, + "summary": "s11", "text": "t11"}, + {"title": "1.2", "node_id": "0002", "start_index": 6, "end_index": 9, + "summary": "s12", "text": "t12"}]}, + {"title": "Ch 2", "node_id": "0003", "start_index": 9, "end_index": 9, + "summary": "same page", "text": "ch2 own", "nodes": [ + {"title": "2.1", "node_id": "0004", "start_index": 9, "end_index": 12, + "summary": "s21", "text": "t21"}]}] + assert list(result[0]) == ["title", "node_id", "start_index", "end_index", + "summary", "text", "nodes"] + assert urls[-1] == "https://api.pageindex.ai/doc/pi-1/metadata/" + + +def test_cloud_get_tree_takes_served_ranges(monkeypatch): + """A server that sends start_index/end_index has its ranges taken as + given, with no metadata request.""" + client = PageIndexClient(api_key="secret") + wire = {"status": "completed", "retrieval_ready": True, "result": [ + {"title": "Ch", "node_id": "0000", "page_index": 2, "start_index": 2, + "end_index": 3, "prefix_summary": "opening", "nodes": [ + {"title": "S", "node_id": "0001", "page_index": 4, "start_index": 4, + "end_index": 8, "summary": "s"}]}]} + urls = [] + + def handler(method, url, kw): + urls.append(url) + return FakeResponse(wire) + _patch_requests(monkeypatch, handler) + + assert client.get_tree("pi-1", node_summary=True)["result"] == [ + {"title": "Ch", "node_id": "0000", "start_index": 2, "end_index": 3, + "summary": "opening", "nodes": [ + {"title": "S", "node_id": "0001", "start_index": 4, "end_index": 8, + "summary": "s"}]}] + assert len(urls) == 1 + + def test_cloud_error_and_empty_delete(cloud, monkeypatch): client, calls, fake = cloud _patch_requests(monkeypatch, @@ -2424,8 +2549,8 @@ async def blank(model, prompt): def test_summarize_tree_partial_empty_reply_absorbed(monkeypatch): - """One blank reply among good ones stays the documented per-node - absorption: blank summary, run survives.""" + """One blank reply among good ones is absorbed per node: the node falls + back to its own text, the run survives.""" async def flaky(model, prompt): if "alpha" in prompt: return "" @@ -2435,7 +2560,7 @@ async def flaky(model, prompt): structure = [{"title": "A", "start_index": 1, "end_index": 1}, {"title": "B", "start_index": 2, "end_index": 2}] out = asyncio.run(pageindex.utils.summarize_tree(structure, pdf_pages)) - assert [n["summary"] for n in out] == ["", "ok"] + assert [n["summary"] for n in out] == [" ".join(["alpha"] * 300)[:600], "ok"] def test_generate_doc_description_absorbs_context_overflow(monkeypatch): diff --git a/tests/test_flash_extraction.py b/tests/test_flash_extraction.py index c53d3aa6b..c2fc301df 100644 --- a/tests/test_flash_extraction.py +++ b/tests/test_flash_extraction.py @@ -707,9 +707,12 @@ def test_every_page_is_in_a_node(tmp_path): covered = {page for node in structure for page in range(node["start_index"], node["end_index"] + 1)} assert covered == set(range(1, pages + 1)), pdf.name - assert structure[0] == {"title": "Preface", "node_id": "0000", - "start_index": 1, "end_index": 1}, pdf.name - assert structure[1]["node_id"] == "0001", pdf.name + preface, first = structure[0], structure[1] + assert (preface["title"], preface["node_id"], preface["start_index"]) == ( + "Preface", "0000", 1), pdf.name + # it runs onto the first section's page unless that heading is known to open it + assert preface["end_index"] in (first["start_index"] - 1, first["start_index"]), pdf.name + assert first["node_id"] == "0001", pdf.name def test_preface_page_is_retrievable(tmp_path, monkeypatch): @@ -732,8 +735,8 @@ async def no_summary(*args, **kwargs): client = PageIndexClient(storage_path=str(tmp_path / "store")) doc_id = client.submit_document(str(pdf), mode="flash")["doc_id"] tree = client.get_tree(doc_id)["result"] - assert [(node["title"], node["page_index"]) for node in tree] == [ - ("Preface", 1), ("Budget", 2), ("Team", 3)] + assert [(node["title"], node["start_index"], node["end_index"]) for node in tree] == [ + ("Preface", 1, 1), ("Budget", 2, 2), ("Team", 3, 3)] assert "Skylark" in tree[0]["text"] diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py new file mode 100644 index 000000000..36066a2d5 --- /dev/null +++ b/tests/test_tree_format.py @@ -0,0 +1,229 @@ +"""The tree every local index writes: intro nodes, covering ranges, leaf-only +splitting, and summaries every node gets, a parent's built from its children's.""" + +import asyncio +import importlib +from types import SimpleNamespace + +import pageindex.flash +import pageindex.tree_optimize as tree_optimize +import pageindex.utils as utils +from conftest import build_pdf +from pageindex import PageIndexClient + +classic = importlib.import_module("pageindex.page_index_classic") + + +def shape(nodes): + return [(n["title"], n["start_index"], n["end_index"]) for n in utils._subtree(nodes)] + + +def test_intro_node_holds_the_pages_a_parent_opens_with(): + lines = [["body"] for _ in range(40)] + lines[11] = ["3.1 Scope", "body"] # 3.1 opens its page, 4.1 starts mid-page + tree = [{"title": "Ch 3", "start_index": 10, "end_index": 30, "nodes": [ + {"title": "3.1 Scope", "start_index": 12, "end_index": 20}, + {"title": "3.2", "start_index": 21, "end_index": 30, "nodes": [ + {"title": "3.2.1", "start_index": 21, "end_index": 30}]}]}, + {"title": "", "start_index": 31, "end_index": 40, "nodes": [ + {"title": "4.1", "start_index": 33, "end_index": 40}]}] + + tree_optimize.add_intro_nodes(tree, lines) + + assert shape(tree) == [ + ("Ch 3", 10, 30), ("Ch 3 (intro)", 10, 11), ("3.1 Scope", 12, 20), + ("3.2", 21, 30), ("3.2.1", 21, 30), # a first child on its parent's page: no intro + ("", 31, 40), ("Intro", 31, 33), ("4.1", 33, 40)] + assert tree_optimize.add_intro_nodes(tree, lines) == tree + # a standard parent ends where its first child starts; a split intro keeps its title + split = [{"title": "Ch 3 (intro)", "start_index": 10, "end_index": 11, "nodes": [ + {"title": "Background", "start_index": 12, "end_index": 14}]}] + assert shape(tree_optimize.add_intro_nodes(split))[1] == ("Ch 3 (intro)", 10, 11) + # a heading with nothing to match cannot be placed on its page, so the page is shared + assert not tree_optimize.heading_at_page_start([["第一章 总则"]], 1, "第一章 总则") + + +def test_expand_gives_a_split_node_its_intro(monkeypatch): + body = "body " * 250 + pages = [body] * 12 + pages[5] = "Sub One\n" + body + pages[8] = "Sub Two\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 12, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "X", "start_index": 4, "end_index": 12, "node_id": "0002"}]}] + + async def propose(model, prompt): + return {"subsections": [{"title": "Sub One", "page": 6}, {"title": "Sub Two", "page": 9}]} + + async def summarize(model, prompt): + return '{"summary": "ok"}' + monkeypatch.setattr(tree_optimize, "ask_model", propose) + monkeypatch.setattr(utils, "llm_acompletion", summarize) + + async def run(): + scheduler = utils.SummaryScheduler(tree, [(page, 0) for page in pages], model="m") + await tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, + on_final=scheduler.mark_final) + await scheduler.finish() + asyncio.run(run()) + + x = tree[0]["nodes"][1] + assert [(n["title"], n["start_index"], n["end_index"], n["node_id"]) for n in x["nodes"]] == [ + ("X (intro)", 4, 5, "0003"), ("Sub One", 6, 8, "0004"), ("Sub Two", 9, 12, "0005")] + assert all(n["summary"] == "ok" for n in utils._subtree(tree)) + + +def test_merge_folds_an_intro_without_listing_its_title(): + tree = [{"title": "P", "start_index": 1, "end_index": 4, "nodes": [ + {"title": "P (intro)", "start_index": 1, "end_index": 1}, + {"title": "C", "start_index": 2, "end_index": 4}]}] + tree_optimize.merge_tree(tree) + assert "nodes" not in tree[0] and tree[0]["key_items"] == ["C"] + + +LARGE = SimpleNamespace(max_page_num_each_node=10, max_token_num_each_node=20000, model="m") + + +def test_large_node_split_keeps_existing_children(monkeypatch): + async def meta_processor(*args, **kwargs): + raise AssertionError("a parent's subsections must not be replaced") + monkeypatch.setattr(classic, "meta_processor", meta_processor) + node = {"title": "Ch", "start_index": 1, "end_index": 20, + "nodes": [{"title": "S", "start_index": 20, "end_index": 25}]} + + asyncio.run(classic.process_large_node_recursively(node, [("page", 5000)] * 25, LARGE)) + + assert [child["title"] for child in node["nodes"]] == ["S"] + + +def test_split_intro_skips_its_parents_heading(monkeypatch): + async def meta_processor(*args, **kwargs): + return [{"title": "Ch", "physical_index": 1}, + {"title": "Background", "physical_index": 5}, + {"title": "Scope", "physical_index": 9}] + + async def appear(items, *args, **kwargs): + for item in items: + item["appear_start"] = "yes" + return items + monkeypatch.setattr(classic, "meta_processor", meta_processor) + monkeypatch.setattr(classic, "check_title_appearance_in_start_concurrent", appear) + intro = {"title": "Ch (intro)", "start_index": 1, "end_index": 14} + + asyncio.run(classic.process_large_node_recursively(intro, [("page", 5000)] * 14, LARGE)) + + assert [child["title"] for child in intro["nodes"]] == ["Background", "Scope"] + + +def test_standard_index_stores_intros_covering_ranges_and_section_summaries(tmp_path, monkeypatch): + texts = ["Opening words"] + [f"page {n}" for n in range(2, 41)] + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(texts)) + + async def tree_parser(page_list, opt, doc=None, logger=None): + return [{"title": "P", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "P (intro)", "start_index": 1, "end_index": 2}, + {"title": "C1", "start_index": 3, "end_index": 20}, + {"title": "C2", "start_index": 21, "end_index": 21, "nodes": [ + {"title": "C2.1", "start_index": 21, "end_index": 30}, + {"title": "C2.2", "start_index": 31, "end_index": 40}]}]}] + prompts = [] + + async def reply(model, prompt): + prompts.append(prompt) + return "whole " + prompt.split("Section Title: ")[1].split("\n")[0] + monkeypatch.setattr(classic, "tree_parser", tree_parser) + monkeypatch.setattr(utils, "llm_acompletion", reply) + monkeypatch.setattr(utils, "llm_completion", lambda model, prompt, **kw: "d") + opt = utils.ConfigLoader().load({"model": "m", "if_add_node_id": "yes", + "if_add_node_summary": "yes", "if_add_node_text": "yes", + "if_add_doc_description": "yes"}) + + result = classic.page_index_main(str(pdf), opt, logger=SimpleNamespace(info=print), + page_list=[(text, 1) for text in texts]) + + p = result["structure"][0] + c2 = p["nodes"][2] + assert shape([p]) == [("P", 1, 40), ("P (intro)", 1, 2), ("C1", 3, 20), + ("C2", 21, 40), ("C2.1", 21, 30), ("C2.2", 31, 40)] + assert (p["text"], p["summary"]) == ("", "whole P") # its intro holds pages 1-2 + assert "Opening words" in p["nodes"][0]["text"] + # C2's first child starts on C2's page; C2 keeps that page as its opening + assert (c2["text"], c2["summary"]) == ("page 21", "whole C2") + # short leaves keep their own text; only the parents are asked, deepest first + assert [n["summary"] for n in p["nodes"][:2]] == [ + "Opening wordspage 2", "".join(f"page {n}" for n in range(3, 21))] + assert [q.split("Section Title: ")[1].split("\n")[0] for q in prompts] == ["C2", "P"] + assert '"summary": "whole C2"' in prompts[-1] + + +def test_flash_gives_parents_their_intro_nodes(tmp_path, monkeypatch): + import pageindex.flash.api as flash_api + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["x"])) + monkeypatch.setattr(flash_api, "extract_toc", lambda pdf, **kw: { + "structure": [{"title": "R", "node_id": "0000", "start_index": 1, "end_index": 6, "nodes": [ + {"title": "A", "node_id": "0001", "start_index": 3, "end_index": 4}, + {"title": "X", "node_id": "0002", "start_index": 5, "end_index": 6}]}], + "page_texts": ["Cover", "Foreword", "A\nalpha", "alpha", "X\nbody", "body"]}) + + result = flash_api.page_index_flash(str(pdf), summary=False, optimize=False) + + assert [(n["title"], n["node_id"], n["start_index"], n["end_index"]) + for n in utils._subtree(result["structure"])] == [ + ("R", "0000", 1, 6), ("R (intro)", "0001", 1, 2), ("A", "0002", 3, 4), + ("X", "0003", 5, 6)] + + +def test_flash_preface_runs_onto_a_page_its_first_heading_does_not_open(tmp_path, monkeypatch): + import pageindex.flash.api as flash_api + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["x"])) + ends = [] + for third in ["A\nalpha", "Contents, continued\nA\nalpha"]: + monkeypatch.setattr(flash_api, "extract_toc", lambda pdf, third=third, **kw: { + "structure": [{"title": "A", "node_id": "0000", "start_index": 3, "end_index": 4}], + "page_texts": ["Cover", "Contents", third, "alpha"]}) + preface = flash_api.page_index_flash(str(pdf), summary=False, optimize=False)["structure"][0] + ends.append((preface["title"], preface["node_id"], preface["end_index"])) + assert ends == [("Preface", "0000", 2), ("Preface", "0000", 3)] + + +def test_summarize_tree_parent_falls_back_to_its_subsection_titles(monkeypatch): + async def reply(model, prompt): + return "" if "Section Title" in prompt else '{"summary": "ok"}' + monkeypatch.setattr(utils, "llm_acompletion", reply) + structure = [{"title": "R", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "A", "start_index": 1, "end_index": 1}, + {"title": "B", "start_index": 2, "end_index": 2}]}] + + out = asyncio.run(utils.summarize_tree(structure, [("alpha " * 300, 0), ("beta " * 300, 0)])) + + assert [n["summary"] for n in utils._subtree(out)] == ["A; B", "ok", "ok"] + + +def test_get_tree_parent_text_shares_its_first_childs_page_unless_an_intro_holds_it( + tmp_path, monkeypatch): + pdf = tmp_path / "report.pdf" + pdf.write_bytes(build_pdf(["Opening words", "Alpha body", "Beta body"])) + structure = [{"title": "Report", "node_id": "0000", "start_index": 1, "end_index": 3, + "summary": "report", "nodes": [ + {"title": "Report (intro)", "node_id": "0001", "start_index": 1, + "end_index": 1, "summary": "opening"}, + {"title": "Alpha", "node_id": "0002", "start_index": 2, + "end_index": 3, "summary": "alpha", "nodes": [ + {"title": "Alpha one", "node_id": "0003", "start_index": 2, + "end_index": 2, "summary": "one"}, + {"title": "Alpha two", "node_id": "0004", "start_index": 3, + "end_index": 3, "summary": "two"}]}]}] + monkeypatch.setattr(pageindex.flash, "page_index_flash", lambda pdf, **kw: { + "doc_name": "report.pdf", "structure": structure}) + monkeypatch.setattr(utils, "llm_completion", lambda model, prompt, **kw: "d") + client = PageIndexClient(storage_path=str(tmp_path / "store")) + doc_id = client.submit_document(str(pdf), mode="flash")["doc_id"] + + root = client.get_tree(doc_id)["result"][0] + + assert [n["text"] for n in utils._subtree([root])] == [ + "", "Opening words", "Alpha body", "Alpha body", "Beta body"] From 50caf1582b06e8118afbd380f9c3b4fe5a818c71 Mon Sep 17 00:00:00 2001 From: Ray Date: Sun, 4 Oct 2026 18:07:25 +0800 Subject: [PATCH 2/4] fix: unified-tree follow-ups in flash expand and the local agent tree Flash expand priced a candidate level without the intro attach_children then adds. An intro sharing its first child's page could make the committed node cost exactly its span, so the next merge folded it back after its summary was final, and the whole index raised "dropped or changed after it was marked final" (PRML 5.1, the 2023 annual report). expand_cost now prices the level with its intro. heading_at_page_start took any first line containing the heading as the heading opening its page. A running header repeating the section title cut intros and the Preface a page early, so the opening text above the real heading was in no node. The first line must now start with the heading, no later line may start with it, and a heading with no Latin letter is never placed. An uncertain page is shared. Expand on an intro or the Preface could list the heading printed on its shared last page (the next node's) or atop it (its parent's) as a new child. A proposal whose page and title, leading number aside, match a node already in the tree is dropped. The local get_document_structure tool read the stored tree through LocalAPI.raw_tree and missed get_tree's end < start clamp. raw_tree is gone; both modes read client.get_tree. A test runs the standard tree_parser path end to end: intro insertion before and after the large-node split, and the covering pass on the default tail. --- pageindex/agent_tools.py | 9 ++-- pageindex/local_api.py | 4 -- pageindex/tree_optimize.py | 39 ++++++++++------ tests/test_agent_tools.py | 12 +++++ tests/test_tree_format.py | 91 ++++++++++++++++++++++++++++++++++++++ 5 files changed, 132 insertions(+), 23 deletions(-) diff --git a/pageindex/agent_tools.py b/pageindex/agent_tools.py index f18abe8de..afc14c9f7 100644 --- a/pageindex/agent_tools.py +++ b/pageindex/agent_tools.py @@ -920,12 +920,9 @@ def _get_document_structure(client, doc_name: str, waited and entry.get("status") != "failed") try: - raw_tree = getattr(getattr(client, "_api", None), "raw_tree", None) - tree = raw_tree(entry["id"]) if raw_tree is not None else None - if tree is None: - # _format_structure strips text anyway — don't download it. - tree = client.get_tree(entry["id"], node_summary=True, - include_text=False).get("result") + # _format_structure strips text anyway — don't download it. + tree = client.get_tree(entry["id"], node_summary=True, + include_text=False).get("result") except PageIndexAPIError as exc: return _failure( f"Failed to retrieve document structure: {exc}", diff --git a/pageindex/local_api.py b/pageindex/local_api.py index 4802634b8..9fe63d97e 100644 --- a/pageindex/local_api.py +++ b/pageindex/local_api.py @@ -269,10 +269,6 @@ def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: add_node_text(structure, pdf_pages) return structure - def raw_tree(self, doc_id: str) -> list | None: - """Stored tree verbatim, every key kept.""" - return self._store.get_tree(doc_id) - def get_tree(self, doc_id: str, node_summary: bool = False, include_text: bool = True) -> dict[str, Any]: meta = self._require_doc(doc_id, "Failed to get tree result") diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index e607d5f00..9842fb350 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -11,7 +11,7 @@ trigger: S(v) > TRIGGER_PAGES (cost control on generation, not the rule) collapse_cost = S(v) - expand_cost = R(v) + max(S_residual(v), max_i S(c_i)) + expand_cost = R(v) + max_i S(c_i) the c_i include the intro attach_children adds expand iff expand_cost < collapse_cost (ties keep collapsed) expand_gain = collapse_cost - expand_cost @@ -172,12 +172,14 @@ def is_frontier(node): def heading_at_page_start(lines, page_no, heading): """Is the heading the first line on its page? No when that cannot be told - (a heading with no Latin letter or digit to match), so the page is shared.""" + (a heading with no Latin letter to match, or a first line that a later line + repeats, as a running header does), so the page is shared.""" page = lines[page_no - 1] key = normalize(heading) - if not page or not key: + if not page or not re.search("[a-z]", key): return False - return key in normalize(page[0]) + starts = [(normalize(line) + " ").startswith(key + " ") for line in page] + return starts[0] and not any(starts[1:]) def assign_ends(node, children, lines): @@ -304,14 +306,13 @@ def tree_cost_via_frontier(node, routing=ROUTING_COST): return max(d * routing + s for d, s, _ in entries) if entries else 0 -def expand_cost(node, children, routing=ROUTING_COST): - """Cost after one-step lookahead, children treated as collapsed.""" - covered = set() - for child in children: - covered |= set(range(child["start_index"], child["end_index"] + 1)) - residual = len(pages_of(node) - covered) - scans = [child["end_index"] - child["start_index"] + 1 for child in children] - return routing + max([residual] + scans), residual +def expand_cost(node, children, lines, routing=ROUTING_COST): + """Cost after one-step lookahead, children treated as collapsed, priced with + the intro node attach_children would give them.""" + trial = dict(node, nodes=children) + residual = S_residual(trial) + add_intro_nodes([trial], lines) + return tree_cost(trial, routing), residual # -------------------------------------------------------------------------- @@ -676,6 +677,13 @@ async def propose_children(node, pages, args): return accepted +def heading_key(node): + """A node's page and heading, compared without the number printed before it; + None when no text is left to compare.""" + title = re.sub(r"^(?:[0-9]+ )+", "", normalize(node["title"])) + return (node["start_index"], title) if title else None + + async def expand(structure, pages, lines, args, log, frozen): """One-step lookahead on every collapsed node over the trigger, recursively. @@ -725,6 +733,11 @@ async def process(node): if cached: candidates.append(("cache", cached)) candidates.extend(llm_candidates) + # a heading the tree holds already: a neighbor's on a shared page, a parent's atop its intro + taken = {heading_key(n) for n, _ in flatten(structure)} - {None} + candidates = [(source, [c for c in children if heading_key(c) not in taken]) + for source, children in candidates] + candidates = [(source, children) for source, children in candidates if children] if not candidates: note(args.progress, f" -> no children found, kept collapsed") @@ -737,7 +750,7 @@ async def process(node): scored = [] for source, children in candidates: sized = assign_ends(node, children, lines) - cost, residual = expand_cost(node, sized, args.routing) + cost, residual = expand_cost(node, sized, lines, args.routing) scored.append({"source": source, "children": sized, "expand_cost": cost, "S_residual": residual}) scored.sort(key=lambda s: s["expand_cost"]) diff --git a/tests/test_agent_tools.py b/tests/test_agent_tools.py index 6fa04f6f3..3f1b1c823 100644 --- a/tests/test_agent_tools.py +++ b/tests/test_agent_tools.py @@ -287,6 +287,18 @@ def test_structure_strips_text_and_orders_keys(client, store_path): assert root["nodes"][0]["end_index"] == 1 +def test_structure_shows_the_ranges_get_tree_serves(client, store_path): + # a leaf stored ending before it starts: its successor was judged to open the same page + tree = [{"title": "Doc", "node_id": "0000", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "A", "node_id": "0001", "start_index": 2, "end_index": 1}, + {"title": "B", "node_id": "0002", "start_index": 2, "end_index": 2}]}] + seed_doc(store_path, "pi-a", "report.pdf", tree=tree) + payload, _ = run(client, "get_document_structure", doc_name="report.pdf") + served = client.get_tree("pi-a", include_text=False)["result"] + assert [(n["start_index"], n["end_index"]) for n in payload["structure"][0]["nodes"]] == [ + (n["start_index"], n["end_index"]) for n in served[0]["nodes"]] == [(2, 2), (2, 2)] + + def test_structure_multipart_pagination(client, store_path): big_tree = [{ "title": f"Chapter {index}", "node_id": f"{index:04d}", diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index 36066a2d5..c539fe048 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -74,6 +74,63 @@ async def run(): assert all(n["summary"] == "ok" for n in utils._subtree(tree)) +def test_a_running_header_or_a_word_prefix_does_not_open_the_page(): + page = ["4.1. Discriminant Functions 181", "opening words", "4.1. Discriminant Functions"] + assert not tree_optimize.heading_at_page_start([page], 1, "4.1. Discriminant Functions") + assert tree_optimize.heading_at_page_start([page[2:]], 1, "4.1. Discriminant Functions") + assert not tree_optimize.heading_at_page_start([["Filed 03/04/24", "I. Background"]], 1, "I.") + + +def test_expand_prices_a_level_with_the_intro_it_gets(monkeypatch): + body = "body " * 250 + pages = [body] * 12 + pages[10] = body + "\nSub Late\n" + body # mid-page: the intro shares page 11 + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 12, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "X", "start_index": 4, "end_index": 12, "node_id": "0002"}]}] + + async def propose(model, prompt): + return {"subsections": [{"title": "Sub Late", "page": 11}]} + + async def summarize(model, prompt): + return '{"summary": "ok"}' + monkeypatch.setattr(tree_optimize, "ask_model", propose) + monkeypatch.setattr(utils, "llm_acompletion", summarize) + + async def run(): + scheduler = utils.SummaryScheduler(tree, [(page, 0) for page in pages], model="m") + await tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, + on_final=scheduler.mark_final) + await scheduler.finish() + asyncio.run(run()) + + # intro 4-11 + Sub Late 11-12 costs 1 + 8 = 9 pages, no better than reading X's 9 + assert "nodes" not in tree[0]["nodes"][1] + + +def test_expand_skips_the_heading_of_a_neighbor_sharing_the_last_page(monkeypatch): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + pages[11] = body + "\n3 Results\n" + body # mid-page: Methods runs onto page 12 + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + + async def propose(model, prompt): + if "Section title: Methods\n" in prompt: + return {"subsections": [{"title": "Setup", "page": 6}, {"title": "Results", "page": 12}]} + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True)) + + assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] + + def test_merge_folds_an_intro_without_listing_its_title(): tree = [{"title": "P", "start_index": 1, "end_index": 4, "nodes": [ {"title": "P (intro)", "start_index": 1, "end_index": 1}, @@ -158,6 +215,40 @@ async def reply(model, prompt): assert '"summary": "whole C2"' in prompts[-1] +def test_standard_index_gives_toc_and_split_parents_intros(tmp_path, monkeypatch): + texts = [f"page {n}" for n in range(1, 41)] + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(texts)) + + splits = {1: [{"structure": "1", "title": "P", "physical_index": 1}, # P's long opening + {"structure": "2", "title": "Background", "physical_index": 8}], + 20: [{"structure": "1", "title": "C2", "physical_index": 20}, # the large leaf C2 + {"structure": "2", "title": "C2.a", "physical_index": 25}, + {"structure": "3", "title": "C2.b", "physical_index": 32}]} + + async def meta_processor(page_list, mode=None, start_index=1, **kwargs): + if len(page_list) == len(texts): + return [{"structure": "1", "title": "P", "physical_index": 1}, + {"structure": "1.1", "title": "C1", "physical_index": 15}, + {"structure": "1.2", "title": "C2", "physical_index": 20}] + return splits[start_index] + + async def appear(items, *args, **kwargs): + return items + monkeypatch.setattr(classic, "check_toc", lambda page_list, opt: {"toc_content": None}) + monkeypatch.setattr(classic, "meta_processor", meta_processor) + monkeypatch.setattr(classic, "check_title_appearance_in_start_concurrent", appear) + opt = utils.ConfigLoader().load({"model": "m", "if_add_node_summary": "no"}) + + result = classic.page_index_main(str(pdf), opt, logger=SimpleNamespace(info=print), + page_list=[(text, 3000) for text in texts]) + + assert shape(result["structure"]) == [ + ("P", 1, 40), ("P (intro)", 1, 15), ("P (intro)", 1, 8), ("Background", 8, 15), + ("C1", 15, 20), + ("C2", 20, 40), ("C2 (intro)", 20, 25), ("C2.a", 25, 32), ("C2.b", 32, 40)] + + def test_flash_gives_parents_their_intro_nodes(tmp_path, monkeypatch): import pageindex.flash.api as flash_api pdf = tmp_path / "doc.pdf" From ff4cfd0989e3732e43ddad390a2c3f014a04d344 Mon Sep 17 00:00:00 2001 From: Ray Date: Mon, 5 Oct 2026 15:16:12 +0800 Subject: [PATCH 3/4] fix: expand keeps a heading for the node it is printed under 50caf15 dropped a proposed child whose page and title, leading number aside, matched any node in the live tree. Sibling expansions run concurrently, so the result hung on reply order: when Methods' reply landed before Results', Methods kept "3.1 Discussion" from their shared page and Results lost its own. Stripping every leading number also took "3.2 Results" for its parent "3 Results" and dropped it. A proposal is now dropped when it already is a node, in the tree as expand found it or made by the node's own ancestors, which no concurrent branch can change; or when, on a page the node shares, it is printed above the node's heading or at and below the next node's. A heading not found on its page decides nothing. A leading number is ignored only when one of the two titles lacks it. The check runs before the empty-reply retry, and the tree is read once per pass instead of once per node. A summary reply counted as answered whenever it was non-empty, so a model returning {"summary": ""} everywhere stored a document of fallback summaries without the all-failed error. It now counts only when a summary parses out of it. post_processing ended a TOC item a page before its start when the next item opened the same page; the end is now at least the start. --- pageindex/tree_optimize.py | 89 ++++++++++++++++++++++++++++++-------- pageindex/utils.py | 6 +-- tests/test_client.py | 5 ++- tests/test_tree_format.py | 70 ++++++++++++++++++++++++++++++ 4 files changed, 146 insertions(+), 24 deletions(-) diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index 9842fb350..5537993ed 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -677,11 +677,55 @@ async def propose_children(node, pages, args): return accepted -def heading_key(node): - """A node's page and heading, compared without the number printed before it; - None when no text is left to compare.""" - title = re.sub(r"^(?:[0-9]+ )+", "", normalize(node["title"])) - return (node["start_index"], title) if title else None +def same_heading(a, b): + """Whether two titles name one heading: equal once normalized, or equal but + for a leading number only one of them prints.""" + a, b = normalize(a), normalize(b) + bare_a, bare_b = (re.sub(r"^(?:[0-9]+ )+", "", t) for t in (a, b)) + return bool(a) and (a == b or (bare_a == bare_b and (a == bare_a or b == bare_b))) + + +def headings(node): + """The headings printed for a node. A same-page fusion keeps them in its + key_items: the summary pass may rewrite the fused title while expand runs.""" + return node["key_items"] if node.get("_same_page") else [node["title"]] + + +def own_children(node, children, lines, known, ancestors, nxt): + """The proposed children printed inside the node's own text. + + Dropped: a heading that already is a node, in the tree as expand found it + (`known`, page -> headings) or made by the node's own ancestors; and on a + page the node shares, anything printed above its heading or at and below + the next node's. Other branches grow concurrently, so nothing they add is + read. A heading not found on its page decides nothing. + """ + start, end = node["start_index"], subtree_end(node) + lineage = [n for a in ancestors for n in [a] + a["nodes"]] + above = [t for n in [node] + ancestors if n["start_index"] == start for t in headings(n)] + after = headings(nxt) if nxt is not None and nxt["start_index"] == end else [] + + def found(page, matches): + page_lines = lines[page - 1] if lines and page <= len(lines) else [] + return [i for i, line in enumerate(page_lines) if matches(line)] + + def is_node(page, title): + titles = known.get(page, []) + [t for n in lineage if n["start_index"] == page + for t in headings(n)] + return any(same_heading(title, t) for t in titles) + + top = found(start, lambda line: any(same_heading(line, t) for t in above)) + bottom = found(end, lambda line: any(same_heading(line, t) for t in after)) if after else [] + kept = [] + for child in children: + page, key = child["start_index"], normalize(child["title"]) + printed = found(page, lambda line: key in normalize(line)) if key else [] + if (is_node(page, child["title"]) + or page == start and top and printed and printed[-1] < top[0] + or page == end and bottom and printed and printed[0] >= bottom[-1]): + continue + kept.append(child) + return kept async def expand(structure, pages, lines, args, log, frozen): @@ -695,7 +739,7 @@ async def expand(structure, pages, lines, args, log, frozen): changed = False semaphore = asyncio.Semaphore(args.concurrency) - async def proposals_for(node): + async def proposals_for(node, own): """The model half of one node's lookahead: the empty-retry ladder and absorbed errors run inside the task; log entries come back so a node's entries stay contiguous under concurrency.""" @@ -704,7 +748,7 @@ async def proposals_for(node): attempts += 1 try: async with semaphore: - proposed = await propose_children(node, pages, args) + proposed = own(await propose_children(node, pages, args)) except Exception as exc: if _is_unrecoverable(exc): raise # every remaining node would fail identically @@ -717,7 +761,7 @@ async def proposals_for(node): break # an empty answer is retried, not trusted return llm_candidates, attempts, entries - async def process(node): + async def process(node, ancestors, nxt): nonlocal changed if not is_frontier(node) or node.get("node_id") in frozen: return @@ -726,18 +770,16 @@ async def process(node): return # below the trigger, stay collapsed note(args.progress, f" expand {node.get('node_id'):>8} S={span} " f"pages {node['start_index']}-{subtree_end(node)} ...") - llm_candidates, attempts, entries = await proposals_for(node) + + def own(children): + return own_children(node, children, lines, known, ancestors, nxt) + llm_candidates, attempts, entries = await proposals_for(node, own) log.extend(entries) candidates = [] - cached = children_from_cache(node, args.cache, args.kinds) + cached = own(children_from_cache(node, args.cache, args.kinds)) if cached: candidates.append(("cache", cached)) candidates.extend(llm_candidates) - # a heading the tree holds already: a neighbor's on a shared page, a parent's atop its intro - taken = {heading_key(n) for n, _ in flatten(structure)} - {None} - candidates = [(source, [c for c in children if heading_key(c) not in taken]) - for source, children in candidates] - candidates = [(source, children) for source, children in candidates if children] if not candidates: note(args.progress, f" -> no children found, kept collapsed") @@ -790,15 +832,24 @@ async def process(node): merge_same_page([node], log) # settle after the fusion: mark_final snapshots the children, finish() rejects a later change args.settled([node]) - results = await asyncio.gather(*(process(child) - for child in node["nodes"]), + children = node["nodes"] + results = await asyncio.gather(*(process(child, [node] + ancestors, after) + for child, after in zip(children, children[1:] + [nxt])), return_exceptions=True) for result in results: if isinstance(result, BaseException): raise result - results = await asyncio.gather(*(process(node) - for node, _ in flatten(structure)), + # the tree as it stands now: expand only ever adds below a leaf it is + # processing, so the nodes around another leaf never change under it + flat = list(flatten(structure)) + known, ancestry = {}, {} + for node, parent in flat: + known.setdefault(node["start_index"], []).extend(headings(node)) + ancestry[id(node)] = [parent] + ancestry[id(parent)] if parent else [] + results = await asyncio.gather(*(process(node, ancestry[id(node)], after) + for (node, _), after in + zip(flat, [n for n, _ in flat[1:]] + [None])), return_exceptions=True) for result in results: if isinstance(result, BaseException): diff --git a/pageindex/utils.py b/pageindex/utils.py index 4981a63c5..73f5c3170 100644 --- a/pageindex/utils.py +++ b/pageindex/utils.py @@ -591,9 +591,9 @@ def post_processing(structure, end_physical_index): item['start_index'] = item.get('physical_index') if i < len(structure) - 1: if structure[i + 1].get('appear_start') == 'yes': - item['end_index'] = structure[i + 1]['physical_index']-1 + item['end_index'] = max(item['start_index'], structure[i + 1]['physical_index']-1) else: - item['end_index'] = structure[i + 1]['physical_index'] + item['end_index'] = max(item['start_index'], structure[i + 1]['physical_index']) else: item['end_index'] = end_physical_index tree = list_to_tree(structure) @@ -1011,7 +1011,7 @@ async def _ask(self, prompt, prio): self._asked = True async with self._gate.slot(prio): reply = await llm_acompletion(self._model, prompt) - if reply: + if parse_summary(reply): self._answered = True return reply diff --git a/tests/test_client.py b/tests/test_client.py index 814415ca6..6cceb98cc 100644 --- a/tests/test_client.py +++ b/tests/test_client.py @@ -2532,12 +2532,13 @@ def test_format_tree_node_keeps_key_items(): # ── retry-ladder and summary fail-loud edges (twelfth review) ── -def test_summarize_tree_all_empty_replies_fail_loud(monkeypatch): +@pytest.mark.parametrize("reply", ["", '{"summary": ""}']) +def test_summarize_tree_all_empty_replies_fail_loud(monkeypatch, reply): """Empty-content replies (content filter, spent output cap) must not vouch for the model: a raw-text short leaf cannot carry the run when every model reply comes back blank.""" async def blank(model, prompt): - return "" + return reply monkeypatch.setattr(pageindex.utils, "llm_acompletion", blank) pdf_pages = [("tiny", 1), ("beta " * 300, 300)] structure = [{"title": "R", "start_index": 1, "end_index": 2, diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index c539fe048..81dc5776d 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -5,6 +5,8 @@ import importlib from types import SimpleNamespace +import pytest + import pageindex.flash import pageindex.tree_optimize as tree_optimize import pageindex.utils as utils @@ -131,6 +133,66 @@ async def propose(model, prompt): assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] +def test_a_proposed_child_must_be_printed_inside_its_nodes_own_text(): + lines = [["body"] for _ in range(20)] + lines[11] = ["2.3 Limits", "3 Results", "Scope", "3.1 Data", "3.2 Results"] + methods = {"title": "2 Methods", "start_index": 4, "end_index": 12} + results = {"title": "3 Results", "start_index": 12, "end_index": 20} + root = {"title": "R", "start_index": 1, "end_index": 20, "nodes": [methods, results]} + known = {} + for node, _ in tree_optimize.flatten([root]): + known.setdefault(node["start_index"], []).append(node["title"]) + + def kept(node, ancestors, nxt, *titles): + children = [{"title": title, "start_index": 12} for title in titles] + return [c["title"] for c in + tree_optimize.own_children(node, children, lines, known, ancestors, nxt)] + + # on a shared last page: the next node's heading, and what is printed below it + assert kept(methods, [root], results, "2.3 Limits", "Results", "3.1 Data") == ["2.3 Limits"] + # on a shared first page: what is printed above the node's heading; a number tells headings apart + assert kept(results, [root], None, "2.3 Limits", "Results", "3.1 Data", "3.2 Results") == [ + "3.1 Data", "3.2 Results"] + # an intro sits under its parent's heading and above the siblings expand made with it + data = {"title": "3.1 Data", "start_index": 12, "end_index": 20} + intro = {"title": "3 Results (intro)", "start_index": 12, "end_index": 12} + results["nodes"] = [intro, data] + assert kept(intro, [results, root], data, "2.3 Limits", "Results", "Scope", "3.1 Data") == ["Scope"] + # a next heading not found on the page decides nothing + assert kept(methods, [root], dict(results, title="Findings"), "2.3 Limits", "3.2 Results") == [ + "2.3 Limits", "3.2 Results"] + + +@pytest.mark.parametrize("methods_delay", [0, 0.05]) +def test_expand_gives_one_tree_whichever_reply_lands_first(monkeypatch, methods_delay): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + pages[11] = body + "\n3 Results\nlead\n3.1 Data\n" + body + pages[14] = "3.2 Analysis\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + + async def propose(model, prompt): + if "Section title: Methods\n" in prompt: + await asyncio.sleep(methods_delay) + return {"subsections": [{"title": "Setup", "page": 6}, {"title": "3.1 Data", "page": 12}]} + if "Section title: 3 Results\n" in prompt: + await asyncio.sleep(0.05 - methods_delay) + return {"subsections": [{"title": "3.1 Data", "page": 12}, {"title": "3.2 Analysis", "page": 15}]} + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True)) + + assert shape(tree[0]["nodes"][1:]) == [ + ("Methods", 4, 12), ("Methods (intro)", 4, 5), ("Setup", 6, 12), + ("3 Results", 12, 20), ("3.1 Data", 12, 14), ("3.2 Analysis", 15, 20)] + + def test_merge_folds_an_intro_without_listing_its_title(): tree = [{"title": "P", "start_index": 1, "end_index": 4, "nodes": [ {"title": "P (intro)", "start_index": 1, "end_index": 1}, @@ -139,6 +201,14 @@ def test_merge_folds_an_intro_without_listing_its_title(): assert "nodes" not in tree[0] and tree[0]["key_items"] == ["C"] +def test_a_toc_item_whose_page_the_next_opens_still_ends_on_its_own_page(): + items = [{"structure": "1", "title": "Overview", "physical_index": 5}, + {"structure": "2", "title": "Scope", "physical_index": 5, "appear_start": "yes"}, + {"structure": "3", "title": "Terms", "physical_index": 9}] + assert shape(utils.post_processing(items, 12)) == [ + ("Overview", 5, 5), ("Scope", 5, 9), ("Terms", 9, 12)] + + LARGE = SimpleNamespace(max_page_num_each_node=10, max_token_num_each_node=20000, model="m") From cdcaf65842b594502f905cd1b752ed320d318653 Mon Sep 17 00:00:00 2001 From: Ray Date: Mon, 5 Oct 2026 15:20:34 +0800 Subject: [PATCH 4/4] test: pin the digit-only heading rule and intro idempotency heading_at_page_start never places a heading without a Latin letter, so a bare "2" (often a page number) shares its page; nothing tested that. The add_intro_nodes idempotency check compared the tree with itself, the same object the call returns, so it could not fail; it now compares a second pass over a copy with the first. --- tests/test_tree_format.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py index 81dc5776d..afd4ff84f 100644 --- a/tests/test_tree_format.py +++ b/tests/test_tree_format.py @@ -2,6 +2,7 @@ splitting, and summaries every node gets, a parent's built from its children's.""" import asyncio +import copy import importlib from types import SimpleNamespace @@ -36,13 +37,14 @@ def test_intro_node_holds_the_pages_a_parent_opens_with(): ("Ch 3", 10, 30), ("Ch 3 (intro)", 10, 11), ("3.1 Scope", 12, 20), ("3.2", 21, 30), ("3.2.1", 21, 30), # a first child on its parent's page: no intro ("", 31, 40), ("Intro", 31, 33), ("4.1", 33, 40)] - assert tree_optimize.add_intro_nodes(tree, lines) == tree + assert shape(tree_optimize.add_intro_nodes(copy.deepcopy(tree), lines)) == shape(tree) # a standard parent ends where its first child starts; a split intro keeps its title split = [{"title": "Ch 3 (intro)", "start_index": 10, "end_index": 11, "nodes": [ {"title": "Background", "start_index": 12, "end_index": 14}]}] assert shape(tree_optimize.add_intro_nodes(split))[1] == ("Ch 3 (intro)", 10, 11) # a heading with nothing to match cannot be placed on its page, so the page is shared assert not tree_optimize.heading_at_page_start([["第一章 总则"]], 1, "第一章 总则") + assert not tree_optimize.heading_at_page_start([["2", "Body text"]], 1, "2") # or a page number def test_expand_gives_a_split_node_its_intro(monkeypatch):