diff --git a/pageindex/agent_tools.py b/pageindex/agent_tools.py index f18abe8de..afc14c9f7 100644 --- a/pageindex/agent_tools.py +++ b/pageindex/agent_tools.py @@ -920,12 +920,9 @@ def _get_document_structure(client, doc_name: str, waited and entry.get("status") != "failed") try: - raw_tree = getattr(getattr(client, "_api", None), "raw_tree", None) - tree = raw_tree(entry["id"]) if raw_tree is not None else None - if tree is None: - # _format_structure strips text anyway — don't download it. - tree = client.get_tree(entry["id"], node_summary=True, - include_text=False).get("result") + # _format_structure strips text anyway — don't download it. + tree = client.get_tree(entry["id"], node_summary=True, + include_text=False).get("result") except PageIndexAPIError as exc: return _failure( f"Failed to retrieve document structure: {exc}", diff --git a/pageindex/client.py b/pageindex/client.py index 4dea417c1..0d77269bf 100644 --- a/pageindex/client.py +++ b/pageindex/client.py @@ -949,14 +949,22 @@ def get_tree(self, doc_id: str, node_summary: bool = False, Returns: dict: {'doc_id', 'status', 'retrieval_ready', 'result', ...} where - result nodes are {'title', 'node_id', 'page_index', ('summary' / - 'prefix_summary',) ('text',) 'nodes'}. + result nodes are {'title', 'node_id', 'start_index', 'end_index', + ('summary',) ('text',) 'nodes'} in both modes. Each node's summary + describes its own pages start_index..end_index; its text is the + part no child holds, though a parent's may share the page its + first child starts on. """ tree = self._api.get_tree(doc_id=doc_id, node_summary=node_summary, include_text=include_text) - if not include_text and tree.get("result"): - from .utils import remove_fields - tree["result"] = remove_fields(tree["result"], fields=["text"]) + if tree.get("result"): + from .utils import _subtree, remove_fields, unify_tree + page_count = None + if any("end_index" not in node for node in _subtree(tree["result"])): + page_count = self._api.get_document(doc_id).get("pageNum") + tree["result"] = unify_tree(tree["result"], page_count) + if not include_text: + tree["result"] = remove_fields(tree["result"], fields=["text"]) return tree def get_document_structure(self, doc_id: str) -> list[dict[str, Any]]: @@ -975,7 +983,7 @@ def is_retrieval_ready(self, doc_id: str) -> bool: failures, timeouts) propagate. """ try: - result = self.get_tree(doc_id) + result = self._api.get_tree(doc_id) return result.get("retrieval_ready", False) except PageIndexAPIError: return False diff --git a/pageindex/flash/api.py b/pageindex/flash/api.py index bc8c71446..5e3e397e1 100644 --- a/pageindex/flash/api.py +++ b/pageindex/flash/api.py @@ -89,9 +89,7 @@ async def _optimize_async(structure, page_texts, do_expand, model, on_final=None the summaries use. """ from ..tree_optimize import optimize - lines = [[line_text.strip() for line_text in (page_text or "").splitlines() - if line_text.strip()] - for page_text in page_texts] + lines = _page_lines(page_texts) outcome = await optimize(structure, page_texts, lines, model=model, do_expand=do_expand, page_count=len(page_texts), on_final=on_final, concurrency=concurrency) @@ -134,14 +132,31 @@ def _page_nodes(page_texts: list[str]) -> list[dict]: return nodes -def _add_preface(structure: list[dict]) -> None: - """The pages before a hierarchy that starts late become a Preface node, as in standard mode.""" +def _page_lines(page_texts: list[str]) -> list[list[str]]: + return [[line_text.strip() for line_text in (page_text or "").splitlines() + if line_text.strip()] + for page_text in page_texts] + + +def _add_intros(structure: list[dict], lines: list[list[str]]) -> None: + """The pages a parent opens with before its first child become its intro node.""" + from ..tree_optimize import add_intro_nodes from ..utils import write_node_id - structure.insert(0, {"title": "Preface", "start_index": 1, - "end_index": structure[0]["start_index"] - 1}) + add_intro_nodes(structure, lines) write_node_id(structure) +def _add_preface(structure: list[dict], lines: list[list[str]]) -> None: + """The pages before a hierarchy that starts late become a Preface node, as in + standard mode. It runs onto the first section's page unless that heading opens it.""" + from ..tree_optimize import heading_at_page_start + first = structure[0] + opens = first["start_index"] <= len(lines) and heading_at_page_start( + lines, first["start_index"], first["title"]) + structure.insert(0, {"title": "Preface", "start_index": 1, + "end_index": first["start_index"] - (1 if opens else 0)}) + + def flash_rejection_reason(result: dict, standard_hint: str = "mode='standard'") -> str | None: """Why a managed pipeline should refuse this flash result, or None to accept it. @@ -166,7 +181,7 @@ def page_index_flash(pdf, summary=True, summary_model=None, optimize: str | bool | None = None, optimize_expand=None, optimize_model=None, summary_concurrency=None, use_embedded_toc=True, summary_max_words=None) -> dict: - """Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """ + """Build a PageIndex tree structure from a PDF using layout statistics. The tree extraction itself uses no LLM; by default an LLM writes node summaries and expands the tree (``summary=False, optimize=False`` runs fully LLM-free). Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: ``"full"`` for merge + LLM expand (a model unreachable after the retry ladder — a missing credential included — fails the run loudly from expand itself; a per-prompt rejection leaves just that node collapsed), ``"merge"`` for deterministic merge only, ``False`` to disable. ``True`` is accepted as ``"full"`` for backward compatibility; defaults to ``"full"``. Expand needs readable page text, so a bookmark-only or scanned PDF runs the merge half only (``expands`` reports 0). optimize_expand: deprecated — use ``optimize``. Honored only when ``optimize`` is not passed (or is the legacy ``True``): ``False`` maps to ``"merge"``, ``True`` to ``"full"``. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: cap on simultaneous indexing model calls per lane: the summaries, and expand up to its own ceiling of 32 (the lanes overlap, so up to cap + min(32, cap) calls run at once); None uses the library defaults (64 and 32). use_embedded_toc: if True, consume the PDF's embedded bookmarks when trustworthy: deep bookmarks become the frame and the detected sections they lack are grafted back in after noise filtering, coarse ones become the chapter frame with detected nodes re-hung under them (deeper sparse entries are filled in when the page text confirms them, and garbled extracted titles are repaired from the bookmark strings), garbage ones are ignored. On by default; pass False for the pure detected structure. summary_max_words: word cap each model-written node summary is asked to stay within (short leaves keep their raw text); None uses the library default (150). Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of ``{"title", "node_id", "start_index", "end_index"}`` dicts; ``"nodes"`` holds the children where there are any and ``"summary"`` appears when summaries ran; page indexes are 1-based; a hierarchy that starts after page 1 is preceded by a ``Preface`` node covering the pages before it, as in standard mode, and the first section's page too unless that heading opens it; a parent whose first child starts on a later page opens with a child titled ``" (intro)"`` holding the pages before it; a parent's range and summary cover its whole subtree) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). ``toc_source`` says where the structure came from: ``"detected"`` (layout), ``"bookmarks"`` (the embedded outline), ``"hybrid"`` (bookmarks framing the detected sections), ``"pages"`` (no hierarchy found, so one node per page titled ``Page N``; left unsummarized and unoptimized when there are more than ``FLAT_TREE_MAX_NODES`` pages, a size the local client and CLI refuse) or ``"unreadable"`` (no page carries text; ``structure`` is empty). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics; a refused flat tree carries neither it nor node summaries. """ for name, value in (("summary_concurrency", summary_concurrency), ("summary_max_words", summary_max_words)): if value is not None and not (isinstance(value, numbers.Integral) and int(value) >= 1): @@ -195,7 +210,9 @@ def page_index_flash(pdf, summary=True, summary_model=None, result["structure"] = structure result["toc_source"] = "pages" if structure else "unreadable" elif structure[0]["start_index"] > 1: - _add_preface(structure) + _add_preface(structure, _page_lines(result.get("page_texts") or [])) + if structure: + _add_intros(structure, _page_lines(result.get("page_texts") or [])) if result.get("toc_source") == "pages" and len(structure) > FLAT_TREE_MAX_NODES: # the managed pipelines refuse a flat tree this size; skip the model passes result.pop("page_texts", None) diff --git a/pageindex/local_api.py b/pageindex/local_api.py index 025ab716d..9fe63d97e 100644 --- a/pageindex/local_api.py +++ b/pageindex/local_api.py @@ -260,6 +260,7 @@ def _index_flash(self, file_path: str) -> tuple[list, str | None]: # ── tree / ocr ── def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: + """Each node's own text (see utils.own_pages).""" from .utils import add_node_text structure = self._require_data( self._store.get_tree(doc_id), error_prefix) @@ -268,11 +269,6 @@ def _load_tree_with_text(self, doc_id: str, error_prefix: str) -> list: add_node_text(structure, pdf_pages) return structure - def raw_tree(self, doc_id: str) -> list | None: - """Stored tree verbatim — keeps start_index/end_index, which - get_tree's cloud wire shape renames and drops.""" - return self._store.get_tree(doc_id) - def get_tree(self, doc_id: str, node_summary: bool = False, include_text: bool = True) -> dict[str, Any]: meta = self._require_doc(doc_id, "Failed to get tree result") @@ -398,17 +394,15 @@ def _format_tree_node(node: dict, node_summary: bool) -> dict: out = { "title": node.get("title", ""), "node_id": node.get("node_id"), - "page_index": node.get("start_index"), + "start_index": node.get("start_index"), + "end_index": node.get("end_index"), } if node.get("key_items"): out["key_items"] = node["key_items"] if node_summary: summary = node.get("summary") if summary is not None: - if children: - out["prefix_summary"] = summary - else: - out["summary"] = summary + out["summary"] = summary if "text" in node: out["text"] = node["text"] if children: diff --git a/pageindex/page_index_classic.py b/pageindex/page_index_classic.py index d680154eb..7788e236b 100644 --- a/pageindex/page_index_classic.py +++ b/pageindex/page_index_classic.py @@ -6,7 +6,7 @@ import re from .utils import * from .naming import sanitize_filename as sanitize_upload_filename -from .tree_optimize import merge_tree +from .tree_optimize import add_intro_nodes, merge_tree import os ######################### Hardening for prompt injection patterns #################################################### @@ -1169,7 +1169,8 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None) node_page_list = page_list[node['start_index']-1:node['end_index']] token_num = sum([page[1] for page in node_page_list]) - if node['end_index'] - node['start_index'] > opt.max_page_num_each_node and token_num >= opt.max_token_num_each_node: + if (not node.get('nodes') and node['end_index'] - node['start_index'] > opt.max_page_num_each_node + and token_num >= opt.max_token_num_each_node): print('large node:', node['title'], 'start_index:', node['start_index'], 'end_index:', node['end_index'], 'token_num:', token_num) node_toc_tree = await meta_processor(node_page_list, mode='process_no_toc', start_index=node['start_index'], opt=opt, logger=logger) @@ -1178,7 +1179,9 @@ async def process_large_node_recursively(node, page_list, opt=None, logger=None) # Filter out items with None physical_index before post_processing valid_node_toc_items = [item for item in node_toc_tree if item.get('physical_index') is not None] - if valid_node_toc_items and node['title'].strip() == valid_node_toc_items[0]['title'].strip(): + # an intro node's pages open with its parent's heading + title = node['title'].strip() + if valid_node_toc_items and valid_node_toc_items[0]['title'].strip() in (title, title.removesuffix(INTRO_SUFFIX)): node['nodes'] = post_processing(valid_node_toc_items[1:], node['end_index']) node['end_index'] = valid_node_toc_items[1]['start_index'] if len(valid_node_toc_items) > 1 else node['end_index'] else: @@ -1221,14 +1224,15 @@ async def tree_parser(page_list, opt, doc=None, logger=None): # Filter out items with None physical_index before post_processings valid_toc_items = [item for item in toc_with_page_number if item.get('physical_index') is not None] - toc_tree = post_processing(valid_toc_items, len(page_list)) + toc_tree = add_intro_nodes(post_processing(valid_toc_items, len(page_list))) tasks = [ process_large_node_recursively(node, page_list, opt, logger=logger) for node in toc_tree ] await asyncio.gather(*tasks) - - return toc_tree + + # a split leaf is now a parent that may open before its first child + return add_intro_nodes(toc_tree) def page_index_main(doc, opt=None, logger=None, page_list=None): @@ -1257,21 +1261,19 @@ async def page_index_builder(): if opt.if_add_node_text == 'yes': add_node_text(structure, page_list) if opt.if_add_node_summary == 'yes': - if opt.if_add_node_text == 'no': - add_node_text(structure, page_list) - await generate_summaries_for_structure(structure, model=getattr(opt, 'summary_model', None) or opt.model) - if opt.if_add_node_text == 'no': - remove_structure_text(structure) + await summarize_tree(structure, page_list, model=getattr(opt, 'summary_model', None) or opt.model) if opt.if_add_doc_description == 'yes': # Create a clean structure without unnecessary fields for description generation clean_structure = create_clean_structure_for_description(structure) doc_description = generate_doc_description(clean_structure, model=getattr(opt, 'summary_model', None) or opt.model) + cover_subtree_ranges(structure) structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes']) return { 'doc_name': get_pdf_name(doc), 'doc_description': doc_description, 'structure': structure, } + cover_subtree_ranges(structure) structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes']) return { 'doc_name': get_pdf_name(doc), diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index ddf397c7d..6f97ca083 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -11,7 +11,7 @@ trigger: S(v) > TRIGGER_PAGES (cost control on generation, not the rule) collapse_cost = S(v) - expand_cost = R(v) + max(S_residual(v), max_i S(c_i)) + expand_cost = R(v) + max_i S(c_i) the c_i include the intro attach_children adds expand iff expand_cost < collapse_cost (ties keep collapsed) expand_gain = collapse_cost - expand_cost @@ -62,8 +62,8 @@ import sys from types import SimpleNamespace -from .utils import (ConfigLoader, _is_unrecoverable, llm_acompletion, - strip_internal_keys) +from .utils import (ConfigLoader, _is_unrecoverable, intro_title, is_intro, + llm_acompletion, strip_internal_keys) TRIGGER_PAGES = 5 # only look ahead on nodes larger than this ROUTING_COST = 1 # R(v), in pages @@ -171,11 +171,15 @@ def is_frontier(node): def heading_at_page_start(lines, page_no, heading): - """Is the heading the first line on its page?""" + """Is the heading the first line on its page? No when that cannot be told + (a heading with no Latin letter to match, or a first line that a later line + repeats, as a running header does), so the page is shared.""" page = lines[page_no - 1] - if not page: + key = normalize(heading) + if not page or not re.search("[a-z]", key): return False - return normalize(heading) in normalize(page[0]) + starts = [(normalize(line) + " ").startswith(key + " ") for line in page] + return starts[0] and not any(starts[1:]) def assign_ends(node, children, lines): @@ -202,9 +206,30 @@ def attach_children(node, children, lines): # so gaining children leaves it unchanged sized = assign_ends(node, children, lines) node["nodes"] = sized + add_intro_nodes([node], lines) + if len(node["nodes"]) > len(sized): # an intro went in; numbered like the proposals + node["nodes"][0]["node_id"] = f"{node['node_id']}.0" return sized +def add_intro_nodes(structure, lines=None): + """Give every parent whose first child starts on a later page an intro child + for the pages before it, titled " (intro)". It ends where the + parent's own pages do, and no later than the first child's page, the page + before it when that child's heading opens its page.""" + for node, _ in list(flatten(structure)): + children = node.get("nodes") or [] + first = children[0]["start_index"] if children else None + if first is None or first <= node["start_index"]: + continue + opens = bool(lines) and first <= len(lines) and heading_at_page_start( + lines, first, children[0]["title"]) + end = max(node["start_index"], min(node["end_index"], first - 1 if opens else first)) + node["nodes"] = [{"title": intro_title(node.get("title")), + "start_index": node["start_index"], "end_index": end}] + children + return structure + + def relabel(structure, width=4): """Renumber every node_id in document order: 0000, 0001, 0002, ... @@ -281,14 +306,13 @@ def tree_cost_via_frontier(node, routing=ROUTING_COST): return max(d * routing + s for d, s, _ in entries) if entries else 0 -def expand_cost(node, children, routing=ROUTING_COST): - """Cost after one-step lookahead, children treated as collapsed.""" - covered = set() - for child in children: - covered |= set(range(child["start_index"], child["end_index"] + 1)) - residual = len(pages_of(node) - covered) - scans = [child["end_index"] - child["start_index"] + 1 for child in children] - return routing + max([residual] + scans), residual +def expand_cost(node, children, lines, routing=ROUTING_COST): + """Cost after one-step lookahead, children treated as collapsed, priced with + the intro node attach_children would give them.""" + trial = dict(node, nodes=children) + residual = S_residual(trial) + add_intro_nodes([trial], lines) + return tree_cost(trial, routing), residual # -------------------------------------------------------------------------- @@ -558,8 +582,9 @@ def visit(node): # titles are routing information; keep them on the parent, in document # order, carrying forward anything an earlier merge already folded in titles = [] - for child, _ in flatten(node["nodes"]): - titles.append(child["title"]) + for child, parent in flatten(node["nodes"]): + if not is_intro(parent or node, child): # its parent's heading again + titles.append(child["title"]) titles.extend(child.get("key_items") or []) log.append({"op": "merge", "node_id": node.get("node_id"), "S": span, "tree_cost": cost, "frontier_cost": checked, @@ -652,6 +677,57 @@ async def propose_children(node, pages, args): return accepted +def same_heading(a, b): + """Whether two titles name one heading: equal once normalized, or equal but + for a leading number only one of them prints.""" + a, b = normalize(a), normalize(b) + bare_a, bare_b = (re.sub(r"^(?:[0-9]+ )+", "", t) for t in (a, b)) + return bool(a) and (a == b or (bare_a == bare_b and (a == bare_a or b == bare_b))) + + +def headings(node): + """The headings printed for a node. A same-page fusion keeps them in its + key_items: the summary pass may rewrite the fused title while expand runs.""" + return node["key_items"] if node.get("_same_page") else [node["title"]] + + +def own_children(node, children, lines, known, ancestors, nxt): + """The proposed children printed inside the node's own text. + + Dropped: a heading that already is a node, in the tree as expand found it + (`known`, page -> headings) or made by the node's own ancestors; and on a + page the node shares, anything printed above its heading or at and below + the next node's. Other branches grow concurrently, so nothing they add is + read. A heading not found on its page decides nothing. + """ + start, end = node["start_index"], subtree_end(node) + lineage = [n for a in ancestors for n in [a] + a["nodes"]] + above = [t for n in [node] + ancestors if n["start_index"] == start for t in headings(n)] + after = headings(nxt) if nxt is not None and nxt["start_index"] == end else [] + + def found(page, matches): + page_lines = lines[page - 1] if page <= len(lines) else [] + return [i for i, line in enumerate(page_lines) if matches(line)] + + def is_node(page, title): + titles = known.get(page, []) + [t for n in lineage if n["start_index"] == page + for t in headings(n)] + return any(same_heading(title, t) for t in titles) + + top = found(start, lambda line: any(same_heading(line, t) for t in above)) + bottom = found(end, lambda line: any(same_heading(line, t) for t in after)) if after else [] + kept = [] + for child in children: + page, key = child["start_index"], normalize(child["title"]) + printed = found(page, lambda line: key in normalize(line)) if key else [] + if (is_node(page, child["title"]) + or page == start and top and printed and printed[-1] < top[0] + or page == end and bottom and printed and printed[0] >= bottom[-1]): + continue + kept.append(child) + return kept + + async def expand(structure, pages, lines, args, log, frozen): """One-step lookahead on every collapsed node over the trigger, recursively. @@ -663,7 +739,7 @@ async def expand(structure, pages, lines, args, log, frozen): changed = False semaphore = asyncio.Semaphore(args.concurrency) - async def proposals_for(node): + async def proposals_for(node, own): """The model half of one node's lookahead: the empty-retry ladder and absorbed errors run inside the task; log entries come back so a node's entries stay contiguous under concurrency.""" @@ -680,12 +756,13 @@ async def proposals_for(node): "decision": "error", "attempt": attempts, "detail": f"{type(exc).__name__}: {exc}"}) continue + proposed = own(proposed) if proposed: llm_candidates.append((f"llm:{attempts}", proposed)) break # an empty answer is retried, not trusted return llm_candidates, attempts, entries - async def process(node): + async def process(node, ancestors, nxt): nonlocal changed if not is_frontier(node) or node.get("node_id") in frozen: return @@ -694,10 +771,13 @@ async def process(node): return # below the trigger, stay collapsed note(args.progress, f" expand {node.get('node_id'):>8} S={span} " f"pages {node['start_index']}-{subtree_end(node)} ...") - llm_candidates, attempts, entries = await proposals_for(node) + + def own(children): + return own_children(node, children, lines, known, ancestors, nxt) + llm_candidates, attempts, entries = await proposals_for(node, own) log.extend(entries) candidates = [] - cached = children_from_cache(node, args.cache, args.kinds) + cached = own(children_from_cache(node, args.cache, args.kinds)) if cached: candidates.append(("cache", cached)) candidates.extend(llm_candidates) @@ -713,7 +793,7 @@ async def process(node): scored = [] for source, children in candidates: sized = assign_ends(node, children, lines) - cost, residual = expand_cost(node, sized, args.routing) + cost, residual = expand_cost(node, sized, lines, args.routing) scored.append({"source": source, "children": sized, "expand_cost": cost, "S_residual": residual}) scored.sort(key=lambda s: s["expand_cost"]) @@ -753,15 +833,24 @@ async def process(node): merge_same_page([node], log) # settle after the fusion: mark_final snapshots the children, finish() rejects a later change args.settled([node]) - results = await asyncio.gather(*(process(child) - for child in node["nodes"]), + children = node["nodes"] + results = await asyncio.gather(*(process(child, [node] + ancestors, after) + for child, after in zip(children, children[1:] + [nxt])), return_exceptions=True) for result in results: if isinstance(result, BaseException): raise result - results = await asyncio.gather(*(process(node) - for node, _ in flatten(structure)), + # the tree as it stands now: expand only ever adds below a leaf it is + # processing, so the nodes around another leaf never change under it + flat = list(flatten(structure)) + known, ancestry = {}, {} + for node, parent in flat: + known.setdefault(node["start_index"], []).extend(headings(node)) + ancestry[id(node)] = [parent] + ancestry[id(parent)] if parent else [] + results = await asyncio.gather(*(process(node, ancestry[id(node)], after) + for (node, _), after in + zip(flat, [n for n, _ in flat[1:]] + [None])), return_exceptions=True) for result in results: if isinstance(result, BaseException): diff --git a/pageindex/utils.py b/pageindex/utils.py index 31ee768b1..73f5c3170 100644 --- a/pageindex/utils.py +++ b/pageindex/utils.py @@ -591,9 +591,9 @@ def post_processing(structure, end_physical_index): item['start_index'] = item.get('physical_index') if i < len(structure) - 1: if structure[i + 1].get('appear_start') == 'yes': - item['end_index'] = structure[i + 1]['physical_index']-1 + item['end_index'] = max(item['start_index'], structure[i + 1]['physical_index']-1) else: - item['end_index'] = structure[i + 1]['physical_index'] + item['end_index'] = max(item['start_index'], structure[i + 1]['physical_index']) else: item['end_index'] = end_physical_index tree = list_to_tree(structure) @@ -708,8 +708,7 @@ def convert_page_to_int(data): def add_node_text(node, pdf_pages): if isinstance(node, dict): - start_page = node.get('start_index') - end_page = node.get('end_index') + start_page, end_page = own_pages(node) node['text'] = get_text_of_pdf_pages(pdf_pages, start_page, end_page) if 'nodes' in node: add_node_text(node['nodes'], pdf_pages) @@ -721,8 +720,7 @@ def add_node_text(node, pdf_pages): def add_node_text_with_labels(node, pdf_pages): if isinstance(node, dict): - start_page = node.get('start_index') - end_page = node.get('end_index') + start_page, end_page = own_pages(node) node['text'] = get_text_of_pdf_pages_with_labels(pdf_pages, start_page, end_page) if 'nodes' in node: add_node_text_with_labels(node['nodes'], pdf_pages) @@ -743,6 +741,63 @@ async def generate_node_summary(node, model=None): return response +FALLBACK_SUMMARY_CHARS = 600 + + +def fallback_summary(node, text=""): + """The summary of a node the model left unsummarized, so no node goes + without one: a parent's subsection titles, a leaf's own text.""" + children = node.get('nodes') or [] + if children: + summary = "; ".join(child['title'] for child in children if child.get('title')) + else: + summary = " ".join(text.split()) + return summary[:FALLBACK_SUMMARY_CHARS] or node.get('title') or "" + + +INTRO_SUFFIX = " (intro)" + + +def intro_title(title): + """The title of a parent's intro node; a split intro keeps its own.""" + title = (title or "").strip() + if title and not title.endswith(INTRO_SUFFIX): + title += INTRO_SUFFIX + return title or "Intro" + + +def is_intro(parent, child): + """Whether child is the intro node holding parent's opening pages.""" + return (child.get('start_index') == parent.get('start_index') + and child.get('title') == intro_title(parent.get('title'))) + + +def own_pages(node): + """The first and last page of a node's own text. A parent's runs onto the + page its first child starts on, where that child's heading may sit mid-page; + it has none (end None) when its intro holds those pages.""" + start, end = node.get('start_index'), node.get('end_index') + children = node.get('nodes') or [] + if children: + first = children[0].get('start_index') + if is_intro(node, children[0]): + end = None + elif end is not None and first is not None: + end = min(end, first) + return start, end + + +def cover_subtree_ranges(structure): + """Widen every parent's range to cover its whole subtree.""" + for node in structure if isinstance(structure, list) else [structure]: + children = node.get('nodes') or [] + if children: + cover_subtree_ranges(children) + node['start_index'] = min([node['start_index']] + [c['start_index'] for c in children]) + node['end_index'] = max([node['end_index']] + [c['end_index'] for c in children]) + return structure + + async def generate_summaries_for_structure(structure, model=None): nodes = structure_to_list(structure) tasks = [generate_node_summary(node, model=model) for node in nodes] @@ -956,7 +1011,7 @@ async def _ask(self, prompt, prio): self._asked = True async with self._gate.slot(prio): reply = await llm_acompletion(self._model, prompt) - if reply: + if parse_summary(reply): self._answered = True return reply @@ -1040,6 +1095,9 @@ async def _visit(self, node, depth): node['summary'] = "" if _is_unrecoverable(e): raise + if not node['summary']: + node['summary'] = fallback_summary(node, "" if children else get_text_of_pdf_pages( + self._pdf_pages, node['start_index'], node['end_index'])) async def finish(self): """Wait for every summary; fails loud if the model never answered.""" @@ -1252,9 +1310,58 @@ def load(self, user_opt=None) -> config: _resolve_models(merged) return config(**merged) +_TREE_WIRE_KEYS = frozenset({"title", "node_id", "start_index", "end_index", "page_index", + "summary", "prefix_summary", "text", "nodes"}) + + +def unify_tree(nodes, page_count=None): + """get_tree's nodes in the field names both modes return. + + The structure is the index's own: nodes are neither added nor moved, so + every node keeps the summary written for its own pages start_index .. + end_index. A node's first page comes as start_index (local) or + page_index (cloud), and a parent's summary as summary or prefix_summary. + Without an end_index (an older server) a node ends on the page where the + next node in reading order starts, the last on page_count. + """ + flat = list(_subtree(nodes)) + starts = [node.get("start_index", node.get("page_index")) for node in flat] + ranges = {} + for i, node in enumerate(flat): + start, end = starts[i], node.get("end_index") + if end is None: + end = starts[i + 1] if i + 1 < len(flat) else page_count + if end is None or (start is not None and end < start): + end = start + ranges[id(node)] = (start, end) + + def build(siblings): + out = [] + for node in siblings: + unified = {"title": node.get("title", "")} + if "node_id" in node: + unified["node_id"] = node["node_id"] + unified["start_index"], unified["end_index"] = ranges[id(node)] + unified.update((key, value) for key, value in node.items() + if key not in _TREE_WIRE_KEYS) + for key in ("summary", "prefix_summary"): + if key in node: + unified["summary"] = node[key] + if "text" in node: + unified["text"] = node["text"] + if node.get("nodes"): + unified["nodes"] = build(node["nodes"]) + out.append(unified) + return out + + return build(nodes) + + def create_node_mapping(tree, include_page_ranges=False, max_page=None): """Map node_id to node; with include_page_ranges, to {"node", "start_index", - "end_index"} (end = next node's page_index, or max_page for the last node).""" + "end_index"}: the node's own range as get_tree returns it, or, for a tree + that carries page_index only, end = next node's page_index (max_page for + the last node).""" def get_all_nodes(tree): if isinstance(tree, dict): return [tree] + [node for child in tree.get('nodes') or [] for node in get_all_nodes(child)] @@ -1268,10 +1375,14 @@ def get_all_nodes(tree): mapping = {} for i, node in enumerate(all_nodes): if node.get("node_id"): - end_page = all_nodes[i + 1].get("page_index") if i + 1 < len(all_nodes) else max_page + if "end_index" in node: + start_page, end_page = node["start_index"], node["end_index"] + else: + start_page = node["page_index"] + end_page = all_nodes[i + 1].get("page_index") if i + 1 < len(all_nodes) else max_page mapping[node["node_id"]] = { "node": node, - "start_index": node["page_index"], + "start_index": start_page, "end_index": end_page, } return mapping diff --git a/tests/test_agent_tools.py b/tests/test_agent_tools.py index 6fa04f6f3..3f1b1c823 100644 --- a/tests/test_agent_tools.py +++ b/tests/test_agent_tools.py @@ -287,6 +287,18 @@ def test_structure_strips_text_and_orders_keys(client, store_path): assert root["nodes"][0]["end_index"] == 1 +def test_structure_shows_the_ranges_get_tree_serves(client, store_path): + # a leaf stored ending before it starts: its successor was judged to open the same page + tree = [{"title": "Doc", "node_id": "0000", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "A", "node_id": "0001", "start_index": 2, "end_index": 1}, + {"title": "B", "node_id": "0002", "start_index": 2, "end_index": 2}]}] + seed_doc(store_path, "pi-a", "report.pdf", tree=tree) + payload, _ = run(client, "get_document_structure", doc_name="report.pdf") + served = client.get_tree("pi-a", include_text=False)["result"] + assert [(n["start_index"], n["end_index"]) for n in payload["structure"][0]["nodes"]] == [ + (n["start_index"], n["end_index"]) for n in served[0]["nodes"]] == [(2, 2), (2, 2)] + + def test_structure_multipart_pagination(client, store_path): big_tree = [{ "title": f"Chapter {index}", "node_id": f"{index:04d}", diff --git a/tests/test_client.py b/tests/test_client.py index 66b069cd3..6cceb98cc 100644 --- a/tests/test_client.py +++ b/tests/test_client.py @@ -549,24 +549,24 @@ def test_submit_and_get_tree(local_client, indexed_doc, tmp_path, monkeypatch): assert tree["status"] == "completed" assert tree["retrieval_ready"] is True root = tree["result"][0] - assert root["page_index"] == 1 - assert "start_index" not in root and "end_index" not in root - assert root["prefix_summary"] == "root summary" - assert "summary" not in root - child = root["nodes"][0] + assert (root["start_index"], root["end_index"]) == (1, 2) + assert "page_index" not in root and "prefix_summary" not in root + assert root["summary"] == "root summary" + assert "Hello page one" in root["text"] + child, = root["nodes"] + assert (child["start_index"], child["end_index"]) == (2, 2) assert child["summary"] == "child summary" assert child["text"] == "Second page about bananas" no_summary = local_client.get_tree(indexed_doc)["result"][0] - assert "summary" not in no_summary and "prefix_summary" not in no_summary + assert all("summary" not in node for node in [no_summary, *no_summary["nodes"]]) def test_get_tree_include_text_false(local_client, indexed_doc): tree = local_client.get_tree(indexed_doc, include_text=False) root = tree["result"][0] - assert "text" not in root - assert "text" not in root["nodes"][0] - assert root["page_index"] == 1 + assert all("text" not in node for node in [root, *root["nodes"]]) + assert (root["start_index"], root["end_index"]) == (1, 2) with_text = local_client.get_tree(indexed_doc)["result"][0] assert "text" in with_text @@ -576,10 +576,72 @@ def test_get_document_structure(local_client, indexed_doc): result = local_client.get_document_structure(indexed_doc) assert isinstance(result, list) root = result[0] - assert "text" not in root - assert "text" not in root["nodes"][0] - assert "prefix_summary" in root - assert root["nodes"][0]["summary"] == "child summary" + assert all("text" not in node for node in [root, *root["nodes"]]) + assert (root["start_index"], root["end_index"]) == (1, 2) + assert [node.get("summary") for node in [root, *root["nodes"]]] == [ + "root summary", "child summary"] + + +@pytest.mark.parametrize("mode", ["flash", "standard"]) +def test_get_tree_keeps_each_index_self_consistent(local_client, tmp_path, + monkeypatch, mode): + """get_tree adds and moves no node: every node keeps the summary written + for its own range, a flash parent's range covering its subtree and a + standard parent's ending where its first child starts.""" + from conftest import build_pdf + pdf = tmp_path / "report.pdf" + pdf.write_bytes(build_pdf(["Opening words", "Alpha body", "Beta body"])) + structure = [{"title": "Report", "node_id": "0000", "start_index": 1, + "end_index": 3 if mode == "flash" else 2, "summary": "report", + "nodes": [{"title": "Alpha", "node_id": "0001", "start_index": 2, + "end_index": 2, "summary": "alpha"}, + {"title": "Beta", "node_id": "0002", "start_index": 3, + "end_index": 3, "summary": "beta"}]}] + result = {"doc_name": "report.pdf", "doc_description": "d", "structure": structure} + monkeypatch.setattr(pageindex.flash, "page_index_flash", lambda pdf, **kw: result) + monkeypatch.setattr(page_index_module, "page_index_main", lambda *a, **kw: result) + monkeypatch.setattr(pageindex.utils, "llm_completion", lambda model, prompt, **kw: "d") + doc_id = local_client.submit_document(str(pdf), mode=mode)["doc_id"] + + root = local_client.get_tree(doc_id, node_summary=True)["result"][0] + assert [(node["title"], node["start_index"], node["end_index"], node["summary"]) + for node in [root, *root["nodes"]]] == [ + ("Report", 1, 3 if mode == "flash" else 2, "report"), + ("Alpha", 2, 2, "alpha"), ("Beta", 3, 3, "beta")] + # the parent's text runs onto its first child's page, where that heading may sit mid-page + assert "Alpha body" in root["text"] and "Beta body" not in root["text"] + + +def test_unify_tree_edges(): + from pageindex.utils import unify_tree + tree = [{"title": "A", "page_index": 3, "prefix_summary": "a", + "nodes": [{"title": "A.1", "page_index": 5, "summary": "a1"}, + {"title": "A.2", "page_index": 7, "prefix_summary": "a2", + "nodes": [{"title": "A.2.1", "page_index": 8}]}]}] + once = unify_tree(tree, 9) + # without end_index each node ends where the next one in reading order + # starts: a parent's range is its own opening, as its summary describes + assert once == [ + {"title": "A", "start_index": 3, "end_index": 5, "summary": "a", "nodes": [ + {"title": "A.1", "start_index": 5, "end_index": 7, "summary": "a1"}, + {"title": "A.2", "start_index": 7, "end_index": 8, "summary": "a2", "nodes": [ + {"title": "A.2.1", "start_index": 8, "end_index": 9}]}]}] + assert unify_tree(once, 9) == once + # a node with no page (a hand-edited store) passes through without raising + assert unify_tree([{"title": "B", "start_index": None}], None) == [ + {"title": "B", "start_index": None, "end_index": None}] + + +def test_create_node_mapping_reads_get_tree_ranges(): + from pageindex.utils import create_node_mapping + tree = [{"title": "A", "node_id": "0000", "start_index": 1, "end_index": 9, + "nodes": [{"title": "B", "node_id": "0001", "start_index": 4, "end_index": 9}]}] + mapping = create_node_mapping(tree, include_page_ranges=True, max_page=12) + assert [(m["start_index"], m["end_index"]) for m in mapping.values()] == [(1, 9), (4, 9)] + legacy = [{"title": "A", "node_id": "0000", "page_index": 1, + "nodes": [{"title": "B", "node_id": "0001", "page_index": 4}]}] + mapping = create_node_mapping(legacy, include_page_ranges=True, max_page=12) + assert [(m["start_index"], m["end_index"]) for m in mapping.values()] == [(1, 4), (4, 12)] def test_get_page_content(local_client, indexed_doc): @@ -1696,6 +1758,69 @@ def test_cloud_request_wiring(cloud, sample_pdf): assert calls[-1]["headers"] == {"api_key": "other"} +def test_cloud_get_tree_unified_shape(monkeypatch): + """An older server's tree wire (page_index only, a parent's + prefix_summary) leaves the SDK in local's field names: a node ends where + the next one in reading order starts, the last on the document's page + count, so a parent's range is the opening its summary describes.""" + client = PageIndexClient(api_key="secret") + wire = {"doc_id": "pi-1", "status": "completed", "retrieval_ready": True, + "metadata": None, "features": {}, "result": [ + {"title": "Ch 1", "node_id": "0000", "page_index": 2, + "prefix_summary": "opening", "text": "ch1 own", "nodes": [ + {"title": "1.1", "node_id": "0001", "page_index": 4, "summary": "s11", "text": "t11"}, + {"title": "1.2", "node_id": "0002", "page_index": 6, "summary": "s12", "text": "t12"}]}, + {"title": "Ch 2", "node_id": "0003", "page_index": 9, + "prefix_summary": "same page", "text": "ch2 own", "nodes": [ + {"title": "2.1", "node_id": "0004", "page_index": 9, "summary": "s21", "text": "t21"}]}]} + urls = [] + + def handler(method, url, kw): + urls.append(url) + return FakeResponse(wire if "type=tree" in url else {"pageNum": 12}) + _patch_requests(monkeypatch, handler) + + result = client.get_tree("pi-1", node_summary=True)["result"] + assert result == [ + {"title": "Ch 1", "node_id": "0000", "start_index": 2, "end_index": 4, + "summary": "opening", "text": "ch1 own", "nodes": [ + {"title": "1.1", "node_id": "0001", "start_index": 4, "end_index": 6, + "summary": "s11", "text": "t11"}, + {"title": "1.2", "node_id": "0002", "start_index": 6, "end_index": 9, + "summary": "s12", "text": "t12"}]}, + {"title": "Ch 2", "node_id": "0003", "start_index": 9, "end_index": 9, + "summary": "same page", "text": "ch2 own", "nodes": [ + {"title": "2.1", "node_id": "0004", "start_index": 9, "end_index": 12, + "summary": "s21", "text": "t21"}]}] + assert list(result[0]) == ["title", "node_id", "start_index", "end_index", + "summary", "text", "nodes"] + assert urls[-1] == "https://api.pageindex.ai/doc/pi-1/metadata/" + + +def test_cloud_get_tree_takes_served_ranges(monkeypatch): + """A server that sends start_index/end_index has its ranges taken as + given, with no metadata request.""" + client = PageIndexClient(api_key="secret") + wire = {"status": "completed", "retrieval_ready": True, "result": [ + {"title": "Ch", "node_id": "0000", "page_index": 2, "start_index": 2, + "end_index": 3, "prefix_summary": "opening", "nodes": [ + {"title": "S", "node_id": "0001", "page_index": 4, "start_index": 4, + "end_index": 8, "summary": "s"}]}]} + urls = [] + + def handler(method, url, kw): + urls.append(url) + return FakeResponse(wire) + _patch_requests(monkeypatch, handler) + + assert client.get_tree("pi-1", node_summary=True)["result"] == [ + {"title": "Ch", "node_id": "0000", "start_index": 2, "end_index": 3, + "summary": "opening", "nodes": [ + {"title": "S", "node_id": "0001", "start_index": 4, "end_index": 8, + "summary": "s"}]}] + assert len(urls) == 1 + + def test_cloud_error_and_empty_delete(cloud, monkeypatch): client, calls, fake = cloud _patch_requests(monkeypatch, @@ -2407,12 +2532,13 @@ def test_format_tree_node_keeps_key_items(): # ── retry-ladder and summary fail-loud edges (twelfth review) ── -def test_summarize_tree_all_empty_replies_fail_loud(monkeypatch): +@pytest.mark.parametrize("reply", ["", '{"summary": ""}']) +def test_summarize_tree_all_empty_replies_fail_loud(monkeypatch, reply): """Empty-content replies (content filter, spent output cap) must not vouch for the model: a raw-text short leaf cannot carry the run when every model reply comes back blank.""" async def blank(model, prompt): - return "" + return reply monkeypatch.setattr(pageindex.utils, "llm_acompletion", blank) pdf_pages = [("tiny", 1), ("beta " * 300, 300)] structure = [{"title": "R", "start_index": 1, "end_index": 2, @@ -2424,8 +2550,8 @@ async def blank(model, prompt): def test_summarize_tree_partial_empty_reply_absorbed(monkeypatch): - """One blank reply among good ones stays the documented per-node - absorption: blank summary, run survives.""" + """One blank reply among good ones is absorbed per node: the node falls + back to its own text, the run survives.""" async def flaky(model, prompt): if "alpha" in prompt: return "" @@ -2435,7 +2561,7 @@ async def flaky(model, prompt): structure = [{"title": "A", "start_index": 1, "end_index": 1}, {"title": "B", "start_index": 2, "end_index": 2}] out = asyncio.run(pageindex.utils.summarize_tree(structure, pdf_pages)) - assert [n["summary"] for n in out] == ["", "ok"] + assert [n["summary"] for n in out] == [" ".join(["alpha"] * 300)[:600], "ok"] def test_generate_doc_description_absorbs_context_overflow(monkeypatch): diff --git a/tests/test_flash_extraction.py b/tests/test_flash_extraction.py index c53d3aa6b..c2fc301df 100644 --- a/tests/test_flash_extraction.py +++ b/tests/test_flash_extraction.py @@ -707,9 +707,12 @@ def test_every_page_is_in_a_node(tmp_path): covered = {page for node in structure for page in range(node["start_index"], node["end_index"] + 1)} assert covered == set(range(1, pages + 1)), pdf.name - assert structure[0] == {"title": "Preface", "node_id": "0000", - "start_index": 1, "end_index": 1}, pdf.name - assert structure[1]["node_id"] == "0001", pdf.name + preface, first = structure[0], structure[1] + assert (preface["title"], preface["node_id"], preface["start_index"]) == ( + "Preface", "0000", 1), pdf.name + # it runs onto the first section's page unless that heading is known to open it + assert preface["end_index"] in (first["start_index"] - 1, first["start_index"]), pdf.name + assert first["node_id"] == "0001", pdf.name def test_preface_page_is_retrievable(tmp_path, monkeypatch): @@ -732,8 +735,8 @@ async def no_summary(*args, **kwargs): client = PageIndexClient(storage_path=str(tmp_path / "store")) doc_id = client.submit_document(str(pdf), mode="flash")["doc_id"] tree = client.get_tree(doc_id)["result"] - assert [(node["title"], node["page_index"]) for node in tree] == [ - ("Preface", 1), ("Budget", 2), ("Team", 3)] + assert [(node["title"], node["start_index"], node["end_index"]) for node in tree] == [ + ("Preface", 1, 1), ("Budget", 2, 2), ("Team", 3, 3)] assert "Skylark" in tree[0]["text"] diff --git a/tests/test_tree_format.py b/tests/test_tree_format.py new file mode 100644 index 000000000..37c61ad24 --- /dev/null +++ b/tests/test_tree_format.py @@ -0,0 +1,492 @@ +"""The tree every local index writes: intro nodes, covering ranges, leaf-only +splitting, and summaries every node gets, a parent's built from its children's.""" + +import asyncio +import copy +import importlib +from types import SimpleNamespace + +import pytest + +import pageindex.flash +import pageindex.tree_optimize as tree_optimize +import pageindex.utils as utils +from conftest import build_pdf +from pageindex import PageIndexClient + +classic = importlib.import_module("pageindex.page_index_classic") + + +def shape(nodes): + return [(n["title"], n["start_index"], n["end_index"]) for n in utils._subtree(nodes)] + + +def test_intro_node_holds_the_pages_a_parent_opens_with(): + lines = [["body"] for _ in range(40)] + lines[11] = ["3.1 Scope", "body"] # 3.1 opens its page, 4.1 starts mid-page + tree = [{"title": "Ch 3", "start_index": 10, "end_index": 30, "nodes": [ + {"title": "3.1 Scope", "start_index": 12, "end_index": 20}, + {"title": "3.2", "start_index": 21, "end_index": 30, "nodes": [ + {"title": "3.2.1", "start_index": 21, "end_index": 30}]}]}, + {"title": "", "start_index": 31, "end_index": 40, "nodes": [ + {"title": "4.1", "start_index": 33, "end_index": 40}]}] + + tree_optimize.add_intro_nodes(tree, lines) + + assert shape(tree) == [ + ("Ch 3", 10, 30), ("Ch 3 (intro)", 10, 11), ("3.1 Scope", 12, 20), + ("3.2", 21, 30), ("3.2.1", 21, 30), # a first child on its parent's page: no intro + ("", 31, 40), ("Intro", 31, 33), ("4.1", 33, 40)] + assert shape(tree_optimize.add_intro_nodes(copy.deepcopy(tree), lines)) == shape(tree) + # a standard parent ends where its first child starts; a split intro keeps its title + split = [{"title": "Ch 3 (intro)", "start_index": 10, "end_index": 11, "nodes": [ + {"title": "Background", "start_index": 12, "end_index": 14}]}] + assert shape(tree_optimize.add_intro_nodes(split))[1] == ("Ch 3 (intro)", 10, 11) + # a heading with nothing to match cannot be placed on its page, so the page is shared + assert not tree_optimize.heading_at_page_start([["第一章 总则"]], 1, "第一章 总则") + # nor can digits alone, which may be a page number + assert not tree_optimize.heading_at_page_start([["2", "Body text"]], 1, "2") + + +def test_expand_gives_a_split_node_its_intro(monkeypatch): + body = "body " * 250 + pages = [body] * 12 + pages[5] = "Sub One\n" + body + pages[8] = "Sub Two\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 12, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "X", "start_index": 4, "end_index": 12, "node_id": "0002"}]}] + + async def propose(model, prompt): + return {"subsections": [{"title": "Sub One", "page": 6}, {"title": "Sub Two", "page": 9}]} + + async def summarize(model, prompt): + return '{"summary": "ok"}' + monkeypatch.setattr(tree_optimize, "ask_model", propose) + monkeypatch.setattr(utils, "llm_acompletion", summarize) + + async def run(): + scheduler = utils.SummaryScheduler(tree, [(page, 0) for page in pages], model="m") + await tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, + on_final=scheduler.mark_final) + await scheduler.finish() + asyncio.run(run()) + + x = tree[0]["nodes"][1] + assert [(n["title"], n["start_index"], n["end_index"], n["node_id"]) for n in x["nodes"]] == [ + ("X (intro)", 4, 5, "0003"), ("Sub One", 6, 8, "0004"), ("Sub Two", 9, 12, "0005")] + assert all(n["summary"] == "ok" for n in utils._subtree(tree)) + + +def test_a_running_header_or_a_word_prefix_does_not_open_the_page(): + page = ["4.1. Discriminant Functions 181", "opening words", "4.1. Discriminant Functions"] + assert not tree_optimize.heading_at_page_start([page], 1, "4.1. Discriminant Functions") + assert tree_optimize.heading_at_page_start([page[2:]], 1, "4.1. Discriminant Functions") + assert not tree_optimize.heading_at_page_start([["Filed 03/04/24", "I. Background"]], 1, "I.") + + +def test_expand_prices_a_level_with_the_intro_it_gets(monkeypatch): + body = "body " * 250 + pages = [body] * 12 + pages[10] = body + "\nSub Late\n" + body # mid-page: the intro shares page 11 + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 12, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "X", "start_index": 4, "end_index": 12, "node_id": "0002"}]}] + + async def propose(model, prompt): + return {"subsections": [{"title": "Sub Late", "page": 11}]} + + async def summarize(model, prompt): + return '{"summary": "ok"}' + monkeypatch.setattr(tree_optimize, "ask_model", propose) + monkeypatch.setattr(utils, "llm_acompletion", summarize) + + async def run(): + scheduler = utils.SummaryScheduler(tree, [(page, 0) for page in pages], model="m") + await tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, + on_final=scheduler.mark_final) + await scheduler.finish() + asyncio.run(run()) + + # intro 4-11 + Sub Late 11-12 costs 1 + 8 = 9 pages, no better than reading X's 9 + assert "nodes" not in tree[0]["nodes"][1] + + +def test_expand_skips_the_heading_of_a_neighbor_sharing_the_last_page(monkeypatch): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + # mid-page, so Methods runs onto page 12; run in, so only the tree knows the headings below it + pages[11] = body + "\n3 Results. We report three findings.\n3.1 Data\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003", "nodes": [ + {"title": "3.1 Data", "start_index": 12, "end_index": 15, "node_id": "0004"}, + {"title": "3.2 More", "start_index": 16, "end_index": 20, "node_id": "0005"}]}]}] + + replies = [[{"title": "Results", "page": 12}, {"title": "3.1 Data", "page": 12}], + [{"title": "Setup", "page": 6}, {"title": "Results", "page": 12}, + {"title": "3.1 Data", "page": 12}]] + + async def propose(model, prompt): + if "Section title: Methods\n" in prompt: # a reply the filter empties is asked again + return {"subsections": replies.pop(0)} + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True)) + + assert not replies + assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] + + +def test_expand_filters_cached_headings_like_proposed_ones(monkeypatch): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + pages[11] = body + "\n3 Results\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + cache = {6: [{"title": "Setup", "kind": "section"}], + 12: [{"title": "3 Results", "kind": "section"}]} + + async def propose(model, prompt): + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, cache=cache)) + + assert shape(tree[0]["nodes"][1]["nodes"]) == [("Methods (intro)", 4, 5), ("Setup", 6, 12)] + + +def test_expand_hands_each_new_child_its_ancestors_and_next_node(monkeypatch): + body = "body " * 250 + pages = [body] * 30 + pages[3] = "P\nP.a Early\n" + body + pages[6] = "P.b Later\n" + body + pages[9] = body + "\nQ\nQ.0 Overview\n" + body + pages[12] = "Q.i Intro part\n" + body + pages[16] = "Q.1 First\n" + body + pages[19] = "Q.2 Second\n" + body + pages[24] = body + "\nS\nS.1 Part\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 30, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "P", "start_index": 4, "end_index": 25, "node_id": "0002"}, + {"title": "S", "start_index": 25, "end_index": 30, "node_id": "0003"}]}] + replies = { + "P": [("Q", 10)], + # Q, made in this pass, is P's intro's next node: what follows its heading is not the intro's + "P (intro)": [("P.a Early", 4), ("P.b Later", 7), ("Q.0 Overview", 10)], + # Q's next node is P's: S + "Q": [("Q.1 First", 17), ("Q.2 Second", 20), ("S.1 Part", 25)], + # Q's intro sits under Q, an ancestor made in this pass + "Q (intro)": [("Q", 10), ("Q.0 Overview", 10), ("Q.i Intro part", 13)]} + + async def propose(model, prompt): + title = prompt.split("Section title: ")[1].split("\n")[0] + return {"subsections": [{"title": t, "page": p} for t, p in replies.get(title, [])]} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True, do_relabel=False)) + + assert shape(tree) == [ + ("R", 1, 30), ("A", 1, 3), ("P", 4, 25), + ("P (intro)", 4, 10), ("P.a Early", 4, 6), ("P.b Later", 7, 10), + ("Q", 10, 25), ("Q (intro)", 10, 16), ("Q.0 Overview", 10, 12), ("Q.i Intro part", 13, 16), + ("Q.1 First", 17, 19), ("Q.2 Second", 20, 25), ("S", 25, 30)] + + +def test_a_proposed_child_must_be_printed_inside_its_nodes_own_text(): + lines = [["body"] for _ in range(20)] + lines[11] = ["2.3 Limits", "3 Results", "Scope", "3.1 Data", "3.2 Results"] + methods = {"title": "2 Methods", "start_index": 4, "end_index": 12} + results = {"title": "3 Results", "start_index": 12, "end_index": 20} + root = {"title": "R", "start_index": 1, "end_index": 20, "nodes": [methods, results]} + known = {} + for node, _ in tree_optimize.flatten([root]): + known.setdefault(node["start_index"], []).append(node["title"]) + + def kept(node, ancestors, nxt, *titles, known=known): + children = [{"title": title, "start_index": 12} for title in titles] + return [c["title"] for c in + tree_optimize.own_children(node, children, lines, known, ancestors, nxt)] + + # on a shared last page: the next node's heading, and what is printed below it + assert kept(methods, [root], results, "2.3 Limits", "Results", "3.1 Data") == ["2.3 Limits"] + # a heading that already is a node, wherever it sits on the page and in the tree + assert kept(methods, [root], None, "2.3 Limits", "Results") == ["2.3 Limits"] + assert kept(methods, [root], None, "2.3 Limits", "3.1 Data", known={12: ["3.1 Data"]}) == [ + "2.3 Limits"] + # on a shared first page: what is printed above the node's heading; a number tells headings apart + assert kept(results, [root], None, "2.3 Limits", "Results", "3.1 Data", "3.2 Results") == [ + "3.1 Data", "3.2 Results"] + # an intro sits under its parent's heading and above the siblings expand made with it + data = {"title": "3.1 Data", "start_index": 12, "end_index": 20} + intro = {"title": "3 Results (intro)", "start_index": 12, "end_index": 12} + results["nodes"] = [intro, data] + assert kept(intro, [results, root], data, "2.3 Limits", "Results", "Scope", "3.1 Data") == ["Scope"] + # ... and knows them through its lineage when this pass made them + assert kept(intro, [results, root], None, "Results", "Scope", "3.1 Data", known={}) == ["Scope"] + # a next heading not found on the page decides nothing + assert kept(methods, [root], dict(results, title="Findings"), "2.3 Limits", "3.2 Results") == [ + "2.3 Limits", "3.2 Results"] + + +def test_own_children_errs_toward_keeping_where_a_heading_repeats(): + methods = {"title": "2 Methods", "start_index": 4, "end_index": 12} + results = {"title": "3 Results", "start_index": 12, "end_index": 20} + + def kept(node, nxt, page, *titles): + lines = [["body"] for _ in range(20)] + lines[11] = page + children = [{"title": title, "start_index": 12} for title in titles] + return [c["title"] for c in tree_optimize.own_children(node, children, lines, {}, [], nxt)] + + # a running header repeats the next heading: the heading is its last line + assert kept(methods, results, ["3 Results", "2.4 Tail", "3 Results", "3.1 Data"], + "2.4 Tail", "3.1 Data") == ["2.4 Tail"] + # a child also mentioned above the node's heading: the child is its last line + assert kept(results, None, ["see 3.1 Data", "3 Results", "3.1 Data"], "3.1 Data") == ["3.1 Data"] + # the next heading spelled otherwise, on the very line the proposal is printed on + assert kept(methods, dict(results, title="IV. Results"), ["Tail", "IV. Results"], + "Tail", "Results") == ["Tail"] + # a same-page fusion is found by its headings, not the title the summary pass rewrites + fused = dict(results, title="Rewritten", _same_page=True, key_items=["3 Results"]) + assert kept(methods, fused, ["2.4 Tail", "3 Results", "3.1 Data"], "2.4 Tail", "3.1 Data") == [ + "2.4 Tail"] + + +@pytest.mark.parametrize("methods_delay", [0, 0.05]) +def test_expand_gives_one_tree_whichever_reply_lands_first(monkeypatch, methods_delay): + body = "body " * 250 + pages = [body] * 20 + pages[5] = "Setup\n" + body + pages[11] = body + "\n3 Results\nlead\n3.1 Data\n" + body + pages[14] = "3.2 Analysis\n" + body + lines = [[line for line in page.splitlines() if line.strip()] for page in pages] + tree = [{"title": "R", "start_index": 1, "end_index": 20, "node_id": "0000", "nodes": [ + {"title": "A", "start_index": 1, "end_index": 3, "node_id": "0001"}, + {"title": "Methods", "start_index": 4, "end_index": 12, "node_id": "0002"}, + {"title": "3 Results", "start_index": 12, "end_index": 20, "node_id": "0003"}]}] + + async def propose(model, prompt): + if "Section title: Methods\n" in prompt: + await asyncio.sleep(methods_delay) + return {"subsections": [{"title": "Setup", "page": 6}, {"title": "3.1 Data", "page": 12}]} + if "Section title: 3 Results\n" in prompt: + await asyncio.sleep(0.05 - methods_delay) + return {"subsections": [{"title": "3.1 Data", "page": 12}, {"title": "3.2 Analysis", "page": 15}]} + return {"subsections": []} + monkeypatch.setattr(tree_optimize, "ask_model", propose) + + asyncio.run(tree_optimize.optimize(tree, pages, lines, model="m", do_expand=True)) + + assert shape(tree[0]["nodes"][1:]) == [ + ("Methods", 4, 12), ("Methods (intro)", 4, 5), ("Setup", 6, 12), + ("3 Results", 12, 20), ("3.1 Data", 12, 14), ("3.2 Analysis", 15, 20)] + + +def test_merge_folds_an_intro_without_listing_its_title(): + tree = [{"title": "P", "start_index": 1, "end_index": 4, "nodes": [ + {"title": "P (intro)", "start_index": 1, "end_index": 1}, + {"title": "C", "start_index": 2, "end_index": 4}]}] + tree_optimize.merge_tree(tree) + assert "nodes" not in tree[0] and tree[0]["key_items"] == ["C"] + + +def test_a_toc_item_whose_page_the_next_opens_still_ends_on_its_own_page(): + items = [{"structure": "1", "title": "Overview", "physical_index": 5}, + {"structure": "2", "title": "Scope", "physical_index": 5, "appear_start": "yes"}, + {"structure": "3", "title": "Terms", "physical_index": 9}, + {"structure": "4", "title": "Annex", "physical_index": 8}] # listed out of order + assert shape(utils.post_processing(items, 12)) == [ + ("Overview", 5, 5), ("Scope", 5, 9), ("Terms", 9, 9), ("Annex", 8, 12)] + + +LARGE = SimpleNamespace(max_page_num_each_node=10, max_token_num_each_node=20000, model="m") + + +def test_large_node_split_keeps_existing_children(monkeypatch): + async def meta_processor(*args, **kwargs): + raise AssertionError("a parent's subsections must not be replaced") + monkeypatch.setattr(classic, "meta_processor", meta_processor) + node = {"title": "Ch", "start_index": 1, "end_index": 20, + "nodes": [{"title": "S", "start_index": 20, "end_index": 25}]} + + asyncio.run(classic.process_large_node_recursively(node, [("page", 5000)] * 25, LARGE)) + + assert [child["title"] for child in node["nodes"]] == ["S"] + + +def test_split_intro_skips_its_parents_heading(monkeypatch): + async def meta_processor(*args, **kwargs): + return [{"title": "Ch", "physical_index": 1}, + {"title": "Background", "physical_index": 5}, + {"title": "Scope", "physical_index": 9}] + + async def appear(items, *args, **kwargs): + for item in items: + item["appear_start"] = "yes" + return items + monkeypatch.setattr(classic, "meta_processor", meta_processor) + monkeypatch.setattr(classic, "check_title_appearance_in_start_concurrent", appear) + intro = {"title": "Ch (intro)", "start_index": 1, "end_index": 14} + + asyncio.run(classic.process_large_node_recursively(intro, [("page", 5000)] * 14, LARGE)) + + assert [child["title"] for child in intro["nodes"]] == ["Background", "Scope"] + + +def test_standard_index_stores_intros_covering_ranges_and_section_summaries(tmp_path, monkeypatch): + texts = ["Opening words"] + [f"page {n}" for n in range(2, 41)] + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(texts)) + + async def tree_parser(page_list, opt, doc=None, logger=None): + return [{"title": "P", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "P (intro)", "start_index": 1, "end_index": 2}, + {"title": "C1", "start_index": 3, "end_index": 20}, + {"title": "C2", "start_index": 21, "end_index": 21, "nodes": [ + {"title": "C2.1", "start_index": 21, "end_index": 30}, + {"title": "C2.2", "start_index": 31, "end_index": 40}]}]}] + prompts = [] + + async def reply(model, prompt): + prompts.append(prompt) + return "whole " + prompt.split("Section Title: ")[1].split("\n")[0] + monkeypatch.setattr(classic, "tree_parser", tree_parser) + monkeypatch.setattr(utils, "llm_acompletion", reply) + monkeypatch.setattr(utils, "llm_completion", lambda model, prompt, **kw: "d") + opt = utils.ConfigLoader().load({"model": "m", "if_add_node_id": "yes", + "if_add_node_summary": "yes", "if_add_node_text": "yes", + "if_add_doc_description": "yes"}) + + result = classic.page_index_main(str(pdf), opt, logger=SimpleNamespace(info=print), + page_list=[(text, 1) for text in texts]) + + p = result["structure"][0] + c2 = p["nodes"][2] + assert shape([p]) == [("P", 1, 40), ("P (intro)", 1, 2), ("C1", 3, 20), + ("C2", 21, 40), ("C2.1", 21, 30), ("C2.2", 31, 40)] + assert (p["text"], p["summary"]) == ("", "whole P") # its intro holds pages 1-2 + assert "Opening words" in p["nodes"][0]["text"] + # C2's first child starts on C2's page; C2 keeps that page as its opening + assert (c2["text"], c2["summary"]) == ("page 21", "whole C2") + # short leaves keep their own text; only the parents are asked, deepest first + assert [n["summary"] for n in p["nodes"][:2]] == [ + "Opening wordspage 2", "".join(f"page {n}" for n in range(3, 21))] + assert [q.split("Section Title: ")[1].split("\n")[0] for q in prompts] == ["C2", "P"] + assert '"summary": "whole C2"' in prompts[-1] + + +def test_standard_index_gives_toc_and_split_parents_intros(tmp_path, monkeypatch): + texts = [f"page {n}" for n in range(1, 41)] + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(texts)) + + splits = {1: [{"structure": "1", "title": "P", "physical_index": 1}, # P's long opening + {"structure": "2", "title": "Background", "physical_index": 8}], + 20: [{"structure": "1", "title": "C2", "physical_index": 20}, # the large leaf C2 + {"structure": "2", "title": "C2.a", "physical_index": 25}, + {"structure": "3", "title": "C2.b", "physical_index": 32}]} + + async def meta_processor(page_list, mode=None, start_index=1, **kwargs): + if len(page_list) == len(texts): + return [{"structure": "1", "title": "P", "physical_index": 1}, + {"structure": "1.1", "title": "C1", "physical_index": 15}, + {"structure": "1.2", "title": "C2", "physical_index": 20}] + return splits[start_index] + + async def appear(items, *args, **kwargs): + return items + monkeypatch.setattr(classic, "check_toc", lambda page_list, opt: {"toc_content": None}) + monkeypatch.setattr(classic, "meta_processor", meta_processor) + monkeypatch.setattr(classic, "check_title_appearance_in_start_concurrent", appear) + opt = utils.ConfigLoader().load({"model": "m", "if_add_node_summary": "no"}) + + result = classic.page_index_main(str(pdf), opt, logger=SimpleNamespace(info=print), + page_list=[(text, 3000) for text in texts]) + + assert shape(result["structure"]) == [ + ("P", 1, 40), ("P (intro)", 1, 15), ("P (intro)", 1, 8), ("Background", 8, 15), + ("C1", 15, 20), + ("C2", 20, 40), ("C2 (intro)", 20, 25), ("C2.a", 25, 32), ("C2.b", 32, 40)] + + +def test_flash_gives_parents_their_intro_nodes(tmp_path, monkeypatch): + import pageindex.flash.api as flash_api + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["x"])) + monkeypatch.setattr(flash_api, "extract_toc", lambda pdf, **kw: { + "structure": [{"title": "R", "node_id": "0000", "start_index": 1, "end_index": 6, "nodes": [ + {"title": "A", "node_id": "0001", "start_index": 3, "end_index": 4}, + {"title": "X", "node_id": "0002", "start_index": 5, "end_index": 6}]}], + "page_texts": ["Cover", "Foreword", "A\nalpha", "alpha", "X\nbody", "body"]}) + + result = flash_api.page_index_flash(str(pdf), summary=False, optimize=False) + + assert [(n["title"], n["node_id"], n["start_index"], n["end_index"]) + for n in utils._subtree(result["structure"])] == [ + ("R", "0000", 1, 6), ("R (intro)", "0001", 1, 2), ("A", "0002", 3, 4), + ("X", "0003", 5, 6)] + + +def test_flash_preface_runs_onto_a_page_its_first_heading_does_not_open(tmp_path, monkeypatch): + import pageindex.flash.api as flash_api + pdf = tmp_path / "doc.pdf" + pdf.write_bytes(build_pdf(["x"])) + ends = [] + for third in ["A\nalpha", "Contents, continued\nA\nalpha"]: + monkeypatch.setattr(flash_api, "extract_toc", lambda pdf, third=third, **kw: { + "structure": [{"title": "A", "node_id": "0000", "start_index": 3, "end_index": 4}], + "page_texts": ["Cover", "Contents", third, "alpha"]}) + preface = flash_api.page_index_flash(str(pdf), summary=False, optimize=False)["structure"][0] + ends.append((preface["title"], preface["node_id"], preface["end_index"])) + assert ends == [("Preface", "0000", 2), ("Preface", "0000", 3)] + + +def test_summarize_tree_parent_falls_back_to_its_subsection_titles(monkeypatch): + async def reply(model, prompt): + return "" if "Section Title" in prompt else '{"summary": "ok"}' + monkeypatch.setattr(utils, "llm_acompletion", reply) + structure = [{"title": "R", "start_index": 1, "end_index": 2, "nodes": [ + {"title": "A", "start_index": 1, "end_index": 1}, + {"title": "B", "start_index": 2, "end_index": 2}]}] + + out = asyncio.run(utils.summarize_tree(structure, [("alpha " * 300, 0), ("beta " * 300, 0)])) + + assert [n["summary"] for n in utils._subtree(out)] == ["A; B", "ok", "ok"] + + +def test_get_tree_parent_text_shares_its_first_childs_page_unless_an_intro_holds_it( + tmp_path, monkeypatch): + pdf = tmp_path / "report.pdf" + pdf.write_bytes(build_pdf(["Opening words", "Alpha body", "Beta body"])) + structure = [{"title": "Report", "node_id": "0000", "start_index": 1, "end_index": 3, + "summary": "report", "nodes": [ + {"title": "Report (intro)", "node_id": "0001", "start_index": 1, + "end_index": 1, "summary": "opening"}, + {"title": "Alpha", "node_id": "0002", "start_index": 2, + "end_index": 3, "summary": "alpha", "nodes": [ + {"title": "Alpha one", "node_id": "0003", "start_index": 2, + "end_index": 2, "summary": "one"}, + {"title": "Alpha two", "node_id": "0004", "start_index": 3, + "end_index": 3, "summary": "two"}]}]}] + monkeypatch.setattr(pageindex.flash, "page_index_flash", lambda pdf, **kw: { + "doc_name": "report.pdf", "structure": structure}) + monkeypatch.setattr(utils, "llm_completion", lambda model, prompt, **kw: "d") + client = PageIndexClient(storage_path=str(tmp_path / "store")) + doc_id = client.submit_document(str(pdf), mode="flash")["doc_id"] + + root = client.get_tree(doc_id)["result"][0] + + assert [n["text"] for n in utils._subtree([root])] == [ + "", "Opening words", "Alpha body", "Alpha body", "Beta body"]