From 4a116daa2fefdbf6f6dd330c75c22ef2ab364caf Mon Sep 17 00:00:00 2001 From: Ray Date: Sun, 2 Aug 2026 17:56:37 +0800 Subject: [PATCH] Replace thinning with merge --- README.md | 2 +- pageindex/flash/README.md | 1 + pageindex/flash/api.py | 17 +++++++++-------- pageindex/page_index.py | 7 ++++--- pageindex/tree_optimize.py | 15 +++++++++++++-- pageindex/utils.py | 1 + run_pageindex.py | 5 +++-- 7 files changed, 32 insertions(+), 16 deletions(-) diff --git a/README.md b/README.md index 752f91d..5ce0ca5 100644 --- a/README.md +++ b/README.md @@ -205,7 +205,7 @@ python3 run_pageindex.py --md_path /path/to/your/document.md > python3 run_pageindex.py --flash --pdf_path /path/to/your/document.pdf > ``` > -> Add `--optimize` to refine the tree structure for more efficient retrieval (`--optimize merge` skips the LLM expansion pass). +> Add `--optimize` to refine the tree structure for more efficient retrieval (with an LLM expansion pass). ## 🚀 Agentic Vectorless RAG: An Example diff --git a/pageindex/flash/README.md b/pageindex/flash/README.md index 675c2ac..b8b4772 100644 --- a/pageindex/flash/README.md +++ b/pageindex/flash/README.md @@ -30,6 +30,7 @@ missing, non-PDF, encrypted, empty, or unreadable file. "node_id": str, # 4-digit, zero-padded "start_index": int, "end_index": int, + "key_items": [str], # titles of merged-away subsections; absent when none "nodes": [...], # absent on leaf nodes } ], diff --git a/pageindex/flash/api.py b/pageindex/flash/api.py index 8041698..1088c6f 100644 --- a/pageindex/flash/api.py +++ b/pageindex/flash/api.py @@ -68,9 +68,10 @@ def _validate_pdf(pdf): return pdf -def _thin(structure): - from ..utils import page_level_thinning, write_node_id - page_level_thinning(structure) +def _merge(structure): + from ..tree_optimize import merge_tree + from ..utils import write_node_id + merge_tree(structure) write_node_id(structure) @@ -82,9 +83,9 @@ async def _summarize(structure, page_list, model, concurrency=None): def _optimize(structure, page_texts, do_expand, model): """Merge/expand refinement between extraction and summaries. - Supersedes ``_thin``: merge collapses everything thinning would, but keeps - the dropped titles as ``key_items``. Summaries run after, so they describe - the final tree. Expand reads the same page text the summaries use. + Beyond the merge the default path runs anyway, this adds LLM expand and + reports before/after search-cost metrics. Summaries run after, so they + describe the final tree. Expand reads the same page text the summaries use. """ import asyncio from ..tree_optimize import optimize @@ -101,7 +102,7 @@ def _optimize(structure, page_texts, do_expand, model): def page_index_flash(pdf, summary=True, summary_model=None, optimize=False, optimize_expand=True, optimize_model=None, summary_concurrency=None) -> dict: - """Build a PageIndex tree structure from a PDF using layout statistics, without an LLM. Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: if True, refine the tree for search cost (merge + expand) before summaries. optimize_expand: if False, optimization only performs deterministic merge; summary generation is unchanged. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics. """ + """Build a PageIndex tree structure from a PDF using layout statistics, without an LLM. Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: if True, additionally expand oversized sections with an LLM and report search-cost metrics; a deterministic merge always runs, collapsing subtrees whose structure does not beat a linear scan and keeping the removed titles on the parent as ``key_items``. optimize_expand: if False, skip the LLM expansion and only report merge metrics. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics. """ result = extract_toc(_validate_pdf(pdf)) structure = result.get("structure", []) if optimize and structure: @@ -109,7 +110,7 @@ def page_index_flash(pdf, summary=True, summary_model=None, optimize_expand, optimize_model or summary_model) elif structure: - _thin(structure) + _merge(structure) if summary and structure: import asyncio from ..utils import ConfigLoader diff --git a/pageindex/page_index.py b/pageindex/page_index.py index 4731f1e..c0b3ea9 100644 --- a/pageindex/page_index.py +++ b/pageindex/page_index.py @@ -5,6 +5,7 @@ import math import random import re from .utils import * +from .tree_optimize import merge_tree import os from concurrent.futures import ThreadPoolExecutor, as_completed @@ -1246,7 +1247,7 @@ def page_index_main(doc, opt=None): async def page_index_builder(): structure = await tree_parser(page_list, opt, doc=doc, logger=logger) - page_level_thinning(structure) + merge_tree(structure) if opt.if_add_node_id == 'yes': write_node_id(structure) if opt.if_add_node_text == 'yes': @@ -1261,13 +1262,13 @@ def page_index_main(doc, opt=None): # Create a clean structure without unnecessary fields for description generation clean_structure = create_clean_structure_for_description(structure) doc_description = generate_doc_description(clean_structure, model=getattr(opt, 'summary_model', None) or opt.model) - structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'summary', 'text', 'nodes']) + structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes']) return { 'doc_name': get_pdf_name(doc), 'doc_description': doc_description, 'structure': structure, } - structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'summary', 'text', 'nodes']) + structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes']) return { 'doc_name': get_pdf_name(doc), 'structure': structure, diff --git a/pageindex/tree_optimize.py b/pageindex/tree_optimize.py index 9e8dd8b..64ebdd3 100644 --- a/pageindex/tree_optimize.py +++ b/pageindex/tree_optimize.py @@ -477,7 +477,8 @@ def merge(structure, routing, log, frozen, progress=False): checked = tree_cost_via_frontier(node, routing) span = S(node) if span <= cost: - removed = [c["node_id"] for c, _ in flatten(node["nodes"])] + # trees arrive here before ids are assigned in the main pipeline + removed = [c.get("node_id") for c, _ in flatten(node["nodes"])] # titles are routing information; keep them on the parent, in document # order, carrying forward anything an earlier merge already folded in titles = [] @@ -496,7 +497,7 @@ def merge(structure, routing, log, frozen, progress=False): node["key_items"] = titles frozen.add(node.get("node_id")) changed = True - note(progress, f" merge {node.get('node_id'):>8} " + note(progress, f" merge {node.get('node_id') or '-':>8} " f"S={span} <= tree_cost={cost} dropped {len(removed)} node(s)") for root in list(structure): @@ -504,6 +505,16 @@ def merge(structure, routing, log, frozen, progress=False): return changed +def merge_tree(structure): + """Deterministic merge over a structure list; the no-LLM default path. + + One bottom-up pass reaches the fixpoint: every decision is made after the + subtree below it is final. + """ + merge(structure, ROUTING_COST, [], set()) + return structure + + # -------------------------------------------------------------------------- # EXPAND # -------------------------------------------------------------------------- diff --git a/pageindex/utils.py b/pageindex/utils.py index 67f9a8d..f715e40 100644 --- a/pageindex/utils.py +++ b/pageindex/utils.py @@ -814,6 +814,7 @@ def format_structure(structure, order=None): def page_level_thinning(structure, thinning_threshold_node_num=20, min_pages_for_large_tree=3): + """Legacy; superseded by tree_optimize.merge_tree.""" def count_nodes(nodes): total = 0 for node in nodes: diff --git a/run_pageindex.py b/run_pageindex.py index d7cfc13..3a8c9f8 100644 --- a/run_pageindex.py +++ b/run_pageindex.py @@ -13,8 +13,9 @@ if __name__ == "__main__": parser.add_argument('--flash', action='store_true', help='Use PageIndex Flash (with --pdf_path)') parser.add_argument('--optimize', nargs='?', const='full', choices=['full', 'merge'], default=None, - help='Refine the tree for search cost: merge + LLM expand; ' - 'pass `merge` to skip the expansion pass (PDF only)') + help='Refine the tree with an LLM expansion pass and report search-cost ' + 'metrics; pass `merge` to skip expansion and only report the ' + 'deterministic merge every run performs (PDF only)') parser.add_argument('--model', type=str, default=None, help='Model to use (overrides config.yaml)') parser.add_argument('--summary-model', type=str, default=None,