Replace thinning with merge

This commit is contained in:
Ray
2026-08-02 17:56:54 +08:00
parent 3f33a53b50
commit 4a116daa2f
7 changed files with 32 additions and 16 deletions
+1 -1
View File
@@ -205,7 +205,7 @@ python3 run_pageindex.py --md_path /path/to/your/document.md
> python3 run_pageindex.py --flash --pdf_path /path/to/your/document.pdf
> ```
>
> Add `--optimize` to refine the tree structure for more efficient retrieval (`--optimize merge` skips the LLM expansion pass).
> Add `--optimize` to refine the tree structure for more efficient retrieval (with an LLM expansion pass).
## 🚀 Agentic Vectorless RAG: An Example
+1
View File
@@ -30,6 +30,7 @@ missing, non-PDF, encrypted, empty, or unreadable file.
"node_id": str, # 4-digit, zero-padded
"start_index": int,
"end_index": int,
"key_items": [str], # titles of merged-away subsections; absent when none
"nodes": [...], # absent on leaf nodes
}
],
+9 -8
View File
@@ -68,9 +68,10 @@ def _validate_pdf(pdf):
return pdf
def _thin(structure):
from ..utils import page_level_thinning, write_node_id
page_level_thinning(structure)
def _merge(structure):
from ..tree_optimize import merge_tree
from ..utils import write_node_id
merge_tree(structure)
write_node_id(structure)
@@ -82,9 +83,9 @@ async def _summarize(structure, page_list, model, concurrency=None):
def _optimize(structure, page_texts, do_expand, model):
"""Merge/expand refinement between extraction and summaries.
Supersedes ``_thin``: merge collapses everything thinning would, but keeps
the dropped titles as ``key_items``. Summaries run after, so they describe
the final tree. Expand reads the same page text the summaries use.
Beyond the merge the default path runs anyway, this adds LLM expand and
reports before/after search-cost metrics. Summaries run after, so they
describe the final tree. Expand reads the same page text the summaries use.
"""
import asyncio
from ..tree_optimize import optimize
@@ -101,7 +102,7 @@ def _optimize(structure, page_texts, do_expand, model):
def page_index_flash(pdf, summary=True, summary_model=None,
optimize=False, optimize_expand=True,
optimize_model=None, summary_concurrency=None) -> dict:
"""Build a PageIndex tree structure from a PDF using layout statistics, without an LLM. Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: if True, refine the tree for search cost (merge + expand) before summaries. optimize_expand: if False, optimization only performs deterministic merge; summary generation is unchanged. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics. """
"""Build a PageIndex tree structure from a PDF using layout statistics, without an LLM. Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: if True, additionally expand oversized sections with an LLM and report search-cost metrics; a deterministic merge always runs, collapsing subtrees whose structure does not beat a linear scan and keeping the removed titles on the parent as ``key_items``. optimize_expand: if False, skip the LLM expansion and only report merge metrics. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics. """
result = extract_toc(_validate_pdf(pdf))
structure = result.get("structure", [])
if optimize and structure:
@@ -109,7 +110,7 @@ def page_index_flash(pdf, summary=True, summary_model=None,
optimize_expand,
optimize_model or summary_model)
elif structure:
_thin(structure)
_merge(structure)
if summary and structure:
import asyncio
from ..utils import ConfigLoader
+4 -3
View File
@@ -5,6 +5,7 @@ import math
import random
import re
from .utils import *
from .tree_optimize import merge_tree
import os
from concurrent.futures import ThreadPoolExecutor, as_completed
@@ -1246,7 +1247,7 @@ def page_index_main(doc, opt=None):
async def page_index_builder():
structure = await tree_parser(page_list, opt, doc=doc, logger=logger)
page_level_thinning(structure)
merge_tree(structure)
if opt.if_add_node_id == 'yes':
write_node_id(structure)
if opt.if_add_node_text == 'yes':
@@ -1261,13 +1262,13 @@ def page_index_main(doc, opt=None):
# Create a clean structure without unnecessary fields for description generation
clean_structure = create_clean_structure_for_description(structure)
doc_description = generate_doc_description(clean_structure, model=getattr(opt, 'summary_model', None) or opt.model)
structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'summary', 'text', 'nodes'])
structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes'])
return {
'doc_name': get_pdf_name(doc),
'doc_description': doc_description,
'structure': structure,
}
structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'summary', 'text', 'nodes'])
structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes'])
return {
'doc_name': get_pdf_name(doc),
'structure': structure,
+13 -2
View File
@@ -477,7 +477,8 @@ def merge(structure, routing, log, frozen, progress=False):
checked = tree_cost_via_frontier(node, routing)
span = S(node)
if span <= cost:
removed = [c["node_id"] for c, _ in flatten(node["nodes"])]
# trees arrive here before ids are assigned in the main pipeline
removed = [c.get("node_id") for c, _ in flatten(node["nodes"])]
# titles are routing information; keep them on the parent, in document
# order, carrying forward anything an earlier merge already folded in
titles = []
@@ -496,7 +497,7 @@ def merge(structure, routing, log, frozen, progress=False):
node["key_items"] = titles
frozen.add(node.get("node_id"))
changed = True
note(progress, f" merge {node.get('node_id'):>8} "
note(progress, f" merge {node.get('node_id') or '-':>8} "
f"S={span} <= tree_cost={cost} dropped {len(removed)} node(s)")
for root in list(structure):
@@ -504,6 +505,16 @@ def merge(structure, routing, log, frozen, progress=False):
return changed
def merge_tree(structure):
"""Deterministic merge over a structure list; the no-LLM default path.
One bottom-up pass reaches the fixpoint: every decision is made after the
subtree below it is final.
"""
merge(structure, ROUTING_COST, [], set())
return structure
# --------------------------------------------------------------------------
# EXPAND
# --------------------------------------------------------------------------
+1
View File
@@ -814,6 +814,7 @@ def format_structure(structure, order=None):
def page_level_thinning(structure, thinning_threshold_node_num=20, min_pages_for_large_tree=3):
"""Legacy; superseded by tree_optimize.merge_tree."""
def count_nodes(nodes):
total = 0
for node in nodes:
+3 -2
View File
@@ -13,8 +13,9 @@ if __name__ == "__main__":
parser.add_argument('--flash', action='store_true', help='Use PageIndex Flash (with --pdf_path)')
parser.add_argument('--optimize', nargs='?', const='full', choices=['full', 'merge'],
default=None,
help='Refine the tree for search cost: merge + LLM expand; '
'pass `merge` to skip the expansion pass (PDF only)')
help='Refine the tree with an LLM expansion pass and report search-cost '
'metrics; pass `merge` to skip expansion and only report the '
'deterministic merge every run performs (PDF only)')
parser.add_argument('--model', type=str, default=None, help='Model to use (overrides config.yaml)')
parser.add_argument('--summary-model', type=str, default=None,