mirror of
https://github.com/VectifyAI/PageIndex.git
synced 2026-10-02 07:44:37 +08:00
Replace thinning with merge
This commit is contained in:
@@ -205,7 +205,7 @@ python3 run_pageindex.py --md_path /path/to/your/document.md
|
||||
> python3 run_pageindex.py --flash --pdf_path /path/to/your/document.pdf
|
||||
> ```
|
||||
>
|
||||
> Add `--optimize` to refine the tree structure for more efficient retrieval (`--optimize merge` skips the LLM expansion pass).
|
||||
> Add `--optimize` to refine the tree structure for more efficient retrieval (with an LLM expansion pass).
|
||||
|
||||
## 🚀 Agentic Vectorless RAG: An Example
|
||||
|
||||
|
||||
@@ -30,6 +30,7 @@ missing, non-PDF, encrypted, empty, or unreadable file.
|
||||
"node_id": str, # 4-digit, zero-padded
|
||||
"start_index": int,
|
||||
"end_index": int,
|
||||
"key_items": [str], # titles of merged-away subsections; absent when none
|
||||
"nodes": [...], # absent on leaf nodes
|
||||
}
|
||||
],
|
||||
|
||||
@@ -68,9 +68,10 @@ def _validate_pdf(pdf):
|
||||
return pdf
|
||||
|
||||
|
||||
def _thin(structure):
|
||||
from ..utils import page_level_thinning, write_node_id
|
||||
page_level_thinning(structure)
|
||||
def _merge(structure):
|
||||
from ..tree_optimize import merge_tree
|
||||
from ..utils import write_node_id
|
||||
merge_tree(structure)
|
||||
write_node_id(structure)
|
||||
|
||||
|
||||
@@ -82,9 +83,9 @@ async def _summarize(structure, page_list, model, concurrency=None):
|
||||
def _optimize(structure, page_texts, do_expand, model):
|
||||
"""Merge/expand refinement between extraction and summaries.
|
||||
|
||||
Supersedes ``_thin``: merge collapses everything thinning would, but keeps
|
||||
the dropped titles as ``key_items``. Summaries run after, so they describe
|
||||
the final tree. Expand reads the same page text the summaries use.
|
||||
Beyond the merge the default path runs anyway, this adds LLM expand and
|
||||
reports before/after search-cost metrics. Summaries run after, so they
|
||||
describe the final tree. Expand reads the same page text the summaries use.
|
||||
"""
|
||||
import asyncio
|
||||
from ..tree_optimize import optimize
|
||||
@@ -101,7 +102,7 @@ def _optimize(structure, page_texts, do_expand, model):
|
||||
def page_index_flash(pdf, summary=True, summary_model=None,
|
||||
optimize=False, optimize_expand=True,
|
||||
optimize_model=None, summary_concurrency=None) -> dict:
|
||||
"""Build a PageIndex tree structure from a PDF using layout statistics, without an LLM. Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: if True, refine the tree for search cost (merge + expand) before summaries. optimize_expand: if False, optimization only performs deterministic merge; summary generation is unchanged. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics. """
|
||||
"""Build a PageIndex tree structure from a PDF using layout statistics, without an LLM. Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: if True, additionally expand oversized sections with an LLM and report search-cost metrics; a deterministic merge always runs, collapsing subtrees whose structure does not beat a linear scan and keeping the removed titles on the parent as ``key_items``. optimize_expand: if False, skip the LLM expansion and only report merge metrics. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics. """
|
||||
result = extract_toc(_validate_pdf(pdf))
|
||||
structure = result.get("structure", [])
|
||||
if optimize and structure:
|
||||
@@ -109,7 +110,7 @@ def page_index_flash(pdf, summary=True, summary_model=None,
|
||||
optimize_expand,
|
||||
optimize_model or summary_model)
|
||||
elif structure:
|
||||
_thin(structure)
|
||||
_merge(structure)
|
||||
if summary and structure:
|
||||
import asyncio
|
||||
from ..utils import ConfigLoader
|
||||
|
||||
@@ -5,6 +5,7 @@ import math
|
||||
import random
|
||||
import re
|
||||
from .utils import *
|
||||
from .tree_optimize import merge_tree
|
||||
import os
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
|
||||
@@ -1246,7 +1247,7 @@ def page_index_main(doc, opt=None):
|
||||
|
||||
async def page_index_builder():
|
||||
structure = await tree_parser(page_list, opt, doc=doc, logger=logger)
|
||||
page_level_thinning(structure)
|
||||
merge_tree(structure)
|
||||
if opt.if_add_node_id == 'yes':
|
||||
write_node_id(structure)
|
||||
if opt.if_add_node_text == 'yes':
|
||||
@@ -1261,13 +1262,13 @@ def page_index_main(doc, opt=None):
|
||||
# Create a clean structure without unnecessary fields for description generation
|
||||
clean_structure = create_clean_structure_for_description(structure)
|
||||
doc_description = generate_doc_description(clean_structure, model=getattr(opt, 'summary_model', None) or opt.model)
|
||||
structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'summary', 'text', 'nodes'])
|
||||
structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes'])
|
||||
return {
|
||||
'doc_name': get_pdf_name(doc),
|
||||
'doc_description': doc_description,
|
||||
'structure': structure,
|
||||
}
|
||||
structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'summary', 'text', 'nodes'])
|
||||
structure = format_structure(structure, order=['title', 'node_id', 'start_index', 'end_index', 'key_items', 'summary', 'text', 'nodes'])
|
||||
return {
|
||||
'doc_name': get_pdf_name(doc),
|
||||
'structure': structure,
|
||||
|
||||
@@ -477,7 +477,8 @@ def merge(structure, routing, log, frozen, progress=False):
|
||||
checked = tree_cost_via_frontier(node, routing)
|
||||
span = S(node)
|
||||
if span <= cost:
|
||||
removed = [c["node_id"] for c, _ in flatten(node["nodes"])]
|
||||
# trees arrive here before ids are assigned in the main pipeline
|
||||
removed = [c.get("node_id") for c, _ in flatten(node["nodes"])]
|
||||
# titles are routing information; keep them on the parent, in document
|
||||
# order, carrying forward anything an earlier merge already folded in
|
||||
titles = []
|
||||
@@ -496,7 +497,7 @@ def merge(structure, routing, log, frozen, progress=False):
|
||||
node["key_items"] = titles
|
||||
frozen.add(node.get("node_id"))
|
||||
changed = True
|
||||
note(progress, f" merge {node.get('node_id'):>8} "
|
||||
note(progress, f" merge {node.get('node_id') or '-':>8} "
|
||||
f"S={span} <= tree_cost={cost} dropped {len(removed)} node(s)")
|
||||
|
||||
for root in list(structure):
|
||||
@@ -504,6 +505,16 @@ def merge(structure, routing, log, frozen, progress=False):
|
||||
return changed
|
||||
|
||||
|
||||
def merge_tree(structure):
|
||||
"""Deterministic merge over a structure list; the no-LLM default path.
|
||||
|
||||
One bottom-up pass reaches the fixpoint: every decision is made after the
|
||||
subtree below it is final.
|
||||
"""
|
||||
merge(structure, ROUTING_COST, [], set())
|
||||
return structure
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# EXPAND
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
@@ -814,6 +814,7 @@ def format_structure(structure, order=None):
|
||||
|
||||
|
||||
def page_level_thinning(structure, thinning_threshold_node_num=20, min_pages_for_large_tree=3):
|
||||
"""Legacy; superseded by tree_optimize.merge_tree."""
|
||||
def count_nodes(nodes):
|
||||
total = 0
|
||||
for node in nodes:
|
||||
|
||||
+3
-2
@@ -13,8 +13,9 @@ if __name__ == "__main__":
|
||||
parser.add_argument('--flash', action='store_true', help='Use PageIndex Flash (with --pdf_path)')
|
||||
parser.add_argument('--optimize', nargs='?', const='full', choices=['full', 'merge'],
|
||||
default=None,
|
||||
help='Refine the tree for search cost: merge + LLM expand; '
|
||||
'pass `merge` to skip the expansion pass (PDF only)')
|
||||
help='Refine the tree with an LLM expansion pass and report search-cost '
|
||||
'metrics; pass `merge` to skip expansion and only report the '
|
||||
'deterministic merge every run performs (PDF only)')
|
||||
|
||||
parser.add_argument('--model', type=str, default=None, help='Model to use (overrides config.yaml)')
|
||||
parser.add_argument('--summary-model', type=str, default=None,
|
||||
|
||||
Reference in New Issue
Block a user