Keep merge inside --optimize (#377)

This commit is contained in:
Ray
2026-08-02 21:52:32 +08:00
committed by GitHub
parent 2a29ac5aa0
commit 1b2fdadf51
3 changed files with 6 additions and 15 deletions
+3 -2
View File
@@ -30,14 +30,15 @@ missing, non-PDF, encrypted, empty, or unreadable file.
"node_id": str, # 4-digit, zero-padded
"start_index": int,
"end_index": int,
"key_items": [str], # titles of merged-away subsections; absent when none
"key_items": [str], # with --optimize: titles of merged-away subsections
"nodes": [...], # absent on leaf nodes
}
],
}
```
Page indexes are 1-based. `nodes` nests the same shape recursively.
Page indexes are 1-based. `nodes` nests the same shape recursively. Without
`--optimize` the extracted tree is returned as-is.
## Benchmark
+1 -10
View File
@@ -68,13 +68,6 @@ def _validate_pdf(pdf):
return pdf
def _merge(structure):
from ..tree_optimize import merge_tree
from ..utils import write_node_id
merge_tree(structure)
write_node_id(structure)
async def _summarize(structure, page_list, model, concurrency=None):
from ..utils import summarize_tree
await summarize_tree(structure, page_list, model=model, concurrency=concurrency)
@@ -105,15 +98,13 @@ def _optimize(structure, page_texts, do_expand, model):
def page_index_flash(pdf, summary=True, summary_model=None,
optimize=False, optimize_expand=True,
optimize_model=None, summary_concurrency=None) -> dict:
"""Build a PageIndex tree structure from a PDF using layout statistics, without an LLM. Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: if True, additionally expand oversized sections with an LLM and report search-cost metrics; a deterministic merge always runs, collapsing subtrees whose structure does not beat a linear scan and keeping the removed titles on the parent as ``key_items``. optimize_expand: if False, skip the LLM expansion and only report merge metrics. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics. """
"""Build a PageIndex tree structure from a PDF using layout statistics, without an LLM. Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. optimize: if True, refine the tree for search cost before summaries: a deterministic merge collapses subtrees whose structure does not beat a linear scan, keeping the removed titles on the parent as ``key_items``, then an LLM pass expands oversized sections. Without it the extracted tree is returned unchanged. optimize_expand: if False, run the merge but skip the LLM expansion. optimize_model: the LLM model for expand (defaults to the summary model). summary_concurrency: maximum simultaneous summary model calls; None uses the library default. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). With ``optimize`` an ``optimize`` key reports merge/expand counts and before/after search-cost metrics. """
result = extract_toc(_validate_pdf(pdf))
structure = result.get("structure", [])
if optimize and structure:
result["optimize"] = _optimize(structure, result.get("page_texts") or [],
optimize_expand,
optimize_model or summary_model)
elif structure:
_merge(structure)
if summary and structure:
import asyncio
from ..utils import ConfigLoader
+2 -3
View File
@@ -13,9 +13,8 @@ if __name__ == "__main__":
parser.add_argument('--flash', action='store_true', help='Use PageIndex Flash (with --pdf_path)')
parser.add_argument('--optimize', nargs='?', const='full', choices=['full', 'merge'],
default=None,
help='Refine the tree with an LLM expansion pass and report search-cost '
'metrics; pass `merge` to skip expansion and only report the '
'deterministic merge every run performs (PDF only)')
help='Refine the tree for search cost: a deterministic merge, then an '
'LLM expansion pass; pass `merge` to run the merge alone (PDF only)')
parser.add_argument('--model', type=str, default=None, help='Model to use (overrides config.yaml)')
parser.add_argument('--summary-model', type=str, default=None,