mirror of
https://github.com/VectifyAI/PageIndex.git
synced 2026-10-02 07:44:37 +08:00
Add PageIndex Flash
This commit is contained in:
@@ -0,0 +1,50 @@
|
||||
# PageIndex Flash
|
||||
|
||||
Builds a PageIndex tree structure from a PDF using layout statistics alone.
|
||||
No LLM, no API key, no OCR, no network. Runs in seconds, fully offline.
|
||||
|
||||
## Usage
|
||||
|
||||
```python
|
||||
from pageindex.flash import page_index_flash
|
||||
|
||||
tree = page_index_flash("paper.pdf")
|
||||
```
|
||||
|
||||
```bash
|
||||
python3 run_pageindex.py --pdf_path document.pdf --flash
|
||||
```
|
||||
|
||||
Accepts a path (`str` or `pathlib.Path`) or an `io.BytesIO` stream. Raises on a
|
||||
missing, non-PDF, encrypted, empty, or unreadable file.
|
||||
|
||||
## Output
|
||||
|
||||
```python
|
||||
{
|
||||
"doc_name": str,
|
||||
"doc_title": str,
|
||||
"structure": [
|
||||
{
|
||||
"title": str,
|
||||
"node_id": str, # 4-digit, zero-padded
|
||||
"start_index": int,
|
||||
"end_index": int,
|
||||
"nodes": [...], # absent on leaf nodes
|
||||
}
|
||||
],
|
||||
}
|
||||
```
|
||||
|
||||
Page indexes are 1-based. `nodes` nests the same shape recursively.
|
||||
|
||||
## Limits
|
||||
|
||||
- Scanned PDFs without embedded text are not supported.
|
||||
- Encrypted PDFs need preprocessing first.
|
||||
- Headings drawn as vector paths, or very decorative layouts, can be missed.
|
||||
- Titles are taken from the document text as-is.
|
||||
|
||||
## Dependencies
|
||||
|
||||
`pypdfium2`, `PyPDF2`, `regex`, `sortedcontainers`.
|
||||
@@ -0,0 +1,5 @@
|
||||
"""PageIndex Flash: LLM-free tree structure extraction from PDF layout statistics."""
|
||||
|
||||
from .api import page_index_flash
|
||||
|
||||
__all__ = ["page_index_flash"]
|
||||
@@ -0,0 +1,104 @@
|
||||
"""Public API for PageIndex Flash. The only supported entry point is :func:`page_index_flash`. Everything else in this package is internal pipeline machinery."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from typing import BinaryIO
|
||||
|
||||
import pypdfium2 as pdfium
|
||||
|
||||
from .main import extract_toc
|
||||
|
||||
|
||||
def _is_pdfium_password_error(exc: Exception) -> bool:
|
||||
msg = str(exc).lower()
|
||||
return "password" in msg or "security" in msg or "encrypted" in msg
|
||||
|
||||
|
||||
def _validate_path(path: Path) -> str:
|
||||
if not path.exists():
|
||||
raise FileNotFoundError(f"PDF file not found: {path}")
|
||||
if not path.is_file():
|
||||
raise ValueError(f"PDF path is not a file: {path}")
|
||||
if path.suffix.lower() != ".pdf":
|
||||
raise ValueError(f"PDF file must have a .pdf extension: {path}")
|
||||
with path.open("rb") as score_value:
|
||||
if score_value.read(5) != b"%PDF-":
|
||||
raise ValueError(f"File does not look like a PDF: {path}")
|
||||
return str(path)
|
||||
|
||||
|
||||
def _validate_stream(stream: BinaryIO) -> BinaryIO:
|
||||
try:
|
||||
pos = stream.tell()
|
||||
head = stream.read(5)
|
||||
stream.seek(pos)
|
||||
except Exception as exc: # noqa: BLE001 - normalize stream capability errors
|
||||
raise TypeError("PDF stream must be seekable and readable") from exc
|
||||
if head != b"%PDF-":
|
||||
raise ValueError("Input stream does not look like a PDF")
|
||||
return stream
|
||||
|
||||
|
||||
def _validate_pdf(pdf):
|
||||
if isinstance(pdf, (str, Path)):
|
||||
handle = _validate_path(Path(pdf))
|
||||
restore = None
|
||||
elif isinstance(pdf, BytesIO):
|
||||
handle = _validate_stream(pdf)
|
||||
restore = pdf.tell()
|
||||
else:
|
||||
raise TypeError("page_index_flash(pdf) expects a PDF path or io.BytesIO stream")
|
||||
|
||||
doc = None
|
||||
try:
|
||||
doc = pdfium.PdfDocument(handle)
|
||||
if len(doc) == 0:
|
||||
raise ValueError("PDF contains no pages")
|
||||
except pdfium.PdfiumError as exc:
|
||||
if _is_pdfium_password_error(exc):
|
||||
raise ValueError("PDF is encrypted or password-protected") from exc
|
||||
raise ValueError(f"Could not open PDF: {exc}") from exc
|
||||
finally:
|
||||
if doc is not None:
|
||||
doc.close()
|
||||
if restore is not None:
|
||||
pdf.seek(restore)
|
||||
return pdf
|
||||
|
||||
|
||||
def _thin(structure):
|
||||
from ..utils import page_level_thinning, write_node_id
|
||||
page_level_thinning(structure)
|
||||
write_node_id(structure)
|
||||
|
||||
|
||||
async def _summarize(structure, page_list, model):
|
||||
from ..utils import add_node_text, generate_summaries_for_structure, remove_structure_text
|
||||
add_node_text(structure, page_list)
|
||||
await generate_summaries_for_structure(structure, model=model)
|
||||
remove_structure_text(structure)
|
||||
|
||||
|
||||
def page_index_flash(pdf, summary=True, summary_model=None) -> dict:
|
||||
"""Build a PageIndex tree structure from a PDF using layout statistics, without an LLM. Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). """
|
||||
result = extract_toc(_validate_pdf(pdf))
|
||||
structure = result.get("structure", [])
|
||||
if structure:
|
||||
_thin(structure)
|
||||
if summary and structure:
|
||||
import asyncio
|
||||
from ..utils import ConfigLoader
|
||||
if summary_model is None:
|
||||
cfg = ConfigLoader().load()
|
||||
summary_model = getattr(cfg, 'summary_model', None) or cfg.model
|
||||
page_texts = result.pop("page_texts", [])
|
||||
page_list = [(text, 0) for text in page_texts]
|
||||
asyncio.run(_summarize(structure, page_list, summary_model))
|
||||
else:
|
||||
result.pop("page_texts", None)
|
||||
return result
|
||||
|
||||
|
||||
__all__ = ["page_index_flash"]
|
||||
@@ -0,0 +1,56 @@
|
||||
"""Block clustering. This module walks page lines in reading order, extends nearby compatible
|
||||
blocks, starts a new block when no neighbor fits, and then splits simple
|
||||
"heading + body" two-line blocks where the first line is a standalone section
|
||||
heading. The clustering pass must return blocks, not raw lines. Reading-order assignment
|
||||
then uses each block's first line to find the column index; doing that on raw
|
||||
lines would read an unrelated first-span flag.
|
||||
"""
|
||||
|
||||
from typing import Optional
|
||||
|
||||
from sortedcontainers import SortedKeyList
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from ..model import (
|
||||
style_key,
|
||||
magnitude_ratio,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
x_centers_close,
|
||||
Rect,
|
||||
last_span,
|
||||
avg_char_width,
|
||||
EMPTY_RECT,
|
||||
left_edge_key,
|
||||
reading_order_key,
|
||||
numbering_kind,
|
||||
Line,
|
||||
case_signal,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
is_upper_dominant,
|
||||
Block,
|
||||
_max_nan_propagating,
|
||||
)
|
||||
from ..stats import DocStats, PageStats
|
||||
from ..tokens import set_case_fold, TrieConfig, build_trie, tokenize_block
|
||||
|
||||
from .join_rules import (
|
||||
_DICT_PATH,
|
||||
_DICTS,
|
||||
SECTION_HEADING_TRIE,
|
||||
BlockClusterContext,
|
||||
should_join_line_to_block,
|
||||
)
|
||||
from .build import (
|
||||
split_heading_body_blocks,
|
||||
_set_add,
|
||||
cluster_lines_into_blocks,
|
||||
)
|
||||
|
||||
__all__ = ["BlockClusterContext", "should_join_line_to_block", "cluster_lines_into_blocks", "split_heading_body_blocks", "SECTION_HEADING_TRIE"]
|
||||
@@ -0,0 +1,173 @@
|
||||
"""Clusters lines into blocks and splits heading-body blocks."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from sortedcontainers import SortedKeyList
|
||||
|
||||
from ..model import (
|
||||
style_key,
|
||||
magnitude_ratio,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
x_centers_close,
|
||||
Rect,
|
||||
last_span,
|
||||
avg_char_width,
|
||||
EMPTY_RECT,
|
||||
left_edge_key,
|
||||
reading_order_key,
|
||||
numbering_kind,
|
||||
Line,
|
||||
case_signal,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
is_upper_dominant,
|
||||
Block,
|
||||
_max_nan_propagating,
|
||||
)
|
||||
from ..tokens import set_case_fold, TrieConfig, build_trie, tokenize_block
|
||||
|
||||
from .join_rules import (
|
||||
SECTION_HEADING_TRIE,
|
||||
BlockClusterContext,
|
||||
should_join_line_to_block,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Two-line block split post-process #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def split_heading_body_blocks(input_blocks: list[Block]) -> list[Block]:
|
||||
"""Split blocks whose first line is a section heading followed by body text."""
|
||||
from ..labels import trie_matches_all, advance_past_line
|
||||
split_output_blocks: list[Block] = []
|
||||
for input_block in input_blocks:
|
||||
first_line = input_block.line() # first line
|
||||
# Skip blocks that obviously aren't "heading + body":
|
||||
# - 1-line blocks
|
||||
# - small/short blocks
|
||||
# - first-span style == last-span style AND wide first line
|
||||
if (
|
||||
input_block.line_count() <= 1
|
||||
or (input_block.bbox_height() >= 0.6 * input_block.bbox_width() and input_block.char_count() < 20 * input_block.line_count())
|
||||
or (style_key(first_span_of(input_block)) == style_key(last_span(last_line_of(input_block))) and first_line.bbox_width() > 0.5 * input_block.bbox_width())
|
||||
):
|
||||
split_output_blocks.append(input_block)
|
||||
continue
|
||||
block_tokens = tokenize_block(input_block)
|
||||
first_line_tokens = block_tokens.slice(0, advance_past_line(block_tokens, first_line, 0))
|
||||
split_token = block_tokens.token_at(first_line_tokens.length)
|
||||
if split_token is None or split_token.primary_slot == 3:
|
||||
split_output_blocks.append(input_block)
|
||||
continue
|
||||
if not trie_matches_all(SECTION_HEADING_TRIE, first_line_tokens):
|
||||
split_output_blocks.append(input_block)
|
||||
continue
|
||||
# Split: first block holds the heading line; second holds the rest.
|
||||
split_heading_block = Block()
|
||||
split_heading_block.add_line(first_line)
|
||||
split_body_block = Block()
|
||||
for line_idx in range(1, input_block.line_count()):
|
||||
split_body_block.add_line(input_block.primary_slot[line_idx])
|
||||
split_output_blocks.append(split_heading_block)
|
||||
split_output_blocks.append(split_body_block)
|
||||
return split_output_blocks
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Block-clustering driver #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _set_add(tree: SortedKeyList, block: Block) -> None:
|
||||
"""Sorted-set insertion semantics: when another block has the same left-edge ordering key, the new block is ignored instead of kept as a multiset duplicate."""
|
||||
idx = tree.bisect_left(block)
|
||||
if idx < len(tree) and left_edge_key(tree[idx]) == left_edge_key(block): # type: ignore[arg-type]
|
||||
return # key collision -> sorted set.add drops the element
|
||||
tree.add(block)
|
||||
|
||||
|
||||
def cluster_lines_into_blocks(ctx: BlockClusterContext) -> list[Block]:
|
||||
"""Walk lines, extend existing blocks when compatible, otherwise open a block. Returns blocks sorted bottom, then top, then left, then right before reading-order assignment."""
|
||||
# Tree of *blocks* sorted by (left, right, top desc, bottom desc)
|
||||
tree: SortedKeyList = SortedKeyList(key=left_edge_key)
|
||||
clustered_blocks: list[Block] = []
|
||||
|
||||
lines = ctx.secondary_slot
|
||||
line_count = len(lines)
|
||||
for line_index in range(line_count):
|
||||
candidate_line = lines[line_index]
|
||||
next_line = lines[line_index + 1] if line_index + 1 < line_count else None
|
||||
|
||||
# The new line wrapped as a block (used as the tree key for lookups).
|
||||
seed_block = Block().add_line(candidate_line)
|
||||
|
||||
# Collect candidate blocks whose horizontal interval overlaps e_line.
|
||||
# * predecessors: walk backwards from g_seed_block's left, gather
|
||||
# blocks whose right edge >= e_line.left.
|
||||
# * successors: walk forwards, gather blocks whose left edge <= e_line.right.
|
||||
candidate_blocks: list[Block] = []
|
||||
# Predecessors by decreasing block-order key.
|
||||
# Predecessor walk starts at the largest key <= the seed key.
|
||||
idx_pred = tree.bisect_right(seed_block)
|
||||
block = idx_pred - 1
|
||||
while block >= 0:
|
||||
existing_block: Block = tree[block] # type: ignore[assignment]
|
||||
if existing_block.right_edge() < candidate_line.left_edge():
|
||||
break
|
||||
candidate_blocks.append(existing_block)
|
||||
block -= 1
|
||||
# Successors by increasing block-order key.
|
||||
# Successor walk starts at the smallest key >= the seed key. An exact
|
||||
# key-equal node is intentionally visited by both walks.
|
||||
idx_succ = tree.bisect_left(seed_block)
|
||||
block = idx_succ
|
||||
while block < len(tree):
|
||||
existing_block = tree[block] # type: ignore[assignment]
|
||||
if existing_block.left_edge() > candidate_line.right_edge():
|
||||
break
|
||||
candidate_blocks.append(existing_block)
|
||||
block += 1
|
||||
|
||||
# Sort candidates by bottom, then top, left, and right.
|
||||
candidate_blocks.sort(key=lambda block: (block.bottom_edge(), block.top_edge(), block.left_edge(), block.right_edge()))
|
||||
|
||||
did_join = False
|
||||
# Capture the first candidate (closest) before mutating the list
|
||||
first_candidate = candidate_blocks[0] if candidate_blocks else None
|
||||
for existing_block in candidate_blocks:
|
||||
if not did_join and first_candidate is not None and should_join_line_to_block(
|
||||
ctx, existing_block, candidate_line, next_line, first_candidate
|
||||
):
|
||||
# Join: remove m from tree, extend with e_line, re-add.
|
||||
try:
|
||||
tree.remove(existing_block)
|
||||
except ValueError:
|
||||
pass
|
||||
existing_block.add_line(candidate_line)
|
||||
_set_add(tree, existing_block)
|
||||
did_join = True
|
||||
else:
|
||||
# Doesn't take this line -- block is "closed", emit it.
|
||||
clustered_blocks.append(existing_block)
|
||||
try:
|
||||
tree.remove(existing_block)
|
||||
except ValueError:
|
||||
pass
|
||||
if not did_join:
|
||||
_set_add(tree, seed_block)
|
||||
|
||||
# Drain remaining open blocks
|
||||
for block in tree:
|
||||
clustered_blocks.append(block)
|
||||
|
||||
# Post-process to split 2-line "heading+body" blocks when the first line
|
||||
# matches section, abstract, or references keywords.
|
||||
clustered_blocks = split_heading_body_blocks(clustered_blocks)
|
||||
clustered_blocks.sort(key=reading_order_key)
|
||||
return clustered_blocks
|
||||
@@ -0,0 +1,326 @@
|
||||
"""Line-to-block joining rules and the section-heading trie."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Optional
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from ..model import (
|
||||
style_key,
|
||||
magnitude_ratio,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
x_centers_close,
|
||||
Rect,
|
||||
last_span,
|
||||
avg_char_width,
|
||||
EMPTY_RECT,
|
||||
left_edge_key,
|
||||
reading_order_key,
|
||||
numbering_kind,
|
||||
Line,
|
||||
case_signal,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
is_upper_dominant,
|
||||
Block,
|
||||
_max_nan_propagating,
|
||||
)
|
||||
from ..stats import DocStats, PageStats
|
||||
from ..tokens import set_case_fold, TrieConfig, build_trie, tokenize_block
|
||||
|
||||
|
||||
# Combined heading trie used to detect "first line is a section header" patterns
|
||||
# when splitting two-line blocks.
|
||||
_DICT_PATH = Path(__file__).parent.parent / "data" / "dictionaries.json"
|
||||
_DICTS = json.loads(_DICT_PATH.read_text(encoding="utf-8"))
|
||||
SECTION_HEADING_TRIE = build_trie(
|
||||
list(_DICTS.get("section_keywords", []))
|
||||
+ list(_DICTS.get("abstract_keywords", []))
|
||||
+ list(_DICTS.get("references", [])),
|
||||
set_case_fold(TrieConfig(), True),
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Block-clustering context bundle #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class BlockClusterContext:
|
||||
"""Block-clustering context. Fields: j document statistics o page bbox g page statistics h lines to cluster v column rectangles """
|
||||
|
||||
__slots__ = ("tertiary_slot", "auxiliary_slot", "primary_slot", "secondary_slot", "state_slot")
|
||||
|
||||
def __init__(self, doc_stats: DocStats, page_bbox: Rect, page_stats: PageStats, lines: list, columns: list):
|
||||
self.tertiary_slot = doc_stats
|
||||
self.auxiliary_slot = page_bbox
|
||||
self.primary_slot = page_stats
|
||||
self.secondary_slot = lines
|
||||
self.state_slot = columns
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Should a line join an existing block? #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def should_join_line_to_block(
|
||||
block_cluster_ctx: BlockClusterContext,
|
||||
other_block: Block,
|
||||
|
||||
candidate_line: Line,
|
||||
|
||||
previous_line: Optional[Line],
|
||||
|
||||
first_candidate_block: Block,
|
||||
|
||||
) -> bool:
|
||||
"""Return True iff the candidate line should be appended to the current block."""
|
||||
# -- Step 1: reject incompatible skew ----------
|
||||
if abs(other_block.skew_frac() - candidate_line.skew_frac()) > 1:
|
||||
return False
|
||||
|
||||
# -- Step 2: size + alignment gates ----------------------------------
|
||||
font_size_delta = candidate_line.avg_font_size() - other_block.avg_font_size()
|
||||
|
||||
left_edges_aligned = left_aligned(other_block, candidate_line, 1)
|
||||
|
||||
both_edges_aligned = left_edges_aligned or (other_block.line_count() == 1 and left_aligned(other_block, candidate_line, 8 * avg_char_width(other_block.line())))
|
||||
|
||||
right_edges_aligned = right_aligned(other_block, candidate_line, 2)
|
||||
|
||||
both_edges_aligned = both_edges_aligned and right_edges_aligned
|
||||
# m = min size-excess over page body; k = min size-excess over doc body
|
||||
page_body_font_delta = min(candidate_line.avg_font_size() - block_cluster_ctx.primary_slot.primary_slot, other_block.avg_font_size() - block_cluster_ctx.primary_slot.primary_slot)
|
||||
|
||||
doc_body_font_delta = min(candidate_line.avg_font_size() - block_cluster_ctx.tertiary_slot.primary_slot, other_block.avg_font_size() - block_cluster_ctx.tertiary_slot.primary_slot)
|
||||
|
||||
|
||||
block_last_span = last_span(last_line_of(other_block))
|
||||
|
||||
line_first_span = candidate_line.primary_slot[0]
|
||||
|
||||
|
||||
if (
|
||||
abs(font_size_delta) > page_body_font_delta
|
||||
and abs(font_size_delta) > doc_body_font_delta - 2
|
||||
and not (style_key(block_last_span) == style_key(line_first_span) and block_last_span.char_count() > 1 and line_first_span.char_count() > 1)
|
||||
and (
|
||||
font_size_delta > 2
|
||||
or (font_size_delta > 1 and not both_edges_aligned)
|
||||
or font_size_delta < -5
|
||||
or (font_size_delta < -2 and candidate_line.char_count() >= 5)
|
||||
or (font_size_delta < -1 and candidate_line.char_count() >= 20 and not both_edges_aligned)
|
||||
)
|
||||
):
|
||||
return False
|
||||
|
||||
# -- Step 3: font / bold mismatch ------------------------------------
|
||||
block_last_line = last_line_of(other_block)
|
||||
|
||||
width_ratio = magnitude_ratio(other_block.bbox_width(), candidate_line.bbox_width())
|
||||
|
||||
bold_mismatch = (block_last_span.primary_slot != line_first_span.primary_slot)
|
||||
|
||||
font_mismatch = (
|
||||
block_last_span.font_name != line_first_span.font_name
|
||||
and dominant_style_of(other_block) != style_key(line_first_span)
|
||||
)
|
||||
|
||||
|
||||
if font_mismatch or bold_mismatch:
|
||||
if bold_mismatch and width_ratio > 2:
|
||||
return False
|
||||
if (block_last_line.char_stats.secondary_slot == 1 or block_last_line.char_stats.secondary_slot == 2) and (
|
||||
candidate_line.char_stats.secondary_slot == 2 or width_ratio > 4
|
||||
):
|
||||
return False
|
||||
if block_last_line.char_stats.tertiary_slot == 6 or other_block.bbox_width() > 1.5 * block_last_line.bbox_width():
|
||||
return False
|
||||
|
||||
if other_block.bold_frac() > 0.9 and candidate_line.bold_frac() < 0.8 and width_ratio > 2:
|
||||
return False
|
||||
|
||||
# -- Step 4: spatial gates -------------------------------------------
|
||||
centers_aligned = center_aligned(other_block, candidate_line, 1)
|
||||
|
||||
if not centers_aligned:
|
||||
vertical_gap = other_block.bottom_edge() - candidate_line.top_edge()
|
||||
|
||||
horizontal_offset = candidate_line.left_edge() - other_block.left_edge()
|
||||
|
||||
if (vertical_gap > -1 and horizontal_offset > 0.33 * other_block.bbox_width()) or horizontal_offset > 0.98 * other_block.bbox_width():
|
||||
return False
|
||||
if candidate_line.center_x() < other_block.left_edge():
|
||||
return False
|
||||
|
||||
# -- Step 5: tolerance base ------------------------------------------
|
||||
bottom_edge_gap = other_block.bottom_edge() - candidate_line.bottom_edge()
|
||||
|
||||
join_tolerance = (
|
||||
_max_nan_propagating(1.3 * (other_block.top_edge() - other_block.bottom_edge()) / other_block.line_count(), block_cluster_ctx.primary_slot.tertiary_slot)
|
||||
+ 1.3 * other_block.avg_font_size()
|
||||
) / 2.0
|
||||
|
||||
|
||||
# -- Step 6: case-flip "hanging indent" detector ---------------------
|
||||
block_case_signal = case_signal(other_block.char_stats)
|
||||
|
||||
line_case_signal = case_signal(candidate_line.char_stats)
|
||||
|
||||
# Capture the old block-last span before comparing both sides of the case
|
||||
# transition.
|
||||
case_signal_flip = (
|
||||
((block_case_signal == 1 and line_case_signal == -1) or (line_case_signal == 1 and block_case_signal == -1))
|
||||
and letter_count(candidate_line.char_stats) >= 3
|
||||
and (is_upper_dominant(other_block.char_stats) != is_upper_dominant(line_first_span.char_stats) or letter_count(line_first_span.char_stats) < 3)
|
||||
and (is_upper_dominant(block_last_span.char_stats) != is_upper_dominant(candidate_line.char_stats) or letter_count(block_last_span.char_stats) < 3)
|
||||
)
|
||||
|
||||
|
||||
if (
|
||||
not font_mismatch and not bold_mismatch and not case_signal_flip
|
||||
and (width_ratio <= 1.2 or left_aligned(block_last_line, candidate_line, 0.1))
|
||||
# Preserve the no-guard width-ratio edge case: a zero-width block still
|
||||
# allows a positive-width last line to increase the join tolerance.
|
||||
and (block_last_line.bbox_width() / other_block.bbox_width() > 0.9 if other_block.bbox_width() != 0 else block_last_line.bbox_width() > 0)
|
||||
):
|
||||
join_tolerance *= 1.3
|
||||
if page_body_font_delta > 0.5 * block_cluster_ctx.primary_slot.primary_slot and not case_signal_flip:
|
||||
join_tolerance *= 2
|
||||
|
||||
# -- Step 7: column alignment ----------------------------------------
|
||||
column_rect = (block_cluster_ctx.state_slot[candidate_line.measure_slot] if (0 <= candidate_line.measure_slot < len(block_cluster_ctx.state_slot)) else None) or EMPTY_RECT
|
||||
|
||||
line_left_aligned_to_column = left_aligned(candidate_line, column_rect, 4.5)
|
||||
|
||||
line_right_aligned_to_column = right_aligned(candidate_line, column_rect, 4.5)
|
||||
|
||||
block_left_aligned_to_column = left_aligned(other_block, column_rect, 4.5)
|
||||
|
||||
block_right_aligned_to_column = right_aligned(other_block, column_rect, 4.5)
|
||||
|
||||
block_column_justified = (
|
||||
block_left_aligned_to_column == block_right_aligned_to_column
|
||||
and other_block.alignment_slot
|
||||
and x_centers_close(block_cluster_ctx.auxiliary_slot, other_block)
|
||||
)
|
||||
|
||||
line_column_centered = (
|
||||
line_left_aligned_to_column == line_right_aligned_to_column
|
||||
and (x_centers_close(block_cluster_ctx.auxiliary_slot, candidate_line) or (block_column_justified and centers_aligned))
|
||||
)
|
||||
|
||||
|
||||
# -- Step 8: alignment multipliers -----------------------------------
|
||||
if (
|
||||
block_column_justified and line_column_centered
|
||||
and other_block.bbox_width() > 0.5 * candidate_line.bbox_width()
|
||||
and (previous_line is None or candidate_line.bottom_edge() - previous_line.bottom_edge() >= bottom_edge_gap)
|
||||
and not font_mismatch
|
||||
):
|
||||
join_tolerance *= 1.3
|
||||
if previous_line is not None and (
|
||||
(other_block.bold_frac() > previous_line.bold_frac() and candidate_line.bold_frac() > previous_line.bold_frac())
|
||||
or (other_block.avg_font_size() > previous_line.bbox_height() + 1 and candidate_line.bbox_height() > previous_line.bbox_height() + 1)
|
||||
):
|
||||
join_tolerance = max(join_tolerance, candidate_line.bottom_edge() - previous_line.top_edge())
|
||||
elif block_right_aligned_to_column and line_left_aligned_to_column:
|
||||
join_tolerance *= 1.3 if other_block.line_count() <= 1 else 1.2
|
||||
elif block_left_aligned_to_column and line_left_aligned_to_column:
|
||||
join_tolerance *= 1.1
|
||||
elif block_right_aligned_to_column:
|
||||
if other_block.line_count() <= 1:
|
||||
join_tolerance *= 1.1
|
||||
if candidate_line.char_stats.secondary_slot == 3:
|
||||
join_tolerance *= 1.1
|
||||
if other_block.line_count() <= 1 and candidate_line.char_stats.secondary_slot == 3:
|
||||
join_tolerance *= 1.1
|
||||
if (
|
||||
candidate_line.left_edge() > other_block.left_edge()
|
||||
and candidate_line.left_edge() <= other_block.left_edge() + 0.1 * other_block.bbox_width()
|
||||
and (other_block.line_count() <= 1 or left_aligned(candidate_line, block_last_line, 1))
|
||||
):
|
||||
join_tolerance *= 1.2
|
||||
elif candidate_line.bbox_width() < 0.9 * block_last_line.bbox_width() and center_aligned(other_block, candidate_line, 1):
|
||||
join_tolerance *= 1.1
|
||||
|
||||
if left_edges_aligned and candidate_line.bbox_width() < 0.5 * other_block.bbox_width() and other_block.char_stats.tertiary_slot != 6 and candidate_line.char_stats.tertiary_slot == 6:
|
||||
join_tolerance *= 1.3
|
||||
|
||||
# -- Step 9: numbering pattern checks --------------------------------
|
||||
block_numbering_kind = numbering_kind(other_block.line())
|
||||
|
||||
block_has_numbering = (
|
||||
numbering_kind(other_block.line()) != 0
|
||||
and first_span_of(other_block).bbox_height() >= 0.8 * other_block.avg_font_size()
|
||||
)
|
||||
|
||||
block_starts_with_digit = block_has_numbering and block_numbering_kind == 1
|
||||
|
||||
line_numbering_kind = numbering_kind(candidate_line)
|
||||
|
||||
line_has_numbering = (
|
||||
numbering_kind(candidate_line) != 0
|
||||
and candidate_line.primary_slot[0].bbox_height() >= 0.8 * candidate_line.avg_font_size()
|
||||
)
|
||||
|
||||
line_starts_with_digit = line_has_numbering and line_numbering_kind == 1
|
||||
|
||||
|
||||
if block_starts_with_digit and not line_starts_with_digit and font_size_delta <= -0.5:
|
||||
join_tolerance /= 2
|
||||
elif (
|
||||
(block_starts_with_digit and (bold_mismatch or font_size_delta <= -0.5))
|
||||
or (line_starts_with_digit and (bold_mismatch or font_size_delta >= 0.5))
|
||||
):
|
||||
join_tolerance /= 1.5
|
||||
elif block_starts_with_digit and candidate_line.left_edge() >= other_block.left_edge() and 0.9 * candidate_line.bbox_width() > other_block.bbox_width():
|
||||
join_tolerance /= 1.5
|
||||
elif block_has_numbering and candidate_line.left_edge() >= other_block.left_edge() and 0.9 * candidate_line.bbox_width() > other_block.bbox_width():
|
||||
join_tolerance /= 1.3
|
||||
elif (block_starts_with_digit and candidate_line.char_stats.secondary_slot != 3 or line_starts_with_digit) and font_mismatch:
|
||||
join_tolerance /= 1.3
|
||||
elif block_starts_with_digit and left_edges_aligned and candidate_line.char_stats.secondary_slot == 2:
|
||||
join_tolerance /= 1.3
|
||||
elif (
|
||||
(block_has_numbering and (font_mismatch or bold_mismatch or font_size_delta <= -0.5 or (left_edges_aligned and candidate_line.char_stats.secondary_slot == 2)))
|
||||
or (line_has_numbering and (font_mismatch or bold_mismatch or font_size_delta >= 0.5))
|
||||
):
|
||||
join_tolerance /= 1.1
|
||||
|
||||
if block_has_numbering and line_has_numbering:
|
||||
join_tolerance /= 1.3
|
||||
|
||||
# -- Step 10: hanging-indent + neighbour patches ---------------------
|
||||
block_first_letter = other_block.line().alignment_slot
|
||||
|
||||
if (
|
||||
block_numbering_kind == 1
|
||||
and line_numbering_kind != 1
|
||||
and not left_edges_aligned
|
||||
and block_first_letter is not None
|
||||
and left_aligned(block_first_letter, candidate_line, 1)
|
||||
):
|
||||
join_tolerance *= 2
|
||||
|
||||
if case_signal_flip:
|
||||
join_tolerance /= 1.1
|
||||
if other_block.line_count() == 1 or not left_edges_aligned:
|
||||
divisor = 3 if width_ratio > 3 else (1.5 if width_ratio > 1.5 else 1)
|
||||
join_tolerance /= divisor
|
||||
if (is_upper_dominant(other_block.char_stats) and block_has_numbering) or (is_upper_dominant(candidate_line.char_stats) and line_has_numbering):
|
||||
join_tolerance /= 2
|
||||
if font_mismatch or bold_mismatch:
|
||||
join_tolerance /= 1.5
|
||||
|
||||
if other_block is not first_candidate_block and bottom_edge_gap > 1.1 * (first_candidate_block.bottom_edge() - candidate_line.bottom_edge()):
|
||||
join_tolerance /= 2
|
||||
|
||||
return bottom_edge_gap <= join_tolerance
|
||||
@@ -0,0 +1,131 @@
|
||||
"""
|
||||
Block classification for header/footer, watermark, boilerplate, TOC-page, and
|
||||
reference-list marking. The module combines recurrence hashes, page-number
|
||||
patterns, body-paragraph gates, cross-page geometry, and numeric-column
|
||||
clustering. The dot-leader and page-number gates intentionally use Unicode
|
||||
number properties so fullwidth and non-Latin digits are handled consistently.
|
||||
"""
|
||||
|
||||
import json
|
||||
import math
|
||||
import regex as regex_module # Unicode \p{...} property classes
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_round_half_up_to_int,
|
||||
magnitude_ratio,
|
||||
intervals_overlap,
|
||||
y_overlaps,
|
||||
center_aligned,
|
||||
to_number,
|
||||
last_span,
|
||||
heading_score,
|
||||
text_of_line,
|
||||
Line,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
is_word_category,
|
||||
block_text,
|
||||
deaccented_text,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
punct_count,
|
||||
info_weight,
|
||||
is_upper_dominant,
|
||||
is_caps_heavy,
|
||||
alignment_code,
|
||||
Block,
|
||||
)
|
||||
from ..stats import style_key, DocStats, weighted_percentile, column_index_of, char_script_bucket
|
||||
from ..tokens import (
|
||||
is_trimmable_token,
|
||||
token_numeric_value,
|
||||
Token,
|
||||
TokenView,
|
||||
wrap_tokens,
|
||||
enumerate_tokens,
|
||||
jenkins_hash,
|
||||
trie_prefix_match,
|
||||
strip_trie_match,
|
||||
strip_leading_if_in,
|
||||
COMMA_CHARS,
|
||||
strip_trailing_comma,
|
||||
trim_trailing_punct,
|
||||
set_case_fold,
|
||||
TrieConfig,
|
||||
build_trie,
|
||||
LineTokenizer,
|
||||
tokenize_block,
|
||||
BuiltTrie,
|
||||
trie_full_match,
|
||||
is_char_token,
|
||||
is_word_token,
|
||||
)
|
||||
|
||||
from .keyword_tables import (
|
||||
_DICT_PATH,
|
||||
_DICTS,
|
||||
_dict_trie,
|
||||
COPYRIGHT_TRIE,
|
||||
VOLUME_WORDS_TRIE,
|
||||
TOC_TITLES_TRIE,
|
||||
FIGURE_KEYWORDS_TRIE,
|
||||
_TABLE_KEYWORDS_TRIE,
|
||||
TABLE_KEYWORDS_TRIE,
|
||||
_CHART_KEYWORDS_TRIE,
|
||||
CHART_KEYWORDS_TRIE,
|
||||
APPENDIX_SECTION_TRIE,
|
||||
INTRODUCTION_SECTION_TRIE,
|
||||
BOX_KEYWORD_TRIE,
|
||||
KEYWORDS_SECTION_TRIE,
|
||||
_BOILERPLATE_PHRASES_PATH,
|
||||
BOILERPLATE_TRIE,
|
||||
DOT_LEADER_ROW_RE,
|
||||
PAGE_NUMBER_ONLY_RE,
|
||||
_search_trie,
|
||||
_normalize_text_key,
|
||||
)
|
||||
from .body_text import (
|
||||
record_recurring_text,
|
||||
is_body_paragraph,
|
||||
span_style_text_key,
|
||||
normalized_block_text,
|
||||
_ROMAN_NUMERALS,
|
||||
span_page_number,
|
||||
longest_word_and_number,
|
||||
)
|
||||
from .header_footer import (
|
||||
PageMarkState,
|
||||
record_marked_block,
|
||||
is_header_positioned,
|
||||
has_adjacent_page_numbers,
|
||||
mark_header_footer,
|
||||
walk_from_page_edge,
|
||||
find_cross_page_match,
|
||||
HeaderFooterContext,
|
||||
bounded_edit_distance,
|
||||
detect_header_footer,
|
||||
)
|
||||
from .toc_boilerplate import (
|
||||
mark_watermarks,
|
||||
_institution_thesis_words,
|
||||
INSTITUTION_THESIS_TRIE,
|
||||
PROFESSOR_TITLES_TRIE,
|
||||
is_boilerplate_block,
|
||||
NumberColumnCluster,
|
||||
extract_number_column,
|
||||
pick_nearer_cluster,
|
||||
detect_toc_range,
|
||||
mark_toc_and_boilerplate,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"is_body_paragraph", "record_recurring_text", "span_style_text_key", "normalized_block_text", "span_page_number", "longest_word_and_number", "record_marked_block", "PageMarkState", "is_header_positioned", "has_adjacent_page_numbers", "mark_header_footer", "walk_from_page_edge", "find_cross_page_match",
|
||||
"HeaderFooterContext", "detect_header_footer", "mark_watermarks",
|
||||
"is_boilerplate_block", "NumberColumnCluster", "extract_number_column", "pick_nearer_cluster", "detect_toc_range", "mark_toc_and_boilerplate",
|
||||
"bounded_edit_distance",
|
||||
"COPYRIGHT_TRIE", "VOLUME_WORDS_TRIE", "TOC_TITLES_TRIE", "FIGURE_KEYWORDS_TRIE", "TABLE_KEYWORDS_TRIE", "CHART_KEYWORDS_TRIE", "APPENDIX_SECTION_TRIE", "INTRODUCTION_SECTION_TRIE", "BOX_KEYWORD_TRIE", "KEYWORDS_SECTION_TRIE",
|
||||
]
|
||||
@@ -0,0 +1,209 @@
|
||||
"""Body-paragraph classification and recurring-text recording."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_round_half_up_to_int,
|
||||
magnitude_ratio,
|
||||
intervals_overlap,
|
||||
y_overlaps,
|
||||
center_aligned,
|
||||
to_number,
|
||||
last_span,
|
||||
heading_score,
|
||||
text_of_line,
|
||||
Line,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
is_word_category,
|
||||
block_text,
|
||||
deaccented_text,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
punct_count,
|
||||
info_weight,
|
||||
is_upper_dominant,
|
||||
is_caps_heavy,
|
||||
alignment_code,
|
||||
Block,
|
||||
)
|
||||
from ..stats import style_key, DocStats, weighted_percentile, column_index_of, char_script_bucket
|
||||
from ..tokens import (
|
||||
is_trimmable_token,
|
||||
token_numeric_value,
|
||||
Token,
|
||||
TokenView,
|
||||
wrap_tokens,
|
||||
enumerate_tokens,
|
||||
jenkins_hash,
|
||||
trie_prefix_match,
|
||||
strip_trie_match,
|
||||
strip_leading_if_in,
|
||||
COMMA_CHARS,
|
||||
strip_trailing_comma,
|
||||
trim_trailing_punct,
|
||||
set_case_fold,
|
||||
TrieConfig,
|
||||
build_trie,
|
||||
LineTokenizer,
|
||||
tokenize_block,
|
||||
BuiltTrie,
|
||||
trie_full_match,
|
||||
is_char_token,
|
||||
is_word_token,
|
||||
)
|
||||
|
||||
from .keyword_tables import (
|
||||
BOILERPLATE_TRIE,
|
||||
PAGE_NUMBER_ONLY_RE,
|
||||
_normalize_text_key,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Recurring-text histogram updater #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def record_recurring_text(doc, other_text: str) -> None:
|
||||
"""Increment the recurring-text histogram under the Jenkins lookup2 hash key. Empty normalized text is a valid key and must not be skipped."""
|
||||
key = jenkins_hash(other_text)
|
||||
doc.tertiary_slot[key] = doc.tertiary_slot.get(key, 0) + 1
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Body-paragraph predicate #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
# The phrase gate is intentionally narrow. Broad Latin keyword matching
|
||||
# over-rejects normal body paragraphs, for example sentences starting with
|
||||
# "figure", and then lets figure captions be treated as headings.
|
||||
|
||||
|
||||
def is_body_paragraph(doc_stats: DocStats, page, block: Block) -> bool:
|
||||
"""Return whether ``block`` is a substantive body paragraph."""
|
||||
if block.weighted_ratio_primary < 0.6:
|
||||
return False
|
||||
width = info_weight(block.char_stats)
|
||||
lines = block.line_count()
|
||||
sentence_punct = block.char_stats.primary_slot[6]
|
||||
|
||||
# The width-per-line ratio uses IEEE-style division. For a zero-line block,
|
||||
# d/e is +inf
|
||||
# (d>0) or NaN (d==0), so every ``d/e < k`` test is False and the block is
|
||||
# NOT rejected here (it falls through to the char-count gate below, which
|
||||
# rejects an empty block). This is not an early return.
|
||||
dw_per_line = (
|
||||
width / lines if lines != 0
|
||||
else (math.inf if width > 0 else math.nan)
|
||||
)
|
||||
if (
|
||||
dw_per_line < 15
|
||||
or (lines >= 10 and dw_per_line < 20)
|
||||
or (lines >= 10 and dw_per_line < 25 and sentence_punct < lines / 8)
|
||||
or (lines >= 20 and dw_per_line < 40 and sentence_punct < lines / 20)
|
||||
):
|
||||
return False
|
||||
|
||||
block_width = block.bbox_width()
|
||||
if lines >= 4:
|
||||
short = 0
|
||||
for state_item in block:
|
||||
if state_item.bbox_width() < 0.75 * block_width and not state_item.primary_slot[0].state_slot.startswith("•"):
|
||||
short += 1
|
||||
if short >= lines / 2 and sentence_punct < lines / 8:
|
||||
return False
|
||||
|
||||
if block_width < page.bounds.bbox_width() / 7:
|
||||
return False
|
||||
|
||||
chars = block.char_count()
|
||||
if chars < 40 or (lines >= 3 and alignment_code(block) == 3) or letter_count(block.char_stats) < 0.1 * chars:
|
||||
return False
|
||||
|
||||
size = block.avg_font_size()
|
||||
body_size = min(page.primary_slot.primary_slot, doc_stats.primary_slot)
|
||||
min_value = min(doc_stats.primary_slot, max(page.bounds.bbox_height(), page.bounds.bbox_width()) / 60)
|
||||
min_value = min(0.7 * min_value, min_value - 3)
|
||||
# Boilerplate phrases are rejected as non-body even when they otherwise look
|
||||
# paragraph-like. This keeps acknowledgement/copyright/proceedings language
|
||||
# out of body-density calculations without broad keyword matching.
|
||||
|
||||
if size < body_size - 2 or size < min_value or trie_prefix_match(BOILERPLATE_TRIE, tokenize_block(block)):
|
||||
return False
|
||||
|
||||
if chars >= 250 and lines >= 4:
|
||||
return True
|
||||
if block_width < page.bounds.bbox_width() / 5 or size < body_size - 0.5:
|
||||
return False
|
||||
if chars >= 100 and lines >= 2 and sentence_punct >= 2:
|
||||
return True
|
||||
if (chars >= 100 or block.char_stats.tertiary_slot == 6) and (
|
||||
size >= page.primary_slot.primary_slot - 0.5 or size > doc_stats.primary_slot - 0.1
|
||||
):
|
||||
return first_span_of(block).font_name == page.primary_slot.state_slot or last_span(last_line_of(block)).font_name == page.primary_slot.state_slot
|
||||
return False
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Header/footer helper keys and predicates #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def span_style_text_key(span) -> str:
|
||||
"""Span style hash including text content: font name, rounded height, bold flag, lowercase text."""
|
||||
# Use exact half-up integer rounding; Python f"{x:.0f}" uses half-even.
|
||||
return f"{span.font_name} {_round_half_up_to_int(span.bbox_height())} {'B' if span.primary_slot else 'R'} {span.text.lower()}"
|
||||
|
||||
|
||||
def normalized_block_text(block: Block) -> str:
|
||||
"""block normalized-text hash."""
|
||||
out = []
|
||||
for token in tokenize_block(block):
|
||||
out.append(_normalize_text_key(token.str.lower()))
|
||||
return "".join(out)
|
||||
|
||||
|
||||
# Roman numeral lookup used for page-number-like header/footer spans.
|
||||
_ROMAN_NUMERALS = {
|
||||
"I": 1, "II": 2, "III": 3, "IV": 4, "V": 5, "VI": 6, "VII": 7,
|
||||
"VIII": 8, "IX": 9, "X": 10, "XI": 11, "XII": 12, "XIII": 13,
|
||||
"XIV": 14, "XV": 15, "XVI": 16, "XVII": 17, "XVIII": 18, "XIX": 19, "XX": 20,
|
||||
}
|
||||
|
||||
|
||||
def span_page_number(span) -> Optional[int]:
|
||||
"""Extract a page number from a span using a digit gate, then Roman numeral lookup."""
|
||||
text = span.text
|
||||
match = PAGE_NUMBER_ONLY_RE.match(text)
|
||||
if match:
|
||||
page_number = to_number(match.group(1))
|
||||
if not math.isnan(page_number) and page_number > 0 and page_number < 1e6 and page_number == math.ceil(page_number):
|
||||
return int(page_number)
|
||||
return None
|
||||
return _ROMAN_NUMERALS.get(text.upper())
|
||||
|
||||
|
||||
def longest_word_and_number(block: Block) -> list[str]:
|
||||
"""extract longest letter-word and longest digit-string. Returns a list of 0-2 strings: lowercased longest word (if >3 chars), then the longest digit-string (raw). """
|
||||
longest_word: Optional[str] = None
|
||||
longest_number: Optional[str] = None
|
||||
for tok in tokenize_block(block):
|
||||
if tok.type == 2:
|
||||
if longest_word is None or len(tok.str) > len(longest_word):
|
||||
longest_word = tok.str
|
||||
elif tok.type == 1:
|
||||
if longest_number is None or len(tok.str) > len(longest_number):
|
||||
longest_number = tok.str
|
||||
out: list[str] = []
|
||||
if longest_word and len(longest_word) > 3:
|
||||
out.append(_normalize_text_key(longest_word.lower()))
|
||||
if longest_number:
|
||||
out.append(longest_number)
|
||||
return out
|
||||
@@ -0,0 +1,481 @@
|
||||
"""Header and footer detection via cross-page recurrence."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_round_half_up_to_int,
|
||||
magnitude_ratio,
|
||||
intervals_overlap,
|
||||
y_overlaps,
|
||||
center_aligned,
|
||||
to_number,
|
||||
last_span,
|
||||
heading_score,
|
||||
text_of_line,
|
||||
Line,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
is_word_category,
|
||||
block_text,
|
||||
deaccented_text,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
punct_count,
|
||||
info_weight,
|
||||
is_upper_dominant,
|
||||
is_caps_heavy,
|
||||
alignment_code,
|
||||
Block,
|
||||
)
|
||||
from ..stats import style_key, DocStats, weighted_percentile, column_index_of, char_script_bucket
|
||||
from ..tokens import (
|
||||
is_trimmable_token,
|
||||
token_numeric_value,
|
||||
Token,
|
||||
TokenView,
|
||||
wrap_tokens,
|
||||
enumerate_tokens,
|
||||
jenkins_hash,
|
||||
trie_prefix_match,
|
||||
strip_trie_match,
|
||||
strip_leading_if_in,
|
||||
COMMA_CHARS,
|
||||
strip_trailing_comma,
|
||||
trim_trailing_punct,
|
||||
set_case_fold,
|
||||
TrieConfig,
|
||||
build_trie,
|
||||
LineTokenizer,
|
||||
tokenize_block,
|
||||
BuiltTrie,
|
||||
trie_full_match,
|
||||
is_char_token,
|
||||
is_word_token,
|
||||
)
|
||||
|
||||
from .keyword_tables import (
|
||||
COPYRIGHT_TRIE,
|
||||
VOLUME_WORDS_TRIE,
|
||||
FIGURE_KEYWORDS_TRIE,
|
||||
TABLE_KEYWORDS_TRIE,
|
||||
CHART_KEYWORDS_TRIE,
|
||||
_search_trie,
|
||||
)
|
||||
from .body_text import (
|
||||
record_recurring_text,
|
||||
is_body_paragraph,
|
||||
span_style_text_key,
|
||||
normalized_block_text,
|
||||
span_page_number,
|
||||
longest_word_and_number,
|
||||
)
|
||||
|
||||
|
||||
class PageMarkState:
|
||||
"""Per-page classification state: first classified index, max heading score, and classified character count."""
|
||||
|
||||
__slots__ = ("primary_slot", "secondary_slot", "tertiary_slot")
|
||||
|
||||
def __init__(self):
|
||||
self.primary_slot = -1
|
||||
self.secondary_slot = 0
|
||||
self.tertiary_slot = 0
|
||||
|
||||
|
||||
def record_marked_block(state: PageMarkState, idx: int, block: Block) -> None:
|
||||
"""Update per-page state after classifying ``block``."""
|
||||
state.primary_slot = idx
|
||||
state.secondary_slot = max(state.secondary_slot, heading_score(block))
|
||||
state.tertiary_slot += block.char_count()
|
||||
|
||||
|
||||
def is_header_positioned(ctx, other_block: Block, candidate_block: Optional[Block]) -> bool:
|
||||
"""Return whether a block is header-positioned relative to the reference block, with content-density gates."""
|
||||
if candidate_block is None:
|
||||
cond = True
|
||||
elif ctx.primary_slot == 1:
|
||||
cond = other_block.top_edge() > candidate_block.bottom_edge()
|
||||
else:
|
||||
cond = other_block.bottom_edge() < candidate_block.top_edge()
|
||||
return cond and other_block.line_count() == 1 and info_weight(other_block.char_stats) >= 8 and letter_count(other_block.char_stats) >= 5 and other_block.char_stats.primary_slot[1] >= 1
|
||||
|
||||
|
||||
def has_adjacent_page_numbers(ctx, page: int, candidate_number: int, reference_flag: bool) -> bool:
|
||||
"""Return whether nearby pages show a strong ``n±1`` / ``n±2`` / ``n±4`` page-number pattern."""
|
||||
page_index = page - 1
|
||||
page_count = len(ctx.secondary_slot.primary_slot)
|
||||
adjacent_one = (
|
||||
(page_index - 1 >= 0 and (candidate_number - 1) in ctx.tertiary_slot[page_index - 1])
|
||||
or (page_index + 1 < page_count and (candidate_number + 1) in ctx.tertiary_slot[page_index + 1])
|
||||
)
|
||||
adjacent_two = (
|
||||
(page_index - 2 >= 0 and (candidate_number - 2) in ctx.tertiary_slot[page_index - 2])
|
||||
or (page_index + 2 < page_count and (candidate_number + 2) in ctx.tertiary_slot[page_index + 2])
|
||||
)
|
||||
if not adjacent_one and not adjacent_two:
|
||||
return False
|
||||
if adjacent_one and adjacent_two:
|
||||
return True
|
||||
adjacent_four = (
|
||||
(page_index - 4 >= 0 and (candidate_number - 4) in ctx.tertiary_slot[page_index - 4])
|
||||
or (page_index + 4 < page_count and (candidate_number + 4) in ctx.tertiary_slot[page_index + 4])
|
||||
)
|
||||
if candidate_number > page / 2 - 30:
|
||||
return adjacent_one or (not reference_flag and adjacent_two) or (adjacent_two and adjacent_four)
|
||||
return bool(adjacent_two and adjacent_four)
|
||||
|
||||
|
||||
def mark_header_footer(ctx, other_block: Block) -> None:
|
||||
"""mark block as classified + bump ghost-text count."""
|
||||
record_recurring_text(ctx.secondary_slot, deaccented_text(other_block))
|
||||
other_block.type = ctx.primary_slot
|
||||
|
||||
|
||||
def walk_from_page_edge(ctx, blocks: list[Block], callback) -> None:
|
||||
"""direction-aware iteration. HEADER (g=1) walks blocks in normal order from top; FOOTER (g=2) walks in reverse from bottom. ``callback`` returns True to halt. """
|
||||
if ctx.primary_slot == 1:
|
||||
for page in blocks:
|
||||
if callback(page):
|
||||
break
|
||||
else:
|
||||
block_index = len(blocks) - 1
|
||||
while block_index >= 0:
|
||||
if callback(blocks[block_index]):
|
||||
break
|
||||
block_index -= 1
|
||||
|
||||
|
||||
def find_cross_page_match(ctx, page, block: Block, text_key: str, ref: Block) -> Optional[Block]:
|
||||
"""Find a matching block on a nearby page by exact normalized text, then by longest word/number pieces."""
|
||||
entries = ctx.auxiliary_slot.get(text_key) or []
|
||||
for entry in entries:
|
||||
entry_page_index = entry["page_index"]
|
||||
entry_block: Block = entry["block"]
|
||||
if entry_page_index < page.page_index - 3:
|
||||
continue
|
||||
if entry_page_index == page.page_index:
|
||||
continue
|
||||
if entry_page_index > page.page_index + 3:
|
||||
break
|
||||
distance_sq = entry_block.left_edge() - block.left_edge()
|
||||
left_delta = entry_block.top_edge() - block.top_edge()
|
||||
right_delta = entry_block.right_edge() - block.right_edge()
|
||||
bottom_delta = entry_block.bottom_edge() - block.bottom_edge()
|
||||
distance_sq = distance_sq * distance_sq + left_delta * left_delta + right_delta * right_delta + bottom_delta * bottom_delta
|
||||
size = page.primary_slot.primary_slot
|
||||
if not (
|
||||
distance_sq >= 100
|
||||
or (distance_sq >= 1 and (
|
||||
(page.page_index == 1 and heading_score(block) >= size + 0.5)
|
||||
or (entry_page_index == 1 and heading_score(entry_block) >= size + 0.5)
|
||||
))
|
||||
):
|
||||
return entry_block
|
||||
|
||||
if is_header_positioned(ctx, block, ref):
|
||||
for key in longest_word_and_number(block):
|
||||
map_value = ctx.measure_slot.get(key)
|
||||
if map_value is None or len(map_value) < max(4, len(ctx.secondary_slot.primary_slot) / 4):
|
||||
continue
|
||||
target = heading_score(block)
|
||||
for nearby_page_index in range(page.page_index - 2, page.page_index + 3):
|
||||
if nearby_page_index == page.page_index:
|
||||
continue
|
||||
nearby_entry = map_value.get(nearby_page_index)
|
||||
if nearby_entry is None:
|
||||
continue
|
||||
body_font_size = page.primary_slot.primary_slot
|
||||
if (abs(target - heading_score(nearby_entry["block"])) > 1
|
||||
or (page.page_index == 1 and target >= body_font_size + 0.5)
|
||||
or (nearby_page_index == 1 and heading_score(nearby_entry["block"]) >= body_font_size + 0.5)):
|
||||
continue
|
||||
threshold = min(len(text_key), len(nearby_entry["text_key"])) / 5
|
||||
if bounded_edit_distance(text_key, nearby_entry["text_key"], threshold) >= threshold:
|
||||
continue
|
||||
return nearby_entry["block"]
|
||||
return None
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Header/footer detection context #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class HeaderFooterContext:
|
||||
"""Per-pass header/footer state."""
|
||||
|
||||
__slots__ = ("secondary_slot", "primary_slot", "previous_slot", "option_slot", "tertiary_slot", "auxiliary_slot", "measure_slot", "state_slot")
|
||||
|
||||
def __init__(self, doc, candidate_number: int):
|
||||
self.secondary_slot = doc
|
||||
self.primary_slot = candidate_number
|
||||
self.previous_slot = "HEADER" if candidate_number == 1 else "FOOTER"
|
||||
self.option_slot: dict[str, int] = {} # span style/text key -> page count
|
||||
self.tertiary_slot: list[set[int]] = [] # per-page page-number set
|
||||
self.auxiliary_slot: dict[str, list[dict]] = {} # normalized text key -> location/block records
|
||||
self.measure_slot: dict[str, dict[int, dict]] = {} # word/number key -> page -> text/block record
|
||||
self.state_slot: list[list[Block]] = [] # per-page candidate blocks
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Bounded edit distance for fuzzy block-key comparison. #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def bounded_edit_distance(text: str, other_text: str, candidate_item: float) -> float:
|
||||
"""Bounded banded edit distance. Returns the limit when the strings differ by more than that many edits; otherwise returns the exact Levenshtein distance."""
|
||||
candidate_item = max(len(text), len(other_text)) if candidate_item <= 0 else math.ceil(candidate_item)
|
||||
if len(text) <= 0:
|
||||
return min(len(other_text), candidate_item)
|
||||
if len(other_text) <= 0:
|
||||
return min(len(text), candidate_item)
|
||||
if len(text) < len(other_text):
|
||||
text, other_text = other_text, text # a is the longer string (columns)
|
||||
if len(text) - len(other_text) >= candidate_item:
|
||||
return candidate_item
|
||||
reference_item = 0 # leftmost band column
|
||||
entry_item = 0 # rightmost band column
|
||||
score_value = [0] * (len(text) + 1) # previous row
|
||||
group_value = [0] * (len(text) + 1) # current row
|
||||
for state_item in range(len(text) + 1): # seed row 0, but only out to column c
|
||||
score_value[state_item] = state_item
|
||||
if state_item > candidate_item:
|
||||
break
|
||||
entry_item = state_item
|
||||
for state_item in range(1, len(other_text) + 1):
|
||||
compare_char = other_text[state_item - 1]
|
||||
key_value = len(text) # leftmost column kept < c this row
|
||||
measure_item = 0 # rightmost column kept < c this row
|
||||
for line_value in range(reference_item, min(entry_item + 1, len(text)) + 1):
|
||||
if line_value == reference_item:
|
||||
group_value[line_value] = 1 + score_value[line_value]
|
||||
elif text[line_value - 1] == compare_char:
|
||||
group_value[line_value] = score_value[line_value - 1]
|
||||
else:
|
||||
group_value[line_value] = 1 + min(group_value[line_value - 1], score_value[line_value - 1])
|
||||
if line_value <= entry_item:
|
||||
group_value[line_value] = min(group_value[line_value], 1 + score_value[line_value])
|
||||
if group_value[line_value] < candidate_item:
|
||||
key_value = min(key_value, line_value)
|
||||
measure_item = line_value
|
||||
if key_value > measure_item: # whole band reached c -> distance >= c
|
||||
return candidate_item
|
||||
score_value, group_value = group_value, score_value
|
||||
reference_item = key_value
|
||||
entry_item = measure_item
|
||||
return min(score_value[entry_item] + len(text) - entry_item, candidate_item)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Header / footer detection #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def detect_header_footer(ctx: HeaderFooterContext) -> None:
|
||||
"""Run the three-pass header/footer detector."""
|
||||
# ----- Pass 1: per-page candidate collection -----------------------
|
||||
for page in ctx.secondary_slot.primary_slot:
|
||||
seen_style_keys: set[str] = set()
|
||||
page_numbers: set[int] = set()
|
||||
ctx.tertiary_slot.append(page_numbers)
|
||||
page_candidates: list[Block] = []
|
||||
ctx.state_slot.append(page_candidates)
|
||||
first_substantive_ref: list[Optional[Block]] = [None] # closure-friendly
|
||||
|
||||
def walk_cb(block: Block) -> bool:
|
||||
if block.skew_frac() >= 1 or block.area() <= 0:
|
||||
return False
|
||||
# Page height should be positive. If a degenerate page appears, keep
|
||||
# IEEE-style Infinity/NaN behavior so the comparisons below stay inert.
|
||||
den = page.bounds.bbox_height()
|
||||
num = block.top_edge() if ctx.primary_slot == 1 else block.bottom_edge()
|
||||
relative = (num / den) if den else (math.copysign(math.inf, num) if num else math.nan)
|
||||
if (ctx.primary_slot == 1 and relative < 0.8) or (ctx.primary_slot == 2 and relative > 0.2):
|
||||
pass_value = False
|
||||
else:
|
||||
tokens = tokenize_block(block)
|
||||
if _search_trie(COPYRIGHT_TRIE, tokens):
|
||||
pass_value = True
|
||||
elif (
|
||||
block.line_count() >= 3
|
||||
or info_weight(block.char_stats) * (1 + block.bold_frac()) >= 200
|
||||
or trie_prefix_match(FIGURE_KEYWORDS_TRIE, tokens)
|
||||
or trie_prefix_match(CHART_KEYWORDS_TRIE, tokens)
|
||||
or trie_prefix_match(TABLE_KEYWORDS_TRIE, tokens)
|
||||
):
|
||||
pass_value = False
|
||||
else:
|
||||
pass_value = True
|
||||
if not pass_value:
|
||||
return True
|
||||
if letter_count(block.char_stats) >= 5 and first_substantive_ref[0] is None:
|
||||
first_substantive_ref[0] = block
|
||||
ref = first_substantive_ref[0]
|
||||
if block.type == 0 and block.char_count() > 0:
|
||||
page_candidates.append(block)
|
||||
if letter_count(block.char_stats) >= 5:
|
||||
text_key = normalized_block_text(block)
|
||||
item_list = ctx.auxiliary_slot.get(text_key)
|
||||
if item_list is None:
|
||||
item_list = []
|
||||
ctx.auxiliary_slot[text_key] = item_list
|
||||
item_list.append({"page_index": page.page_index, "block": block})
|
||||
if is_header_positioned(ctx, block, ref):
|
||||
for key in longest_word_and_number(block):
|
||||
inner = ctx.measure_slot.get(key)
|
||||
if inner is None:
|
||||
inner = {}
|
||||
ctx.measure_slot[key] = inner
|
||||
if page.page_index not in inner:
|
||||
inner[page.page_index] = {"text_key": text_key, "block": block}
|
||||
for line in block:
|
||||
for span in line:
|
||||
if span.char_count() <= 0:
|
||||
continue
|
||||
ok = span_style_text_key(span)
|
||||
if block.char_count() >= 4 and ok not in seen_style_keys:
|
||||
ctx.option_slot[ok] = ctx.option_slot.get(ok, 0) + 1
|
||||
seen_style_keys.add(ok)
|
||||
detected_page_number = span_page_number(span)
|
||||
if detected_page_number is not None:
|
||||
page_numbers.add(detected_page_number)
|
||||
return False
|
||||
|
||||
walk_from_page_edge(ctx, page.output_slot, walk_cb)
|
||||
|
||||
# ----- Pass 2: per-page rejection sweep ----------------------------
|
||||
text_counts: dict[str, int] = {}
|
||||
samples: list[tuple[float, float]] = []
|
||||
|
||||
for page in ctx.secondary_slot.primary_slot:
|
||||
candidates = ctx.state_slot[page.page_index - 1]
|
||||
state = PageMarkState()
|
||||
seen_page_number = False
|
||||
first_substantive: list[Optional[Block]] = [None]
|
||||
for candidate_index in range(len(candidates)):
|
||||
candidate_block = candidates[candidate_index]
|
||||
if candidate_block.char_count() <= 0:
|
||||
continue
|
||||
if candidate_block.type == ctx.primary_slot:
|
||||
record_marked_block(state, candidate_index, candidate_block)
|
||||
continue
|
||||
if candidate_block.type != 0:
|
||||
continue
|
||||
candidate_tokens = tokenize_block(candidate_block)
|
||||
# Copyright terms must appear at the start of the block, not merely
|
||||
# anywhere inside it.
|
||||
if candidate_tokens.length < 10 and trie_prefix_match(COPYRIGHT_TRIE, candidate_tokens):
|
||||
mark_header_footer(ctx, candidate_block)
|
||||
record_marked_block(state, candidate_index, candidate_block)
|
||||
continue
|
||||
if letter_count(candidate_block.char_stats) >= 5:
|
||||
if first_substantive[0] is None:
|
||||
first_substantive[0] = candidate_block
|
||||
pk_hash = normalized_block_text(candidate_block)
|
||||
match = find_cross_page_match(ctx, page, candidate_block, pk_hash, first_substantive[0])
|
||||
if match is not None:
|
||||
mark_header_footer(ctx, candidate_block)
|
||||
record_marked_block(state, candidate_index, candidate_block)
|
||||
other_unmarked = match.type != ctx.primary_slot
|
||||
if other_unmarked:
|
||||
mark_header_footer(ctx, match)
|
||||
if len(pk_hash) >= 5:
|
||||
previous_count = text_counts.get(pk_hash, 0)
|
||||
text_counts[pk_hash] = 2 if (previous_count or other_unmarked) else 1
|
||||
continue
|
||||
style_threshold = max(2.0, min(len(ctx.secondary_slot.primary_slot) / 3.0, 5.0))
|
||||
chars = 0
|
||||
for candidate_line in candidate_block:
|
||||
for candidate_span in candidate_line:
|
||||
if candidate_span.char_count() <= 0:
|
||||
continue
|
||||
style_hash = span_style_text_key(candidate_span)
|
||||
if ctx.option_slot.get(style_hash, 0) >= style_threshold:
|
||||
chars += candidate_span.char_count()
|
||||
continue
|
||||
page_number = span_page_number(candidate_span)
|
||||
if page_number is not None and has_adjacent_page_numbers(ctx, page.page_index, page_number, seen_page_number):
|
||||
seen_page_number = True
|
||||
chars += candidate_span.char_count()
|
||||
if chars >= candidate_block.char_count():
|
||||
mark_header_footer(ctx, candidate_block)
|
||||
record_marked_block(state, candidate_index, candidate_block)
|
||||
pk_again = normalized_block_text(candidate_block)
|
||||
if len(pk_again) >= 5:
|
||||
text_counts[pk_again] = text_counts.get(pk_again, 0) + 1
|
||||
|
||||
if state.primary_slot < 0:
|
||||
continue
|
||||
first_block = candidates[state.primary_slot]
|
||||
samples.append((first_block.center_y(), float(state.tertiary_slot)))
|
||||
|
||||
# Also classify earlier non-confirmed blocks
|
||||
for index in range(state.primary_slot):
|
||||
block = candidates[index]
|
||||
if block.type == ctx.primary_slot:
|
||||
continue
|
||||
if ctx.primary_slot == 1 and block.bottom_edge() < first_block.bottom_edge():
|
||||
continue
|
||||
if block.bbox_width() >= page.bounds.bbox_width() / 2:
|
||||
continue
|
||||
if is_body_paragraph(ctx.secondary_slot.secondary_slot, page, block):
|
||||
continue
|
||||
if letter_count(block.char_stats) > 0 and heading_score(block) >= state.secondary_slot + 1:
|
||||
continue
|
||||
block.type = ctx.primary_slot
|
||||
record_recurring_text(ctx.secondary_slot, deaccented_text(block))
|
||||
|
||||
# ----- Pass 3: cutoff line + top-3 text-hash sweep -----------------
|
||||
if len(samples) < len(ctx.secondary_slot.primary_slot) / 20:
|
||||
return
|
||||
cutoff = weighted_percentile(samples, 20 if ctx.primary_slot == 1 else 80)
|
||||
|
||||
top: list[tuple[str, int]] = []
|
||||
for text_hash, count in text_counts.items():
|
||||
if count < len(ctx.secondary_slot.primary_slot) / 20:
|
||||
continue
|
||||
top.append((text_hash, count))
|
||||
if not top:
|
||||
return
|
||||
top.sort(key=lambda item_pair: -item_pair[1])
|
||||
if len(top) > 3:
|
||||
top = top[:3]
|
||||
|
||||
for page in ctx.secondary_slot.primary_slot:
|
||||
for recurring_block in ctx.state_slot[page.page_index - 1]:
|
||||
if ctx.primary_slot == 1 and recurring_block.top_edge() < cutoff:
|
||||
break
|
||||
if ctx.primary_slot == 2 and recurring_block.bottom_edge() > cutoff:
|
||||
break
|
||||
if recurring_block.type != 0:
|
||||
continue
|
||||
recurring_tokens = tokenize_block(recurring_block)
|
||||
stripped = _search_trie(VOLUME_WORDS_TRIE, recurring_tokens)
|
||||
if stripped is not None:
|
||||
# ``stripped.end`` is absolute in the forward token view, so this
|
||||
# drops the matched volume phrase and keeps the tail.
|
||||
tail = recurring_tokens.slice(stripped.end)
|
||||
head_tok = tail.token_at(0) if tail.length > 0 else None
|
||||
if head_tok is not None and head_tok.type == 1:
|
||||
recurring_block.type = ctx.primary_slot
|
||||
record_recurring_text(ctx.secondary_slot, deaccented_text(recurring_block))
|
||||
# Deliberately fall through: the same block can also match the
|
||||
# top recurring-text sweep below.
|
||||
if letter_count(recurring_block.char_stats) < 5:
|
||||
continue
|
||||
if page.page_index <= 1 and heading_score(recurring_block) > ctx.secondary_slot.secondary_slot.primary_slot + 1:
|
||||
continue
|
||||
text_key = normalized_block_text(recurring_block)
|
||||
for text_hash, _ in top:
|
||||
threshold = min(len(text_key), len(text_hash)) / 2.0
|
||||
if bounded_edit_distance(text_key, text_hash, threshold) >= threshold:
|
||||
continue
|
||||
# No break: every sufficiently similar recurring key contributes
|
||||
# to the recurring-text histogram.
|
||||
recurring_block.type = ctx.primary_slot
|
||||
record_recurring_text(ctx.secondary_slot, deaccented_text(recurring_block))
|
||||
@@ -0,0 +1,113 @@
|
||||
"""Dictionary-backed keyword tries and shared regexes."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import regex as regex_module # Unicode \p{...} property classes
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_round_half_up_to_int,
|
||||
magnitude_ratio,
|
||||
intervals_overlap,
|
||||
y_overlaps,
|
||||
center_aligned,
|
||||
to_number,
|
||||
last_span,
|
||||
heading_score,
|
||||
text_of_line,
|
||||
Line,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
is_word_category,
|
||||
block_text,
|
||||
deaccented_text,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
punct_count,
|
||||
info_weight,
|
||||
is_upper_dominant,
|
||||
is_caps_heavy,
|
||||
alignment_code,
|
||||
Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
is_trimmable_token,
|
||||
token_numeric_value,
|
||||
Token,
|
||||
TokenView,
|
||||
wrap_tokens,
|
||||
enumerate_tokens,
|
||||
jenkins_hash,
|
||||
trie_prefix_match,
|
||||
strip_trie_match,
|
||||
strip_leading_if_in,
|
||||
COMMA_CHARS,
|
||||
strip_trailing_comma,
|
||||
trim_trailing_punct,
|
||||
set_case_fold,
|
||||
TrieConfig,
|
||||
build_trie,
|
||||
LineTokenizer,
|
||||
tokenize_block,
|
||||
BuiltTrie,
|
||||
trie_full_match,
|
||||
is_char_token,
|
||||
is_word_token,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Load dictionaries (built into tries on first use) #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
_DICT_PATH = Path(__file__).parent.parent / "data" / "dictionaries.json"
|
||||
_DICTS = json.loads(_DICT_PATH.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def _dict_trie(key: str) -> BuiltTrie:
|
||||
"""Build a case-folded trie from a dictionary entry."""
|
||||
return build_trie(_DICTS.get(key, []), set_case_fold(TrieConfig(), True))
|
||||
|
||||
|
||||
COPYRIGHT_TRIE = build_trie(["Copyright", "©"], set_case_fold(TrieConfig(), True)) # inline list
|
||||
VOLUME_WORDS_TRIE = _dict_trie("volume_words")
|
||||
TOC_TITLES_TRIE = _dict_trie("toc_titles")
|
||||
FIGURE_KEYWORDS_TRIE = _dict_trie("ai_section_keywords")
|
||||
_TABLE_KEYWORDS_TRIE = _dict_trie("table_keywords")
|
||||
TABLE_KEYWORDS_TRIE = _TABLE_KEYWORDS_TRIE
|
||||
|
||||
_CHART_KEYWORDS_TRIE = _dict_trie("chart_keywords")
|
||||
CHART_KEYWORDS_TRIE = _CHART_KEYWORDS_TRIE
|
||||
APPENDIX_SECTION_TRIE = _dict_trie("appendices_dict")
|
||||
INTRODUCTION_SECTION_TRIE = _dict_trie("introduction_dict")
|
||||
BOX_KEYWORD_TRIE = build_trie(["box"], set_case_fold(TrieConfig(), True)) # inline list
|
||||
KEYWORDS_SECTION_TRIE = _dict_trie("keywords_dict")
|
||||
|
||||
# Multilingual boilerplate phrase trie: publisher and proceeding headers plus
|
||||
# stock acknowledgement openers such as "First of all I would like to thank".
|
||||
# Used by the body-paragraph gate to reject boilerplate as non-body.
|
||||
# Phrase list stored as a data asset.
|
||||
_BOILERPLATE_PHRASES_PATH = Path(__file__).parent.parent / "data" / "boilerplate_phrases.json"
|
||||
BOILERPLATE_TRIE = build_trie(json.loads(_BOILERPLATE_PHRASES_PATH.read_text(encoding="utf-8")), set_case_fold(TrieConfig(), True))
|
||||
|
||||
# Regular expressions for the dot-leader and page-number gates (Unicode \p{Number} -> ``regex`` module).
|
||||
# Leading class is ASCII 1-9 + fullwidth 1-9 (U+FF11-FF19); it must NOT admit
|
||||
# fullwidth zero U+FF10, so it is [1-91-9], not [1-90-9].
|
||||
DOT_LEADER_ROW_RE = regex_module.compile(r"([.][" + _UNICODE_WHITESPACE_CLASS + r"]*){5,}[" + _UNICODE_WHITESPACE_CLASS + r"]*[1-91-9]\p{Number}*\Z")
|
||||
PAGE_NUMBER_ONLY_RE = regex_module.compile(r"^[ |]*([1-91-9]\p{Number}*)[ |]*\Z")
|
||||
|
||||
|
||||
def _search_trie(trie: BuiltTrie, tokens) -> Optional[TokenView]:
|
||||
"""Return the shortest earliest Aho-Corasick trie match for ``tokens``."""
|
||||
from ..tokens import aho_corasick_tokens as _real_bh
|
||||
return _real_bh(trie, tokens)
|
||||
|
||||
|
||||
def _normalize_text_key(text: str) -> str:
|
||||
"""Strip diacritics only; callers lowercase first when a case-folded key is needed."""
|
||||
return _strip_diacritics(text)
|
||||
@@ -0,0 +1,375 @@
|
||||
"""Watermark, boilerplate, and TOC-range detection."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_round_half_up_to_int,
|
||||
magnitude_ratio,
|
||||
intervals_overlap,
|
||||
y_overlaps,
|
||||
center_aligned,
|
||||
to_number,
|
||||
last_span,
|
||||
heading_score,
|
||||
text_of_line,
|
||||
Line,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
is_word_category,
|
||||
block_text,
|
||||
deaccented_text,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
punct_count,
|
||||
info_weight,
|
||||
is_upper_dominant,
|
||||
is_caps_heavy,
|
||||
alignment_code,
|
||||
Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
is_trimmable_token,
|
||||
token_numeric_value,
|
||||
Token,
|
||||
TokenView,
|
||||
wrap_tokens,
|
||||
enumerate_tokens,
|
||||
jenkins_hash,
|
||||
trie_prefix_match,
|
||||
strip_trie_match,
|
||||
strip_leading_if_in,
|
||||
COMMA_CHARS,
|
||||
strip_trailing_comma,
|
||||
trim_trailing_punct,
|
||||
set_case_fold,
|
||||
TrieConfig,
|
||||
build_trie,
|
||||
LineTokenizer,
|
||||
tokenize_block,
|
||||
BuiltTrie,
|
||||
trie_full_match,
|
||||
is_char_token,
|
||||
is_word_token,
|
||||
)
|
||||
|
||||
from .keyword_tables import (
|
||||
_DICTS,
|
||||
_dict_trie,
|
||||
TOC_TITLES_TRIE,
|
||||
DOT_LEADER_ROW_RE,
|
||||
_search_trie,
|
||||
)
|
||||
from .body_text import (
|
||||
is_body_paragraph,
|
||||
normalized_block_text,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Side-rail watermark detector #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def mark_watermarks(doc) -> None:
|
||||
"""Bucket skewed side-rail blocks by normalized text; recurring groups are marked as watermarks."""
|
||||
buckets: dict[str, list[Block]] = {}
|
||||
for page in doc.primary_slot:
|
||||
for block in page.output_slot:
|
||||
if block.skew_frac() < 1:
|
||||
continue
|
||||
if letter_count(block.char_stats) < 5:
|
||||
continue
|
||||
# Skip blocks in the central 80% of the page width.
|
||||
horizontal_offset = block.center_x()
|
||||
page_width = page.bounds.bbox_width()
|
||||
if 0.1 * page_width < horizontal_offset < 0.9 * page_width:
|
||||
continue
|
||||
# No empty-string guard: empty normalized-text keys bucket together.
|
||||
key = normalized_block_text(block)
|
||||
buckets.setdefault(key, []).append(block)
|
||||
for group in buckets.values():
|
||||
if len(group) < 3:
|
||||
continue
|
||||
for block in group:
|
||||
block.type = 12
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Boilerplate block predicate #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
# Institution/thesis trie combines institution words with thesis-specific terms.
|
||||
_institution_thesis_words = list(_DICTS.get("institution_words", [])) + list(_DICTS.get("nk_thesis_words", []))
|
||||
INSTITUTION_THESIS_TRIE = build_trie(_institution_thesis_words, set_case_fold(TrieConfig(), True))
|
||||
PROFESSOR_TITLES_TRIE = _dict_trie("professor_titles")
|
||||
|
||||
|
||||
def is_boilerplate_block(block: Block) -> bool:
|
||||
"""line/block looks like boilerplate (committee members, author affiliations, journal volume info etc.)."""
|
||||
tokens = tokenize_block(block)
|
||||
if info_weight(block.char_stats) >= 200 or tokens.length >= 100:
|
||||
return False
|
||||
if _search_trie(INSTITUTION_THESIS_TRIE, tokens):
|
||||
return True
|
||||
# Strip leading lines that match professor/title boilerplate.
|
||||
while tokens.length > 0:
|
||||
# Count tokens belonging to the first token's line and strip that line.
|
||||
first_line = tokens.token_at(0).line() if tokens.token_at(0) else None
|
||||
if first_line is None:
|
||||
break
|
||||
line_end = 0
|
||||
while line_end < tokens.length:
|
||||
tok = tokens.token_at(line_end)
|
||||
if tok is None or tok.line() is not first_line:
|
||||
break
|
||||
line_end += 1
|
||||
if not trie_prefix_match(PROFESSOR_TITLES_TRIE, tokens.slice(0, line_end)):
|
||||
return False
|
||||
tokens = tokens.slice(line_end)
|
||||
return True
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# TOC-page detection chain #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class NumberColumnCluster:
|
||||
"""numeric-leading-token cluster."""
|
||||
|
||||
__slots__ = ("anchor_x", "width", "secondary_slot", "primary_slot", "length", "tertiary_slot")
|
||||
|
||||
def __init__(self, anchor_x_value: float, width: float, reference_number: int, next_number: int, length: int, limit_flag: bool):
|
||||
self.anchor_x = anchor_x_value # anchor x-position
|
||||
self.width = width # cluster typical width
|
||||
self.secondary_slot = reference_number # first value seen
|
||||
self.primary_slot = next_number # last value seen
|
||||
self.length = length
|
||||
self.tertiary_slot = limit_flag # is increasing
|
||||
|
||||
|
||||
def extract_number_column(block) -> Optional[NumberColumnCluster]:
|
||||
"""Extract a numeric-leading cluster if block lines form an increasing page-number sequence."""
|
||||
column = 0
|
||||
last_number = 0
|
||||
sequence_length = 0
|
||||
for line in block:
|
||||
line_number = to_number(text_of_line(line))
|
||||
if math.isnan(line_number):
|
||||
return None
|
||||
if not (line_number > 0 and line_number < 1e6 and line_number == math.ceil(line_number)) or line_number >= 1e4 or last_number > line_number:
|
||||
return NumberColumnCluster(block.center_x(), block.bbox_width(), column, last_number, sequence_length, False)
|
||||
if column <= 0:
|
||||
column = int(line_number)
|
||||
last_number = int(line_number)
|
||||
sequence_length += 1
|
||||
return NumberColumnCluster(block.center_x(), block.bbox_width(), column, last_number, sequence_length, True)
|
||||
|
||||
|
||||
def pick_nearer_cluster(cluster: NumberColumnCluster, other_cluster: Optional[NumberColumnCluster], other: Optional[NumberColumnCluster]) -> Optional[NumberColumnCluster]:
|
||||
"""Pick the closer neighbor cluster within the current cluster width."""
|
||||
distance = (cluster.anchor_x - other_cluster.anchor_x) if other_cluster is not None else math.inf
|
||||
candidate_distance = (other.anchor_x - cluster.anchor_x) if other is not None else math.inf
|
||||
if distance > cluster.width and candidate_distance > cluster.width:
|
||||
return None
|
||||
return other_cluster if distance < candidate_distance else other
|
||||
|
||||
|
||||
def detect_toc_range(doc, page, index) -> Optional[dict]:
|
||||
"""Detect a TOC-like block range within ``page``. The detector combines dot-leader rows, blocks ending in dot-leader page numbers, contents-like titles, and same-x-range numeric clusters. Returns a ``{start_index, end_index}`` range or ``None``."""
|
||||
blocks = page.output_slot
|
||||
lines = 0
|
||||
dot_leader_blocks = 0
|
||||
weight = 0.0
|
||||
last_multiline = -1
|
||||
contents = -1
|
||||
pre_contents = -1
|
||||
last_toc = -1
|
||||
seen_body = False
|
||||
body_stop_y = page.bounds.top_edge()
|
||||
is_last_page = (index == page.page_index - 1) if isinstance(index, int) and index >= 0 else False
|
||||
clusters: list[NumberColumnCluster] = [] # sorted by anchor_x
|
||||
|
||||
def _add_cluster(cluster: NumberColumnCluster) -> None:
|
||||
# Sorted-set semantics: an equal anchor_x is a no-op.
|
||||
import bisect
|
||||
keys = [existing_cluster.anchor_x for existing_cluster in clusters]
|
||||
insert_index = bisect.bisect_left(keys, cluster.anchor_x)
|
||||
if insert_index < len(clusters) and clusters[insert_index].anchor_x == cluster.anchor_x:
|
||||
return
|
||||
clusters.insert(insert_index, cluster)
|
||||
|
||||
def _next_number_column_cluster(cluster: NumberColumnCluster) -> Optional[NumberColumnCluster]:
|
||||
# Non-strict successor: an equal anchor_x entry is returned.
|
||||
import bisect
|
||||
keys = [column.anchor_x for column in clusters]
|
||||
cluster_index = bisect.bisect_left(keys, cluster.anchor_x)
|
||||
return clusters[cluster_index] if cluster_index < len(clusters) else None
|
||||
|
||||
def _prev_number_column_cluster(cluster: NumberColumnCluster) -> Optional[NumberColumnCluster]:
|
||||
# Non-strict predecessor: an equal anchor_x entry is returned.
|
||||
import bisect
|
||||
keys = [column.anchor_x for column in clusters]
|
||||
cluster_index = bisect.bisect_right(keys, cluster.anchor_x)
|
||||
return clusters[cluster_index - 1] if cluster_index > 0 else None
|
||||
|
||||
def _remove(cluster: NumberColumnCluster) -> None:
|
||||
try:
|
||||
clusters.remove(cluster)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
for codepoint, block in enumerate(blocks):
|
||||
is_toc = False
|
||||
# Count dot-leader rows across all lines in the block.
|
||||
|
||||
for line in block:
|
||||
if DOT_LEADER_ROW_RE.search(text_of_line(line)):
|
||||
is_toc = True
|
||||
if last_toc >= 0:
|
||||
last_toc = codepoint
|
||||
else:
|
||||
lines += 1
|
||||
if lines >= 5 or (lines >= 3 and is_last_page):
|
||||
last_toc = codepoint
|
||||
# A dot-leader on the last line also starts or extends the TOC range.
|
||||
|
||||
last_line = block.primary_slot[-1] if block.primary_slot else None
|
||||
if last_line is not None and DOT_LEADER_ROW_RE.search(text_of_line(last_line)):
|
||||
is_toc = True
|
||||
if last_toc >= 0:
|
||||
last_toc = codepoint
|
||||
continue
|
||||
else:
|
||||
dot_leader_blocks += 1
|
||||
weight += info_weight(block.char_stats)
|
||||
if is_last_page and dot_leader_blocks >= 2 and weight >= 0.8 * page.primary_slot.secondary_slot:
|
||||
last_toc = codepoint
|
||||
# Body block tracking
|
||||
if not is_toc and is_body_paragraph(doc.secondary_slot, page, block):
|
||||
seen_body = True
|
||||
body_stop_y = min(body_stop_y, block.bottom_edge())
|
||||
if block.line_count() > 1 and not is_toc:
|
||||
last_multiline = codepoint
|
||||
# "Contents"-like title must consume the whole block, not just a prefix.
|
||||
if contents < 0 and block.line_count() <= 1 and trie_full_match(TOC_TITLES_TRIE, tokenize_block(block)):
|
||||
contents = codepoint
|
||||
pre_contents = last_multiline
|
||||
if not seen_body and block.top_edge() > 3 * page.bounds.bbox_height() / 4:
|
||||
return {"start_index": pre_contents + 1, "end_index": len(blocks) - 1}
|
||||
# Numeric-column clustering on unclassified blocks below the body line.
|
||||
if block.right_edge() < page.bounds.center_x():
|
||||
continue
|
||||
if block.top_edge() > body_stop_y:
|
||||
continue
|
||||
# Extract even from an empty-looking block; the extractor decides whether
|
||||
# a usable numeric sequence exists.
|
||||
cluster = extract_number_column(block)
|
||||
if cluster is not None:
|
||||
successor = _next_number_column_cluster(cluster)
|
||||
predecessor = _prev_number_column_cluster(cluster)
|
||||
picked = pick_nearer_cluster(cluster, predecessor, successor)
|
||||
if picked is not None:
|
||||
_remove(picked)
|
||||
new_value = NumberColumnCluster(
|
||||
picked.anchor_x,
|
||||
picked.width,
|
||||
picked.secondary_slot,
|
||||
cluster.primary_slot,
|
||||
picked.length + cluster.length,
|
||||
picked.tertiary_slot and cluster.tertiary_slot and picked.primary_slot <= cluster.secondary_slot,
|
||||
)
|
||||
else:
|
||||
new_value = cluster
|
||||
_add_cluster(new_value)
|
||||
if (new_value.tertiary_slot
|
||||
and (new_value.length >= 10 or (new_value.length >= 5 and is_last_page))
|
||||
and new_value.primary_slot - new_value.secondary_slot > 0.01 * new_value.primary_slot):
|
||||
last_toc = codepoint
|
||||
|
||||
if last_toc < 0:
|
||||
return None
|
||||
return {"start_index": pre_contents + 1 if pre_contents >= 0 else 0, "end_index": last_toc}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# TOC pages, references lists, and figure/table captions #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def mark_toc_and_boilerplate(doc) -> None:
|
||||
"""Mark TOC blocks as type=9 and captions/boilerplate as type=12."""
|
||||
previous_toc_page = -math.inf
|
||||
for page in doc.primary_slot:
|
||||
if (
|
||||
page.page_index - 1 >= len(doc.primary_slot) / 2
|
||||
and page.primary_slot.secondary_slot >= 0.9 * doc.secondary_slot.secondary_slot
|
||||
):
|
||||
continue
|
||||
# "Most-boilerplate" check for front-matter pages.
|
||||
|
||||
if (
|
||||
page.page_index > 1
|
||||
and page.page_index < 50
|
||||
and page.primary_slot.secondary_slot < max(200, min(0.75 * doc.secondary_slot.secondary_slot, 1000))
|
||||
):
|
||||
total = 0.0
|
||||
body_paragraph = 0.0
|
||||
body_paragraph_lines = 0
|
||||
for line_or_block in page.output_slot:
|
||||
if line_or_block.type != 0 or line_or_block.skew_frac() >= 1:
|
||||
continue
|
||||
width_value = info_weight(line_or_block.char_stats) * heading_score(line_or_block)
|
||||
total += width_value
|
||||
if is_boilerplate_block(line_or_block):
|
||||
body_paragraph += width_value
|
||||
body_paragraph_lines += 1
|
||||
if body_paragraph >= 0.8 * total and body_paragraph_lines >= 3:
|
||||
for block in page.output_slot:
|
||||
block.type = 12
|
||||
continue
|
||||
toc = detect_toc_range(doc, page, previous_toc_page)
|
||||
if toc is None:
|
||||
continue
|
||||
previous_toc_page = page.page_index
|
||||
start = toc["start_index"]
|
||||
end = toc["end_index"]
|
||||
blocks = page.output_slot
|
||||
# Track centered-block count, weighted font sum, total weight, and the
|
||||
# running bottom edge used by walk-forward break conditions.
|
||||
centered_flag = 0
|
||||
width_flag = 0.0
|
||||
width = 0.0
|
||||
walk_break_y = page.bounds.top_edge()
|
||||
for idx in range(start, len(blocks)):
|
||||
block = blocks[idx]
|
||||
score = heading_score(block)
|
||||
if idx <= end:
|
||||
if block.isolated_centered:
|
||||
centered_flag += 1
|
||||
heading_weight = info_weight(block.char_stats)
|
||||
width_flag += block.avg_font_size() * heading_weight
|
||||
width += heading_weight
|
||||
walk_break_y = min(walk_break_y, block.bottom_edge())
|
||||
block.type = 9
|
||||
continue
|
||||
# Walk forward with four break conditions: prominent heading, centered
|
||||
# block, dense body text, or a large vertical gap to a prominent block.
|
||||
if block.skew_frac() < 1 and score > doc.secondary_slot.primary_slot + 4 and score > page.primary_slot.primary_slot + 4:
|
||||
break
|
||||
if centered_flag <= 1 and block.isolated_centered:
|
||||
break
|
||||
if info_weight(block.char_stats) > 300 and block.char_stats.primary_slot[6] > 2 and block.weighted_ratio_secondary > 0.5:
|
||||
break
|
||||
if width > 0:
|
||||
average = width_flag / width
|
||||
if walk_break_y - block.top_edge() > average and block.skew_frac() < 1 and score > average + 1.5:
|
||||
break
|
||||
walk_break_y = min(walk_break_y, block.bottom_edge())
|
||||
block.type = 9
|
||||
@@ -0,0 +1,68 @@
|
||||
"""Line clustering pipeline.
|
||||
|
||||
The initial pass walks spans in document order and groups them into lines using
|
||||
an in-line continuation test, while also collapsing overstrike duplicates
|
||||
(artificial-bold rendering where the same glyph is painted twice). The merge
|
||||
pass inserts lines into a sorted structure keyed by top-desc reading order,
|
||||
looks up predecessor/successor neighbors, and either merges the new line into a
|
||||
neighbor or keeps it separate. Neighbor lookup is inclusive of an exact
|
||||
reading-order key match, so the successor uses ``bisect_left`` and the
|
||||
predecessor uses ``bisect_right - 1``.
|
||||
"""
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Optional
|
||||
|
||||
from sortedcontainers import SortedKeyList
|
||||
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
avg_char_width2,
|
||||
Span,
|
||||
magnitude_ratio,
|
||||
same_x_extent,
|
||||
same_y_extent,
|
||||
append_span,
|
||||
last_span,
|
||||
avg_char_width,
|
||||
raw_text_of_line,
|
||||
text_of_line,
|
||||
reading_order_key,
|
||||
left_edge_key,
|
||||
numbering_kind,
|
||||
Line,
|
||||
letter_count,
|
||||
is_upper_dominant,
|
||||
)
|
||||
|
||||
from .merge_rules import (
|
||||
TRAILING_DOT_LEADER_RE,
|
||||
span_continues_line,
|
||||
vertical_distance_in_line_heights,
|
||||
pick_closer_neighbor,
|
||||
should_merge_lines,
|
||||
)
|
||||
from .build import (
|
||||
_skip_mark_only,
|
||||
build_initial_lines,
|
||||
_is_label_stack,
|
||||
LinesContainer,
|
||||
_set_add,
|
||||
cluster_lines,
|
||||
)
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Combined helper #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
__all__ = [
|
||||
"span_continues_line",
|
||||
"vertical_distance_in_line_heights",
|
||||
"pick_closer_neighbor",
|
||||
"should_merge_lines",
|
||||
"build_initial_lines",
|
||||
"cluster_lines",
|
||||
"TRAILING_DOT_LEADER_RE",
|
||||
]
|
||||
@@ -0,0 +1,211 @@
|
||||
"""Builds initial lines and clusters them into merged lines."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Optional
|
||||
|
||||
from sortedcontainers import SortedKeyList
|
||||
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
avg_char_width2,
|
||||
Span,
|
||||
magnitude_ratio,
|
||||
same_x_extent,
|
||||
same_y_extent,
|
||||
append_span,
|
||||
last_span,
|
||||
avg_char_width,
|
||||
raw_text_of_line,
|
||||
text_of_line,
|
||||
reading_order_key,
|
||||
left_edge_key,
|
||||
numbering_kind,
|
||||
Line,
|
||||
letter_count,
|
||||
is_upper_dominant,
|
||||
)
|
||||
|
||||
from .merge_rules import (
|
||||
span_continues_line,
|
||||
pick_closer_neighbor,
|
||||
should_merge_lines,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Initial line builder.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _skip_mark_only(span: Span, page_area: float) -> bool:
|
||||
"""Return True for mark-heavy tiny glyphs whose area is below one part per million of the page area."""
|
||||
return (span.char_count() - span.char_stats.primary_slot[5]) > 1 and span.area() < page_area * 1e-6
|
||||
|
||||
|
||||
def build_initial_lines(spans: list[Span], page_bbox) -> list[Line]:
|
||||
"""Build initial lines from flat spans. Returns the list of initial lines. """
|
||||
line: list[Line] = []
|
||||
pending_line = Line()
|
||||
pending_span: Optional[Span] = None
|
||||
page_area = page_bbox.area()
|
||||
|
||||
for span in spans:
|
||||
if span.text == "" or _skip_mark_only(span, page_area):
|
||||
continue
|
||||
if pending_span is not None:
|
||||
# Overstrike duplicate detection: same trimmed text, both edges +
|
||||
# both top/bottom within 10% of f's geometry -> f gets the bold
|
||||
# bit and h is discarded.
|
||||
if (
|
||||
pending_span.char_count() > 0
|
||||
and pending_span.state_slot == span.state_slot
|
||||
and same_x_extent(pending_span, span, 0.1 * pending_span.bbox_width())
|
||||
and same_y_extent(pending_span, span, 0.1 * pending_span.bbox_height())
|
||||
):
|
||||
pending_span.primary_slot = True
|
||||
continue
|
||||
# End current line if e is non-empty AND tn says NOT to continue
|
||||
if not (len(pending_line.primary_slot) <= 0 or span_continues_line(pending_line, pending_span)):
|
||||
line.append(pending_line)
|
||||
pending_line = Line()
|
||||
append_span(pending_line, pending_span)
|
||||
pending_span = span
|
||||
else:
|
||||
pending_span = span
|
||||
|
||||
if pending_span is not None:
|
||||
if not (len(pending_line.primary_slot) <= 0 or span_continues_line(pending_line, pending_span)):
|
||||
line.append(pending_line)
|
||||
pending_line = Line()
|
||||
append_span(pending_line, pending_span)
|
||||
line.append(pending_line)
|
||||
# If no current span exists, the pending line is intentionally dropped.
|
||||
# This path is currently unreachable from the loop logic.
|
||||
return line
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# xn -- line clustering driver #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _is_label_stack(line: Line, other_line: Line, body_ma: float) -> bool:
|
||||
"""Detect a display-sized label stacked directly above the text it labels. The geometry must overlap horizontally while sitting on a different baseline; the upper piece must be display-sized relative to body text and larger than the lower text. This captures chapter numbers and drop caps that should be read before the title below them."""
|
||||
if body_ma <= 0:
|
||||
return False
|
||||
overlap = min(line.right_edge(), other_line.right_edge()) - max(line.left_edge(), other_line.left_edge())
|
||||
frac = overlap / max(1e-6, min(line.bbox_width(), other_line.bbox_width()))
|
||||
vertical_overlap = min(line.top_edge(), other_line.top_edge()) - max(line.bottom_edge(), other_line.bottom_edge())
|
||||
vertical_overlap_fraction = vertical_overlap / max(1e-6, min(line.bbox_height(), other_line.bbox_height()))
|
||||
if not (frac > 0.5 and vertical_overlap_fraction < 0.5):
|
||||
return False
|
||||
upper, lower = (line, other_line) if line.center_y() > other_line.center_y() else (other_line, line)
|
||||
# display-type (>= 2x body) AND larger than the text it sits above
|
||||
# (>= 1.5x lower): a leading label over smaller text. The second clause
|
||||
# drops same-size display stacks (e.g. chart axis numbers over each other).
|
||||
return upper.avg_font_size() >= 2.0 * body_ma and upper.avg_font_size() >= 1.5 * lower.avg_font_size()
|
||||
|
||||
|
||||
@dataclass
|
||||
class LinesContainer:
|
||||
"""Mutable line container used by the clustering pass."""
|
||||
|
||||
primary_slot: list[Line] = field(default_factory=list)
|
||||
|
||||
|
||||
def _set_add(tree: SortedKeyList, line: Line) -> None:
|
||||
"""Set-style insertion into the sorted line index. Lines with identical top, bottom, left, and right ordering keys are dropped instead of duplicated."""
|
||||
idx = tree.bisect_left(line)
|
||||
if idx < len(tree) and reading_order_key(tree[idx]) == reading_order_key(line): # type: ignore[arg-type]
|
||||
return # reading-order key collision -> sorted set insertion drops the element
|
||||
tree.add(line)
|
||||
|
||||
|
||||
def cluster_lines(lines_container: LinesContainer, other_item: float, candidate_items: list) -> list[Line]:
|
||||
"""Mutate the contained line list by merging nearby compatible lines."""
|
||||
# Sort input lines by reading order.
|
||||
lines_container.primary_slot.sort(key=left_edge_key)
|
||||
|
||||
# Body-text reference for the display-size test in _is_label_stack: the
|
||||
# median glyph font size across the page (dominated by body text).
|
||||
merged_accent_spans = sorted(
|
||||
span_value.font_size for line in lines_container.primary_slot for span_value in line.primary_slot
|
||||
if getattr(span_value, "font_size", 0) > 0
|
||||
)
|
||||
body_ma = merged_accent_spans[len(merged_accent_spans) // 2] if merged_accent_spans else 0.0
|
||||
|
||||
# Tree of lines, ordered by reading position (top desc, bottom desc, left, right).
|
||||
tree: SortedKeyList = SortedKeyList(key=reading_order_key)
|
||||
merged_lines: list[Line] = [] # output (lines that won't merge further)
|
||||
|
||||
for candidate_line in lines_container.primary_slot:
|
||||
# Rotated / skewed lines: don't try to cluster, just emit
|
||||
if last_span(candidate_line).previous_slot > 1:
|
||||
merged_lines.append(candidate_line)
|
||||
continue
|
||||
|
||||
# successor (just below f vertically) and predecessor (just above).
|
||||
# predecessor/successor search are INCLUSIVE floor/ceiling, so a reading-order-key-equal line already
|
||||
# in the tree is the zero-distance neighbour: successor = bisect_left
|
||||
# (first key >= f), predecessor = bisect_right - 1 (last key <= f).
|
||||
idx_succ = tree.bisect_left(candidate_line)
|
||||
successor_line = tree[idx_succ] if idx_succ < len(tree) else None
|
||||
idx_pred = tree.bisect_right(candidate_line)
|
||||
line_item = tree[idx_pred - 1] if idx_pred > 0 else None
|
||||
|
||||
neighbor_line = pick_closer_neighbor(line_item, successor_line, candidate_line, other_item)
|
||||
if neighbor_line is None:
|
||||
_set_add(tree, candidate_line)
|
||||
continue
|
||||
|
||||
tree.remove(neighbor_line)
|
||||
neighbor_last_span = last_span(neighbor_line) # last span of k
|
||||
|
||||
# Subscript / overstrike case (single-span f duplicating k's last span)
|
||||
if (
|
||||
len(candidate_line.primary_slot) == 1
|
||||
and len(neighbor_line.primary_slot) <= 5
|
||||
and neighbor_last_span.char_count() > 0
|
||||
and neighbor_last_span.state_slot == candidate_line.primary_slot[0].state_slot
|
||||
and same_x_extent(neighbor_last_span, candidate_line, 0.1 * neighbor_last_span.bbox_width())
|
||||
and same_y_extent(neighbor_last_span, candidate_line, 0.1 * neighbor_last_span.bbox_height())
|
||||
):
|
||||
if abs(neighbor_last_span.left_edge() - candidate_line.left_edge()) < 0.01 and abs(neighbor_last_span.top_edge() - candidate_line.top_edge()) < 0.01:
|
||||
# exact duplicate -> keep the original line unchanged
|
||||
_set_add(tree, neighbor_line)
|
||||
continue
|
||||
# Otherwise create a new line carrying k's spans with m marked bold
|
||||
new_line = Line()
|
||||
neighbor_last_span.primary_slot = True
|
||||
for source_span in neighbor_line:
|
||||
append_span(new_line, source_span)
|
||||
_set_add(tree, new_line)
|
||||
elif should_merge_lines(neighbor_line, candidate_line, candidate_items):
|
||||
# Continuation merge. Normally append f after k (left-to-right).
|
||||
# If the candidate is a display-sized label stacked above the text,
|
||||
# reading order is top-to-bottom, so the label leads. Reorder spans
|
||||
# only; the merge and block/line structure stay unchanged.
|
||||
if _is_label_stack(neighbor_line, candidate_line, body_ma) and candidate_line.center_y() > neighbor_line.center_y():
|
||||
merged = Line()
|
||||
for span in candidate_line:
|
||||
append_span(merged, span)
|
||||
for span in neighbor_line:
|
||||
append_span(merged, span)
|
||||
_set_add(tree, merged)
|
||||
else:
|
||||
for span in candidate_line:
|
||||
append_span(neighbor_line, span)
|
||||
_set_add(tree, neighbor_line)
|
||||
else:
|
||||
# Cannot merge: emit k as a finalized line, start fresh with f
|
||||
merged_lines.append(neighbor_line)
|
||||
_set_add(tree, candidate_line)
|
||||
|
||||
# Drain remaining
|
||||
merged_lines.extend(tree)
|
||||
# Final sort by reading order.
|
||||
merged_lines.sort(key=reading_order_key)
|
||||
lines_container.primary_slot = merged_lines
|
||||
return merged_lines
|
||||
@@ -0,0 +1,190 @@
|
||||
"""Span continuation and line-merge predicates."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
avg_char_width2,
|
||||
Span,
|
||||
magnitude_ratio,
|
||||
same_x_extent,
|
||||
same_y_extent,
|
||||
append_span,
|
||||
last_span,
|
||||
avg_char_width,
|
||||
raw_text_of_line,
|
||||
text_of_line,
|
||||
reading_order_key,
|
||||
left_edge_key,
|
||||
numbering_kind,
|
||||
Line,
|
||||
letter_count,
|
||||
is_upper_dominant,
|
||||
)
|
||||
|
||||
|
||||
# Matches "...." dot-leader trails used in TOC entries: "Chapter 1 ........"
|
||||
TRAILING_DOT_LEADER_RE = re.compile(r"([.][" + _UNICODE_WHITESPACE_CLASS + r"]*){4,}\Z")
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# In-line continuation predicate.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def span_continues_line(line: Line, other_span: Span) -> bool:
|
||||
"""Return whether ``span`` continues the current line. The test requires matching skew, overlapping vertical intervals, and a horizontal gap within a per-character tolerance that widens after sentence-ending punctuation."""
|
||||
if last_span(line).previous_slot != other_span.previous_slot:
|
||||
return False
|
||||
line_center_y = line.center_y() # a's y-center
|
||||
span_center_y = other_span.center_y() # b's y-center
|
||||
# Vertical disjointness check: if both centers fall outside the other box,
|
||||
# the spans are not on the same line.
|
||||
if (line_center_y > other_span.top_edge() or line_center_y < other_span.bottom_edge()) and (span_center_y > line.top_edge() or span_center_y < line.bottom_edge()):
|
||||
return False
|
||||
# tolerance from per-char height
|
||||
tolerance = min(5.0, max(0.1, avg_char_width(line), avg_char_width2(other_span)))
|
||||
wide_tolerance = 2.0 * tolerance
|
||||
# When a's last char is sentence-end punctuation, widen the tolerance
|
||||
if line.char_stats.tertiary_slot == 5:
|
||||
wide_tolerance *= 2.0
|
||||
return other_span.left_edge() > line.right_edge() - wide_tolerance and other_span.left_edge() < line.right_edge() + tolerance
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Neighbor distance and picker.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def vertical_distance_in_line_heights(line: Line, other_line: Line) -> float:
|
||||
"""normalized vertical-center distance between two lines. ``|a.center_y - b.center_y| / max(a.bbox_height, b.bbox_height)``: how many line-heights apart the centres are. Returns 0 when centres coincide. """
|
||||
line_center_y = line.center_y()
|
||||
other_center_y = other_line.center_y()
|
||||
if line_center_y == other_center_y:
|
||||
return 0.0
|
||||
denom = max(line.bbox_height(), other_line.bbox_height())
|
||||
if denom == 0.0:
|
||||
# Empty lines carry an inverted-sentinel bbox. Preserve IEEE division
|
||||
# edge cases so the later distance comparison simply does not merge.
|
||||
diff = line_center_y - other_center_y
|
||||
return float("nan") if diff != diff else float("inf")
|
||||
return abs(line_center_y - other_center_y) / denom
|
||||
|
||||
|
||||
def pick_closer_neighbor(
|
||||
line: Optional[Line],
|
||||
other_line: Optional[Line],
|
||||
candidate_line: Line,
|
||||
reference_item: float,
|
||||
) -> Optional[Line]:
|
||||
"""Pick the closer neighboring line to the current line when it falls within the merge tolerance. Returns the closer candidate when the distance is below the threshold, else ``None``. Either or both candidates may be ``None`` (e.g. c is at the top of the tree -> no predecessor). """
|
||||
if line is None and other_line is None:
|
||||
return None
|
||||
entry_item = vertical_distance_in_line_heights(line, candidate_line) if line is not None else float("inf")
|
||||
second_candidate = vertical_distance_in_line_heights(other_line, candidate_line) if other_line is not None else float("inf")
|
||||
if entry_item >= reference_item and second_candidate >= reference_item:
|
||||
return None
|
||||
return line if entry_item < second_candidate else other_line
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Line merge predicate.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def should_merge_lines(line: Line, other_line: Line, candidate_items: list) -> bool:
|
||||
"""Return whether ``other_line`` should merge into ``line``. The decision compares the horizontal gap against a tolerance based on harmonic mean character width, then adjusts for style mismatch, script category, dot leaders, column membership, short continuations, bracketed starts, sentence endings, and uppercase dominance."""
|
||||
if line.char_count() > 0 and other_line.char_count() > 0:
|
||||
# Different skew/rotation -> never merge
|
||||
if magnitude_ratio(line.previous_slot, other_line.previous_slot) > 2 and abs(line.previous_slot - other_line.previous_slot) > 10:
|
||||
return False
|
||||
|
||||
# Harmonic mean of character heights with no clamp. A zero char-height
|
||||
# contributes an infinite inverse, driving the merge tolerance to zero.
|
||||
line_projection = 1.0 / avg_char_width(line) if avg_char_width(line) != 0 else float("inf")
|
||||
other_projection = 1.0 / avg_char_width(other_line) if avg_char_width(other_line) != 0 else float("inf")
|
||||
harmonic_char_width = 2.0 / (line_projection + other_projection)
|
||||
horizontal_gap = other_line.left_edge() - line.right_edge() # horizontal gap
|
||||
gap_factor = 2.0
|
||||
|
||||
# italic mismatch
|
||||
italic = line.bold_frac() > 0
|
||||
other_italic = other_line.bold_frac() > 0
|
||||
if italic != other_italic:
|
||||
gap_factor /= 1.5
|
||||
|
||||
# last-char category 4 = other-letter (Lo, CJK/syllabics)
|
||||
# OR more than half of a's chars are category 4
|
||||
if line.char_stats.tertiary_slot == 4 or line.char_stats.primary_slot[4] > line.char_count() / 2:
|
||||
gap_factor /= 2.0
|
||||
|
||||
# sentence-end + all-digits + dot leader pattern -> TOC row, don't merge
|
||||
sent_end = line.char_stats.tertiary_slot == 6
|
||||
if sent_end:
|
||||
# candidate numeric-token test: the candidate has digits and all characters are digits
|
||||
|
||||
all_digits = other_line.char_stats.auxiliary_slot > 0 and other_line.char_stats.auxiliary_slot == other_line.char_stats.primary_slot[1]
|
||||
if all_digits and TRAILING_DOT_LEADER_RE.search(raw_text_of_line(line)):
|
||||
gap_factor *= 3.0
|
||||
else:
|
||||
all_digits = False
|
||||
|
||||
# Column-based bonuses ----------------------------------------------------
|
||||
if candidate_items and 0 <= line.measure_slot < len(candidate_items):
|
||||
line_column = candidate_items[line.measure_slot]
|
||||
col_left = line_column.get("left", float("inf"))
|
||||
col_right = line_column.get("right", float("-inf"))
|
||||
else:
|
||||
col_left = float("inf")
|
||||
col_right = float("-inf")
|
||||
|
||||
inside_col = (
|
||||
line.left_edge() >= col_left
|
||||
and line.right_edge() <= col_right
|
||||
and other_line.left_edge() >= col_left
|
||||
and other_line.right_edge() <= col_right
|
||||
)
|
||||
if (line.char_count() < 40 or inside_col) and (
|
||||
same_y_extent(line, other_line, 0.1) or same_y_extent(last_span(line), other_line, 0.1)
|
||||
):
|
||||
gap_factor *= 1.5
|
||||
if line.char_count() < 40 and inside_col:
|
||||
gap_factor *= 2.0
|
||||
|
||||
# At-column-edge demotion
|
||||
if 0 <= other_line.measure_slot < len(candidate_items):
|
||||
other_column = candidate_items[other_line.measure_slot]
|
||||
else:
|
||||
other_column = None
|
||||
if (
|
||||
len(candidate_items) <= 0
|
||||
or (
|
||||
abs(line.right_edge() - col_right) < 5
|
||||
and (line.measure_slot >= len(candidate_items) - 1 or not other_column or abs(other_line.left_edge() - other_column.get("left", float("inf"))) < 5)
|
||||
)
|
||||
):
|
||||
gap_factor /= 2.0
|
||||
|
||||
# Very short leading line with continuation evidence: short, low aspect,
|
||||
# numbering-like, and followed by text with letters. The inside-column flag
|
||||
# controls whether this gets the stronger multiplier.
|
||||
if line.char_count() <= 8 and line.bbox_width() <= 10 * line.avg_font_size() and numbering_kind(line) != 0 and letter_count(other_line.char_stats) > 0:
|
||||
gap_factor *= 3.0 if inside_col else 2.0
|
||||
|
||||
# Bracketed short line or uppercase sentence-period inside a column.
|
||||
if line.char_count() <= 10:
|
||||
text = text_of_line(line)
|
||||
if text.startswith("[") and text.endswith("]"):
|
||||
gap_factor *= 2.0
|
||||
elif inside_col and line.char_stats.secondary_slot == 2 and text.endswith("."):
|
||||
gap_factor *= 2.0
|
||||
|
||||
# Both lines are uppercase-dominant inside the same column.
|
||||
|
||||
if inside_col and is_upper_dominant(line.char_stats) and is_upper_dominant(other_line.char_stats):
|
||||
gap_factor *= 1.5
|
||||
|
||||
return horizontal_gap <= gap_factor * harmonic_char_width
|
||||
@@ -0,0 +1,49 @@
|
||||
"""
|
||||
Column detection via sweep-line gutter scoring and recursive page splitting.
|
||||
|
||||
The detector builds horizontal and vertical sweep events, scores candidate
|
||||
gutters, assigns column indexes to lines, and returns column rectangles used by
|
||||
the second line-clustering pass. Direction code 0 scans vertical positions to
|
||||
find row breaks; direction code 1 scans horizontal positions to find column
|
||||
breaks.
|
||||
"""
|
||||
|
||||
import math
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
Rect, rect_union, EMPTY_RECT, Line, info_weight, text_of_line, numbering_kind, numbering_value, _UNICODE_WHITESPACE_CLASS, _max_nan_propagating, _min_nan_propagating,
|
||||
)
|
||||
|
||||
|
||||
# Detect TOC dot leaders ("... 5", "....3"). Gutter scoring rejects a split
|
||||
# candidate when too many dot-leader lines straddle the gap, because a TOC page
|
||||
# should remain in one reading region.
|
||||
# The regular expression is end-anchored only; use re.search rather than re.match.
|
||||
import re as re_module
|
||||
|
||||
from .gutters import (
|
||||
SweepEvent,
|
||||
SplitCandidate,
|
||||
ColumnDetectionContext,
|
||||
DOT_LEADER_RE,
|
||||
collect_gutter_candidates,
|
||||
_score_gutter_gap,
|
||||
)
|
||||
from .splitting import (
|
||||
assign_column_index,
|
||||
recursive_split,
|
||||
detect_columns,
|
||||
columns_to_x_bounds,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"SweepEvent",
|
||||
"SplitCandidate",
|
||||
"ColumnDetectionContext",
|
||||
"collect_gutter_candidates",
|
||||
"assign_column_index",
|
||||
"recursive_split",
|
||||
"detect_columns",
|
||||
"columns_to_x_bounds",
|
||||
]
|
||||
@@ -0,0 +1,345 @@
|
||||
"""Gutter-gap candidates and scoring for column detection."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
Rect, rect_union, EMPTY_RECT, Line, info_weight, text_of_line, numbering_kind, numbering_value, _UNICODE_WHITESPACE_CLASS, _max_nan_propagating, _min_nan_propagating,
|
||||
)
|
||||
|
||||
|
||||
# Detect TOC dot leaders ("... 5", "....3"). Gutter scoring rejects a split
|
||||
# candidate when too many dot-leader lines straddle the gap, because a TOC page
|
||||
# should remain in one reading region.
|
||||
# The regular expression is end-anchored only; use re.search rather than re.match.
|
||||
import re as re_module
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Sweep event. ``is_start=True`` means "line enters" at a start edge; False means
|
||||
# "line leaves" at an end edge.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class SweepEvent:
|
||||
__slots__ = ("line", "position", "is_start")
|
||||
|
||||
def __init__(self, line: Line, position: float, is_start_flag: bool):
|
||||
self.line = line
|
||||
self.position = position
|
||||
self.is_start = is_start_flag
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Column-split candidate. Direction 0 is a vertical sweep; direction 1 is a
|
||||
# horizontal sweep. Higher score is better.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class SplitCandidate:
|
||||
__slots__ = ("start", "end", "direction", "score")
|
||||
|
||||
def __init__(self, start: float, end: float, direction_value: int, score: float):
|
||||
self.start = start
|
||||
self.end = end
|
||||
self.direction = direction_value
|
||||
self.score = score
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Detection context. Thresholds derived from page geometry and page statistics.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class ColumnDetectionContext:
|
||||
"""Page-level thresholds used while recursively scoring gutter candidates."""
|
||||
|
||||
__slots__ = ("secondary_slot", "primary_slot", "tertiary_slot", "state_slot", "auxiliary_slot", "option_slot", "measure_slot")
|
||||
|
||||
def __init__(self, primary_item, secondary_item, candidate_item):
|
||||
self.secondary_slot = secondary_item
|
||||
self.primary_slot = candidate_item
|
||||
# ``log2(0)`` would be -inf -- guard against empty input.
|
||||
self.tertiary_slot = math.floor(2 * math.log2(len(candidate_item))) if candidate_item else 0
|
||||
self.state_slot = primary_item.bbox_width() / 6.0
|
||||
self.auxiliary_slot = _max_nan_propagating(0.5 * secondary_item.primary_slot, _min_nan_propagating(1.1 * (secondary_item.tertiary_slot - secondary_item.primary_slot), 3.0 * secondary_item.primary_slot))
|
||||
self.option_slot = secondary_item.measure_slot
|
||||
self.measure_slot = 1.5 * secondary_item.primary_slot
|
||||
DOT_LEADER_RE = re_module.compile(r"([.][" + _UNICODE_WHITESPACE_CLASS + r"]*){5,}\Z")
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Score split candidates in a sweep.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def collect_gutter_candidates(
|
||||
context: ColumnDetectionContext,
|
||||
other_rect: Rect, # root rect (page bbox)
|
||||
candidate_rect: Rect, # current sub-rect
|
||||
events: list[SweepEvent], # sorted events
|
||||
direction: int, # direction: 0 vert sweep / 1 horiz sweep
|
||||
extent: float, # extent (height or width)
|
||||
min_gap: float, # minimum gutter size
|
||||
out_candidates: list[SplitCandidate], # output: candidates to append to
|
||||
) -> None:
|
||||
"""Walk adjacent event pairs looking for column gutters. Each candidate gap receives a multiplicative score from line weight, font-size balance, indent/outdent structure, citation markers, and edge proximity; the best viable score wins."""
|
||||
active_count = 0
|
||||
for size_value in range(len(events) - 1):
|
||||
if events[size_value].is_start:
|
||||
active_count += 1
|
||||
else:
|
||||
active_count -= 1
|
||||
if active_count > 0:
|
||||
continue
|
||||
# The gap between adjacent event positions is a candidate gutter.
|
||||
score = _score_gutter_gap(context, other_rect, candidate_rect, events, direction, extent, min_gap, size_value)
|
||||
if score is not None:
|
||||
key_value = events[size_value].position
|
||||
score_value = events[size_value + 1].position
|
||||
out_candidates.append(SplitCandidate(key_value, score_value, direction, score))
|
||||
|
||||
|
||||
def _score_gutter_gap(
|
||||
primary_item: ColumnDetectionContext,
|
||||
other_rect: Rect, # root rect
|
||||
candidate_rect: Rect, # current sub-rect (variable name p follows the extraction rule)
|
||||
|
||||
reference_items: list[SweepEvent], # events
|
||||
next_number: int, # direction
|
||||
extent: float, # extent
|
||||
min_gap: float, # min gutter
|
||||
limit_number: int, # current event index
|
||||
) -> Optional[float]:
|
||||
"""Score one candidate gap at adjacent sweep events, or return None when it is not viable."""
|
||||
key_value = reference_items[limit_number].position
|
||||
score_value = reference_items[limit_number + 1].position
|
||||
item_value = score_value - key_value
|
||||
if item_value < min_gap:
|
||||
return None
|
||||
|
||||
# --- backward pass: lines that close before this gap -----------------
|
||||
measure_item = preceding_max_width = 0
|
||||
secondary_item = reference_item = distance_accumulator = width_value = wide_accumulator = 0.0
|
||||
min_edge = math.inf
|
||||
preceding_max_trailing_edge = -math.inf
|
||||
min_edge_position = math.inf
|
||||
max_edge = -math.inf
|
||||
event_count = 0
|
||||
lower_accumulator = math.inf
|
||||
candidate_item = group_value = max_char_count = state_item = 0
|
||||
sample_item = limit_number
|
||||
while sample_item >= 0:
|
||||
sweep_event = reference_items[sample_item]
|
||||
event_position = sweep_event.position
|
||||
event_is_start = sweep_event.is_start
|
||||
sweep_line = sweep_event.line
|
||||
if event_position < key_value - item_value:
|
||||
break
|
||||
if event_is_start:
|
||||
sample_item -= 1
|
||||
continue
|
||||
measure_item += 1
|
||||
preceding_max_width = max(preceding_max_width, sweep_line.bbox_width())
|
||||
if next_number == 1:
|
||||
leading_edge, trailing_edge = sweep_line.bottom_edge(), sweep_line.top_edge()
|
||||
else:
|
||||
leading_edge, trailing_edge = sweep_line.left_edge(), sweep_line.right_edge()
|
||||
min_edge = min(min_edge, leading_edge)
|
||||
preceding_max_trailing_edge = max(preceding_max_trailing_edge, trailing_edge)
|
||||
entry_item = info_weight(sweep_line.char_stats)
|
||||
secondary_item += entry_item
|
||||
if sweep_line.avg_font_size() > reference_item:
|
||||
reference_item = sweep_line.avg_font_size()
|
||||
distance_accumulator = entry_item
|
||||
elif sweep_line.avg_font_size() == reference_item:
|
||||
distance_accumulator += entry_item
|
||||
if key_value - event_position < 1:
|
||||
width_value += 1
|
||||
wide_accumulator = max(wide_accumulator, sweep_line.bbox_width())
|
||||
min_edge_position = min(min_edge_position, leading_edge)
|
||||
max_edge = max(max_edge, trailing_edge)
|
||||
if numbering_kind(sweep_line) == 1:
|
||||
event_count += 1
|
||||
if numbering_value(sweep_line) == 1:
|
||||
lower_accumulator = min(lower_accumulator, sweep_line.left_edge())
|
||||
if next_number == 1:
|
||||
if sweep_line.char_count() <= 5 and numbering_kind(sweep_line) != 0:
|
||||
candidate_item += 1
|
||||
if sweep_line.char_count() <= 10:
|
||||
text = text_of_line(sweep_line)
|
||||
if (text.startswith("[") and text.endswith("]")) or (
|
||||
sweep_line.char_stats.secondary_slot == 2 and text.endswith(".")
|
||||
):
|
||||
group_value += 1
|
||||
if DOT_LEADER_RE.search(text_of_line(sweep_line)):
|
||||
state_item += 1
|
||||
max_char_count = max(max_char_count, sweep_line.char_count())
|
||||
sample_item -= 1
|
||||
|
||||
# Sanity gates
|
||||
if (
|
||||
candidate_item >= measure_item
|
||||
or candidate_item >= max(2, measure_item / 2)
|
||||
or group_value >= measure_item
|
||||
or state_item >= max(2, measure_item / 2)
|
||||
or (next_number == 1 and max_char_count <= 1)
|
||||
):
|
||||
return None
|
||||
|
||||
# --- forward pass: lines that open after this gap --------------------
|
||||
next_gap = 0.0
|
||||
following_line_count = 0
|
||||
other_gap = math.inf
|
||||
following_max_trailing_edge = -math.inf
|
||||
page_gap = following_max_font_size = after = 0
|
||||
quantity = 0
|
||||
numbering_score = following_numbering_count = following_edge_max_width = 0
|
||||
right_gap = 0.0
|
||||
count_item = limit_number + 1
|
||||
while count_item < len(reference_items):
|
||||
sweep_event = reference_items[count_item]
|
||||
event_position = sweep_event.position
|
||||
event_is_start = sweep_event.is_start
|
||||
sweep_line = sweep_event.line
|
||||
if event_position > score_value + item_value:
|
||||
break
|
||||
if not event_is_start:
|
||||
count_item += 1
|
||||
continue
|
||||
following_line_count += 1
|
||||
next_gap = max(next_gap, sweep_line.bbox_width())
|
||||
if next_number == 1:
|
||||
leading_edge, trailing_edge = sweep_line.bottom_edge(), sweep_line.top_edge()
|
||||
else:
|
||||
leading_edge, trailing_edge = sweep_line.left_edge(), sweep_line.right_edge()
|
||||
other_gap = min(other_gap, leading_edge)
|
||||
following_max_trailing_edge = max(following_max_trailing_edge, trailing_edge)
|
||||
width = info_weight(sweep_line.char_stats)
|
||||
after += width
|
||||
if sweep_line.avg_font_size() > following_max_font_size:
|
||||
following_max_font_size = sweep_line.avg_font_size()
|
||||
page_gap = width
|
||||
elif sweep_line.avg_font_size() == following_max_font_size:
|
||||
page_gap += width
|
||||
if event_position - score_value < 1:
|
||||
quantity += 1
|
||||
following_edge_max_width = max(following_edge_max_width, sweep_line.bbox_width())
|
||||
if numbering_kind(sweep_line) == 1:
|
||||
following_numbering_count += 1
|
||||
numbering_score = _max_nan_propagating(numbering_score, numbering_value(sweep_line))
|
||||
right_gap = _max_nan_propagating(right_gap, sweep_line.top_edge())
|
||||
count_item += 1
|
||||
|
||||
if measure_item <= 0 or following_line_count <= 0:
|
||||
return None
|
||||
|
||||
# Combined scoring mixes the root page rectangle for page-level thresholds
|
||||
# with the current recursive sub-rectangle for split geometry. Root and sub
|
||||
# coincide before the first split, but diverge on genuinely multi-column
|
||||
# pages; keeping both frames is part of the column decision model.
|
||||
root_width = other_rect.bbox_width()
|
||||
height = other_rect.bbox_height()
|
||||
root_center_x = other_rect.center_x()
|
||||
|
||||
if next_number == 1:
|
||||
# vertical sweep: special pre-gate for single-line columns
|
||||
# The top-edge gate compares the sub-rectangle to the root page height.
|
||||
|
||||
if width_value <= 1 and quantity <= 1 and not (
|
||||
candidate_rect.top_edge() < other_rect.bottom_edge() + 0.3 * height
|
||||
and lower_accumulator < math.inf
|
||||
and numbering_score <= 4
|
||||
):
|
||||
return None
|
||||
if (secondary_item <= 100 and after <= 100) and (
|
||||
key_value < other_rect.left + 0.2 * root_width or score_value > other_rect.left + 0.8 * root_width
|
||||
):
|
||||
return None
|
||||
|
||||
mid = (reference_item + following_max_font_size) / 2
|
||||
|
||||
if next_number == 0 and distance_accumulator >= 0.8 * secondary_item and page_gap >= 0.8 * after and (
|
||||
(
|
||||
abs(reference_item - following_max_font_size) < 0.1
|
||||
and reference_item >= primary_item.secondary_slot.primary_slot + 0.5
|
||||
and following_max_font_size >= primary_item.secondary_slot.primary_slot + 0.5
|
||||
and item_value < max(1.3 * mid, min_gap * 2)
|
||||
)
|
||||
or (
|
||||
reference_item >= primary_item.secondary_slot.primary_slot + 2
|
||||
and following_max_font_size >= primary_item.secondary_slot.primary_slot + 2
|
||||
and item_value < max(1.5 * mid, min_gap * 3)
|
||||
)
|
||||
):
|
||||
return None
|
||||
|
||||
line_value = extent * extent * item_value
|
||||
|
||||
if next_number == 1:
|
||||
line_value *= min(width_value, quantity)
|
||||
if secondary_item <= 50 and measure_item <= 1:
|
||||
line_value /= 100
|
||||
candidate_height = candidate_rect.bbox_height()
|
||||
threshold = candidate_rect.top_edge() - 0.2 * candidate_height
|
||||
if following_numbering_count >= 3 and numbering_score >= 6 and right_gap < threshold:
|
||||
# Preserve IEEE division here: a zero denominator yields +inf and a
|
||||
# negative denominator is clamped below. The branch selection depends
|
||||
# on those numeric edge cases.
|
||||
denom = numbering_score - following_numbering_count
|
||||
inv = (1 / denom) if denom != 0 else math.inf
|
||||
factor = max(0.3, min(1.0, inv))
|
||||
factor *= factor
|
||||
line_value *= factor
|
||||
elif candidate_height > height / 2:
|
||||
factor = candidate_height / height
|
||||
factor *= factor
|
||||
line_value *= 1 + factor
|
||||
line_value *= max(1, 2 - abs(root_center_x - (key_value + score_value) / 2) / root_width * 10)
|
||||
|
||||
if next_number == 0:
|
||||
candidate_width = candidate_rect.bbox_width()
|
||||
line_value *= max(wide_accumulator, following_edge_max_width) / candidate_width * (max(preceding_max_width, next_gap) / candidate_width)
|
||||
min_value = min(min_edge, other_gap)
|
||||
max_value = max(preceding_max_trailing_edge, following_max_trailing_edge)
|
||||
if min_value < root_center_x and max_value > root_center_x:
|
||||
left_center_distance = root_center_x - min_value
|
||||
value = max_value - root_center_x
|
||||
line_value *= 1 + min(left_center_distance, value) / max(left_center_distance, value)
|
||||
# Edge bands are measured from the root page rectangle.
|
||||
|
||||
edge_top = other_rect.primary_slot + 0.2 * height
|
||||
edge_bot = other_rect.primary_slot + 0.8 * height
|
||||
if key_value < edge_top or score_value > edge_bot:
|
||||
line_value *= 4
|
||||
# The lower-edge boost uses forward-pass numbering and width counts,
|
||||
# because it is testing the material below the candidate gap.
|
||||
|
||||
if (key_value < edge_top and event_count >= 1 and wide_accumulator < root_width / 4) or (
|
||||
score_value > edge_bot and following_numbering_count >= 1 and following_edge_max_width < root_width / 4
|
||||
):
|
||||
line_value *= 9
|
||||
if 2 * width_value >= limit_number and reference_item > following_max_font_size + 0.5:
|
||||
line_value *= 100
|
||||
if lower_accumulator < math.inf:
|
||||
if lower_accumulator < root_center_x:
|
||||
line_value *= 100
|
||||
elif event_count >= 2 or following_numbering_count >= 2:
|
||||
line_value /= 4
|
||||
size = max(reference_item, following_max_font_size)
|
||||
# This test uses the full sweep extent, not the minimum gutter size.
|
||||
if (
|
||||
extent >= 0.99 * root_width
|
||||
and min_edge_position < root_center_x
|
||||
and max_edge > root_center_x
|
||||
and reference_item >= following_max_font_size + 0.5
|
||||
and item_value > size
|
||||
):
|
||||
line_value *= item_value / size
|
||||
|
||||
if secondary_item < 1 or after < 1:
|
||||
line_value *= 10
|
||||
|
||||
return _max_nan_propagating(0.0, line_value)
|
||||
@@ -0,0 +1,223 @@
|
||||
"""Recursive column splitting and column index assignment."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
Rect, rect_union, EMPTY_RECT, Line, info_weight, text_of_line, numbering_kind, numbering_value, _UNICODE_WHITESPACE_CLASS, _max_nan_propagating, _min_nan_propagating,
|
||||
)
|
||||
|
||||
from .gutters import (
|
||||
SweepEvent,
|
||||
SplitCandidate,
|
||||
ColumnDetectionContext,
|
||||
collect_gutter_candidates,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Assign column indexes to lines whose event is a start edge.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def assign_column_index(items: list[SweepEvent], other_number: int) -> None:
|
||||
"""Assign ``column_index`` to each gutter event that starts a column-owned line."""
|
||||
for column in items:
|
||||
if column.is_start:
|
||||
column.line.measure_slot = other_number
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Recursive split driver.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def recursive_split(
|
||||
context: ColumnDetectionContext,
|
||||
horizontal_events: list[SweepEvent], # horizontal events (sorted by F/L)
|
||||
vertical_events: list[SweepEvent], # vertical events (sorted by C/D)
|
||||
reference_rect: Rect, # root rect
|
||||
current_rect: Rect, # current sub-rect
|
||||
depth: int, # depth
|
||||
column_offset: int, # column-index offset
|
||||
) -> list[Rect]:
|
||||
if depth >= context.tertiary_slot:
|
||||
assign_column_index(horizontal_events, column_offset)
|
||||
return [current_rect]
|
||||
|
||||
split_candidates: list[SplitCandidate] = []
|
||||
if current_rect.bbox_height() >= context.measure_slot:
|
||||
collect_gutter_candidates(context, reference_rect, current_rect, horizontal_events, 1, current_rect.bbox_height(), context.option_slot, split_candidates)
|
||||
if current_rect.bbox_width() >= context.state_slot:
|
||||
collect_gutter_candidates(context, reference_rect, current_rect, vertical_events, 0, current_rect.bbox_width(), context.auxiliary_slot, split_candidates)
|
||||
|
||||
if len(split_candidates) <= 0:
|
||||
if current_rect.bbox_width() < 0.8 * reference_rect.bbox_width():
|
||||
assign_column_index(horizontal_events, column_offset)
|
||||
return [current_rect]
|
||||
# Examine gaps in b for vertical gutters (fallback)
|
||||
key_value = active_overlap_count = 0
|
||||
measure_item = local = 0
|
||||
gap_indices: list[int] = []
|
||||
gap_scan_index = 0
|
||||
while gap_scan_index < len(horizontal_events) - 1:
|
||||
width_value = horizontal_events[gap_scan_index].position
|
||||
event_is_start = horizontal_events[gap_scan_index].is_start
|
||||
candidate_item = horizontal_events[gap_scan_index].line
|
||||
if width_value > reference_rect.left + reference_rect.bbox_width() * 5 / 6:
|
||||
break
|
||||
if event_is_start:
|
||||
active_overlap_count += 1
|
||||
key_value += info_weight(candidate_item.char_stats)
|
||||
else:
|
||||
active_overlap_count -= 1
|
||||
key_value -= info_weight(candidate_item.char_stats)
|
||||
local = max(local, active_overlap_count)
|
||||
measure_item = max(measure_item, key_value)
|
||||
if event_is_start or active_overlap_count > 2 or width_value < reference_rect.left + reference_rect.bbox_width() / 6:
|
||||
gap_scan_index += 1
|
||||
continue
|
||||
previous_gap_index = gap_indices[-1] if gap_indices else None
|
||||
if previous_gap_index is not None and width_value < horizontal_events[previous_gap_index].position + reference_rect.bbox_width() / 10:
|
||||
gap_indices[-1] = gap_scan_index
|
||||
elif local >= 8 and measure_item >= 100:
|
||||
gap_indices.append(gap_scan_index)
|
||||
local = measure_item = 0
|
||||
gap_scan_index += 1
|
||||
|
||||
if len(gap_indices) <= 0 or len(gap_indices) > 2:
|
||||
assign_column_index(horizontal_events, column_offset)
|
||||
return [current_rect]
|
||||
if local < 4 or measure_item < 50:
|
||||
assign_column_index(horizontal_events, column_offset)
|
||||
return [current_rect]
|
||||
# Assign columns based on the discovered gaps
|
||||
out: list[Rect] = []
|
||||
cursor = 0
|
||||
for index in range(len(gap_indices) + 1):
|
||||
pos = gap_indices[index] if index < len(gap_indices) else len(horizontal_events)
|
||||
for event_index in range(cursor, pos):
|
||||
sweep_event = horizontal_events[event_index]
|
||||
if sweep_event.is_start:
|
||||
sweep_event.line.measure_slot = column_offset + len(out)
|
||||
end = horizontal_events[pos].position if pos < len(horizontal_events) else current_rect.right_edge()
|
||||
out.append(Rect(horizontal_events[cursor].position, end, current_rect.top, current_rect.bottom_edge()))
|
||||
cursor = pos + 1
|
||||
return out
|
||||
|
||||
# Pick best candidate split
|
||||
best: Optional[SplitCandidate] = None
|
||||
for count_item in split_candidates:
|
||||
if best is None or best.score < count_item.score:
|
||||
best = count_item
|
||||
assert best is not None # h is non-empty here
|
||||
|
||||
if best.direction == 0:
|
||||
# Horizontal split (vertical gutter): divide events into top / bottom halves
|
||||
upper_left, split_max = math.inf, -math.inf
|
||||
value, lower_right = math.inf, -math.inf
|
||||
upper_horizontal_events: list[SweepEvent] = []
|
||||
lower_horizontal_events: list[SweepEvent] = []
|
||||
for horizontal_event in horizontal_events:
|
||||
line = horizontal_event.line
|
||||
if line.top_edge() > best.start:
|
||||
upper_horizontal_events.append(horizontal_event)
|
||||
upper_left = min(upper_left, line.left_edge())
|
||||
split_max = max(split_max, line.right_edge())
|
||||
elif line.bottom_edge() < best.end:
|
||||
lower_horizontal_events.append(horizontal_event)
|
||||
value = min(value, line.left_edge())
|
||||
lower_right = max(lower_right, line.right_edge())
|
||||
upper_vertical_events: list[SweepEvent] = []
|
||||
split: list[SweepEvent] = []
|
||||
for vertical_event in vertical_events:
|
||||
if vertical_event.position > best.start:
|
||||
upper_vertical_events.append(vertical_event)
|
||||
elif vertical_event.position < best.end:
|
||||
split.append(vertical_event)
|
||||
upper = recursive_split(context, upper_horizontal_events, upper_vertical_events, reference_rect, Rect(upper_left, split_max, current_rect.top, best.end), depth + 1, column_offset)
|
||||
lower = recursive_split(
|
||||
context,
|
||||
lower_horizontal_events,
|
||||
split,
|
||||
reference_rect,
|
||||
Rect(value, lower_right, best.start, current_rect.bottom_edge()),
|
||||
depth + 1,
|
||||
column_offset + len(upper),
|
||||
)
|
||||
return upper + lower
|
||||
|
||||
# Vertical split (horizontal gutter): divide events into left / right halves
|
||||
split_max, left_bottom = -math.inf, math.inf
|
||||
right_top, right_bottom = -math.inf, math.inf
|
||||
left_events: list[SweepEvent] = []
|
||||
right_events: list[SweepEvent] = []
|
||||
left_vert: list[SweepEvent] = []
|
||||
right_vert: list[SweepEvent] = []
|
||||
for split_event in horizontal_events:
|
||||
if split_event.position < best.end:
|
||||
left_events.append(split_event)
|
||||
elif split_event.position > best.start:
|
||||
right_events.append(split_event)
|
||||
for event in vertical_events:
|
||||
line = event.line
|
||||
if line.left_edge() < best.end:
|
||||
left_vert.append(event)
|
||||
split_max = max(split_max, line.top_edge())
|
||||
left_bottom = min(left_bottom, line.bottom_edge())
|
||||
elif line.right_edge() > best.start:
|
||||
right_vert.append(event)
|
||||
right_top = max(right_top, line.top_edge())
|
||||
right_bottom = min(right_bottom, line.bottom_edge())
|
||||
left = recursive_split(
|
||||
context, left_events, left_vert, reference_rect, Rect(current_rect.left, best.start, split_max, left_bottom), depth + 1, column_offset
|
||||
)
|
||||
right = recursive_split(
|
||||
context,
|
||||
right_events,
|
||||
right_vert,
|
||||
reference_rect,
|
||||
Rect(best.end, current_rect.right_edge(), right_top, right_bottom),
|
||||
depth + 1,
|
||||
column_offset + len(left),
|
||||
)
|
||||
return left + right
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Build events, sort them, then start recursive splitting.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def detect_columns(column: ColumnDetectionContext) -> list[Rect]:
|
||||
"""Detect column rectangles and populate each line's column index."""
|
||||
horizontal_events: list[SweepEvent] = []
|
||||
vertical_events: list[SweepEvent] = []
|
||||
bbox = EMPTY_RECT
|
||||
for line in column.primary_slot:
|
||||
if line.bbox_width() <= 0 or line.bbox_height() <= 0:
|
||||
continue
|
||||
bbox = rect_union(bbox, line.secondary_slot)
|
||||
horizontal_events.append(SweepEvent(line, line.left_edge(), True))
|
||||
horizontal_events.append(SweepEvent(line, line.right_edge(), False))
|
||||
vertical_events.append(SweepEvent(line, line.bottom_edge(), True))
|
||||
vertical_events.append(SweepEvent(line, line.top_edge(), False))
|
||||
|
||||
# Sort by position, start events before end events, then line width.
|
||||
horizontal_events.sort(key=lambda split_event: (split_event.position, 0 if split_event.is_start else 1, split_event.line.bbox_width()))
|
||||
# Sort by position, start events before end events, then line height.
|
||||
vertical_events.sort(key=lambda split_event: (split_event.position, 0 if split_event.is_start else 1, split_event.line.bbox_height()))
|
||||
|
||||
return recursive_split(column, horizontal_events, vertical_events, bbox, bbox, 0, 0)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Public helper: produce the {left, right} dict list used by line merging #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def columns_to_x_bounds(column_rects: list[Rect]) -> list[dict]:
|
||||
"""Convert column rectangles to a ``[{left, right}, ...]`` table."""
|
||||
return [{"left": column.left, "right": column.right} for column in column_rects]
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,120 @@
|
||||
"""Per-page heading-candidate detection. This module builds and filters heading candidates from page blocks. It combines
|
||||
numbering recognition, chapter/appendix keywords, local neighbor geometry,
|
||||
font/style signals, cross-page rejection, and page-level candidate filtering
|
||||
before handing candidates to outline assembly.
|
||||
"""
|
||||
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
import regex as regex_module # Unicode \p{...} property classes.
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
from ..outline_assembly import HeadingCandidate, OutlineNode
|
||||
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_trim_unicode_ws,
|
||||
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
|
||||
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
|
||||
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
|
||||
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
from .keyword_tables import (
|
||||
_DICT_PATH,
|
||||
_DICTS,
|
||||
SECTION_KEYWORDS_TRIE,
|
||||
ABSTRACT_KEYWORDS_TRIE,
|
||||
REFERENCES_TRIE,
|
||||
APPENDIX_SECTION_TRIE,
|
||||
INTRODUCTION_SECTION_TRIE,
|
||||
BOX_KEYWORD_TRIE,
|
||||
KEYWORDS_SECTION_TRIE,
|
||||
CHAPTER_WORDS_TRIE,
|
||||
APPENDIX_KEYWORDS_TRIE,
|
||||
_normalize_text_key,
|
||||
ABSTRACT_KEYWORDS_SET,
|
||||
REFERENCES_SET,
|
||||
NUMBERED_PREFIX_RE,
|
||||
DEAD_DIGIT_RE,
|
||||
EQUATION_KEYWORDS_TRIE,
|
||||
ENGLISH_WORD_TO_NUMBER,
|
||||
ROMAN_NUMERAL_MAP,
|
||||
FORMULA_CHAR_WEIGHTS,
|
||||
)
|
||||
from .text_checks import (
|
||||
token_text_of_block,
|
||||
similar_style,
|
||||
is_heading_continuation,
|
||||
matches_abstract,
|
||||
matches_references,
|
||||
vertically_close,
|
||||
is_equation_adjacent_line,
|
||||
has_substantive_content,
|
||||
is_cover_page,
|
||||
clamp,
|
||||
token_to_number,
|
||||
letter_to_ordinal,
|
||||
)
|
||||
from .neighbors import (
|
||||
BlockNeighborCache,
|
||||
compute_bucket_span,
|
||||
neighbor_above,
|
||||
body_neighbor_above,
|
||||
neighbor_right,
|
||||
neighbor_right_peer,
|
||||
closest_body_neighbor_above,
|
||||
PageNeighborMap,
|
||||
)
|
||||
from .candidates import (
|
||||
PageScanState,
|
||||
push_candidate,
|
||||
make_heading_candidate,
|
||||
make_plain_candidate,
|
||||
make_body_heading_candidate,
|
||||
make_numbered_candidate,
|
||||
_di_count,
|
||||
_number_at_token_index,
|
||||
)
|
||||
from .detectors import (
|
||||
detect_numbered_heading,
|
||||
detect_labeled_heading,
|
||||
detect_chapter_appendix,
|
||||
detect_box_heading,
|
||||
classify_heading,
|
||||
is_acceptable_heading,
|
||||
safe_column_index,
|
||||
try_classify_heading,
|
||||
is_too_wide_for_heading,
|
||||
passes_neighbor_check,
|
||||
has_competing_labeled_heading,
|
||||
is_year_string,
|
||||
is_bibliography_entry,
|
||||
)
|
||||
from .style_detectors import (
|
||||
detect_font_heading,
|
||||
detect_heading_with_body,
|
||||
)
|
||||
from .page_scan import (
|
||||
scan_page_headings,
|
||||
DocCandidateCollector,
|
||||
filter_page_candidates,
|
||||
build_doc_heading_candidates,
|
||||
find_section_openers,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"SECTION_KEYWORDS_TRIE", "ABSTRACT_KEYWORDS_TRIE", "ABSTRACT_KEYWORDS_SET", "REFERENCES_TRIE", "REFERENCES_SET", "APPENDIX_SECTION_TRIE", "INTRODUCTION_SECTION_TRIE", "BOX_KEYWORD_TRIE", "KEYWORDS_SECTION_TRIE", "CHAPTER_WORDS_TRIE", "APPENDIX_KEYWORDS_TRIE",
|
||||
"ROMAN_NUMERAL_MAP", "ENGLISH_WORD_TO_NUMBER", "FORMULA_CHAR_WEIGHTS", "NUMBERED_PREFIX_RE", "DEAD_DIGIT_RE",
|
||||
"is_heading_continuation", "similar_style", "matches_abstract", "matches_references", "vertically_close", "is_equation_adjacent_line", "has_substantive_content", "is_cover_page", "token_to_number", "letter_to_ordinal",
|
||||
"BlockNeighborCache", "compute_bucket_span", "neighbor_above", "body_neighbor_above", "neighbor_right", "closest_body_neighbor_above",
|
||||
"PageNeighborMap", "PageScanState", "DocCandidateCollector", "filter_page_candidates",
|
||||
"classify_heading", "is_acceptable_heading", "try_classify_heading", "push_candidate", "detect_numbered_heading", "make_heading_candidate", "detect_labeled_heading", "make_body_heading_candidate", "detect_heading_with_body", "detect_chapter_appendix",
|
||||
"is_too_wide_for_heading", "passes_neighbor_check", "make_plain_candidate", "make_numbered_candidate", "has_competing_labeled_heading", "detect_font_heading", "scan_page_headings", "detect_box_heading",
|
||||
"build_doc_heading_candidates",
|
||||
]
|
||||
@@ -0,0 +1,214 @@
|
||||
"""Page scan state and heading-candidate constructors."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Any, Optional
|
||||
from ..outline_assembly import HeadingCandidate, OutlineNode
|
||||
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_trim_unicode_ws,
|
||||
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
|
||||
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
|
||||
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
|
||||
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
from .text_checks import matches_references
|
||||
from .neighbors import (
|
||||
neighbor_right,
|
||||
closest_body_neighbor_above,
|
||||
PageNeighborMap,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Main per-page heading state #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class PageScanState:
|
||||
"""Per-page heading scan state."""
|
||||
|
||||
__slots__ = ("secondary_slot", "primary_slot", "state_slot", "auxiliary_slot", "tertiary_slot", "option_slot", "measure_slot")
|
||||
|
||||
def __init__(self, doc, page):
|
||||
self.secondary_slot = doc # document state
|
||||
self.primary_slot = page
|
||||
self.state_slot = doc.primary_slot[page.page_index - 2] if page.page_index >= 2 else None # prev page
|
||||
self.auxiliary_slot = page.output_slot # blocks in original order
|
||||
self.tertiary_slot = PageNeighborMap(page) # neighbor map
|
||||
self.option_slot: list[HeadingCandidate] = [] # output candidates
|
||||
self.measure_slot: set = set() # set of block ids already pushed
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Push heading candidate into page state #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def push_candidate(page_scan: PageScanState, candidate: HeadingCandidate) -> None:
|
||||
"""Push a candidate into the page scan state."""
|
||||
page_scan.option_slot.append(candidate)
|
||||
page_scan.measure_slot.add(candidate.group_slot)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Heading-candidate builder.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def make_heading_candidate(page_scan: PageScanState, type_: int, block: Block, item_list: list[int],
|
||||
tokens: Optional[TokenView], title_tokens: Optional[TokenView], has_numbering_flag: bool = False) -> HeadingCandidate:
|
||||
"""Build a heading candidate and apply the spatial promotion rule."""
|
||||
neighbor = page_scan.tertiary_slot
|
||||
right_neighbor = neighbor_right(neighbor, block)
|
||||
# Spatial promotion to structural numbering: if a right-side neighbour exists, the block
|
||||
# has high skew (real horizontal text), its title ends in a colon-like
|
||||
# symbol, and its last line is nearly as wide as and right-aligned to
|
||||
# the neighbour -> promote the flag to true.
|
||||
if (
|
||||
not has_numbering_flag and right_neighbor is not None
|
||||
and block.previous_slot > 0.9 and title_tokens is not None
|
||||
):
|
||||
last_title_token = last_token(title_tokens)
|
||||
if last_title_token is not None and is_trimmable_token(last_title_token):
|
||||
mh_block = last_line_of(block)
|
||||
if (
|
||||
mh_block.bbox_width() > 0.7 * right_neighbor.bbox_width()
|
||||
and abs(mh_block.right_edge() - right_neighbor.right_edge()) < 2 * avg_char_width(mh_block)
|
||||
):
|
||||
has_numbering_flag = True
|
||||
prominent_flag = (
|
||||
type_ == 7
|
||||
or (len(item_list) > 0 and title_tokens is not None and matches_references(title_tokens))
|
||||
)
|
||||
return HeadingCandidate(
|
||||
type_=type_,
|
||||
page=page_scan.primary_slot,
|
||||
group_value=block,
|
||||
anchor=closest_body_neighbor_above(neighbor, block),
|
||||
numbering_value=item_list,
|
||||
tokens=tokens,
|
||||
title_tokens=title_tokens,
|
||||
has_numbering_flag=has_numbering_flag,
|
||||
prominent_flag=prominent_flag,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Shorthand heading-candidate builders #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def make_plain_candidate(page_scan: PageScanState, type_: int, block: Block) -> HeadingCandidate:
|
||||
"""Build a type-only candidate using the full block text."""
|
||||
return make_heading_candidate(page_scan, type_, block, [], None, tokenize_block(block), False)
|
||||
|
||||
|
||||
def make_body_heading_candidate(page_scan: PageScanState, type_: int, block: Block, tokens: TokenView) -> HeadingCandidate:
|
||||
"""Build a candidate from body-heading tokens."""
|
||||
return make_heading_candidate(page_scan, type_, block, [], None, trim_trailing_punct(tokens), True)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Composed-number heading-candidate builder #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def make_numbered_candidate(page_scan: PageScanState, block: Block, item_list: list[int], tokens: TokenView, title_tokens: TokenView) -> Optional[HeadingCandidate]:
|
||||
"""Build a numbered-heading candidate after the full reject-guard chain. The guard rejects empty numbering, weak single-token numbering, unsupported top-of-page continuations, alignment failures, and trailing-number continuation conflicts."""
|
||||
from ..labels import extract_structural_number # numbering-prefix detector
|
||||
|
||||
# Basic reject branch for empty, weak, or top-of-page continuation markers.
|
||||
if title_tokens.length <= 0:
|
||||
return None
|
||||
first_title_token = first_token(title_tokens)
|
||||
if (title_tokens.length == 1 and first_title_token is not None
|
||||
and first_title_token.primary_slot != 2 and first_title_token.primary_slot != 4 and first_title_token.secondary_slot != 2
|
||||
and not block.isolated_centered):
|
||||
return None
|
||||
if len(item_list) == 1 and item_list[0] == 1 and block.top_edge() < 0.3 * page_scan.primary_slot.bounds.bbox_height():
|
||||
from ..heading_detection import neighbor_right
|
||||
if neighbor_right(page_scan.tertiary_slot, block) is None:
|
||||
last_title_token = last_token(title_tokens)
|
||||
if last_title_token is not None and last_title_token.anchor_ranges and last_title_token.anchor_ranges[-1].line is last_line_of(block):
|
||||
return None
|
||||
|
||||
# If basic guards didn't trigger, examine multi-line patterns.
|
||||
reject = False
|
||||
if block.line_count() > 1:
|
||||
second_line = block.primary_slot[1]
|
||||
first_number_token = first_token(tokens)
|
||||
first_title_token = first_token(title_tokens)
|
||||
if first_number_token is not None and first_title_token is not None:
|
||||
left = first_anchor_span(first_number_token).left_edge()
|
||||
title_left = first_anchor_span(first_title_token).left_edge()
|
||||
if not (left < title_left and second_line.left_edge() > (left + title_left) / 2):
|
||||
# Check trailing tokens for c+1 continuation
|
||||
trailing_tokens = tokenize_block(block)
|
||||
trailing_tokens = trailing_tokens.slice(_di_count(trailing_tokens, block.line()))
|
||||
trailing_tokens = extract_structural_number(trailing_tokens)
|
||||
if trailing_tokens is None or trailing_tokens.length <= 0:
|
||||
reject = False
|
||||
elif block.measure_slot:
|
||||
reject = True
|
||||
else:
|
||||
if len(item_list) == 1 and trailing_tokens.length <= 2:
|
||||
trailing_first_token = trailing_tokens.token_at(0)
|
||||
if trailing_first_token is not None:
|
||||
value = token_numeric_value(trailing_first_token)
|
||||
# Strict equality on the raw Number, no truncation
|
||||
# (a fractional value never
|
||||
# equals the integer c[0]+1).
|
||||
reject = (not math.isnan(value) and value == item_list[0] + 1)
|
||||
else:
|
||||
reject = False
|
||||
else:
|
||||
reject = False
|
||||
if reject:
|
||||
return None
|
||||
return make_heading_candidate(page_scan, 1, block, item_list, tokens, title_tokens, False)
|
||||
|
||||
|
||||
def _di_count(tokens: TokenView, line) -> int:
|
||||
"""count tokens belonging to ``line`` starting from index 0."""
|
||||
count_item = 0
|
||||
for index_value in range(tokens.length):
|
||||
token_value = tokens.token_at(index_value)
|
||||
if token_value is None or token_value.line() is not line:
|
||||
break
|
||||
count_item += 1
|
||||
return count_item
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Numbered heading detector #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _number_at_token_index(tokens: TokenView, index: int) -> int:
|
||||
"""Try to extract a numbering value at index ``b_idx`` of a token view. Returns 0 if not a number-followed-by-separator, else the number. """
|
||||
if tokens.length < index + 2:
|
||||
return 0
|
||||
token = tokens.token_at(index)
|
||||
if token is None or token.type != 1:
|
||||
return 0
|
||||
next_tok = tokens.token_at(index + 1)
|
||||
if next_tok is None:
|
||||
return 0
|
||||
from ..labels import PERIOD_CHARS as period_chars
|
||||
if not (
|
||||
next_tok.str in period_chars
|
||||
or next_tok.str in (")", "]", ".", "。", "。", ")", "]", "】")
|
||||
):
|
||||
return 0
|
||||
val = token_numeric_value(token)
|
||||
if math.isnan(val) or val <= 0 or val >= 1000:
|
||||
return 0
|
||||
return int(val)
|
||||
@@ -0,0 +1,492 @@
|
||||
"""Numbered, labeled, chapter/appendix, and box heading detectors plus acceptability checks."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Any, Optional
|
||||
from ..outline_assembly import HeadingCandidate, OutlineNode
|
||||
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_trim_unicode_ws,
|
||||
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
|
||||
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
|
||||
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
|
||||
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
from .keyword_tables import (
|
||||
APPENDIX_SECTION_TRIE,
|
||||
BOX_KEYWORD_TRIE,
|
||||
CHAPTER_WORDS_TRIE,
|
||||
APPENDIX_KEYWORDS_TRIE,
|
||||
ROMAN_NUMERAL_MAP,
|
||||
)
|
||||
from .text_checks import (
|
||||
similar_style,
|
||||
matches_abstract,
|
||||
matches_references,
|
||||
has_substantive_content,
|
||||
clamp,
|
||||
token_to_number,
|
||||
letter_to_ordinal,
|
||||
)
|
||||
from .neighbors import (
|
||||
neighbor_above,
|
||||
body_neighbor_above,
|
||||
neighbor_right,
|
||||
closest_body_neighbor_above,
|
||||
)
|
||||
from .candidates import (
|
||||
PageScanState,
|
||||
make_heading_candidate,
|
||||
make_plain_candidate,
|
||||
make_numbered_candidate,
|
||||
)
|
||||
|
||||
|
||||
def detect_numbered_heading(page_scan: PageScanState, block: Block, tokens: TokenView) -> Optional[HeadingCandidate]:
|
||||
"""Identify "1.2.3" / "[1]" style numbered heading prefixes."""
|
||||
item_list: list[int] = []
|
||||
at_value = None
|
||||
for entry in enumerate_tokens(tokens):
|
||||
index = entry["index"]
|
||||
token = entry["token"]
|
||||
if token.type == 1:
|
||||
if len(item_list) >= 4:
|
||||
break
|
||||
val = token_numeric_value(token)
|
||||
if math.isnan(val) or val <= 0 or val >= 20:
|
||||
break
|
||||
if len(token.str) >= 3:
|
||||
break
|
||||
item_list.append(int(val))
|
||||
if not token.boundary_slot:
|
||||
continue
|
||||
at_value = tokens.token_at(index + 1)
|
||||
if (
|
||||
index + 2 < tokens.length
|
||||
and at_value is not None
|
||||
and (is_word_token(at_value) or at_value.type == 6)
|
||||
and tokens.token_at(index + 2) is not None
|
||||
and tokens.token_at(index + 2).type == 1
|
||||
):
|
||||
break
|
||||
# Strong numbering, prominent style, or a viable separator token is
|
||||
# enough to build a numbered-heading candidate.
|
||||
if (
|
||||
len(item_list) > 1
|
||||
or heading_score(block) > page_scan.primary_slot.primary_slot.primary_slot + 1
|
||||
or (not block.measure_slot and at_value is not None and (
|
||||
at_value.primary_slot in (2, 4) or at_value.secondary_slot == 2 or at_value.type == 4
|
||||
or at_value.str == "." or at_value.str == "|"
|
||||
))
|
||||
):
|
||||
return make_numbered_candidate(
|
||||
page_scan, block, item_list,
|
||||
tokens.slice(0, index + 1),
|
||||
tokens.slice(index + 1),
|
||||
)
|
||||
return None
|
||||
# Accept period-like punctuation or a symbol token as a numbering
|
||||
# separator.
|
||||
if token.str in (".", ".", "。", "。") or token.type == 4:
|
||||
prev = tokens.token_at(index - 1) # token_at(-1) returns None
|
||||
if prev is None or prev.type != 1:
|
||||
break
|
||||
if not token.boundary_slot:
|
||||
continue
|
||||
return make_numbered_candidate(
|
||||
page_scan, block, item_list,
|
||||
tokens.slice(0, index + 1), tokens.slice(index + 1),
|
||||
)
|
||||
if len(item_list) <= 0 or token.type != 2:
|
||||
break
|
||||
if token.primary_slot not in (2, 4):
|
||||
break
|
||||
if (
|
||||
len(item_list) > 1
|
||||
or len(token.str) >= 3
|
||||
or tokens.length - index >= 3
|
||||
):
|
||||
return make_numbered_candidate(
|
||||
page_scan, block, item_list,
|
||||
tokens.slice(0, index), tokens.slice(index),
|
||||
)
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Complex numbering format detector.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def detect_labeled_heading(page_scan: PageScanState, block: Block, tokens: TokenView) -> Optional[HeadingCandidate]:
|
||||
"""Detect Roman, letter, CJK, and mixed-numbering headings."""
|
||||
if tokens.length <= 1:
|
||||
return None
|
||||
first = tokens.token_at(0)
|
||||
second = tokens.token_at(1)
|
||||
if first is None or second is None:
|
||||
return None
|
||||
# Roman numeral path
|
||||
roman = ROMAN_NUMERAL_MAP.get(first.str)
|
||||
if roman is not None and is_word_token(second) and second.str in "..。。:)":
|
||||
prefix = tokens.slice(0, 2)
|
||||
return make_heading_candidate(page_scan, 2, block, [roman], prefix, tokens.slice(prefix.length))
|
||||
# CJK number path
|
||||
cjk_pos = "一二三四五六七八九十".find(first.str)
|
||||
if cjk_pos >= 0 and is_word_token(second):
|
||||
prefix = tokens.slice(0, 2)
|
||||
rest = tokens.slice(prefix.length)
|
||||
if rest.length <= 0:
|
||||
return None
|
||||
return make_heading_candidate(page_scan, 3, block, [cjk_pos + 1], prefix, rest)
|
||||
# Letter path
|
||||
if tokens.length <= 1 or (block.char_stats.secondary_slot == 3 and (block.line_count() > 1 or is_punct_category(block.char_stats.tertiary_slot))):
|
||||
return None
|
||||
letter_val = letter_to_ordinal(first.str)
|
||||
if letter_val is None:
|
||||
return None
|
||||
if second.str == "." or second.str == ")":
|
||||
value = letter_val
|
||||
else:
|
||||
first_anchor = first_anchor_span(first)
|
||||
second_anchor = first_anchor_span(second)
|
||||
if (
|
||||
not first.boundary_slot
|
||||
or first_anchor is second_anchor
|
||||
or second_anchor.left_edge() < first_anchor.right_edge() + first_anchor.bbox_width()
|
||||
or heading_score(block) < page_scan.primary_slot.primary_slot.primary_slot + 1
|
||||
or letter_count(block.char_stats) / tokens.length < 2
|
||||
):
|
||||
return None
|
||||
value = letter_val
|
||||
item_list: list[int] = [value]
|
||||
prefix = tokens.slice(0, 2 if is_word_token(second) else 1)
|
||||
rest = tokens.slice(prefix.length)
|
||||
if second.str == "." and not second.boundary_slot and rest.length >= 2:
|
||||
first_rest = first_token(rest)
|
||||
if first_rest is not None and first_rest.type == 1:
|
||||
heading = token_numeric_value(first_rest)
|
||||
if math.isnan(heading) or heading <= 0 or heading >= 20:
|
||||
return None
|
||||
item_list.append(int(heading))
|
||||
rest = rest.slice(1)
|
||||
first_rest = first_token(rest)
|
||||
if rest.length > 0 and first_rest is not None and is_word_token(first_rest):
|
||||
rest = rest.slice(1)
|
||||
if rest.length <= 0:
|
||||
return None
|
||||
prefix = tokens.slice(0, tokens.length - rest.length)
|
||||
return make_heading_candidate(page_scan, 4, block, item_list, prefix, rest)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Chapter, appendix, and box-style dispatch.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def detect_chapter_appendix(page_scan: PageScanState, other_block: Block) -> Optional[HeadingCandidate]:
|
||||
""". Match "Chapter X" / "Appendix X" / box-N / etc."""
|
||||
candidate_item = heading_score(other_block)
|
||||
if candidate_item <= page_scan.primary_slot.primary_slot.primary_slot + 0.1:
|
||||
return None
|
||||
flag = (
|
||||
other_block.isolated_centered or candidate_item > page_scan.secondary_slot.secondary_slot.primary_slot + 0.1
|
||||
and (other_block.bold_frac() > 0.9 or is_upper_dominant(other_block.char_stats) or candidate_item > 1.5 * page_scan.secondary_slot.secondary_slot.primary_slot)
|
||||
)
|
||||
tokens = tokenize_block(other_block)
|
||||
match = None
|
||||
if flag:
|
||||
match = trie_prefix_match(CHAPTER_WORDS_TRIE, tokens)
|
||||
if flag and match is not None:
|
||||
value = token_to_number(tokens.token_at(match.length))
|
||||
if value is None:
|
||||
return None
|
||||
prefix = tokens.slice(0, skip_bracketed_word(tokens, match.length + 1))
|
||||
return make_heading_candidate(page_scan, 8, other_block, [value], prefix, tokens.slice(prefix.length))
|
||||
if flag:
|
||||
match = trie_prefix_match(APPENDIX_SECTION_TRIE, tokens)
|
||||
if match is not None:
|
||||
prefix = tokens.slice(0, skip_bracketed_word(tokens, match.length))
|
||||
return make_heading_candidate(page_scan, 9, other_block, [], prefix, tokens.slice(prefix.length))
|
||||
match = trie_prefix_match(APPENDIX_KEYWORDS_TRIE, tokens)
|
||||
if match is not None:
|
||||
next_item = tokens.token_at(match.length)
|
||||
val = token_to_number(next_item) or (letter_to_ordinal(next_item.str) if next_item is not None else None)
|
||||
if not flag and val is None:
|
||||
return None
|
||||
item_list = [val] if val is not None else []
|
||||
prefix = tokens.slice(0, skip_bracketed_word(tokens, match.length + (1 if val is not None else 0)))
|
||||
return make_heading_candidate(page_scan, 10, other_block, item_list, prefix, tokens.slice(prefix.length))
|
||||
return None
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Box-format heading.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def detect_box_heading(page_scan: PageScanState, other_block: Block) -> Optional[HeadingCandidate]:
|
||||
"""match "Box N" pattern."""
|
||||
tokens = tokenize_block(other_block)
|
||||
match = trie_prefix_match(BOX_KEYWORD_TRIE, tokens)
|
||||
if match is None:
|
||||
return None
|
||||
rest = tokens.slice(match.length)
|
||||
if rest.length <= 0 or rest.token_at(0).type != 1:
|
||||
return None
|
||||
val = token_numeric_value(rest.token_at(0))
|
||||
if math.isnan(val) or val <= 0:
|
||||
return None
|
||||
prefix = tokens.slice(0, skip_bracketed_word(tokens, match.length + 1))
|
||||
return make_heading_candidate(page_scan, 12, other_block, [int(val)], prefix, tokens.slice(prefix.length))
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Heading-type dispatcher.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def classify_heading(page_scan: PageScanState, other_block: Block) -> HeadingCandidate:
|
||||
""". Sequential dispatch through type detectors; fallback to the font-position classifier."""
|
||||
tokens = tokenize_block(other_block)
|
||||
heading = detect_chapter_appendix(page_scan, other_block)
|
||||
if heading is None:
|
||||
heading = detect_box_heading(page_scan, other_block)
|
||||
if heading is None:
|
||||
heading = detect_numbered_heading(page_scan, other_block, tokens)
|
||||
if heading is None:
|
||||
heading = detect_labeled_heading(page_scan, other_block, tokens)
|
||||
if heading is not None:
|
||||
return heading
|
||||
type_code = 7 if matches_references(tokens) else (5 if matches_abstract(tokens) else 0)
|
||||
return make_plain_candidate(page_scan, type_code, other_block)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Heading acceptance gate.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def is_acceptable_heading(page_scan: PageScanState, other_heading_candidate: HeadingCandidate) -> bool:
|
||||
""". The big "is this an acceptable heading?" gate."""
|
||||
heading = other_heading_candidate.group_slot
|
||||
if heading.bbox_height() >= 2 * heading.bbox_width() or info_weight(heading.char_stats) <= 3 or heading.line_count() > 5 or heading.char_count() >= 300:
|
||||
return False
|
||||
page_height = page_scan.primary_slot.bounds.bbox_height()
|
||||
if heading.bottom_edge() > 0.95 * page_height:
|
||||
return False
|
||||
score = heading_score(heading)
|
||||
doc_group = page_scan.secondary_slot.secondary_slot.primary_slot
|
||||
if score <= page_scan.primary_slot.primary_slot.primary_slot + 0.5 and score <= doc_group + 0.5 and not other_heading_candidate.is_prominent:
|
||||
return False
|
||||
width = page_scan.primary_slot.bounds.bbox_width()
|
||||
if (
|
||||
(heading.left_edge() > 0.55 * width and score <= doc_group + 5)
|
||||
or heading.left_edge() > 0.75 * width
|
||||
or (
|
||||
heading.left_edge() > 0.4 * width
|
||||
and heading.center_x() > 0.6 * width
|
||||
and page_scan.primary_slot.primary_slot.secondary_slot > min(1000, page_scan.secondary_slot.secondary_slot.secondary_slot)
|
||||
)
|
||||
):
|
||||
return False
|
||||
col_bottom = page_scan.primary_slot.tertiary_slot[safe_column_index(heading)] if 0 <= safe_column_index(heading) < len(page_scan.primary_slot.tertiary_slot) else None
|
||||
if (
|
||||
heading.bbox_width() < 0.2 * width and col_bottom is not None
|
||||
and col_bottom.bbox_width() < 0.2 * width and col_bottom.bbox_height() > 1.5 * col_bottom.bbox_width()
|
||||
):
|
||||
return False
|
||||
neighbor = neighbor_right(page_scan.tertiary_slot, heading)
|
||||
gap = heading.bottom_edge() - neighbor.top_edge() if neighbor is not None else math.inf
|
||||
line_gap = page_scan.primary_slot.primary_slot.tertiary_slot - page_scan.primary_slot.primary_slot.primary_slot
|
||||
if gap < 0.9 * line_gap:
|
||||
return False
|
||||
above = neighbor_above(page_scan.tertiary_slot, heading)
|
||||
above_gap = above.bottom_edge() - heading.top_edge() if above is not None else math.inf
|
||||
if above_gap < 0.9 * line_gap:
|
||||
return False
|
||||
if (
|
||||
page_scan.state_slot is not None
|
||||
and not page_scan.state_slot.measure_slot
|
||||
and (
|
||||
page_scan.state_slot.primary_slot.secondary_slot < clamp(page_scan.secondary_slot.secondary_slot.secondary_slot, 200, 500)
|
||||
or not page_scan.state_slot.state_slot
|
||||
)
|
||||
and score > doc_group + 0.5
|
||||
):
|
||||
return True
|
||||
doc_right_neighbor = page_scan.secondary_slot.secondary_slot.measure_slot
|
||||
previous_page_left_neighbor = page_scan.state_slot.primary_slot.option_slot if page_scan.state_slot is not None else math.nan
|
||||
previous_page_height = page_scan.state_slot.bounds.bbox_height() if page_scan.state_slot is not None else math.nan
|
||||
centered_flag = heading.isolated_centered
|
||||
if (
|
||||
previous_page_left_neighbor <= doc_right_neighbor
|
||||
and (score <= doc_group + 1.5 or (score <= doc_group + 5 and not centered_flag))
|
||||
or heading.weighted_ratio_primary < 0.5 * page_scan.secondary_slot.secondary_slot.auxiliary_slot
|
||||
):
|
||||
return False
|
||||
body_neighbor = closest_body_neighbor_above(page_scan.tertiary_slot, heading)
|
||||
# Reject candidates that are separated from a classified above-neighbor, or
|
||||
# whose own content is more formula-like than heading-like.
|
||||
if body_neighbor is not None and heading.bottom_edge() - body_neighbor.top_edge() > 2 * heading.bbox_height() and body_neighbor.marker_slot != 0:
|
||||
return False
|
||||
previous_block = page_scan.auxiliary_slot[heading.orig_index - 1] if 0 <= heading.orig_index - 1 < len(page_scan.auxiliary_slot) else None
|
||||
next_block = page_scan.auxiliary_slot[heading.orig_index + 1] if 0 <= heading.orig_index + 1 < len(page_scan.auxiliary_slot) else None
|
||||
if has_substantive_content(heading, previous_block, next_block):
|
||||
return False
|
||||
return (
|
||||
(previous_page_left_neighbor > doc_right_neighbor + 0.1 * previous_page_height
|
||||
and (score > doc_group + 2
|
||||
or (previous_page_left_neighbor > doc_right_neighbor + 0.2 * previous_page_height
|
||||
and body_neighbor is not None and neighbor is not None
|
||||
and gap > neighbor.avg_font_size())))
|
||||
or (centered_flag and (neighbor is None or neighbor.marker_slot == 0))
|
||||
or score > 1.5 * doc_group
|
||||
or (len(other_heading_candidate.numbering) == 1 and other_heading_candidate.numbering[0] == 1 and above is None)
|
||||
)
|
||||
|
||||
|
||||
def safe_column_index(block) -> int:
|
||||
"""Safe wrapper for hh that handles missing H field."""
|
||||
from ..stats import column_index_of
|
||||
# Empty containers must return -1; returning 0 would index a real column.
|
||||
return column_index_of(block)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Heading classification + acceptability gate #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def try_classify_heading(page_scan: PageScanState, other_block: Block) -> Optional[HeadingCandidate]:
|
||||
"""Try to build a candidate for a block, then apply rejection gates."""
|
||||
candidate = classify_heading(page_scan, other_block)
|
||||
if len(candidate.numbering) > 1:
|
||||
return None
|
||||
if candidate.type in (8, 9, 10):
|
||||
return candidate
|
||||
if candidate.type == 12:
|
||||
return None
|
||||
return candidate if is_acceptable_heading(page_scan, candidate) else None
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Additional heading detectors and neighbor gates.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def is_too_wide_for_heading(page_scan: PageScanState, other_block: Block) -> bool:
|
||||
""". Block is too wide / central to be a heading."""
|
||||
width = other_block.bbox_width()
|
||||
if width > 0.7 * page_scan.primary_slot.bounds.bbox_width() / 2 or width > 0.7 * page_scan.secondary_slot.secondary_slot.option_slot:
|
||||
return True
|
||||
count = 0
|
||||
for heading in range(page_scan.primary_slot.page_index - 1, page_scan.primary_slot.page_index + 2):
|
||||
if 0 < heading <= len(page_scan.secondary_slot.primary_slot):
|
||||
page = page_scan.secondary_slot.primary_slot[heading - 1]
|
||||
if width > 0.7 * page.primary_slot.previous_slot:
|
||||
count += 1
|
||||
return count >= 2
|
||||
|
||||
|
||||
def passes_neighbor_check(page_scan: PageScanState, other_block: Block) -> bool:
|
||||
"""Block-level neighbor-aware acceptance gate. Returns True when the caller should reject the block."""
|
||||
if is_too_wide_for_heading(page_scan, other_block):
|
||||
return False
|
||||
blocks = page_scan.auxiliary_slot
|
||||
prev_idx = other_block.orig_index - 1
|
||||
candidate_item = blocks[prev_idx] if 0 <= prev_idx < len(blocks) else None
|
||||
overlap = candidate_item is not None and y_overlaps(other_block, candidate_item)
|
||||
if overlap and is_too_wide_for_heading(page_scan, candidate_item):
|
||||
return False
|
||||
next_idx = other_block.orig_index + 1
|
||||
candidate_item = blocks[next_idx] if 0 <= next_idx < len(blocks) else None
|
||||
next_overlap = candidate_item is not None and y_overlaps(other_block, candidate_item)
|
||||
if next_overlap and is_too_wide_for_heading(page_scan, candidate_item):
|
||||
return False
|
||||
if not overlap and not next_overlap:
|
||||
return False
|
||||
candidate_item = neighbor_right(page_scan.tertiary_slot, other_block)
|
||||
if candidate_item is not None and is_too_wide_for_heading(page_scan, candidate_item):
|
||||
return False
|
||||
if candidate_item is not None and not candidate_item.is_body_paragraph and candidate_item.line_count() > 3 and candidate_item.bbox_height() > 0.8 * candidate_item.bbox_width():
|
||||
return True
|
||||
# Compare against the closest above-neighbor with a width threshold derived
|
||||
# from this block's first line.
|
||||
above_or_overlap = closest_body_neighbor_above(page_scan.tertiary_slot, other_block)
|
||||
threshold = 4 * avg_char_width(other_block.line())
|
||||
if (above_or_overlap is not None and candidate_item is not above_or_overlap
|
||||
and x_aligned(other_block, above_or_overlap, threshold)
|
||||
and above_or_overlap.state_slot == 0
|
||||
and is_too_wide_for_heading(page_scan, above_or_overlap)):
|
||||
return False
|
||||
keyword_match = body_neighbor_above(page_scan.tertiary_slot, other_block)
|
||||
if (keyword_match is not None
|
||||
and x_aligned(other_block, keyword_match, threshold)
|
||||
and keyword_match.state_slot == 0
|
||||
and is_too_wide_for_heading(page_scan, keyword_match)):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def has_competing_labeled_heading(page_scan: PageScanState, other_heading_candidate: HeadingCandidate, candidate_block: Block) -> bool:
|
||||
"""Cross-page reject check for competing labeled-heading siblings."""
|
||||
if candidate_block.type != 0 or candidate_block.char_count() >= 500:
|
||||
return False
|
||||
block = other_heading_candidate.group_slot
|
||||
if not similar_style(block, candidate_block) or abs(block.top_edge() - candidate_block.top_edge()) >= 5 * block.bbox_height():
|
||||
return False
|
||||
other_candidate = detect_labeled_heading(page_scan, candidate_block, tokenize_block(candidate_block))
|
||||
if other_candidate is None or other_heading_candidate.type != other_candidate.type:
|
||||
return False
|
||||
return abs(other_candidate.numbering[0] - other_heading_candidate.numbering[0]) >= 1
|
||||
|
||||
|
||||
def is_year_string(text: str) -> bool:
|
||||
"""True iff the text parses to a plausible year (1700..2100)."""
|
||||
value = to_number(text)
|
||||
return not math.isnan(value) and 1700 < value < 2100
|
||||
|
||||
|
||||
def is_bibliography_entry(block: Block, other_number: int = -1) -> bool:
|
||||
"""True iff ``block`` looks like a bibliography entry."""
|
||||
if other_number < 0:
|
||||
other_number = 0
|
||||
for line in block:
|
||||
reference_item = numbering_value(line)
|
||||
if not math.isnan(reference_item) and 0 < reference_item <= 9999:
|
||||
other_number += 1
|
||||
if other_number < 2 and block.char_count() / max(1, other_number) > 300:
|
||||
return False
|
||||
year = 0
|
||||
digit = 0
|
||||
word = 0
|
||||
period_after_word = 0
|
||||
word_state = 0
|
||||
tokens = tokenize_block(block)
|
||||
for entry in enumerate_tokens(tokens):
|
||||
state_item = entry["token"]
|
||||
if is_word_token(state_item):
|
||||
if state_item.type == 3 and word_state == 1:
|
||||
period_after_word += 1
|
||||
word_state = 0
|
||||
elif state_item.type == 1:
|
||||
key_value = token_numeric_value(state_item)
|
||||
if 0 < key_value < 1000:
|
||||
digit += 1
|
||||
elif is_year_string(state_item.str):
|
||||
year += 1
|
||||
elif state_item.type == 2:
|
||||
word += 1
|
||||
word_state += 1
|
||||
if word < 0.1 * tokens.length:
|
||||
return False
|
||||
return digit >= 1.5 * other_number or year >= 0.5 * other_number or period_after_word >= 0.5 * other_number
|
||||
@@ -0,0 +1,96 @@
|
||||
"""Dictionary-backed keyword tries, keyword sets, and numbering tables."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import regex as regex_module # Unicode \p{...} property classes.
|
||||
from pathlib import Path
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_trim_unicode_ws,
|
||||
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
|
||||
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
|
||||
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
|
||||
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Dictionary tries (case-folded) #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
_DICT_PATH = Path(__file__).parent.parent / "data" / "dictionaries.json"
|
||||
_DICTS = json.loads(_DICT_PATH.read_text(encoding="utf-8"))
|
||||
|
||||
SECTION_KEYWORDS_TRIE = build_trie(_DICTS.get("section_keywords", []), set_case_fold(TrieConfig(), True)) # general sections
|
||||
ABSTRACT_KEYWORDS_TRIE = build_trie(_DICTS.get("abstract_keywords", []), set_case_fold(TrieConfig(), True)) # abstract
|
||||
REFERENCES_TRIE = build_trie(_DICTS.get("references", []), set_case_fold(TrieConfig(), True)) # references
|
||||
APPENDIX_SECTION_TRIE = build_trie(_DICTS.get("appendices_dict", []), set_case_fold(TrieConfig(), True)) # appendix
|
||||
INTRODUCTION_SECTION_TRIE = build_trie(_DICTS.get("introduction_dict", []), set_case_fold(TrieConfig(), True)) # introduction
|
||||
BOX_KEYWORD_TRIE = build_trie(["box"], set_case_fold(TrieConfig(), True))
|
||||
KEYWORDS_SECTION_TRIE = build_trie(_DICTS.get("keywords_dict", []), set_case_fold(TrieConfig(), True)) # keywords
|
||||
CHAPTER_WORDS_TRIE = build_trie(_DICTS.get("chapter_words", []), set_case_fold(TrieConfig(), True)) # chapter
|
||||
APPENDIX_KEYWORDS_TRIE = build_trie(_DICTS.get("appendix_keywords", []), set_case_fold(TrieConfig(), True)) # appendix (hi)
|
||||
|
||||
# Whole-text lookup sets use normalized lowercase strings. The normalization is
|
||||
# NFD -> strip combining marks (U+0300-U+036F) -> NFC; it is diacritic stripping,
|
||||
# not compatibility folding.
|
||||
def _normalize_text_key(text: str) -> str:
|
||||
return _strip_diacritics(text)
|
||||
|
||||
# Whole-text lookup sets for abstract and references headings.
|
||||
# Abstract headings are matched diacritic-insensitively; references are not.
|
||||
ABSTRACT_KEYWORDS_SET = frozenset(_strip_diacritics(text_value.lower()) for text_value in _DICTS.get("abstract_keywords", []))
|
||||
REFERENCES_SET = frozenset(text_value.lower() for text_value in _DICTS.get("references", []))
|
||||
|
||||
|
||||
# Numbered heading prefix: leading ASCII/fullwidth 1-9, followed by Unicode
|
||||
# numeric code points, punctuation, and whitespace or uppercase lookahead. The
|
||||
# leading class deliberately excludes fullwidth zero (U+FF10).
|
||||
NUMBERED_PREFIX_RE = regex_module.compile(r"^([1-91-9]\p{Number}*)[ .-](?:[" + _UNICODE_WHITESPACE_CLASS + r"]|\p{Lu})")
|
||||
# Equation separator fallback. This intentionally matches only the literal
|
||||
# string pattern around ``p{Number}``, so the branch remains inert for ordinary
|
||||
# numeric text.
|
||||
DEAD_DIGIT_RE = re.compile(r"^.p\{Number\}+.$")
|
||||
|
||||
# Trie of equation-like keywords ("equation", "eqn", "eq", plus multilingual
|
||||
# variants).
|
||||
EQUATION_KEYWORDS_TRIE = build_trie(
|
||||
[
|
||||
"equation", "equation.", "eqn", "eqn.", "eq", "eq.",
|
||||
"ecuación", "equação", "gleichung", "equazione", "ekvation",
|
||||
"yhtälö", "ligning", "persamaan", "denklem", "ecuația",
|
||||
"equació", "rovnica", "rovnice", "równanie", "vergelijking",
|
||||
"jednadžba", "jöfnu", "võrrand", "vienādojums", "lygtis",
|
||||
"enačba", "egyenlet", "phương trình", "εξίσωση",
|
||||
"方程", "방정식", "уравнение", "рівняння", "раўнанне", "једначина",
|
||||
],
|
||||
set_case_fold(TrieConfig(), True),
|
||||
)
|
||||
|
||||
|
||||
# Roman and English number words used by heading numbering detectors.
|
||||
ENGLISH_WORD_TO_NUMBER = {
|
||||
"one": 1, "two": 2, "three": 3, "four": 4, "five": 5, "six": 6,
|
||||
"seven": 7, "eight": 8, "nine": 9, "ten": 10, "eleven": 11,
|
||||
"twelve": 12, "thirteen": 13, "fourteen": 14, "fifteen": 15,
|
||||
"sixteen": 16, "seventeen": 17, "eighteen": 18, "nineteen": 19, "twenty": 20,
|
||||
}
|
||||
ROMAN_NUMERAL_MAP = {
|
||||
"I": 1, "II": 2, "III": 3, "IV": 4, "V": 5, "VI": 6, "VII": 7,
|
||||
"VIII": 8, "IX": 9, "X": 10, "XI": 11, "XII": 12, "XIII": 13,
|
||||
"XIV": 14, "XV": 15, "XVI": 16, "XVII": 17, "XVIII": 18, "XIX": 19, "XX": 20,
|
||||
}
|
||||
|
||||
|
||||
# Special-character weights used by equation-content scoring.
|
||||
FORMULA_CHAR_WEIGHTS = {
|
||||
"=": 10, "{": 5, "}": 5, "+": 5, "/": 3, "*": 3,
|
||||
"-": 1, "~": 1, "[": 1, "]": 1, "(": 1, ")": 1,
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
"""Per-page block neighborhood maps and neighbor lookups."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Any, Optional
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_trim_unicode_ws,
|
||||
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
|
||||
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
|
||||
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
|
||||
)
|
||||
|
||||
from .text_checks import clamp
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Per-page neighbor map.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class BlockNeighborCache:
|
||||
"""Per-block neighbor cache populated by the page neighbor map."""
|
||||
|
||||
__slots__ = ("state_slot", "tertiary_slot", "measure_slot", "auxiliary_slot", "primary_slot", "secondary_slot", "option_slot")
|
||||
|
||||
def __init__(self):
|
||||
self.state_slot = False # initialized flag
|
||||
self.tertiary_slot = None # closest body block below
|
||||
self.measure_slot = None # block 1-column-left
|
||||
self.auxiliary_slot = None # next block to the right
|
||||
self.primary_slot = None # earlier body block above
|
||||
self.secondary_slot = None # nearest body block above
|
||||
self.option_slot = None # block 1-column-right peer
|
||||
|
||||
|
||||
def compute_bucket_span(neighbor_map, block) -> dict:
|
||||
"""Compute the inclusive horizontal bucket span for a block."""
|
||||
start_bucket = int(clamp(math.floor(block.left_edge() / neighbor_map.tertiary_slot), 0, neighbor_map.secondary_slot - 1))
|
||||
end_bucket = int(clamp(math.ceil(block.right_edge() / neighbor_map.tertiary_slot), 0, neighbor_map.secondary_slot - 1))
|
||||
return {"start_bucket": start_bucket, "end_bucket": end_bucket}
|
||||
|
||||
|
||||
def neighbor_above(neighbor_map, other_block: Block) -> Optional[Block]:
|
||||
"""closest 'j' neighbor (block above)."""
|
||||
width_value = neighbor_map.primary_slot[other_block.orig_index] if other_block.orig_index < len(neighbor_map.primary_slot) else None
|
||||
return width_value.tertiary_slot if width_value is not None else None
|
||||
|
||||
|
||||
def body_neighbor_above(neighbor_map, other_block: Block) -> Optional[Block]:
|
||||
"""closest 'g' neighbor."""
|
||||
width_value = neighbor_map.primary_slot[other_block.orig_index] if other_block.orig_index < len(neighbor_map.primary_slot) else None
|
||||
return width_value.primary_slot if width_value is not None else None
|
||||
|
||||
|
||||
def neighbor_right(neighbor_map, other_block: Block) -> Optional[Block]:
|
||||
"""Closest right-side peer neighbor."""
|
||||
width_value = neighbor_map.primary_slot[other_block.orig_index] if other_block.orig_index < len(neighbor_map.primary_slot) else None
|
||||
return width_value.auxiliary_slot if width_value is not None else None
|
||||
|
||||
|
||||
def neighbor_right_peer(neighbor_map, secondary_item):
|
||||
return neighbor_right(neighbor_map, secondary_item)
|
||||
|
||||
|
||||
def closest_body_neighbor_above(neighbor_map, other_block: Block) -> Optional[Block]:
|
||||
"""Closest stored neighbor above."""
|
||||
width_value = neighbor_map.primary_slot[other_block.orig_index] if other_block.orig_index < len(neighbor_map.primary_slot) else None
|
||||
return width_value.secondary_slot if width_value is not None else None
|
||||
|
||||
|
||||
class PageNeighborMap:
|
||||
"""Per-page horizontal-bucket neighbor map for constant-time nearby-block queries."""
|
||||
|
||||
__slots__ = ("tertiary_slot", "secondary_slot", "primary_slot")
|
||||
|
||||
def __init__(self, page):
|
||||
blocks = page.output_slot
|
||||
self.tertiary_slot = max(5, page.bounds.bbox_width() / 300) # bucket width
|
||||
self.secondary_slot = int(math.floor(page.bounds.bbox_width() / self.tertiary_slot)) # bucket count
|
||||
self.primary_slot: list[Optional[BlockNeighborCache]] = [None] * (max(len(blocks), 1) + 1)
|
||||
# mark buckets crossed by body-marked blocks
|
||||
marked = [False] * self.secondary_slot
|
||||
for candidate_item in blocks:
|
||||
if not candidate_item.is_body_paragraph:
|
||||
continue
|
||||
spans = compute_bucket_span(self, candidate_item)
|
||||
for index in range(spans["start_bucket"], spans["end_bucket"]):
|
||||
if 0 <= index < self.secondary_slot:
|
||||
marked[index] = True
|
||||
recent_height: list[int] = [-1] * self.secondary_slot # most recent body block height at bucket
|
||||
# Reads past the end of this list must behave like an unset slot: -1 is
|
||||
# falsy at the ``>= 0`` tests below just as a missing entry is, and a
|
||||
# write to it extends the list. On a degenerate page with zero buckets
|
||||
# every clamped index is 0, so one slot reproduces that growth.
|
||||
recent_block_index: list[int] = [-1] * max(self.secondary_slot, 1) # most-recent block V (j-direction)
|
||||
recent: list[Optional[Block]] = [None] * self.secondary_slot # most-recent block at bucket
|
||||
pending: list[list[int]] = [[] for _ in range(self.secondary_slot)] # pending V's per bucket
|
||||
|
||||
for block_index, current_block in enumerate(blocks):
|
||||
if (
|
||||
current_block.char_count() <= 0
|
||||
or current_block.skew_frac() > 1
|
||||
or current_block.type in (1, 2, 12)
|
||||
):
|
||||
continue
|
||||
width = BlockNeighborCache()
|
||||
self.primary_slot[current_block.orig_index] = width
|
||||
spans = compute_bucket_span(self, current_block)
|
||||
left = spans["start_bucket"]
|
||||
right = spans["end_bucket"]
|
||||
# H field: block 1-column-left or right
|
||||
value = recent_block_index[left]
|
||||
adjacent_bucket_index = recent_block_index[
|
||||
left - 1 if (left > 0 and current_block.left_edge() < (left + 0.5) * self.tertiary_slot)
|
||||
else (left + 1 if left < self.secondary_slot - 1 else left)
|
||||
]
|
||||
if value >= 0 or adjacent_bucket_index >= 0:
|
||||
same_bucket_block = blocks[value] if 0 <= value < len(blocks) else None
|
||||
adjacent_bucket_block = blocks[adjacent_bucket_index] if 0 <= adjacent_bucket_index < len(blocks) else None
|
||||
if same_bucket_block is not None and (adjacent_bucket_block is None or same_bucket_block.bottom_edge() < adjacent_bucket_block.bottom_edge()):
|
||||
picked_index = value
|
||||
else:
|
||||
picked_index = adjacent_bucket_index
|
||||
if 0 <= picked_index < len(blocks):
|
||||
width.measure_slot = blocks[picked_index]
|
||||
if self.primary_slot[picked_index] is not None:
|
||||
self.primary_slot[picked_index].option_slot = current_block
|
||||
recent_block_index[left] = current_block.orig_index
|
||||
for col in range(left, right):
|
||||
if 0 <= col < self.secondary_slot:
|
||||
width.state_slot = width.state_slot or marked[col]
|
||||
recent_block = recent[col]
|
||||
if recent_block is not None and (width.primary_slot is None or recent_block.bottom_edge() < width.primary_slot.bottom_edge()):
|
||||
width.primary_slot = recent_block
|
||||
previous_body_index = recent_height[col]
|
||||
recent_height[col] = current_block.orig_index
|
||||
if previous_body_index >= 0 and previous_body_index < len(blocks):
|
||||
same_bucket_block = blocks[previous_body_index]
|
||||
if width.tertiary_slot is None or same_bucket_block.bottom_edge() < width.tertiary_slot.bottom_edge():
|
||||
width.tertiary_slot = same_bucket_block
|
||||
previous_cache = self.primary_slot[previous_body_index]
|
||||
if previous_cache is not None and (previous_cache.auxiliary_slot is None or current_block.top_edge() > previous_cache.auxiliary_slot.top_edge()):
|
||||
previous_cache.auxiliary_slot = current_block
|
||||
if current_block.is_body_paragraph:
|
||||
for pending_index in pending[col]:
|
||||
if 0 <= pending_index < len(self.primary_slot):
|
||||
pending_neighbor = self.primary_slot[pending_index]
|
||||
if pending_neighbor is not None and (pending_neighbor.secondary_slot is None or current_block.top_edge() > pending_neighbor.secondary_slot.top_edge()):
|
||||
pending_neighbor.secondary_slot = current_block
|
||||
recent[col] = current_block
|
||||
pending[col].clear()
|
||||
pending[col].append(current_block.orig_index)
|
||||
@@ -0,0 +1,474 @@
|
||||
"""Whole-page heading scan and document-level candidate collection/filtering."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Any, Optional
|
||||
from ..outline_assembly import HeadingCandidate, OutlineNode
|
||||
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_trim_unicode_ws,
|
||||
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
|
||||
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
|
||||
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
|
||||
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
from .keyword_tables import (
|
||||
SECTION_KEYWORDS_TRIE,
|
||||
INTRODUCTION_SECTION_TRIE,
|
||||
)
|
||||
from .text_checks import (
|
||||
is_heading_continuation,
|
||||
matches_abstract,
|
||||
matches_references,
|
||||
has_substantive_content,
|
||||
is_cover_page,
|
||||
)
|
||||
from .neighbors import (
|
||||
neighbor_above,
|
||||
neighbor_right,
|
||||
closest_body_neighbor_above,
|
||||
)
|
||||
from .candidates import (
|
||||
PageScanState,
|
||||
push_candidate,
|
||||
make_plain_candidate,
|
||||
make_body_heading_candidate,
|
||||
)
|
||||
from .detectors import (
|
||||
detect_numbered_heading,
|
||||
detect_labeled_heading,
|
||||
detect_chapter_appendix,
|
||||
try_classify_heading,
|
||||
passes_neighbor_check,
|
||||
has_competing_labeled_heading,
|
||||
)
|
||||
from .style_detectors import (
|
||||
detect_font_heading,
|
||||
detect_heading_with_body,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Main per-page heading scan #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def scan_page_headings(page_scan: PageScanState) -> list[HeadingCandidate]:
|
||||
"""Return heading candidates found on this page."""
|
||||
if is_cover_page(page_scan.secondary_slot, page_scan.primary_slot):
|
||||
return []
|
||||
page_scan.option_slot.clear()
|
||||
page_scan.measure_slot.clear()
|
||||
blocks = page_scan.auxiliary_slot
|
||||
for block in blocks:
|
||||
if block.char_count() <= 0 or block.skew_frac() > 1 or block.type != 0:
|
||||
continue
|
||||
if block.state_slot != 0: # already classified
|
||||
continue
|
||||
above = neighbor_above(page_scan.tertiary_slot, block)
|
||||
if above is not None and above.secondary_slot.contains(block.secondary_slot):
|
||||
continue
|
||||
if block.char_count() <= 1 and block.char_stats.secondary_slot != 4:
|
||||
continue
|
||||
wn_entry = page_scan.tertiary_slot.primary_slot[block.orig_index] if 0 <= block.orig_index < len(page_scan.tertiary_slot.primary_slot) else None
|
||||
has_da_above = wn_entry is not None and wn_entry.state_slot
|
||||
rows = block.line_count()
|
||||
# Try lo for 2-line heading-body patterns
|
||||
if has_da_above and rows > 1 and (
|
||||
(0 < block.bold_frac() < 1)
|
||||
or style_key(first_span_of(block)) != style_key(last_span(last_line_of(block)))
|
||||
):
|
||||
lo_result = detect_heading_with_body(page_scan, block)
|
||||
if lo_result is not None:
|
||||
push_candidate(page_scan, lo_result)
|
||||
continue
|
||||
tokens = tokenize_block(block)
|
||||
first_line = block.line()
|
||||
first_line_tokens = tokens.slice(0, advance_past_line(tokens, first_line, 0))
|
||||
if has_da_above and not block.measure_slot and rows >= 3\
|
||||
and first_line.bbox_width() <= 0.2 * min(block.primary_slot[1].bbox_width(), block.primary_slot[2].bbox_width())\
|
||||
and matches_abstract(first_line_tokens):
|
||||
push_candidate(page_scan, make_body_heading_candidate(page_scan, 5, block, first_line_tokens))
|
||||
continue
|
||||
if block.char_count() >= 200:
|
||||
continue
|
||||
if block.char_count() >= 100 and rows > 1 and block.char_stats.primary_slot[6] - first_line.char_stats.primary_slot[6] > 1:
|
||||
continue
|
||||
size = block.avg_font_size()
|
||||
page_width = page_scan.primary_slot.bounds.bbox_width()
|
||||
lots_caps = block.char_stats.primary_slot[2] >= max(3, letter_count(block.char_stats) / 2)
|
||||
if rows > 4 or (rows >= 3 and not (size >= 1.5 * page_scan.primary_slot.primary_slot.primary_slot or lots_caps)):
|
||||
continue
|
||||
if block.weighted_ratio_primary < 0.5 * page_scan.secondary_slot.secondary_slot.auxiliary_slot:
|
||||
continue
|
||||
if letter_count(block.char_stats) <= 0:
|
||||
continue
|
||||
if block.bold_frac() < 0.1 and not lots_caps and size < page_scan.secondary_slot.secondary_slot.primary_slot - 2:
|
||||
continue
|
||||
layout_gate = detect_chapter_appendix(page_scan, block)
|
||||
if layout_gate is not None:
|
||||
push_candidate(page_scan, layout_gate)
|
||||
continue
|
||||
if passes_neighbor_check(page_scan, block):
|
||||
continue
|
||||
top_gap = above.bottom_edge() - block.top_edge() if above is not None else math.inf
|
||||
isolated = (
|
||||
not block.measure_slot and rows <= 2
|
||||
and (above is None or top_gap > 1.5 * block.avg_font_size()
|
||||
or (above.state_slot != 5 and above.state_slot != 11))
|
||||
)
|
||||
if isolated:
|
||||
if matches_references(tokens):
|
||||
push_candidate(page_scan, make_plain_candidate(page_scan, 7, block))
|
||||
continue
|
||||
if has_da_above and trie_matches_all(INTRODUCTION_SECTION_TRIE, tokens):
|
||||
push_candidate(page_scan, make_plain_candidate(page_scan, 11, block))
|
||||
continue
|
||||
above_index = block.orig_index - 1
|
||||
below_index = block.orig_index + 1
|
||||
previous_block = blocks[above_index] if 0 <= above_index < len(blocks) else None
|
||||
below = blocks[below_index] if 0 <= below_index < len(blocks) else None
|
||||
if not has_da_above and (
|
||||
not (heading_score(block) >= page_scan.primary_slot.primary_slot.primary_slot + 1.5)
|
||||
or (above is not None and above.type != 1)
|
||||
or (previous_block is not None and previous_block.type != 1)
|
||||
or (below is not None and not (below.top_edge() < block.bottom_edge() - size))
|
||||
):
|
||||
continue
|
||||
if block.bold_frac() < 0.1 and not lots_caps\
|
||||
and first_span_of(block).font_name == page_scan.primary_slot.primary_slot.state_slot\
|
||||
and size < page_scan.secondary_slot.secondary_slot.primary_slot - 0.5:
|
||||
continue
|
||||
predecessor = neighbor_right(page_scan.tertiary_slot, block)
|
||||
if predecessor is not None and predecessor.skew_frac() > 1:
|
||||
continue
|
||||
if top_gap < 0:
|
||||
continue
|
||||
predecessor_gap = block.bottom_edge() - predecessor.top_edge() if predecessor is not None else math.inf
|
||||
if predecessor_gap < -0.9 * block.bbox_height():
|
||||
continue
|
||||
line_gap = page_scan.primary_slot.primary_slot.tertiary_slot - page_scan.primary_slot.primary_slot.primary_slot
|
||||
if top_gap < line_gap and size < page_scan.primary_slot.primary_slot.primary_slot - 1:
|
||||
continue
|
||||
# Sibling/peer block pointers from the neighbor cache.
|
||||
right_neighbor_sib = wn_entry.measure_slot if wn_entry is not None else None
|
||||
left_neighbor_sib = wn_entry.option_slot if wn_entry is not None else None
|
||||
|
||||
heading_kind = detect_numbered_heading(page_scan, block, tokens)
|
||||
if heading_kind is not None:
|
||||
from ..stats import column_index_of as _column_index
|
||||
col_idx = _column_index(block) if block.primary_slot else 0
|
||||
column_rect = page_scan.primary_slot.tertiary_slot[col_idx] if 0 <= col_idx < len(page_scan.primary_slot.tertiary_slot) else None
|
||||
# Narrow-column heading-vs-prev-numbering check
|
||||
if (size < page_scan.secondary_slot.secondary_slot.primary_slot - 1
|
||||
and block.bbox_width() < 0.2 * page_width
|
||||
and column_rect is not None
|
||||
and column_rect.bbox_width() < 0.2 * page_width
|
||||
and column_rect.bbox_height() > 1.5 * column_rect.bbox_width()):
|
||||
first_number = heading_kind.numbering[0] if heading_kind.numbering else 0
|
||||
if above is not None:
|
||||
first = first_token(tokenize_block(above))
|
||||
if first is not None and first.type == 1 and token_numeric_value(first) != first_number:
|
||||
continue
|
||||
if predecessor is not None:
|
||||
predecessor_first_token = first_token(tokenize_block(predecessor))
|
||||
if predecessor_first_token is not None and predecessor_first_token.type == 1 and token_numeric_value(predecessor_first_token) != first_number:
|
||||
continue
|
||||
# Prev-block continuation check via Kn (numbered-sequence test)
|
||||
first_token_value = tokens.token_at(0)
|
||||
second_token_value = tokens.token_at(1) if len(tokens) > 1 else None
|
||||
if len(heading_kind.numbering) <= 1 and (
|
||||
(first_token_value is not None and first_token_value.boundary_slot)
|
||||
or (second_token_value is not None and is_word_token(second_token_value))):
|
||||
first_number = heading_kind.numbering[0] if heading_kind.numbering else 0
|
||||
if above is not None and is_heading_continuation(above, block, first_number):
|
||||
continue
|
||||
if (right_neighbor_sib is not None and above is not right_neighbor_sib
|
||||
and not center_aligned(block, right_neighbor_sib, 1)
|
||||
and (above.line_count() < 10 or above.char_count() < 300)
|
||||
and is_heading_continuation(right_neighbor_sib, block, first_number)):
|
||||
continue
|
||||
if predecessor is not None and is_heading_continuation(predecessor, block, first_number):
|
||||
continue
|
||||
if (left_neighbor_sib is not None and predecessor is not left_neighbor_sib
|
||||
and not center_aligned(block, left_neighbor_sib, 1)
|
||||
and (predecessor.line_count() < 10 or predecessor.char_count() < 300)
|
||||
and is_heading_continuation(left_neighbor_sib, block, first_number)):
|
||||
continue
|
||||
# Top-of-page small-font footnote-marker rejection
|
||||
second = tokens.token_at(1) if len(tokens) > 1 else None
|
||||
if (block.top_edge() < page_scan.primary_slot.bounds.bbox_height() / 4
|
||||
and size <= page_scan.primary_slot.primary_slot.primary_slot
|
||||
and first_span_of(block).char_stats.secondary_slot == 1
|
||||
and first_span_of(block).bbox_height() < size - 0.5
|
||||
and len(heading_kind.numbering) <= 1
|
||||
and second is not None and is_char_token(second)):
|
||||
continue
|
||||
push_candidate(page_scan, heading_kind)
|
||||
continue
|
||||
if block.measure_slot:
|
||||
continue
|
||||
if block.char_count() >= 120:
|
||||
continue
|
||||
if top_gap <= line_gap - 0.1:
|
||||
continue
|
||||
heading_signature = detect_labeled_heading(page_scan, block, tokens)
|
||||
if heading_signature is not None:
|
||||
if above is not None and has_competing_labeled_heading(page_scan, heading_signature, above):
|
||||
continue
|
||||
if (right_neighbor_sib is not None and above is not right_neighbor_sib
|
||||
and has_competing_labeled_heading(page_scan, heading_signature, right_neighbor_sib)):
|
||||
continue
|
||||
if predecessor is not None and has_competing_labeled_heading(page_scan, heading_signature, predecessor):
|
||||
continue
|
||||
if (left_neighbor_sib is not None and predecessor is not left_neighbor_sib
|
||||
and has_competing_labeled_heading(page_scan, heading_signature, left_neighbor_sib)):
|
||||
continue
|
||||
push_candidate(page_scan, heading_signature)
|
||||
continue
|
||||
if isolated:
|
||||
caps_heavy = size + 2 * block.bold_frac() >= page_scan.secondary_slot.secondary_slot.primary_slot + 4 or is_upper_dominant(block.char_stats)
|
||||
above_or_overlap = closest_body_neighbor_above(page_scan.tertiary_slot, block)
|
||||
page_width = page_scan.primary_slot.bounds.bbox_width()
|
||||
# Abstract-heading acceptance uses the closest above-overlap block
|
||||
# as the guard; when it exists, the predecessor exists too.
|
||||
type5_cond = caps_heavy or (
|
||||
above_or_overlap is not None
|
||||
and (block.bottom_edge() - above_or_overlap.top_edge() < 3 * (block.bottom_edge() - predecessor.top_edge())
|
||||
or info_weight(predecessor.char_stats) >= 30)
|
||||
)
|
||||
if type5_cond and matches_abstract(tokens):
|
||||
push_candidate(page_scan, make_plain_candidate(page_scan, 5, block))
|
||||
continue
|
||||
type6_cond = (
|
||||
caps_heavy
|
||||
or (above is not None and x_aligned(block, above, 1)
|
||||
and (above.bbox_width() >= page_width / 6
|
||||
or above.bold_frac() > 0.9
|
||||
or is_upper_dominant(above.char_stats)))
|
||||
or (predecessor is not None and x_aligned(block, predecessor, 1)
|
||||
and (predecessor.bbox_width() >= page_width / 6
|
||||
or predecessor.bold_frac() > 0.9
|
||||
or is_upper_dominant(predecessor.char_stats)
|
||||
or (block.previous_slot < 0.1 and predecessor.previous_slot > 0.9)))
|
||||
)
|
||||
if type6_cond and trie_full_match(SECTION_KEYWORDS_TRIE, tokens):
|
||||
push_candidate(page_scan, make_plain_candidate(page_scan, 6, block))
|
||||
continue
|
||||
if info_weight(block.char_stats) <= 3:
|
||||
continue
|
||||
if has_substantive_content(block, previous_block, below):
|
||||
continue
|
||||
outline_context = detect_font_heading(page_scan, block)
|
||||
if outline_context is not None:
|
||||
push_candidate(page_scan, outline_context)
|
||||
return page_scan.option_slot
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Document-wide outline-candidate collector #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class DocCandidateCollector:
|
||||
"""Document-level state aggregating per-page heading candidates."""
|
||||
|
||||
__slots__ = ("previous_slot", "measure_slot", "option_slot", "auxiliary_slot", "primary_slot", "tertiary_slot", "secondary_slot", "state_slot")
|
||||
|
||||
def __init__(self, doc, labeled):
|
||||
self.previous_slot = doc
|
||||
self.measure_slot = labeled
|
||||
self.option_slot = 0
|
||||
self.auxiliary_slot = False
|
||||
self.primary_slot = 0
|
||||
self.secondary_slot = False
|
||||
self.tertiary_slot = False
|
||||
self.state_slot: list[HeadingCandidate] = []
|
||||
|
||||
|
||||
def filter_page_candidates(doc_collector: DocCandidateCollector, page, page_candidates: list[HeadingCandidate]) -> None:
|
||||
"""Per-page candidate filter for noisy pages, title overlap, page headers, and numbering continuity."""
|
||||
from ..outline_assembly import is_script_compatible
|
||||
from ..model import intervals_overlap, is_caps_heavy
|
||||
from ..stats import column_index_of
|
||||
|
||||
# Advance document-level numbering state through outline entries up to this page.
|
||||
while doc_collector.option_slot < len(doc_collector.measure_slot):
|
||||
outline_entry = doc_collector.measure_slot[doc_collector.option_slot]
|
||||
current_candidate = outline_entry.heading
|
||||
if current_candidate.page.page_index > page.page_index:
|
||||
break
|
||||
if current_candidate.type == 2:
|
||||
if not doc_collector.tertiary_slot:
|
||||
doc_collector.tertiary_slot = (len(current_candidate.numbering) > 0 and current_candidate.numbering[0] == 1)
|
||||
elif current_candidate.type == 4:
|
||||
if not doc_collector.secondary_slot:
|
||||
doc_collector.secondary_slot = (len(current_candidate.numbering) > 0 and current_candidate.numbering[0] == 1)
|
||||
elif current_candidate.type == 1 and len(current_candidate.numbering) > 0:
|
||||
doc_collector.primary_slot = max(doc_collector.primary_slot, current_candidate.numbering[0])
|
||||
doc_collector.option_slot += 1
|
||||
|
||||
count = len(page_candidates)
|
||||
if count >= 20:
|
||||
return
|
||||
|
||||
# Sort candidates by column, vertical position, then horizontal position.
|
||||
def _ih_key(heading_candidate: HeadingCandidate):
|
||||
first_line = heading_candidate.group_slot.primary_slot[0] if heading_candidate.group_slot.primary_slot else None
|
||||
column_index = first_line.measure_slot if first_line is not None else -1
|
||||
return (column_index, -heading_candidate.group_slot.top_edge(), -heading_candidate.group_slot.bottom_edge(), heading_candidate.group_slot.left_edge(), heading_candidate.group_slot.right_edge())
|
||||
page_candidates.sort(key=_ih_key)
|
||||
|
||||
accepted: list[HeadingCandidate] = []
|
||||
title: Optional[Block] = None
|
||||
if not doc_collector.auxiliary_slot and getattr(page, "auxiliary_slot", False):
|
||||
for block in page.output_slot:
|
||||
if block.type == 3:
|
||||
title = block
|
||||
break
|
||||
|
||||
min_first_number = math.inf
|
||||
total_bottom = math.inf
|
||||
single_numbering_count = 0
|
||||
for page_candidate in page_candidates:
|
||||
if total_bottom == math.inf and page_candidate.type == 5:
|
||||
total_bottom = page_candidate.group_slot.top_edge() + page_candidate.group_slot.avg_font_size()
|
||||
if page_candidate.type == 1 and len(page_candidate.numbering) > 0:
|
||||
min_first_number = min(min_first_number, page_candidate.numbering[0])
|
||||
if len(page_candidate.numbering) == 1:
|
||||
single_numbering_count += 1
|
||||
|
||||
max_first_number = 0
|
||||
for index in range(count):
|
||||
active_candidate = page_candidates[index]
|
||||
next_item = page_candidates[index + 1] if index + 1 < count else None
|
||||
|
||||
if is_script_compatible(doc_collector.previous_slot.secondary_slot.tertiary_slot, active_candidate):
|
||||
continue
|
||||
|
||||
if not doc_collector.auxiliary_slot and active_candidate.type != 11:
|
||||
bottom = active_candidate.group_slot.top_edge()
|
||||
if (title is not None and bottom > title.top_edge()
|
||||
and intervals_overlap(active_candidate.group_slot.left_edge(), active_candidate.group_slot.right_edge(), title.left_edge(), title.right_edge())):
|
||||
continue
|
||||
if bottom > total_bottom:
|
||||
continue
|
||||
|
||||
if (active_candidate.type == 0 and next_item is not None and next_item.type == 5
|
||||
and active_candidate.group_slot.left_edge() <= next_item.group_slot.right_edge() and active_candidate.group_slot.right_edge() >= next_item.group_slot.left_edge()
|
||||
and active_candidate.group_slot.bottom_edge() - next_item.group_slot.top_edge() < 2 * active_candidate.group_slot.bbox_height()
|
||||
and len(tokenize_block(active_candidate.group_slot)) > 1):
|
||||
continue
|
||||
|
||||
if active_candidate.type == 2:
|
||||
if len(active_candidate.numbering) > 0 and active_candidate.numbering[0] == 1:
|
||||
doc_collector.tertiary_slot = True
|
||||
elif not doc_collector.tertiary_slot:
|
||||
continue
|
||||
accepted.append(active_candidate)
|
||||
continue
|
||||
|
||||
if active_candidate.type == 4:
|
||||
if len(active_candidate.numbering) > 0 and active_candidate.numbering[0] == 1:
|
||||
doc_collector.secondary_slot = True
|
||||
elif not doc_collector.secondary_slot:
|
||||
continue
|
||||
accepted.append(active_candidate)
|
||||
continue
|
||||
|
||||
if active_candidate.type != 1:
|
||||
accepted.append(active_candidate)
|
||||
continue
|
||||
|
||||
# type == 1
|
||||
if single_numbering_count >= 5:
|
||||
continue
|
||||
first_number = active_candidate.numbering[0] if len(active_candidate.numbering) > 0 else 0
|
||||
if (active_candidate.group_slot.bold_frac() < 0.9 and not is_caps_heavy(active_candidate.group_slot)
|
||||
and active_candidate.group_slot.avg_font_size() < page.primary_slot.primary_slot + 1):
|
||||
if doc_collector.primary_slot <= 0 and min_first_number > 1 and len(active_candidate.numbering) <= 1:
|
||||
continue
|
||||
if first_number > 3 * page.page_index:
|
||||
continue
|
||||
max_first_number = max(max_first_number, first_number)
|
||||
accepted.append(active_candidate)
|
||||
|
||||
doc_collector.state_slot.extend(accepted)
|
||||
doc_collector.auxiliary_slot = True
|
||||
doc_collector.primary_slot = max(doc_collector.primary_slot, max_first_number)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Public entry point: build heading candidates for the whole document #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def build_doc_heading_candidates(doc, labeled: Optional[list] = None) -> list[HeadingCandidate]:
|
||||
"""Run per-page heading detection across the document."""
|
||||
doc_collector = DocCandidateCollector(doc, labeled if labeled is not None else [])
|
||||
saw_body = False
|
||||
for page in doc.primary_slot:
|
||||
# Skip initial cover-like pages until the first body-like page is reached.
|
||||
from ..title import is_cover_like_page
|
||||
if not saw_body and is_cover_like_page(doc, page):
|
||||
continue
|
||||
saw_body = True
|
||||
page_vo = PageScanState(doc, page)
|
||||
page_candidates = scan_page_headings(page_vo)
|
||||
filter_page_candidates(doc_collector, page, page_candidates)
|
||||
return doc_collector.state_slot
|
||||
|
||||
|
||||
def find_section_openers(doc, start_page_idx: int) -> list:
|
||||
"""Find the first valid heading on each page, then clique-filter the result."""
|
||||
from ..outline_assembly import is_script_compatible, has_conflict_in_context, OutlineContext, OutlineNode
|
||||
|
||||
item_list: list[HeadingCandidate] = []
|
||||
index = start_page_idx
|
||||
while index < len(doc.primary_slot):
|
||||
page = doc.primary_slot[index]
|
||||
current_candidate: Optional[HeadingCandidate] = None
|
||||
page_scan_state = PageScanState(doc, page)
|
||||
if not page_scan_state.primary_slot.auxiliary_slot:
|
||||
for block in page_scan_state.auxiliary_slot:
|
||||
if block.char_count() <= 0 or block.skew_frac() > 1 or block.type != 0:
|
||||
continue
|
||||
if block.is_body_paragraph or block.top_edge() < 0.5 * page_scan_state.primary_slot.bounds.bbox_height():
|
||||
break
|
||||
if block.marker_slot != 0:
|
||||
break
|
||||
detected_candidate = try_classify_heading(page_scan_state, block)
|
||||
if detected_candidate is not None:
|
||||
current_candidate = detected_candidate
|
||||
break
|
||||
if block.line_count() > 2:
|
||||
break
|
||||
if current_candidate is not None and not is_script_compatible(doc.secondary_slot.tertiary_slot, current_candidate):
|
||||
item_list.append(current_candidate)
|
||||
index += 1
|
||||
|
||||
if len(item_list) <= 1:
|
||||
return []
|
||||
|
||||
# Compare candidates against the full context and against the accepted subset.
|
||||
bundle = OutlineContext(item_list)
|
||||
accepted_context = OutlineContext([])
|
||||
out = []
|
||||
for current_candidate in item_list:
|
||||
if accepted_context.has_nearby_duplicate(current_candidate):
|
||||
current_candidate.group_slot.type = 12
|
||||
continue
|
||||
if has_conflict_in_context(bundle, current_candidate):
|
||||
continue
|
||||
accepted_context.add(current_candidate)
|
||||
out.append(OutlineNode(current_candidate))
|
||||
current_candidate.group_slot.type = 7
|
||||
current_candidate.group_slot.used_as_heading = True
|
||||
return out
|
||||
@@ -0,0 +1,383 @@
|
||||
"""Font-change and body-embedded heading detectors."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Any, Optional
|
||||
from ..outline_assembly import HeadingCandidate, OutlineNode
|
||||
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_trim_unicode_ws,
|
||||
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
|
||||
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
|
||||
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
|
||||
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
from .keyword_tables import (
|
||||
SECTION_KEYWORDS_TRIE,
|
||||
INTRODUCTION_SECTION_TRIE,
|
||||
KEYWORDS_SECTION_TRIE,
|
||||
)
|
||||
from .text_checks import (
|
||||
matches_abstract,
|
||||
vertically_close,
|
||||
)
|
||||
from .neighbors import (
|
||||
neighbor_above,
|
||||
body_neighbor_above,
|
||||
neighbor_right,
|
||||
closest_body_neighbor_above,
|
||||
)
|
||||
from .candidates import (
|
||||
PageScanState,
|
||||
make_heading_candidate,
|
||||
make_plain_candidate,
|
||||
make_body_heading_candidate,
|
||||
)
|
||||
from .detectors import (
|
||||
detect_numbered_heading,
|
||||
detect_labeled_heading,
|
||||
is_bibliography_entry,
|
||||
)
|
||||
|
||||
|
||||
def detect_font_heading(page_scan: PageScanState, other_block: Block) -> Optional[HeadingCandidate]:
|
||||
"""Detailed font/position-based fallback heading classifier."""
|
||||
from ..model import x_aligned, last_span, last_line_of, first_span_of, letter_count, punct_count, dominant_style_of, is_upper_dominant, is_caps_heavy, is_sentence_like, alignment_code
|
||||
from ..tokens import last_token, is_comma_token
|
||||
|
||||
above = neighbor_above(page_scan.tertiary_slot, other_block)
|
||||
top_gap = above.bottom_edge() - other_block.top_edge() if above is not None else math.inf
|
||||
predecessor = neighbor_right(page_scan.tertiary_slot, other_block)
|
||||
predecessor_gap = other_block.bottom_edge() - predecessor.top_edge() if predecessor is not None else math.inf
|
||||
keyword_match = body_neighbor_above(page_scan.tertiary_slot, other_block)
|
||||
above_or_overlap = closest_body_neighbor_above(page_scan.tertiary_slot, other_block)
|
||||
|
||||
# Initial gate: one of On OR bold/centered tall block.
|
||||
if not (
|
||||
vertically_close(keyword_match, other_block) or vertically_close(above_or_overlap, other_block)
|
||||
or (other_block.bottom_edge() >= 0.8 * page_scan.primary_slot.bounds.bbox_height()
|
||||
and other_block.avg_font_size() >= page_scan.secondary_slot.secondary_slot.primary_slot + 1
|
||||
and other_block.bold_frac() > 0.9
|
||||
and (above is None or above.type == 1))
|
||||
):
|
||||
return None
|
||||
|
||||
page = page_scan.primary_slot.primary_slot # page statistics
|
||||
far = 10 * min(other_block.avg_font_size(), page.tertiary_slot)
|
||||
if top_gap < math.inf and top_gap > far and above.state_slot == 0:
|
||||
return None
|
||||
if predecessor is not None and predecessor.state_slot != 0:
|
||||
return None
|
||||
|
||||
# Compound rejection for candidates sitting above non-body predecessors.
|
||||
# Keep the explicit short-circuit structure: each inner predicate requires
|
||||
# the predecessor to exist.
|
||||
|
||||
inner_reject = False
|
||||
if above is not None and above.state_slot != 0:
|
||||
inner_reject = (
|
||||
predecessor_gap > 5 * page.tertiary_slot
|
||||
or (predecessor is not None and not predecessor.is_body_paragraph)
|
||||
or (predecessor is not None and predecessor.bbox_width() < page_scan.primary_slot.bounds.bbox_width() / 5)
|
||||
or (predecessor is not None and predecessor.char_count() < 0.5 * other_block.char_count())
|
||||
or (predecessor is not None and predecessor.weighted_ratio_secondary < 0.33)
|
||||
or (predecessor is not None and predecessor.char_count() < 500
|
||||
and predecessor.weighted_ratio_secondary < 0.5 and alignment_code(predecessor) != 1)
|
||||
or (predecessor is not None and predecessor.char_count() < 250 and predecessor.weighted_ratio_secondary < 0.5)
|
||||
)
|
||||
if (inner_reject
|
||||
or (predecessor is not None and (
|
||||
predecessor.weighted_ratio_primary < 0.67 * page_scan.secondary_slot.secondary_slot.auxiliary_slot
|
||||
or (other_block.char_count() < 30 and predecessor.char_count() < 300
|
||||
and predecessor.weighted_ratio_primary < 0.8 * page_scan.secondary_slot.secondary_slot.auxiliary_slot)))):
|
||||
return None
|
||||
last_tok = last_token(tokenize_block(other_block))
|
||||
if last_tok is not None and is_comma_token(last_tok):
|
||||
return None
|
||||
|
||||
# Branch 1: tall first-line + big-font heading
|
||||
if (predecessor_gap < math.inf and predecessor_gap > 0
|
||||
and other_block.style_slot >= page_scan.secondary_slot.secondary_slot.primary_slot + 2
|
||||
and other_block.avg_font_size() >= page_scan.secondary_slot.secondary_slot.primary_slot + 1.5
|
||||
and other_block.avg_font_size() >= page.primary_slot + 0.5
|
||||
and (above is None or (other_block.style_slot >= above.style_slot and other_block.avg_font_size() >= above.avg_font_size()))
|
||||
and predecessor is not None
|
||||
and other_block.style_slot >= predecessor.style_slot and other_block.avg_font_size() >= predecessor.avg_font_size()):
|
||||
return make_plain_candidate(page_scan, 0, other_block)
|
||||
|
||||
caps_heavy = is_caps_heavy(other_block)
|
||||
# Branch 2 reject: matches body-style and not all-caps, OR clearly
|
||||
# smaller font than predecessor near it.
|
||||
if ((dominant_style_of(other_block) in page_scan.primary_slot.style_slot and not caps_heavy
|
||||
and (page.auxiliary_slot == dominant_style_of(other_block)
|
||||
or (other_block.bold_frac() < 0.9 and other_block.previous_slot < 0.9
|
||||
and alignment_code(other_block) != 3 and not is_sentence_like(other_block))))
|
||||
or (above is not None and predecessor is not None
|
||||
and other_block.avg_font_size() <= predecessor.avg_font_size()
|
||||
and top_gap < predecessor_gap / 4)):
|
||||
return None
|
||||
|
||||
# Branch 3: medium-confidence font-size heading
|
||||
if (predecessor_gap < math.inf and predecessor_gap > 0
|
||||
and other_block.avg_font_size() >= page_scan.secondary_slot.secondary_slot.primary_slot + 0.5
|
||||
and predecessor is not None
|
||||
and other_block.style_slot >= predecessor.style_slot and other_block.avg_font_size() >= predecessor.avg_font_size()
|
||||
and predecessor.avg_font_size() >= page.primary_slot - 0.5
|
||||
and other_block.bbox_width() < 0.95 * predecessor.bbox_width()
|
||||
and predecessor.bbox_width() >= 0.25 * page_scan.primary_slot.bounds.bbox_width()):
|
||||
return make_plain_candidate(page_scan, 0, other_block)
|
||||
|
||||
line_height = page.tertiary_slot - page.primary_slot
|
||||
# Branch 4: moderate-gap large-font heading
|
||||
if (predecessor_gap > line_height and predecessor_gap < 5 * line_height
|
||||
and (above is None or other_block.avg_font_size() >= above.avg_font_size() + 0.5)
|
||||
and predecessor is not None
|
||||
and other_block.avg_font_size() >= predecessor.avg_font_size() + 0.5
|
||||
and other_block.avg_font_size() >= page_scan.secondary_slot.secondary_slot.primary_slot - 0.5
|
||||
and other_block.bbox_width() < 0.95 * predecessor.bbox_width()
|
||||
and predecessor.is_body_paragraph
|
||||
and predecessor.avg_font_size() >= page.primary_slot - 0.5
|
||||
and predecessor.char_stats.secondary_slot != 1):
|
||||
return make_plain_candidate(page_scan, 0, other_block)
|
||||
|
||||
# Reject: many letters with low density signals body para
|
||||
letters = other_block.char_stats.primary_slot[6]
|
||||
if other_block.char_stats.primary_slot[10] != 0:
|
||||
ratio = (letters + other_block.char_stats.primary_slot[8]) / other_block.char_stats.primary_slot[10]
|
||||
else:
|
||||
# IEEE division edge case: positive numerator over zero behaves as +inf,
|
||||
# which keeps the low-density rejection active.
|
||||
ratio = math.inf if (letters + other_block.char_stats.primary_slot[8]) > 0 else math.nan
|
||||
if letters > 1 and ratio > 0.3:
|
||||
return None
|
||||
|
||||
symbol_count = punct_count(other_block.char_stats)
|
||||
letter_total = letter_count(other_block.char_stats)
|
||||
# Same IEEE division edge case as the letter-density ratio above.
|
||||
symbol_ratio = symbol_count / letter_total if letter_total != 0 else (math.inf if symbol_count > 0 else math.nan)
|
||||
if (symbol_count >= 5 and symbol_ratio > 0.2
|
||||
or top_gap < 0.2 * other_block.avg_font_size()
|
||||
or top_gap < min(other_block.avg_font_size(), 0.7 * predecessor_gap)):
|
||||
return None
|
||||
|
||||
centered = other_block.char_stats.secondary_slot == 2
|
||||
neg = -0.2 * last_span(last_line_of(other_block)).bbox_height() if caps_heavy else 0
|
||||
|
||||
# Branch A: tight criteria with neighbor analysis
|
||||
neighbor_heading_cue = (
|
||||
predecessor_gap < math.inf and predecessor_gap > neg
|
||||
and other_block.avg_font_size() >= page.primary_slot - 0.1
|
||||
and predecessor is not None and other_block.avg_font_size() >= predecessor.avg_font_size() - 0.1
|
||||
and other_block.avg_font_size() >= page_scan.secondary_slot.secondary_slot.primary_slot - 0.5
|
||||
and ((predecessor.is_body_paragraph and other_block.bbox_width() < 0.95 * predecessor.bbox_width()
|
||||
and x_aligned(other_block, predecessor, max(1, other_block.bbox_width() / 10))
|
||||
and predecessor_gap < 6 * other_block.bbox_height())
|
||||
or (top_gap < math.inf and above is not None and above.is_body_paragraph
|
||||
and other_block.bbox_width() < 0.95 * above.bbox_width()
|
||||
and x_aligned(other_block, above, max(1, other_block.bbox_width() / 10))
|
||||
and top_gap < 6 * other_block.bbox_height()))
|
||||
and (centered or caps_heavy)
|
||||
and ((other_block.bold_frac() > predecessor.bold_frac() and other_block.bold_frac() > 0.5
|
||||
and (not first_span_of(predecessor).primary_slot
|
||||
or (above is not None and other_block.bold_frac() > above.bold_frac())))
|
||||
or caps_heavy)
|
||||
)
|
||||
# Nearby body text with the dominant style changed is a strong heading cue.
|
||||
difference_style = bool(
|
||||
predecessor is not None and predecessor.is_body_paragraph
|
||||
and predecessor.avg_font_size() > page.primary_slot - 0.5
|
||||
and predecessor_gap > 0 and predecessor_gap < 3 * other_block.bbox_height()
|
||||
and dominant_style_of(other_block) != dominant_style_of(predecessor)
|
||||
)
|
||||
style_change_cue = (
|
||||
difference_style
|
||||
and above is not None and above.is_body_paragraph
|
||||
and top_gap > 0 and top_gap < 3 * other_block.bbox_height()
|
||||
and centered and dominant_style_of(above) == dominant_style_of(predecessor) if predecessor is not None else False
|
||||
)
|
||||
if neighbor_heading_cue or style_change_cue:
|
||||
return make_plain_candidate(page_scan, 0, other_block)
|
||||
|
||||
# Top-like context: there is no above block, or the above block is already a
|
||||
# title/heading marker.
|
||||
topnum = above is None or above.type == 1
|
||||
branch_C1 = (
|
||||
topnum and centered and difference_style
|
||||
and predecessor_gap < other_block.bbox_height()
|
||||
and predecessor is not None and dominant_style_of(predecessor) == page.auxiliary_slot
|
||||
)
|
||||
branch_C2 = (
|
||||
topnum and centered
|
||||
and predecessor is not None and above_or_overlap is not None
|
||||
and predecessor is not above_or_overlap
|
||||
and predecessor.bottom_edge() - above_or_overlap.top_edge() < predecessor.avg_font_size()
|
||||
and dominant_style_of(above_or_overlap) == page.auxiliary_slot and dominant_style_of(other_block) != page.auxiliary_slot
|
||||
and (other_block.avg_font_size() >= predecessor.avg_font_size() + 0.5
|
||||
or (caps_heavy and not is_upper_dominant(predecessor.char_stats)))
|
||||
)
|
||||
branch_C3 = (
|
||||
above is not None
|
||||
and (above.used_as_heading or above in page_scan.measure_slot)
|
||||
and (above.avg_font_size() >= other_block.avg_font_size() + 0.5
|
||||
or (is_upper_dominant(above.char_stats) and not caps_heavy))
|
||||
and centered and difference_style
|
||||
and predecessor is not None and dominant_style_of(predecessor) == page.auxiliary_slot
|
||||
)
|
||||
if branch_C1 or branch_C2 or branch_C3:
|
||||
return make_plain_candidate(page_scan, 0, other_block)
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def detect_heading_with_body(page_scan: PageScanState, other_block: Block) -> Optional[HeadingCandidate]:
|
||||
""". Detect heading-with-body 2-line patterns."""
|
||||
if other_block.line_count() < 2:
|
||||
return None
|
||||
tokens = tokenize_block(other_block)
|
||||
first_line = other_block.line()
|
||||
second_line = other_block.primary_slot[1]
|
||||
split = 0
|
||||
letter_count = 0
|
||||
font = first_line.primary_slot[0].font_name if first_line.primary_slot else ""
|
||||
if first_line.bold_frac() > 0 and first_line.bold_frac() < 1:
|
||||
for entry in enumerate_tokens(tokens):
|
||||
index = entry["index"]
|
||||
anchor_token = entry["token"]
|
||||
if anchor_token.line() is not first_line or not first_anchor_span(anchor_token).primary_slot:
|
||||
break
|
||||
if anchor_token.type == 2 and len(anchor_token.str) > 1:
|
||||
letter_count += 1
|
||||
split = index + 1
|
||||
elif last_span(last_line_of(other_block)).font_name != font:
|
||||
other_count = 0
|
||||
for candidate_line in other_block:
|
||||
if candidate_line is not first_line and candidate_line.primary_slot[0].font_name == font:
|
||||
other_count += 1
|
||||
if other_count > other_block.line_count() / 4:
|
||||
return None
|
||||
for entry in enumerate_tokens(tokens):
|
||||
index = entry["index"]
|
||||
anchor_token = entry["token"]
|
||||
line = anchor_token.line()
|
||||
if first_anchor_span(anchor_token).font_name != font or (line is not first_line and line is not second_line):
|
||||
break
|
||||
if anchor_token.type == 2 and (len(anchor_token.str) > 1 or anchor_token.primary_slot == 4):
|
||||
letter_count += 1
|
||||
split = index + 1
|
||||
if split <= 0 or split >= tokens.length:
|
||||
return None
|
||||
# Allow up to two punctuation-like tokens to stay with the prefix when they
|
||||
# remain on the same line and bracket attachment permits it.
|
||||
token = tokens.token_at(split - 1)
|
||||
next_token = tokens.token_at(split)
|
||||
for _ in range(2):
|
||||
if token is None or next_token is None:
|
||||
return None
|
||||
last_anchor = last_token_anchor(token)
|
||||
if not (is_word_token(next_token)
|
||||
and getattr(last_anchor, "line", None) is next_token.line()
|
||||
and (not token.boundary_slot or next_token.boundary_slot)):
|
||||
break
|
||||
split += 1
|
||||
token = next_token
|
||||
next_token = tokens.token_at(split)
|
||||
if token is None or next_token is None:
|
||||
return None
|
||||
if letter_count <= 0:
|
||||
return None
|
||||
prefix = tokens.slice(0, split)
|
||||
|
||||
# First-token style check for prefix/body split confidence.
|
||||
first = prefix.token_at(0)
|
||||
first_anchor = first_anchor_span(first) if first is not None else None
|
||||
if first_anchor is not None:
|
||||
if not first_anchor.primary_slot and not first_anchor.measure_slot and other_block.previous_slot > 0.5:
|
||||
return None
|
||||
if (not first_anchor.primary_slot and first_anchor.font_size < other_block.avg_font_size() + 1):
|
||||
rest = tokens.slice(split)
|
||||
if rest.length <= 0 or (rest.token_at(0) is not None and rest.token_at(0).primary_slot == 3):
|
||||
return None
|
||||
|
||||
# Reject prefixes that are only section keywords and contain no extra text.
|
||||
hn_match = trie_prefix_match(KEYWORDS_SECTION_TRIE, prefix)
|
||||
if hn_match is not None and len(hn_match) >= letter_count:
|
||||
return None
|
||||
|
||||
# Reuse numbered-heading detection on the prefix.
|
||||
heading_kind = detect_numbered_heading(page_scan, other_block, prefix)
|
||||
if heading_kind is not None and len(heading_kind.numbering) > 1:
|
||||
return make_heading_candidate(page_scan, heading_kind.type, heading_kind.group_slot, heading_kind.numbering, heading_kind.secondary_slot, heading_kind.primary_slot, True)
|
||||
if heading_kind is not None and (trie_matches_all(INTRODUCTION_SECTION_TRIE, heading_kind.primary_slot) or (is_uppercase_dominant(heading_kind.primary_slot) and not is_bibliography_entry(other_block))):
|
||||
return make_heading_candidate(page_scan, heading_kind.type, heading_kind.group_slot, heading_kind.numbering, heading_kind.secondary_slot, heading_kind.primary_slot, True)
|
||||
|
||||
# Reuse the labeled-heading detector on the prefix with body-heading status.
|
||||
if prefix.length > 3:
|
||||
heading_signature = detect_labeled_heading(page_scan, other_block, prefix)
|
||||
if heading_signature is not None:
|
||||
return make_heading_candidate(page_scan, heading_signature.type, heading_signature.group_slot, heading_signature.numbering, heading_signature.secondary_slot, trim_trailing_punct(heading_signature.primary_slot), True)
|
||||
|
||||
if matches_abstract(prefix):
|
||||
return make_body_heading_candidate(page_scan, 5, other_block, prefix)
|
||||
if trie_matches_all(INTRODUCTION_SECTION_TRIE, prefix):
|
||||
return make_body_heading_candidate(page_scan, 11, other_block, prefix)
|
||||
|
||||
# Final font-size and trailing-token reject gates.
|
||||
if first_line.avg_font_size() < page_scan.secondary_slot.secondary_slot.primary_slot - 2:
|
||||
return None
|
||||
if token is not None and is_word_token(token) and not is_trimmable_token(token):
|
||||
return None
|
||||
|
||||
# Body paragraphs can still contain an all-caps heading prefix.
|
||||
if (not is_upper_dominant(other_block.char_stats) and other_block.is_body_paragraph and info_weight(other_block.char_stats) >= 100):
|
||||
all_caps_vf = CharStats(prefix.to_string())
|
||||
if is_upper_dominant(all_caps_vf) and all_caps_vf.primary_slot[2] <= other_block.char_stats.primary_slot[3]:
|
||||
if heading_kind is not None:
|
||||
return make_heading_candidate(page_scan, heading_kind.type, heading_kind.group_slot, heading_kind.numbering, heading_kind.secondary_slot, heading_kind.primary_slot, True)
|
||||
if trie_matches_all(SECTION_KEYWORDS_TRIE, prefix):
|
||||
return make_body_heading_candidate(page_scan, 6, other_block, prefix)
|
||||
return make_body_heading_candidate(page_scan, 0, other_block, prefix)
|
||||
|
||||
# Centered two-line heading branch.
|
||||
above_neighbor = neighbor_above(page_scan.tertiary_slot, other_block)
|
||||
gap = (above_neighbor.bottom_edge() - first_line.top_edge()) if above_neighbor is not None else math.inf
|
||||
intersection = first_line.bottom_edge() - second_line.top_edge()
|
||||
per_char = avg_char_width(first_line)
|
||||
centered_flag = False
|
||||
# When there is no above block, the infinite gap is sufficient for this
|
||||
# branch and later above-block checks must remain guarded.
|
||||
if first_anchor is not None:
|
||||
cond_outer = (
|
||||
first_anchor.measure_slot
|
||||
and not first_anchor_span(next_token).measure_slot if next_token is not None else False
|
||||
)
|
||||
# First-line anchor, second-line anchor, gap, neighbor, and punctuation
|
||||
# checks together identify a centered heading prefix.
|
||||
if (first_anchor.measure_slot
|
||||
and next_token is not None and not first_anchor_span(next_token).measure_slot
|
||||
and (gap > 1.1 * intersection
|
||||
or (last_token(tokenize_block(above_neighbor)) is not None and is_word_token(last_token(tokenize_block(above_neighbor))))
|
||||
or last_line_of(above_neighbor).right_edge() < first_line.right_edge() - 8 * per_char)
|
||||
and (first_line.right_edge() > second_line.right_edge() - 4 * per_char
|
||||
or first_line.char_stats.tertiary_slot != 6
|
||||
or second_line.char_stats.secondary_slot == 3)):
|
||||
for prefix_token in prefix:
|
||||
if prefix_token.primary_slot == 2:
|
||||
centered_flag = True
|
||||
break
|
||||
if prefix_token.type == 2 or prefix_token.boundary_slot:
|
||||
break
|
||||
if (centered_flag
|
||||
and token is not None and is_trimmable_token(token)
|
||||
and next_token is not None and next_token.primary_slot == 2):
|
||||
if trie_matches_all(SECTION_KEYWORDS_TRIE, prefix):
|
||||
return make_body_heading_candidate(page_scan, 6, other_block, prefix)
|
||||
if letter_count > 1:
|
||||
return make_body_heading_candidate(page_scan, 0, other_block, prefix)
|
||||
return None
|
||||
@@ -0,0 +1,221 @@
|
||||
"""Block-text predicates: keyword matches, continuation, content, and number parsing."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Any, Optional
|
||||
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
|
||||
from ..model import (
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_strip_diacritics,
|
||||
_trim_unicode_ws,
|
||||
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
|
||||
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
|
||||
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
|
||||
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
from .keyword_tables import (
|
||||
ABSTRACT_KEYWORDS_TRIE,
|
||||
REFERENCES_TRIE,
|
||||
_normalize_text_key,
|
||||
ABSTRACT_KEYWORDS_SET,
|
||||
REFERENCES_SET,
|
||||
NUMBERED_PREFIX_RE,
|
||||
DEAD_DIGIT_RE,
|
||||
EQUATION_KEYWORDS_TRIE,
|
||||
ENGLISH_WORD_TO_NUMBER,
|
||||
ROMAN_NUMERAL_MAP,
|
||||
FORMULA_CHAR_WEIGHTS,
|
||||
)
|
||||
|
||||
|
||||
def token_text_of_block(block: Block) -> str:
|
||||
"""Tokenize ``block``, join tokens using their stored spacing flags, trim the result, and memoize it on the block."""
|
||||
if block.token_text_cache is not None:
|
||||
return block.token_text_cache
|
||||
block.token_text_cache = _trim_unicode_ws(tokenize_block(block).to_string())
|
||||
|
||||
return block.token_text_cache
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Simple heading and equation predicates.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def similar_style(block: Block, other_block: Block) -> bool:
|
||||
"""Return whether two blocks have very similar bold ratio and font size."""
|
||||
return abs(block.bold_frac() - other_block.bold_frac()) < 0.5 and abs(block.avg_font_size() - other_block.avg_font_size()) < 1
|
||||
|
||||
|
||||
def is_heading_continuation(block: Block, other_block: Block, candidate_number: int) -> bool:
|
||||
"""Return whether ``block`` is the next numbered heading continuation of ``other_block``."""
|
||||
if block.type != 0 or block.char_count() >= 500 or not similar_style(other_block, block):
|
||||
return False
|
||||
text = token_text_of_block(block)
|
||||
if block.left_edge() >= other_block.left_edge() and text.startswith("•"):
|
||||
return True
|
||||
if is_upper_dominant(other_block.char_stats) and is_upper_dominant(block.char_stats) and not left_aligned(other_block, block, 1) and not right_aligned(other_block, block, 1) and center_aligned(other_block, block, 1):
|
||||
return False
|
||||
heading = NUMBERED_PREFIX_RE.match(text)
|
||||
if heading and len(heading.groups()) >= 1:
|
||||
matched_number = to_number(heading.group(1))
|
||||
return abs(candidate_number - matched_number) == 1
|
||||
return False
|
||||
|
||||
|
||||
def matches_abstract(tokens: TokenView) -> bool:
|
||||
"""Token sequence matches abstract keywords or their normalized text set."""
|
||||
if trie_matches_all(ABSTRACT_KEYWORDS_TRIE, tokens):
|
||||
return True
|
||||
if tokens.length > 10:
|
||||
return False
|
||||
normalized = ""
|
||||
for candidate_item in tokens:
|
||||
if is_word_token(candidate_item):
|
||||
continue
|
||||
if candidate_item.type != 2 or len(normalized) + len(candidate_item.str) > 20:
|
||||
return False
|
||||
normalized += _normalize_text_key(candidate_item.str.lower())
|
||||
return normalized in ABSTRACT_KEYWORDS_SET
|
||||
|
||||
|
||||
def matches_references(tokens: TokenView) -> bool:
|
||||
"""Token sequence matches references keywords or their whole-text set."""
|
||||
secondary_item = trie_prefix_match(REFERENCES_TRIE, tokens)
|
||||
if secondary_item is None:
|
||||
if tokens.length <= 15:
|
||||
normalized = ""
|
||||
for candidate_item in tokens:
|
||||
if is_word_token(candidate_item):
|
||||
continue
|
||||
if candidate_item.type != 2 or len(normalized) + len(candidate_item.str) > 20:
|
||||
return False
|
||||
normalized += candidate_item.str.lower()
|
||||
return normalized in REFERENCES_SET
|
||||
return False
|
||||
if secondary_item.length == tokens.length:
|
||||
return True
|
||||
rest = tokens.slice(secondary_item.length)
|
||||
if rest.length == 1:
|
||||
first = rest.token_at(0)
|
||||
if first is not None and is_word_token(first):
|
||||
return True
|
||||
return trie_matches_all(REFERENCES_TRIE, rest)
|
||||
|
||||
|
||||
def vertically_close(block: Optional[Block], other_block: Block) -> bool:
|
||||
"""a is vertically very close to b."""
|
||||
if block is None:
|
||||
return False
|
||||
candidate_item = block.bottom_edge() - other_block.top_edge() if block.top_edge() > other_block.top_edge() else other_block.bottom_edge() - block.top_edge()
|
||||
return candidate_item < 2 * other_block.avg_font_size() or (x_aligned(block, other_block, 1) and candidate_item < 5 * other_block.avg_font_size())
|
||||
|
||||
|
||||
def is_equation_adjacent_line(line: Optional[Line], block: Block) -> bool:
|
||||
"""Return whether a line is adjacent to an equation block: it overlaps and follows the block, matches the equation-separator pattern, or consists entirely of equation-keyword tokens after trimming wrapper punctuation."""
|
||||
from ..labels import extract_structural_number
|
||||
if line is None or line.line_count() != 1:
|
||||
return False
|
||||
if line.left_edge() < block.right_edge() or not y_overlaps(block, line):
|
||||
return False
|
||||
if DEAD_DIGIT_RE.match(block_text(line)):
|
||||
return True # Equation separator match is enough to accept.
|
||||
tokens = tokenize_block(line)
|
||||
# Drop single non-digit chars at both edges when token-count is >= 3.
|
||||
if (tokens.length >= 3
|
||||
and (first := first_token(tokens)) is not None and len(first.str) <= 1
|
||||
and first.type != 1
|
||||
and (last := last_token(tokens)) is not None and len(last.str) <= 1
|
||||
and last.type != 1):
|
||||
tokens = tokens.slice(1, tokens.length - 1)
|
||||
tokens = strip_trie_match(tokens, EQUATION_KEYWORDS_TRIE)
|
||||
yi_match = extract_structural_number(tokens)
|
||||
return yi_match is not None and yi_match.length == tokens.length
|
||||
|
||||
|
||||
def has_substantive_content(block: Block, other_block: Optional[Block], candidate_block: Optional[Block]) -> bool:
|
||||
"""heuristic "this block has substantive content?" score >= 5."""
|
||||
entry_item = 0
|
||||
for token in tokenize_block(block):
|
||||
anchor = first_anchor_span(token)
|
||||
line = token.line()
|
||||
size = line.previous_slot
|
||||
flag = anchor.top_edge() < line.bottom_edge() + 0.8 * size or anchor.bottom_edge() > line.top_edge() - 0.8 * size
|
||||
if token.type == 1:
|
||||
entry_item += 2 if flag else 1
|
||||
continue
|
||||
weight = FORMULA_CHAR_WEIGHTS.get(token.str)
|
||||
if weight is not None:
|
||||
entry_item += (3 if flag else 1) * weight
|
||||
continue
|
||||
if token.type == 6:
|
||||
entry_item += (3 if flag else 1) * 5
|
||||
continue
|
||||
if len(token.str) <= 3 and token.primary_slot != 4:
|
||||
if flag:
|
||||
entry_item += 5 if is_word_token(token) else 1
|
||||
continue
|
||||
if flag:
|
||||
continue
|
||||
len_value = (2 if anchor.primary_slot else 1) * len(token.str)
|
||||
if token.primary_slot == 4:
|
||||
entry_item -= 2 * len_value
|
||||
elif token.primary_slot == 2:
|
||||
entry_item -= len_value
|
||||
elif token.primary_slot == 3:
|
||||
entry_item -= 0.5 * len_value
|
||||
if entry_item < 0:
|
||||
return False
|
||||
if entry_item >= 5:
|
||||
return True
|
||||
return is_equation_adjacent_line(other_block, block) or is_equation_adjacent_line(candidate_block, block)
|
||||
|
||||
|
||||
def is_cover_page(doc, page) -> bool:
|
||||
"""Return whether ``page`` behaves like a cover page: it is title-marked, appears early, and has light content or no body text."""
|
||||
return (
|
||||
page.auxiliary_slot
|
||||
and page.page_index < max(2, len(doc.primary_slot) / 2)
|
||||
and (
|
||||
page.primary_slot.secondary_slot < clamp(0.5 * doc.secondary_slot.secondary_slot, 200, 1000)
|
||||
or not page.state_slot
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def clamp(value: float, lower_bound: float, upper_bound: float) -> float:
|
||||
"""``max(lo, min(hi, v))``. NaN propagates."""
|
||||
measure_item = upper_bound if upper_bound < value else value
|
||||
return lower_bound if lower_bound > measure_item else measure_item
|
||||
|
||||
|
||||
def token_to_number(tok: Optional[Token]) -> Optional[int | float]:
|
||||
"""extract numeric value from a token (digit, Roman, or English)."""
|
||||
if tok is None:
|
||||
return None
|
||||
if tok.type == 1:
|
||||
token = token_numeric_value(tok)
|
||||
if not math.isnan(token) and token > 0:
|
||||
return int(token) if token.is_integer() else token
|
||||
return None
|
||||
return ROMAN_NUMERAL_MAP.get(tok.str) or ENGLISH_WORD_TO_NUMBER.get(tok.str.lower())
|
||||
|
||||
|
||||
def letter_to_ordinal(tok_str: str) -> Optional[int]:
|
||||
"""'a'/'A' -> 1, 'b' -> 2, ..., 'h' -> 8. None otherwise."""
|
||||
if len(tok_str) != 1:
|
||||
return None
|
||||
# Only the FIRST UTF-16 code unit of the lowercased character counts: a
|
||||
# case mapping that expands to several units (U+0130) contributes just its
|
||||
# first, and an astral lowercase contributes its high surrogate.
|
||||
low = tok_str[0].lower()
|
||||
code_unit = ord(low[0])
|
||||
if code_unit > 0xFFFF:
|
||||
code_unit = 0xD800 + ((code_unit - 0x10000) >> 10)
|
||||
value = code_unit - 96
|
||||
return value if 1 <= value <= 8 else None
|
||||
@@ -0,0 +1,48 @@
|
||||
"""Keyword-labeled section and caption-region detection. This module finds blocks that look like figure/table/chart labels or named
|
||||
sections, then extends each label forward or backward to claim the associated
|
||||
body blocks. The resulting regions are used by classification and outline
|
||||
assembly to avoid treating captions or labeled content as ordinary headings.
|
||||
"""
|
||||
|
||||
import regex as regex_module # Unicode \p{...} property classes.
|
||||
from typing import Optional
|
||||
|
||||
from ..classification import FIGURE_KEYWORDS_TRIE, TABLE_KEYWORDS_TRIE, CHART_KEYWORDS_TRIE
|
||||
from ..model import (
|
||||
Rect, rect_union, extend_top_to, extend_bottom_to, EMPTY_RECT, Bounded,
|
||||
_trim_unicode_ws,
|
||||
center_aligned, last_span, heading_score, reading_order_key, numbering_text, Line, last_line_of, first_span_of, dominant_style_of, info_weight, Block,
|
||||
)
|
||||
from ..stats import column_index_of
|
||||
from ..tokens import Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_leading_if_in, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, BuiltTrie, is_word_token
|
||||
|
||||
from .caption_text import (
|
||||
PERIOD_CHARS,
|
||||
STRUCTURAL_NUMBER_RE,
|
||||
is_number_separator,
|
||||
extract_structural_number,
|
||||
format_caption_label,
|
||||
REFERENCE_PHRASE_TRIE,
|
||||
is_uppercase_dominant,
|
||||
trie_matches_all,
|
||||
advance_past_line,
|
||||
skip_bracketed_word,
|
||||
token_case_signal,
|
||||
caption_outranks,
|
||||
)
|
||||
from .caption_regions import (
|
||||
CaptionedRegion,
|
||||
dedupe_caption_entries,
|
||||
extend_caption_region,
|
||||
build_caption_regions,
|
||||
CaptionEntry,
|
||||
CaptionContext,
|
||||
iter_page_blocks,
|
||||
detect_captions,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"PERIOD_CHARS", "STRUCTURAL_NUMBER_RE", "is_number_separator", "extract_structural_number", "format_caption_label", "REFERENCE_PHRASE_TRIE",
|
||||
"is_uppercase_dominant", "trie_matches_all", "advance_past_line", "skip_bracketed_word", "token_case_signal", "caption_outranks",
|
||||
"CaptionEntry", "CaptionedRegion", "CaptionContext", "iter_page_blocks", "detect_captions", "dedupe_caption_entries", "extend_caption_region", "build_caption_regions",
|
||||
]
|
||||
@@ -0,0 +1,366 @@
|
||||
"""Caption region growth, deduplication, and detection."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Optional
|
||||
|
||||
from ..classification import FIGURE_KEYWORDS_TRIE, TABLE_KEYWORDS_TRIE, CHART_KEYWORDS_TRIE
|
||||
from ..model import (
|
||||
Rect, rect_union, extend_top_to, extend_bottom_to, EMPTY_RECT, Bounded,
|
||||
_trim_unicode_ws,
|
||||
center_aligned, last_span, heading_score, reading_order_key, numbering_text, Line, last_line_of, first_span_of, dominant_style_of, info_weight, Block,
|
||||
)
|
||||
from ..stats import column_index_of
|
||||
from ..tokens import Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_leading_if_in, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, BuiltTrie, is_word_token
|
||||
|
||||
from .caption_text import (
|
||||
PERIOD_CHARS,
|
||||
extract_structural_number,
|
||||
format_caption_label,
|
||||
REFERENCE_PHRASE_TRIE,
|
||||
caption_outranks,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Captioned/labeled region wrapper #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class CaptionedRegion(Bounded):
|
||||
"""Captioned or labeled region plus its body blocks. The region stores the document context, page, heading block, body blocks, neighboring block reference, label flag, label type, and an area-weighted score used to choose forward vs backward extension."""
|
||||
|
||||
__slots__ = ("weighted_ratio_primary", "page", "primary_slot", "output_slot", "state_slot", "alignment_slot", "type", "score")
|
||||
|
||||
def __init__(self, primary_item, secondary_item, candidate_item, bbox: Rect, blocks, next_item, flag):
|
||||
super().__init__(bbox)
|
||||
self.weighted_ratio_primary = primary_item
|
||||
self.page = secondary_item
|
||||
self.primary_slot = candidate_item # the original heading block
|
||||
self.output_slot = blocks # list of body blocks
|
||||
self.state_slot = next_item
|
||||
self.alignment_slot = flag
|
||||
# Caption label type is carried by the heading block marker.
|
||||
# ``Block.type`` is a later classification label and is still zero here.
|
||||
self.type = candidate_item.marker_slot
|
||||
# Region score formula.
|
||||
area_pct = 100.0 * self.area() / self.page.bounds.area() if self.page.bounds.area() > 0 else 0.0
|
||||
if area_pct <= 0:
|
||||
score = 0.0
|
||||
else:
|
||||
if (self.state_slot is not None
|
||||
and self.state_slot.top_edge() < self.top_edge()
|
||||
and self.state_slot.right_edge() > self.left_edge()
|
||||
and self.alignment_slot):
|
||||
area_pct /= 5.0
|
||||
if self.type == 4:
|
||||
inner = 0.0
|
||||
for block in self.output_slot:
|
||||
if block.skew_frac() > 1:
|
||||
continue
|
||||
inner += block.area()
|
||||
score = area_pct * max(0.1, 1 - inner / self.area()) if self.area() > 0 else 0.0
|
||||
else:
|
||||
# Span text is a string, so every span contributes its character
|
||||
# count to the caption-region score.
|
||||
count = 1.0
|
||||
for block in self.output_slot:
|
||||
for line in block:
|
||||
for span in line:
|
||||
count += span.char_count()
|
||||
score = count * area_pct
|
||||
self.score = score
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Deduplicate caption entries and keep the best entry for each label.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def dedupe_caption_entries(caption_context: "CaptionContext") -> list["CaptionEntry"]:
|
||||
"""Deduplicate structural-number entries by label while preserving page order."""
|
||||
if not caption_context.state_slot:
|
||||
return caption_context.auxiliary_slot
|
||||
captions_by_label: dict[str, CaptionEntry] = {}
|
||||
for caption in caption_context.auxiliary_slot:
|
||||
if len(caption.primary_slot) <= 1:
|
||||
continue
|
||||
existing = captions_by_label.get(caption.primary_slot)
|
||||
if existing is None or caption_outranks(caption, existing):
|
||||
captions_by_label[caption.primary_slot] = caption
|
||||
out = list(captions_by_label.values())
|
||||
out.sort(key=lambda caption_sort_key: (caption_sort_key.page_index, caption_sort_key.group_slot.reading_order_index))
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Extend a labeled section forward or backward.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def extend_caption_region(
|
||||
caption_context: "CaptionContext",
|
||||
entry: "CaptionEntry",
|
||||
prior_regions: list,
|
||||
page_set: Optional[set],
|
||||
direction: int,
|
||||
) -> Optional[CaptionedRegion]:
|
||||
"""Walk page blocks forward or backward from a labeled entry, accumulating a region until an already-classified block, claimed block, deep body block, fresh top-level heading, or size/gap boundary is reached."""
|
||||
page = caption_context.primary_slot.primary_slot[entry.page_index - 1]
|
||||
origin = entry.group_slot
|
||||
anchor = origin.bottom_edge() if direction > 0 else origin.top_edge()
|
||||
bbox = Rect(origin.left_edge(), origin.right_edge(), anchor, anchor)
|
||||
blocks: list[Block] = []
|
||||
sorted_value = page.secondary_slot
|
||||
index = entry.group_slot.reading_order_index + direction
|
||||
previous: Block = origin
|
||||
|
||||
while 0 <= index < len(sorted_value):
|
||||
caption = sorted_value[index]
|
||||
caption_column = column_index_of(caption)
|
||||
if caption_column < 0:
|
||||
break
|
||||
# layout branch: when crossing the column band, walk page.j (column
|
||||
# rects) to the nearest column that horizontally overlaps the
|
||||
# bbox and extend the bbox vertically to that column's edge.
|
||||
if direction < 0 and caption_column < column_index_of(entry.group_slot) and caption.bottom_edge() < anchor:
|
||||
col_idx = caption_column - 1
|
||||
column_rect = page.tertiary_slot[col_idx] if 0 <= col_idx < len(page.tertiary_slot) else None
|
||||
while column_rect is not None and (
|
||||
column_rect.bottom_edge() < bbox.top_edge()
|
||||
or column_rect.right_edge() < bbox.left_edge()
|
||||
or column_rect.left_edge() > bbox.right_edge()
|
||||
):
|
||||
col_idx -= 1
|
||||
column_rect = page.tertiary_slot[col_idx] if 0 <= col_idx < len(page.tertiary_slot) else None
|
||||
if column_rect is not None:
|
||||
bbox = extend_top_to(bbox, column_rect.bottom_edge())
|
||||
else:
|
||||
bbox = extend_top_to(bbox, page.bounds.top_edge())
|
||||
break
|
||||
if direction > 0 and caption_column > column_index_of(entry.group_slot) and caption.top_edge() > anchor:
|
||||
col_idx = caption_column + 1
|
||||
column_rect = page.tertiary_slot[col_idx] if 0 <= col_idx < len(page.tertiary_slot) else None
|
||||
while column_rect is not None and (
|
||||
column_rect.top_edge() > bbox.bottom_edge()
|
||||
or column_rect.right_edge() < bbox.left_edge()
|
||||
or column_rect.left_edge() > bbox.right_edge()
|
||||
):
|
||||
col_idx += 1
|
||||
column_rect = page.tertiary_slot[col_idx] if 0 <= col_idx < len(page.tertiary_slot) else None
|
||||
if column_rect is not None:
|
||||
bbox = extend_bottom_to(bbox, column_rect.top_edge())
|
||||
else:
|
||||
bbox = extend_bottom_to(bbox, page.bounds.bottom_edge())
|
||||
break
|
||||
# Grow the bbox to include n
|
||||
if direction < 0:
|
||||
bbox = extend_top_to(bbox, caption.bottom_edge())
|
||||
else:
|
||||
bbox = extend_bottom_to(bbox, caption.top_edge())
|
||||
# Stop conditions
|
||||
if caption.type != 0 or caption.reading_order_index in caption_context.secondary_slot:
|
||||
break
|
||||
if page_set is not None and index in page_set:
|
||||
break
|
||||
size = min(caption_context.primary_slot.secondary_slot.primary_slot, entry.group_slot.avg_font_size())
|
||||
if caption.is_body_paragraph and caption.avg_font_size() > min(0.9 * size, size - 1.5):
|
||||
break
|
||||
next_block = sorted_value[index + 1] if index + 1 < len(sorted_value) else None
|
||||
gap = previous.bottom_edge() - caption.top_edge() if direction > 0 else 0
|
||||
line_gap = page.primary_slot.tertiary_slot - page.primary_slot.primary_slot
|
||||
if (
|
||||
direction > 0 and next_block is not None and caption.line_count() <= 4 and caption.char_stats.secondary_slot != 3
|
||||
and gap > line_gap
|
||||
and (previous is entry.group_slot or gap > min(3 * line_gap, caption.bottom_edge() - next_block.top_edge()))
|
||||
):
|
||||
next_item = sorted_value[index + 2] if index + 2 < len(sorted_value) else None
|
||||
if heading_score(caption) >= heading_score(previous) + 0.5 and (next_block.is_body_paragraph or (next_item is not None and next_item.is_body_paragraph)):
|
||||
break
|
||||
# A numbering-like line with enough trailing text can stop this
|
||||
# backward body-paragraph scan.
|
||||
line_text = numbering_text(caption.line())
|
||||
if (line_text
|
||||
and caption.char_stats.secondary_slot == 2
|
||||
and heading_score(caption) >= size
|
||||
and gap > 2 * caption.avg_font_size()
|
||||
and caption.char_count() - len(line_text) > 2):
|
||||
break
|
||||
blocks.append(caption)
|
||||
bbox = rect_union(bbox, caption.secondary_slot)
|
||||
index += direction
|
||||
previous = caption
|
||||
|
||||
if direction < 0 and index < 0:
|
||||
bbox = extend_top_to(bbox, page.bounds.top_edge())
|
||||
elif direction > 0 and index >= len(sorted_value):
|
||||
bbox = extend_bottom_to(bbox, page.bounds.bottom_edge())
|
||||
|
||||
# When a backward extension expands the region, also consume forward
|
||||
# neighbours whose geometric center sits inside the grown bbox.
|
||||
if direction < 0:
|
||||
fwd_idx = entry.group_slot.reading_order_index + 1
|
||||
while fwd_idx < len(sorted_value):
|
||||
block = sorted_value[fwd_idx]
|
||||
center_x = block.center_x()
|
||||
center_y = block.center_y()
|
||||
if (center_x < bbox.left_edge() or center_x > bbox.right_edge()
|
||||
or center_y < bbox.bottom_edge() or center_y > bbox.top_edge()):
|
||||
break
|
||||
blocks.append(block)
|
||||
bbox = rect_union(bbox, block.secondary_slot)
|
||||
fwd_idx += 1
|
||||
|
||||
area = bbox.area()
|
||||
if area <= 0:
|
||||
return None
|
||||
|
||||
# Check overlap with prior regions; if heavy overlap, reject.
|
||||
for prior in prior_regions:
|
||||
overlap_area = max(
|
||||
0.0,
|
||||
min(bbox.right, prior.secondary_slot.right) - max(bbox.left, prior.secondary_slot.left),
|
||||
) * max(
|
||||
0.0,
|
||||
min(bbox.top, prior.secondary_slot.top) - max(bbox.primary_slot, prior.secondary_slot.primary_slot),
|
||||
)
|
||||
if overlap_area >= 0.25 * min(area, prior.area()):
|
||||
return None
|
||||
|
||||
next_block = sorted_value[index] if 0 <= index < len(sorted_value) else None
|
||||
on_page_set = page_set is not None and index in page_set
|
||||
return CaptionedRegion(
|
||||
primary_item=caption_context.primary_slot, secondary_item=page, candidate_item=entry.group_slot,
|
||||
bbox=bbox, blocks=blocks, next_item=next_block, flag=on_page_set,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Extend all deduplicated labeled-section entries.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def build_caption_regions(caption_context: "CaptionContext") -> list[CaptionedRegion]:
|
||||
"""Build caption regions by extending each labeled entry in both directions."""
|
||||
caption_context.tertiary_slot.clear()
|
||||
caption_context.secondary_slot.clear()
|
||||
entries = dedupe_caption_entries(caption_context)
|
||||
for caption in entries:
|
||||
set_value = caption_context.tertiary_slot.get(caption.page_index)
|
||||
if set_value is None:
|
||||
set_value = set()
|
||||
caption_context.tertiary_slot[caption.page_index] = set_value
|
||||
set_value.add(caption.group_slot.reading_order_index)
|
||||
out: list[CaptionedRegion] = []
|
||||
page = 0
|
||||
prior_regions: list[CaptionedRegion] = []
|
||||
for entry in entries:
|
||||
if entry.page_index != page:
|
||||
prior_regions = []
|
||||
caption_context.secondary_slot.clear()
|
||||
page = entry.page_index
|
||||
if len(prior_regions) >= 8:
|
||||
continue
|
||||
page_set = caption_context.tertiary_slot.get(entry.page_index)
|
||||
back = extend_caption_region(caption_context, entry, prior_regions, page_set, -1)
|
||||
forward = extend_caption_region(caption_context, entry, prior_regions, page_set, 1)
|
||||
winner = (
|
||||
back if (back is not None and (forward is None or back.score > forward.score))
|
||||
else forward
|
||||
)
|
||||
if winner is not None:
|
||||
for body_block in winner.output_slot:
|
||||
caption_context.secondary_slot.add(body_block.reading_order_index)
|
||||
prior_regions.append(winner)
|
||||
out.append(winner)
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Labeled-section entry.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class CaptionEntry:
|
||||
"""One labeled-section entry with label, type, page, block, and remainder tokens."""
|
||||
|
||||
__slots__ = ("primary_slot", "type", "page_index", "group_slot", "secondary_slot")
|
||||
|
||||
def __init__(self, label: str, type_: int, page: int, block: Block, remainder: TokenView):
|
||||
self.primary_slot = label
|
||||
self.type = type_
|
||||
self.page_index = page
|
||||
self.group_slot = block
|
||||
self.secondary_slot = remainder
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Labeled-section context.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class CaptionContext:
|
||||
"""Document-level state for labeled-section detection."""
|
||||
|
||||
__slots__ = ("primary_slot", "auxiliary_slot", "state_slot", "tertiary_slot", "secondary_slot")
|
||||
|
||||
def __init__(self, doc):
|
||||
self.primary_slot = doc
|
||||
self.auxiliary_slot: list[CaptionEntry] = []
|
||||
self.state_slot: bool = False
|
||||
self.tertiary_slot: dict = {} # page -> set of heading-block ga
|
||||
self.secondary_slot: set = set() # set of heading-block ga across doc
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Document-wide (page, block) iterator.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def iter_page_blocks(doc):
|
||||
"""Yield ``{'page': page, 'G': block}`` records in reading order."""
|
||||
for page in doc.primary_slot:
|
||||
for block in (page.secondary_slot or []):
|
||||
yield {"page": page, "block": block}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Labeled-section detection driver.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def detect_captions(caption_context: CaptionContext) -> None:
|
||||
"""Find figure, table, and chart labels and record their structural prefixes."""
|
||||
for entry in iter_page_blocks(caption_context.primary_slot):
|
||||
page = entry["page"]
|
||||
block = entry["block"]
|
||||
if block.type != 0:
|
||||
continue
|
||||
tokens = tokenize_block(block)
|
||||
type_value: Optional[int] = None
|
||||
prefix = trie_prefix_match(FIGURE_KEYWORDS_TRIE, tokens)
|
||||
if prefix is not None:
|
||||
type_value = 4
|
||||
else:
|
||||
prefix = trie_prefix_match(TABLE_KEYWORDS_TRIE, tokens)
|
||||
if prefix is not None:
|
||||
type_value = 5
|
||||
else:
|
||||
prefix = trie_prefix_match(CHART_KEYWORDS_TRIE, tokens)
|
||||
if prefix is not None:
|
||||
type_value = 11
|
||||
if type_value is None:
|
||||
continue
|
||||
|
||||
remainder = strip_leading_if_in(tokens.slice(prefix.length), PERIOD_CHARS)
|
||||
number = extract_structural_number(remainder)
|
||||
label = format_caption_label(type_value, number)
|
||||
if number is not None:
|
||||
caption_context.state_slot = True
|
||||
remainder = remainder.slice(number.length)
|
||||
if trie_prefix_match(REFERENCE_PHRASE_TRIE, remainder) is not None:
|
||||
continue
|
||||
page.measure_slot = True
|
||||
caption_context.auxiliary_slot.append(CaptionEntry(label, type_value, page.page_index, block, remainder))
|
||||
# Mark the block's Y category (used by outline.py heading filter)
|
||||
block.marker_slot = type_value
|
||||
@@ -0,0 +1,171 @@
|
||||
"""Caption label text helpers and structural-number parsing."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import regex as regex_module # Unicode \p{...} property classes.
|
||||
from typing import Optional
|
||||
from ..model import (
|
||||
Rect, rect_union, extend_top_to, extend_bottom_to, EMPTY_RECT, Bounded,
|
||||
_trim_unicode_ws,
|
||||
center_aligned, last_span, heading_score, reading_order_key, numbering_text, Line, last_line_of, first_span_of, dominant_style_of, info_weight, Block,
|
||||
)
|
||||
from ..tokens import Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_leading_if_in, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, BuiltTrie, is_word_token
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Helpers #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
PERIOD_CHARS = {".", ".", "。", "。"} # period-character set
|
||||
|
||||
|
||||
# Structural-number pattern: Unicode numeric code points, optional letter
|
||||
# affixes, or Roman numerals. ``\Z`` anchors at the absolute end of string, not
|
||||
# before a trailing newline.
|
||||
STRUCTURAL_NUMBER_RE = regex_module.compile(
|
||||
r"^(?:[A-M]*\p{Number}+[A-Ma-m]?|[A-Ma-m]\p{Number}*|[IVX]+)\Z"
|
||||
)
|
||||
|
||||
|
||||
def is_number_separator(token: Optional[Token], other_flag: bool = True) -> bool:
|
||||
"""Return whether the token is a structural-number separator candidate."""
|
||||
if token is None:
|
||||
return False
|
||||
if token.boundary_slot:
|
||||
return False
|
||||
if token.type == 3:
|
||||
return True
|
||||
if other_flag and token.type == 4:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def extract_structural_number(tokens: TokenView, other_flag: bool = True) -> Optional[TokenView]:
|
||||
"""extract a leading structural-number prefix from tokens. Returns the matched prefix as a token-view slice, or None. """
|
||||
if tokens.length < 1:
|
||||
return None
|
||||
candidate_item = tokens
|
||||
first = tokens.token_at(0)
|
||||
if first is None:
|
||||
return None
|
||||
reference_item = first.str
|
||||
if len(reference_item) == 1 and "A" <= reference_item[0] <= "H":
|
||||
if not is_number_separator(tokens.token_at(1), other_flag):
|
||||
return None
|
||||
candidate_item = tokens.slice(2)
|
||||
if candidate_item.length < 1:
|
||||
return None
|
||||
head = first_token(candidate_item)
|
||||
if head is None or not STRUCTURAL_NUMBER_RE.match(head.str):
|
||||
return None
|
||||
candidate_item = candidate_item.slice(1)
|
||||
while candidate_item.length >= 2 and is_number_separator(candidate_item.token_at(0), other_flag) and STRUCTURAL_NUMBER_RE.match(candidate_item.token_at(1).str): # type: ignore[union-attr]
|
||||
candidate_item = candidate_item.slice(2)
|
||||
return tokens.slice(0, tokens.length - candidate_item.length)
|
||||
|
||||
|
||||
# - format code label
|
||||
def format_caption_label(type_: int, num: Optional[TokenView]) -> str:
|
||||
"""format the section-type letter prefix + number. type_ 4 -> "F", 5 -> "T", 11 -> "Q". Append the number string if any. """
|
||||
if type_ == 4:
|
||||
letter = "F"
|
||||
elif type_ == 5:
|
||||
letter = "T"
|
||||
elif type_ == 11:
|
||||
letter = "Q"
|
||||
else:
|
||||
return ""
|
||||
if num is not None:
|
||||
letter += _trim_unicode_ws(str(num))
|
||||
return letter
|
||||
|
||||
|
||||
# - case-sensitive trie of phrases that indicate "this is a
|
||||
# reference TO a figure/table, not a label OF one".
|
||||
REFERENCE_PHRASE_TRIE = build_trie(["lists the", "presents", "show the", "showed the", "shows"], set_case_fold(TrieConfig(), False))
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Token helpers for caption-entry ranking.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def is_uppercase_dominant(tokens: TokenView) -> bool:
|
||||
"""Return True when the token sequence is dominated by uppercase words. Multi-character lowercase-start words whose second character is not uppercase reject the sequence as body-like text."""
|
||||
from ..tokens import char_category
|
||||
secondary_item = candidate_item = 0
|
||||
for reference_item in tokens:
|
||||
if reference_item.type != 2:
|
||||
continue
|
||||
if reference_item.primary_slot == 2:
|
||||
secondary_item += 1
|
||||
elif reference_item.primary_slot == 3:
|
||||
if len(reference_item.str) > 4 and len(reference_item.str) >= 2 and char_category(reference_item.str[1]) != 2:
|
||||
return False
|
||||
candidate_item += 1
|
||||
return secondary_item > max(2, candidate_item)
|
||||
|
||||
|
||||
def trie_matches_all(trie: BuiltTrie, tokens: TokenView) -> bool:
|
||||
"""tokens fully match ``trie`` (or all but a final word-y token)."""
|
||||
match = trie_prefix_match(trie, tokens)
|
||||
if match is None:
|
||||
return False
|
||||
if match.length == tokens.length:
|
||||
return True
|
||||
if match.length == tokens.length - 1:
|
||||
last = last_token(tokens)
|
||||
return last is not None and is_word_token(last)
|
||||
return False
|
||||
|
||||
|
||||
def advance_past_line(tokens: TokenView, line: Line, index: int) -> int:
|
||||
"""Advance while the token at the current index belongs to ``line``."""
|
||||
while index < tokens.length:
|
||||
tok = tokens.token_at(index)
|
||||
if tok is None:
|
||||
break
|
||||
if tok.line() is not line:
|
||||
break
|
||||
index += 1
|
||||
return index
|
||||
|
||||
|
||||
def skip_bracketed_word(tokens: TokenView, index: int) -> int:
|
||||
"""advance over bracket-attached word token."""
|
||||
tok = tokens.token_at(index)
|
||||
if tok is not None and tok.boundary_slot and is_word_token(tok):
|
||||
return index + 1
|
||||
return index
|
||||
|
||||
|
||||
def token_case_signal(token: Optional[Token]) -> int:
|
||||
"""per-token "direction signal". Returns 2 if g==7/6 (sentence end), 1 if g==2 (uppercase), -1 if g==3 (lowercase), 0 otherwise. """
|
||||
if token is None:
|
||||
return 0
|
||||
token_kind = token.primary_slot
|
||||
if token_kind == 7 or token_kind == 6:
|
||||
return 2
|
||||
if token_kind == 2:
|
||||
return 1
|
||||
if token_kind == 3:
|
||||
return -1
|
||||
return 0
|
||||
|
||||
|
||||
def caption_outranks(caption_entry: "CaptionEntry", other_caption_entry: "CaptionEntry") -> bool:
|
||||
"""Return True when the first caption entry ranks better than the second."""
|
||||
caption = is_uppercase_dominant(tokenize_block(caption_entry.group_slot))
|
||||
other_is_uppercase = is_uppercase_dominant(tokenize_block(other_caption_entry.group_slot))
|
||||
if caption != other_is_uppercase:
|
||||
return caption
|
||||
caption_first_token = first_token(caption_entry.secondary_slot) if caption_entry.secondary_slot.length > 0 else None
|
||||
other_first_token = first_token(other_caption_entry.secondary_slot) if other_caption_entry.secondary_slot.length > 0 else None
|
||||
group = token_case_signal(caption_first_token)
|
||||
other_case_signal = token_case_signal(other_first_token)
|
||||
if group != other_case_signal:
|
||||
return group > other_case_signal
|
||||
if caption_entry.page_index != other_caption_entry.page_index:
|
||||
return caption_entry.page_index < other_caption_entry.page_index
|
||||
return caption_entry.group_slot.reading_order_index < other_caption_entry.group_slot.reading_order_index
|
||||
@@ -0,0 +1,314 @@
|
||||
"""
|
||||
End-to-end orchestrator for the TOC extraction pipeline. Pipeline order: 1. parse character-level spans and page viewport metadata 2. cluster spans into lines 3. compute page statistics 4. detect columns and recluster lines with column awareness 5. remove line-number artifacts and recompute statistics 6. compute document-level statistics 7. cluster lines into blocks and assign reading order 8. classify headers, footers, watermarks, TOC-like pages, captions, references, and body paragraphs 9. detect the document title 10. collect heading candidates and assemble the final outline The ordering is load-bearing: title selection, labeled-section detection,
|
||||
heading candidate collection, and outline assembly each consume annotations
|
||||
from the previous stages. ``to_pageindex_tree`` serializes the final outline
|
||||
into the JSON shape that ``run_pageindex.py`` writes.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import unicodedata
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from typing import Optional, Union
|
||||
|
||||
# (re is used by the title-reject regex below)
|
||||
|
||||
from .blocks import cluster_lines_into_blocks, BlockClusterContext
|
||||
from .classification import is_body_paragraph, detect_header_footer, HeaderFooterContext, mark_watermarks, mark_toc_and_boilerplate
|
||||
from .labels import detect_captions, build_caption_regions, CaptionContext
|
||||
from .model import Rect, numbering_kind, block_text, deaccented_text, Block
|
||||
from .outline_assembly import (
|
||||
build_heading_from_block, is_landscape_or_empty, is_outline_valid, is_chapter_outline_valid, mark_outline_block_types, assemble_outline, compute_max_heading_gap, has_table_or_prominent, OutlineNode, outline_to_dict_tree,
|
||||
)
|
||||
from .parser_pdfium_parallel import parse_charlevel_meta_parallel
|
||||
from .phases import assign_reading_order, PageView, process_page
|
||||
PageView = PageView # re-export for type hints
|
||||
from .stats import compute_doc_stats
|
||||
from .title import detect_title
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# References-section dictionary (load once) #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
_DICT_PATH = Path(__file__).parent / "data" / "dictionaries.json"
|
||||
|
||||
|
||||
def _normalize_text_key(text: str) -> str:
|
||||
return " ".join(unicodedata.normalize("NFKC", text).strip().split()).lower()
|
||||
|
||||
|
||||
_REFS_DICT_RAW = json.loads(_DICT_PATH.read_text(encoding="utf-8"))
|
||||
REFERENCES_KEYWORDS = frozenset(_normalize_text_key(text_value) for text_value in _REFS_DICT_RAW.get("references", []) if text_value)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Document container #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class DocumentState:
|
||||
"""Document-level extraction state: pages, document statistics, and recurring-text frequency map. """
|
||||
|
||||
__slots__ = ("primary_slot", "secondary_slot", "tertiary_slot")
|
||||
|
||||
def __init__(self, pages: list[PageView]):
|
||||
self.primary_slot = pages
|
||||
self.secondary_slot = None # set after document statistics are computed
|
||||
self.tertiary_slot: dict = {}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# References-section detection #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def find_references(doc: DocumentState) -> Optional[tuple[int, Block]]:
|
||||
"""Return ``(page_num, block)`` for the first references heading in reading order."""
|
||||
for page in doc.primary_slot:
|
||||
for block in (page.secondary_slot or []):
|
||||
if block.type != 0:
|
||||
continue
|
||||
normalized = deaccented_text(block)
|
||||
if not normalized or len(normalized) > 80:
|
||||
continue
|
||||
if normalized in REFERENCES_KEYWORDS:
|
||||
return page.page_index, block
|
||||
# Allow short numbered prefix: "12. References"
|
||||
parts = normalized.split()
|
||||
if 1 <= len(parts) <= 4 and parts[-1] in REFERENCES_KEYWORDS:
|
||||
return page.page_index, block
|
||||
return None
|
||||
|
||||
|
||||
def mark_references(doc: DocumentState, ref: Optional[tuple[int, Block]]) -> None:
|
||||
"""Tag the references heading itself + everything after as type=3."""
|
||||
if ref is None:
|
||||
return
|
||||
ref_page, ref_block = ref
|
||||
seen = False
|
||||
for page in doc.primary_slot:
|
||||
if page.page_index < ref_page:
|
||||
continue
|
||||
for block in (page.secondary_slot or []):
|
||||
if not seen and block is ref_block:
|
||||
seen = True
|
||||
block.type = 3
|
||||
continue
|
||||
if seen:
|
||||
block.type = 3
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Repeated-text accumulator #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def page_by_block_lookup(pages, block) -> Optional[PageView]:
|
||||
"""Find which page owns ``block``. Used for wrapping labeled blocks."""
|
||||
for page in pages:
|
||||
if block in (page.secondary_slot or []):
|
||||
return page
|
||||
return None
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# End-to-end entry point #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def extract_toc(
|
||||
doc_handle: Union[str, Path, BytesIO],
|
||||
workers: Optional[int] = None,
|
||||
) -> dict:
|
||||
"""Run the full pipeline. Returns a dict shaped like:: { "doc_name": "...", "doc_title": "...", "structure": [ {"title": "...", "start_index": 1, "end_index": 3, "nodes": [...]}, ... ], "has_abstract_or_references_section": False } ``has_abstract_or_references_section`` is True when any TOP-LEVEL outline entry is an abstract-keyword heading or carries the prominent-heading flag (a references-keyword heading, plain or numbered). The near-empty bail and the valid-outline branch both report False. ``workers`` sets the process count for the per-page parallel parser: None = auto (CPU count - 1), 1 forces the sequential path; output is identical either way. """
|
||||
# ----- 1) Parse PDF -> flat spans per page --------------------------
|
||||
# per-page (view box, /Rotate) comes from the same engine (PDFium) that
|
||||
# produced the block coordinates, so the geometry frame is consistent.
|
||||
parsed, page_meta = parse_charlevel_meta_parallel(doc_handle, workers=workers)
|
||||
|
||||
# ----- 2) Per-page layout classification ----------------------------------
|
||||
# Heading coordinate projection uses the page viewport.
|
||||
pages: list[PageView] = []
|
||||
for index_value, spans in enumerate(parsed):
|
||||
viewport_box_value, rot = page_meta[index_value]
|
||||
viewport_x0, viewport_y0, viewport_x1, viewport_y1 = viewport_box_value
|
||||
# page bbox uses DISPLAYED (post-/Rotate) dims.
|
||||
viewport_width, viewport_height = abs(viewport_x1 - viewport_x0), abs(viewport_y1 - viewport_y0)
|
||||
page_width, page_height = (viewport_height, viewport_width) if rot % 180 == 90 else (viewport_width, viewport_height)
|
||||
page_bbox = Rect(0, page_width, page_height, 0)
|
||||
page = process_page(spans, page_num=index_value + 1, page_bbox=page_bbox)
|
||||
if viewport_box_value is not None:
|
||||
page.viewport_box, page.rot = viewport_box_value, rot
|
||||
pages.append(page)
|
||||
|
||||
# ----- 3) Document-level stats --------------------------------------
|
||||
doc = DocumentState(pages)
|
||||
doc.secondary_slot = compute_doc_stats(pages)
|
||||
|
||||
# ----- 4) Block clustering per page, then reading order -------------
|
||||
for page in pages:
|
||||
ctx = BlockClusterContext(doc.secondary_slot, page.bounds, page.primary_slot, page.lines, page.tertiary_slot)
|
||||
page.blocks = cluster_lines_into_blocks(ctx)
|
||||
assign_reading_order(page, page.blocks)
|
||||
|
||||
# ----- Early empty-outline gate ------------------------------------
|
||||
# Short, near-empty, unsupported-script, or mostly-landscape documents
|
||||
# emit an empty outline rather than a fabricated structure.
|
||||
if (doc.secondary_slot.state_slot <= 300 or doc.secondary_slot.previous_slot <= 200
|
||||
or doc.secondary_slot.tertiary_slot in (0, 2, 10) or is_landscape_or_empty(doc)):
|
||||
if isinstance(doc_handle, (str, Path)):
|
||||
doc_name = Path(str(doc_handle)).name
|
||||
else:
|
||||
doc_name = "document.pdf"
|
||||
return {
|
||||
"doc_name": doc_name,
|
||||
"doc_title": None,
|
||||
"structure": [],
|
||||
"has_abstract_or_references_section": False,
|
||||
}
|
||||
|
||||
# ----- 5) Classification: header / footer / watermark / TOC pages ---
|
||||
detect_header_footer(HeaderFooterContext(doc, 1)) # HEADER
|
||||
detect_header_footer(HeaderFooterContext(doc, 2)) # FOOTER
|
||||
mark_watermarks(doc)
|
||||
mark_toc_and_boilerplate(doc)
|
||||
|
||||
# ----- 6) Body-paragraph flagging (post-classification) -------------
|
||||
# Populates body-paragraph flags, page substantive-body flags,
|
||||
# and per-page body-style hashes.
|
||||
from .model import dominant_style_of as span_style_hash
|
||||
for page in pages:
|
||||
for block in (page.output_slot or []):
|
||||
if block.type == 0:
|
||||
block.is_body_paragraph = is_body_paragraph(doc.secondary_slot, page, block)
|
||||
if block.is_body_paragraph:
|
||||
page.state_slot = True
|
||||
# The empty style hash is significant for later page-level
|
||||
# membership checks, so it must be retained.
|
||||
page.style_slot.add(span_style_hash(block))
|
||||
|
||||
# ----- 7) Title selection ------------------------------------------
|
||||
# Title selection and title-echo marking must run before labeled-section
|
||||
# detection and heading collection so title blocks are excluded from both.
|
||||
from .classification import bounded_edit_distance, _normalize_text_key
|
||||
doc_title: Optional[str] = None
|
||||
title_winner = detect_title(doc)
|
||||
if title_winner is not None:
|
||||
# Emit the full joined title string, preserving inter-block spaces.
|
||||
doc_title = title_winner.to_string()
|
||||
title_winner.page.auxiliary_slot = True
|
||||
for block in title_winner.output_slot:
|
||||
block.type = 3
|
||||
title_norm = _normalize_text_key(title_winner.to_string()).lower()
|
||||
# The body-paragraph break exits only the inner block loop; later
|
||||
# pages are still scanned for title-echo headers.
|
||||
for candidate_page in doc.primary_slot:
|
||||
for candidate_block in (candidate_page.output_slot or []):
|
||||
if candidate_block.type != 0:
|
||||
continue
|
||||
normalized = deaccented_text(candidate_block).lower()
|
||||
if (len(normalized) > 20 and len(title_norm) > 20 and (
|
||||
normalized.startswith(title_norm)
|
||||
or title_norm.startswith(normalized)
|
||||
or title_norm.endswith(normalized))):
|
||||
candidate_page.auxiliary_slot = True
|
||||
candidate_block.type = 3
|
||||
continue
|
||||
threshold = 0.2 * min(len(normalized), len(title_norm))
|
||||
if bounded_edit_distance(normalized, title_norm, threshold) < threshold:
|
||||
candidate_page.auxiliary_slot = True
|
||||
candidate_block.type = 3
|
||||
elif candidate_block.is_body_paragraph:
|
||||
break
|
||||
|
||||
# ----- 8) Keyword-labeled section detection -------------------------
|
||||
# Labeled section regions are built here but extended after heading collection.
|
||||
caption_context = CaptionContext(doc)
|
||||
detect_captions(caption_context)
|
||||
|
||||
# ----- 9) General heading collection --------------------------------
|
||||
# Heading collection runs before labeled regions claim their body blocks.
|
||||
# The start page skips the title page when a title was found.
|
||||
page_lookup: dict[int, int] = {}
|
||||
for page in pages:
|
||||
for block in (page.secondary_slot or []):
|
||||
page_lookup[id(block)] = page.page_index
|
||||
|
||||
from .heading_detection import find_section_openers as _find_section_openers
|
||||
title_page_idx = title_winner.page.page_index if title_winner is not None else 0
|
||||
section_openers = _find_section_openers(doc, title_page_idx)
|
||||
|
||||
# ----- 10) Extend labeled sections and claim body blocks ------------
|
||||
# Each labeled heading keeps its label type; body blocks are marked with
|
||||
# the used-as-heading flag so heading collection skips claimed caption/section bodies.
|
||||
# The head block type is preserved; claimed body blocks are not retyped.
|
||||
caption_regions = build_caption_regions(caption_context)
|
||||
for caption_region in caption_regions:
|
||||
head_block = caption_region.primary_slot
|
||||
head_block.state_slot = head_block.marker_slot
|
||||
for body_block in caption_region.output_slot:
|
||||
body_block.measure_slot = True
|
||||
|
||||
# NOTE: References-section detection -- intentionally absent ---------
|
||||
# Bulk-marking everything after a references heading would hide later
|
||||
# appendix headings in some documents, so references detection remains off.
|
||||
# ref = find_references(doc)
|
||||
# mark_references(doc, ref)
|
||||
|
||||
# NOTE: Ghost-text histogram -- intentionally absent -----------------
|
||||
# Recurring text is counted during header/footer/watermark marking. A
|
||||
# second doc-wide pass would double-count headers and pollute title scoring.
|
||||
|
||||
# ----- 11) Outline assembly and validation gate ---------------------
|
||||
outline_nodes = assemble_outline(doc, section_openers)
|
||||
# Validate the assembled outline. Structured outlines must cover enough
|
||||
# chapters; unstructured outlines are filtered by script and density gap.
|
||||
# The abstract/references signal rides along with this gate: it is False on
|
||||
# the valid-outline branch, and on the other branch it is read off the
|
||||
# possibly-emptied list once the density filter has run.
|
||||
if is_outline_valid(doc, outline_nodes):
|
||||
if not is_chapter_outline_valid(doc, outline_nodes):
|
||||
outline_nodes = []
|
||||
has_abstract_or_references = False
|
||||
else:
|
||||
mark_outline_block_types(outline_nodes)
|
||||
page_count = len(doc.primary_slot)
|
||||
if doc.secondary_slot.tertiary_slot == 7 or (
|
||||
page_count >= 3
|
||||
and compute_max_heading_gap(outline_nodes, 1)["max_gap"] > (0.65 if doc.secondary_slot.tertiary_slot == 4 else 0.85) * page_count
|
||||
):
|
||||
outline_nodes = []
|
||||
has_abstract_or_references = has_table_or_prominent(outline_nodes)
|
||||
if outline_nodes:
|
||||
structure = outline_to_dict_tree(outline_nodes, total_pages=len(pages))
|
||||
else:
|
||||
structure = []
|
||||
|
||||
# ----- 12) Output ---------------------------------------------------
|
||||
if isinstance(doc_handle, (str, Path)):
|
||||
doc_name = Path(str(doc_handle)).name
|
||||
else:
|
||||
doc_name = "document.pdf"
|
||||
|
||||
page_texts = []
|
||||
for page in pages:
|
||||
parts = []
|
||||
for block in (page.secondary_slot or []):
|
||||
parts.append(block_text(block))
|
||||
page_texts.append("\n".join(parts))
|
||||
|
||||
return {
|
||||
"doc_name": doc_name,
|
||||
"doc_title": doc_title,
|
||||
"structure": structure,
|
||||
"has_abstract_or_references_section": has_abstract_or_references,
|
||||
"page_texts": page_texts,
|
||||
}
|
||||
|
||||
|
||||
__all__ = ["extract_toc", "DocumentState", "find_references", "mark_references"]
|
||||
@@ -0,0 +1,121 @@
|
||||
"""
|
||||
Data model for rectangles, spans, lines, blocks, character categories, and
|
||||
alignment predicates. Coordinate convention follows PDF (origin bottom-left, y increases upward).
|
||||
``Rect`` is constructed as ``Rect(left, right, top, bottom)``. A few internal
|
||||
storage fields are implementation details; public callers should use the semantic accessors.
|
||||
"""
|
||||
|
||||
import math
|
||||
import re
|
||||
import unicodedata
|
||||
from decimal import Decimal, ROUND_HALF_UP
|
||||
from typing import Any, Iterator, Optional, Protocol
|
||||
|
||||
import regex as regex_module # supports Unicode \p{...} property classes
|
||||
|
||||
from .char_stats import (
|
||||
_SENTENCE_END_CHARS,
|
||||
_MINUS_SIGN_CHARS,
|
||||
_max_nan_propagating,
|
||||
_min_nan_propagating,
|
||||
char_category,
|
||||
is_word_category,
|
||||
is_punct_category,
|
||||
_UNICODE_WHITESPACE_CHARS,
|
||||
_trim_unicode_ws,
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
_round_half_up_to_int,
|
||||
CharStats,
|
||||
merge_char_stats,
|
||||
letter_count,
|
||||
punct_count,
|
||||
info_weight,
|
||||
is_upper_dominant,
|
||||
)
|
||||
from .rects import (
|
||||
RectLike,
|
||||
Rect,
|
||||
EMPTY_RECT,
|
||||
Bounded,
|
||||
rect_union,
|
||||
rect_intersection,
|
||||
extend_top_to,
|
||||
extend_bottom_to,
|
||||
cmp_left_edge,
|
||||
left_edge_key,
|
||||
cmp_reading_order,
|
||||
reading_order_key,
|
||||
cmp_bottom_edge,
|
||||
magnitude_ratio,
|
||||
same_x_extent,
|
||||
same_y_extent,
|
||||
intervals_overlap,
|
||||
y_overlaps,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
x_aligned,
|
||||
x_centers_close,
|
||||
)
|
||||
from .span_line import (
|
||||
_bold_font_re,
|
||||
_italic_font_re,
|
||||
_font_name_aliases,
|
||||
_subset_prefix_re,
|
||||
Span,
|
||||
Line,
|
||||
append_span,
|
||||
last_span,
|
||||
_HasCharCount,
|
||||
avg_char_width,
|
||||
raw_text_of_line,
|
||||
text_of_line,
|
||||
avg_char_width2,
|
||||
_ONE_DECIMAL_QUANTUM,
|
||||
_format_half_up_one_decimal,
|
||||
style_key,
|
||||
)
|
||||
from .block import (
|
||||
Block,
|
||||
iter_sorted_children,
|
||||
argmax_key,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
dominant_style_of,
|
||||
dominant_font_size,
|
||||
is_caps_heavy,
|
||||
is_sentence_like,
|
||||
heading_score,
|
||||
case_signal,
|
||||
alignment_code,
|
||||
block_text,
|
||||
deaccented_text,
|
||||
_COMBINING_MARKS,
|
||||
_strip_diacritics,
|
||||
)
|
||||
from .numbering import (
|
||||
_NUMBERING_PREFIX_RE,
|
||||
_BRACKETED_NUM_RE,
|
||||
_TO_NUMBER_DEC,
|
||||
_TO_NUMBER_INF,
|
||||
_TO_NUMBER_HEX,
|
||||
_TO_NUMBER_OCT,
|
||||
_TO_NUMBER_BIN,
|
||||
to_number,
|
||||
_detect_numbering,
|
||||
numbering_text,
|
||||
numbering_value,
|
||||
numbering_kind,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"RectLike", "Rect", "Bounded", "EMPTY_RECT", "rect_union", "rect_intersection", "extend_top_to", "extend_bottom_to",
|
||||
"CharStats", "merge_char_stats", "letter_count", "punct_count", "info_weight", "is_upper_dominant",
|
||||
"char_category", "is_word_category", "is_punct_category",
|
||||
"Span", "Line", "Block",
|
||||
"append_span", "last_span", "avg_char_width", "raw_text_of_line", "text_of_line", "avg_char_width2", "style_key", "iter_sorted_children",
|
||||
"magnitude_ratio", "same_x_extent", "same_y_extent", "intervals_overlap", "y_overlaps", "left_aligned", "right_aligned", "center_aligned", "x_aligned", "x_centers_close",
|
||||
"to_number", "numbering_text", "numbering_value", "numbering_kind",
|
||||
"argmax_key", "last_line_of", "first_span_of", "dominant_style_of", "dominant_font_size", "is_caps_heavy", "heading_score", "case_signal", "alignment_code", "block_text", "deaccented_text",
|
||||
"cmp_left_edge", "left_edge_key", "cmp_reading_order", "reading_order_key", "cmp_bottom_edge",
|
||||
]
|
||||
@@ -0,0 +1,288 @@
|
||||
"""Block type with text, style, and alignment helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import re
|
||||
import unicodedata
|
||||
from typing import Any, Iterator, Optional, Protocol
|
||||
|
||||
from .char_stats import (
|
||||
is_punct_category,
|
||||
CharStats,
|
||||
merge_char_stats,
|
||||
letter_count,
|
||||
punct_count,
|
||||
info_weight,
|
||||
is_upper_dominant,
|
||||
)
|
||||
from .rects import (
|
||||
EMPTY_RECT,
|
||||
Bounded,
|
||||
rect_union,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
)
|
||||
from .span_line import (
|
||||
Span,
|
||||
Line,
|
||||
text_of_line,
|
||||
style_key,
|
||||
)
|
||||
|
||||
|
||||
class Block(Bounded):
|
||||
"""A vertically contiguous group of lines that share layout, such as a paragraph or heading run. Adding lines maintains weighted style, size, text, bbox, reading-order, classification, and cache fields."""
|
||||
|
||||
__slots__ = (
|
||||
"primary_slot", "char_stats", "alignment_slot", "weighted_ratio_tertiary", "previous_slot", "weighted_skew", "weighted_font_size", "weighted_ratio_primary", "weighted_ratio_secondary", "style_slot",
|
||||
"style_char_counts", "size_char_counts", "reading_order_index", "orig_index", "type", "isolated_centered", "is_body_paragraph", "measure_slot", "used_as_heading",
|
||||
"state_slot", "marker_slot", "metric_slot",
|
||||
"dominant_style_cache", "dominant_size_cache", "token_text_cache", "deaccented_text_cache", "cache_slot", "tokens_cache",
|
||||
)
|
||||
|
||||
def __init__(self):
|
||||
super().__init__(EMPTY_RECT)
|
||||
self.primary_slot: list = []
|
||||
self.char_stats: CharStats = CharStats("")
|
||||
self.alignment_slot: bool = True
|
||||
self.weighted_ratio_tertiary: float = 0.0
|
||||
self.previous_slot: float = 0.0
|
||||
self.weighted_skew: float = 0.0
|
||||
self.weighted_font_size: float = 0.0
|
||||
self.weighted_ratio_primary: float = 0.0
|
||||
self.weighted_ratio_secondary: float = 0.0
|
||||
self.style_slot: float = 0.0
|
||||
self.style_char_counts: dict = {}
|
||||
self.size_char_counts: dict = {}
|
||||
self.reading_order_index: int = 0
|
||||
self.orig_index: int = 0
|
||||
self.type: int = 0
|
||||
self.isolated_centered: bool = False
|
||||
self.is_body_paragraph: bool = False
|
||||
self.measure_slot: bool = False
|
||||
self.used_as_heading: bool = False
|
||||
self.state_slot: int = 0
|
||||
self.marker_slot: int = 0
|
||||
self.metric_slot: float = 0.0
|
||||
# caches, invalidated on every add_line
|
||||
self.dominant_style_cache: Optional[str] = None
|
||||
self.dominant_size_cache: Optional[float] = None
|
||||
self.token_text_cache: Optional[str] = None
|
||||
self.deaccented_text_cache: Optional[str] = None
|
||||
self.cache_slot: Optional[str] = None
|
||||
self.tokens_cache: Optional[Any] = None
|
||||
|
||||
def __iter__(self):
|
||||
return iter(self.primary_slot)
|
||||
|
||||
def line_count(self) -> int:
|
||||
"""Line count -- ."""
|
||||
return len(self.primary_slot)
|
||||
|
||||
def line(self):
|
||||
"""First line -- ."""
|
||||
return self.primary_slot[0]
|
||||
|
||||
def char_count(self) -> int: # type: ignore[override]
|
||||
"""Total char count across all child lines."""
|
||||
return self.char_stats.auxiliary_slot
|
||||
|
||||
def avg_font_size(self) -> float:
|
||||
"""Weighted average font size -- ."""
|
||||
return self.weighted_font_size
|
||||
|
||||
def bold_frac(self) -> float:
|
||||
"""Weighted bold fraction -- ."""
|
||||
return self.weighted_ratio_tertiary
|
||||
|
||||
def skew_frac(self) -> float:
|
||||
"""Weighted skew fraction -- ."""
|
||||
return self.weighted_skew
|
||||
|
||||
def add_line(self, other_line) -> "Block":
|
||||
"""Add a line while maintaining weighted style, size, character, bbox, and per-style histograms."""
|
||||
self.alignment_slot = self.alignment_slot and (len(self.primary_slot) <= 0 or center_aligned(self, other_line, 1))
|
||||
self.primary_slot.append(other_line)
|
||||
line = info_weight(self.char_stats)
|
||||
added_weight = info_weight(other_line.char_stats)
|
||||
total_weight = line + added_weight
|
||||
if total_weight > 0:
|
||||
self.weighted_ratio_tertiary = (self.weighted_ratio_tertiary * line + other_line.bold_frac() * added_weight) / total_weight
|
||||
self.previous_slot = (self.previous_slot * line + other_line.weighted_ratio_secondary * added_weight) / total_weight
|
||||
self.weighted_skew = (self.weighted_skew * line + other_line.skew_frac() * added_weight) / total_weight
|
||||
self.weighted_font_size = (self.weighted_font_size * line + other_line.avg_font_size() * added_weight) / total_weight
|
||||
self.weighted_ratio_primary = (self.weighted_ratio_primary * line + other_line.cache_slot * added_weight) / total_weight
|
||||
merge_char_stats(self.char_stats, other_line.char_stats)
|
||||
if other_line.char_count() <= 0:
|
||||
return self
|
||||
line = self.area() # area before union
|
||||
self.style_slot = max(self.style_slot, other_line.previous_slot)
|
||||
self.secondary_slot = rect_union(self.secondary_slot, other_line.secondary_slot)
|
||||
added_weight = self.area() # area after union
|
||||
if added_weight > 0:
|
||||
self.weighted_ratio_secondary = (self.weighted_ratio_secondary * line + other_line.cache_slot * other_line.area()) / added_weight
|
||||
for span in other_line:
|
||||
sty = style_key(span)
|
||||
self.style_char_counts[sty] = self.style_char_counts.get(sty, 0) + span.char_count()
|
||||
# Font-size buckets use half-up rounding to one decimal place.
|
||||
# Python round is half-to-even, so use floor(x + 0.5) on the
|
||||
# scaled non-negative font size.
|
||||
size_key = math.floor(span.font_size * 10 + 0.5) / 10
|
||||
self.size_char_counts[size_key] = self.size_char_counts.get(size_key, 0) + span.char_count()
|
||||
# invalidate caches
|
||||
self.dominant_style_cache = self.dominant_size_cache = self.token_text_cache = self.deaccented_text_cache = self.cache_slot = self.tokens_cache = None
|
||||
self.metric_slot = 0.0
|
||||
return self
|
||||
|
||||
|
||||
# Sorted child iterator.
|
||||
|
||||
def iter_sorted_children(primary_item):
|
||||
"""Iterate a page-like object's sorted children as indexed item records."""
|
||||
for idx, item in enumerate(primary_item.secondary_slot):
|
||||
yield {"index": idx, "block": item}
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Block-level accessors and derived text/style helpers #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def argmax_key(items) -> Optional[str]:
|
||||
"""return the key with max value. ``None`` if empty. ``items`` may be a ``dict`` (in which case we iterate ``.items``) or any iterable of ``(key, value)`` pairs. """
|
||||
pairs = items.items() if isinstance(items, dict) else items
|
||||
best: Optional[str] = None
|
||||
candidate_item = float("-inf")
|
||||
for reference_item, entry_item in pairs:
|
||||
if entry_item <= candidate_item:
|
||||
continue
|
||||
best = reference_item
|
||||
candidate_item = entry_item
|
||||
return best
|
||||
|
||||
|
||||
def last_line_of(block: Block) -> Line:
|
||||
"""last child line of a block."""
|
||||
return block.primary_slot[-1]
|
||||
|
||||
|
||||
def first_span_of(block: Block) -> Span:
|
||||
"""first span of a block's first line."""
|
||||
return block.line().primary_slot[0]
|
||||
|
||||
|
||||
def dominant_style_of(block: Block) -> str:
|
||||
"""Cached dominant style hash from the block's style histogram."""
|
||||
if block.dominant_style_cache is None:
|
||||
block.dominant_style_cache = argmax_key(block.style_char_counts) or ""
|
||||
return block.dominant_style_cache
|
||||
|
||||
|
||||
def dominant_font_size(block: Block) -> float:
|
||||
"""Return cached dominant font size from rounded-size character counts."""
|
||||
if block.dominant_size_cache is None:
|
||||
block.dominant_size_cache = float(argmax_key(block.size_char_counts) or 0)
|
||||
return block.dominant_size_cache
|
||||
|
||||
|
||||
def is_caps_heavy(primary_item) -> bool:
|
||||
"""Return True if a line or block is uppercase-dominant."""
|
||||
return is_upper_dominant(primary_item.char_stats) or primary_item.char_stats.primary_slot[2] >= max(2, primary_item.char_stats.auxiliary_slot)
|
||||
|
||||
|
||||
def is_sentence_like(primary_item) -> bool:
|
||||
"""Return whether a block looks like mixed-case body text rather than a heading. The test requires enough tokens, enough uppercase letters, and rejects long lowercase words."""
|
||||
from ..tokens import tokenize_block
|
||||
tokens = tokenize_block(primary_item)
|
||||
if tokens.length < 3 or is_caps_heavy(primary_item):
|
||||
return False
|
||||
upper_count = primary_item.char_stats.primary_slot[2]
|
||||
if upper_count <= 2 or upper_count < tokens.length / 10:
|
||||
return False
|
||||
match = 0
|
||||
for token in tokens:
|
||||
# Skip non-word tokens, short tokens, or g==4 (special)
|
||||
if token.type != 2 or len(token.str) <= 2 or token.primary_slot == 4:
|
||||
continue
|
||||
if token.primary_slot == 2:
|
||||
match += 1
|
||||
elif len(token.str) >= 5:
|
||||
return False
|
||||
return match >= 3
|
||||
|
||||
|
||||
def heading_score(heading) -> float:
|
||||
"""Line/block heading score: dominant font size plus caps-heavy and bold bonuses."""
|
||||
return dominant_font_size(heading) + (2 if is_caps_heavy(heading) else 0) + (1 if heading.weighted_ratio_tertiary > 0.5 else 0)
|
||||
|
||||
|
||||
def case_signal(char_stats: CharStats) -> int:
|
||||
"""Return an uppercase, lowercase, or neutral case signal from character statistics."""
|
||||
if is_upper_dominant(char_stats) and not is_punct_category(char_stats.secondary_slot) and letter_count(char_stats) > 3 * char_stats.auxiliary_slot / 4 and punct_count(char_stats) < 5:
|
||||
return 1
|
||||
if char_stats.primary_slot[3] > 0:
|
||||
return -1
|
||||
return 0
|
||||
|
||||
|
||||
def alignment_code(primary_item) -> int:
|
||||
"""cached block-level alignment code. Returns: 1 fully-justified (every line aligned with the block on left or right) 2 left-aligned (every line shares the block's left) 3 flag-set justified (the block center-alignment flag is set) 4 right-aligned 5 mixed / other """
|
||||
if primary_item.metric_slot != 0 or len(primary_item.primary_slot) <= 0:
|
||||
return primary_item.metric_slot
|
||||
left = True
|
||||
right = True
|
||||
any_value = True
|
||||
for score_value in primary_item.primary_slot:
|
||||
tolerance = max(1.0, score_value.bbox_width() / 20.0)
|
||||
line_left_aligned = left_aligned(primary_item, score_value, tolerance)
|
||||
line_right_aligned = right_aligned(primary_item, score_value, tolerance)
|
||||
if not line_left_aligned:
|
||||
left = False
|
||||
if not line_right_aligned:
|
||||
right = False
|
||||
if not (line_left_aligned or line_right_aligned):
|
||||
any_value = False
|
||||
if left and not right:
|
||||
primary_item.metric_slot = 2
|
||||
elif right and not left:
|
||||
primary_item.metric_slot = 4
|
||||
elif any_value:
|
||||
primary_item.metric_slot = 1
|
||||
elif primary_item.alignment_slot:
|
||||
primary_item.metric_slot = 3
|
||||
else:
|
||||
primary_item.metric_slot = 5
|
||||
return primary_item.metric_slot
|
||||
|
||||
|
||||
def block_text(block: Block) -> str:
|
||||
"""cached joined trimmed text of a block (space-separated)."""
|
||||
if block.cache_slot is not None:
|
||||
return block.cache_slot
|
||||
parts = []
|
||||
for line_index, line_value in enumerate(block.primary_slot):
|
||||
parts.append(text_of_line(line_value))
|
||||
if line_index < len(block.primary_slot) - 1:
|
||||
parts.append(" ")
|
||||
block.cache_slot = "".join(parts)
|
||||
return block.cache_slot
|
||||
|
||||
|
||||
def deaccented_text(block: Block) -> str:
|
||||
"""Cached diacritic-stripped block text; case and spacing are preserved."""
|
||||
if block.deaccented_text_cache is not None:
|
||||
return block.deaccented_text_cache
|
||||
block.deaccented_text_cache = _strip_diacritics(block_text(block))
|
||||
return block.deaccented_text_cache
|
||||
|
||||
|
||||
_COMBINING_MARKS = re.compile("[̀-ͯ]")
|
||||
|
||||
|
||||
def _strip_diacritics(text: str) -> str:
|
||||
"""Strip combining diacritics only while preserving case and internal spacing."""
|
||||
return unicodedata.normalize(
|
||||
"NFC", _COMBINING_MARKS.sub("", unicodedata.normalize("NFD", text))
|
||||
)
|
||||
@@ -0,0 +1,184 @@
|
||||
"""Character categories and per-run character statistics."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import unicodedata
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Character classifier #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
# Character categories used by tokenization:
|
||||
# 0 empty
|
||||
# 1 number (digit / numeral)
|
||||
# 2 uppercase letter (Lu, Lt)
|
||||
# 3 lowercase letter (Ll)
|
||||
# 4 other letter (Lo) -- CJK ideographs, syllabics, etc.
|
||||
# 5 mark (Mc, Me, Mn)
|
||||
# 6 sentence-end punct -- . ? ! 。 。 ? ! .
|
||||
# 7 connector / dash -- _ - — − ⁻ ₋ etc.
|
||||
# 8 other punctuation
|
||||
# 9 math symbol (Sm)
|
||||
# 10 whitespace
|
||||
# 11 other (symbols, format, control, unassigned)
|
||||
|
||||
_SENTENCE_END_CHARS = frozenset(".?!。。?!.")
|
||||
_MINUS_SIGN_CHARS = frozenset("−⁻₋") # minus, superscript/subscript minus
|
||||
|
||||
|
||||
def _max_nan_propagating(value: float, other_item: float) -> float:
|
||||
"""propagates NaN (Python ``max`` swallows it)."""
|
||||
if math.isnan(value) or math.isnan(other_item):
|
||||
return math.nan
|
||||
return value if value >= other_item else other_item
|
||||
|
||||
|
||||
def _min_nan_propagating(value: float, other_item: float) -> float:
|
||||
"""propagates NaN (Python ``min`` swallows it)."""
|
||||
if math.isnan(value) or math.isnan(other_item):
|
||||
return math.nan
|
||||
return value if value <= other_item else other_item
|
||||
|
||||
|
||||
def char_category(char_value: str) -> int:
|
||||
"""Return the tokenizer character category code from Unicode General_Category."""
|
||||
if not char_value:
|
||||
return 0
|
||||
cat = unicodedata.category(char_value)
|
||||
# Letters ------------------------------------------------------------------
|
||||
if cat == "Ll":
|
||||
return 3
|
||||
if cat == "Lu" or cat == "Lt":
|
||||
return 2
|
||||
if cat == "Lo":
|
||||
return 4
|
||||
# Whitespace ---------------------------------------------------------------
|
||||
# The whitespace set is the Unicode WhiteSpace + LineTerminator set:
|
||||
# the C0 set \t\n\v\f\r, the BOM , and
|
||||
# Unicode Space/Line/Paragraph separators (Zs/Zl/Zp). NOT Python's
|
||||
# str.isspace, which also matches the C0 separators U+001C-U+001F and NEL
|
||||
# U+0085, which this tokenizer intentionally excludes, and misses .
|
||||
if char_value in "\t\n\x0b\x0c\r" or char_value == "\ufeff" or cat in ("Zs", "Zl", "Zp"):
|
||||
return 10
|
||||
# Sentence-end punctuation -------------------------------------------------
|
||||
if char_value in _SENTENCE_END_CHARS:
|
||||
return 6
|
||||
# Dash / connector punctuation ---------------------------------------------
|
||||
if cat in ("Pc", "Pd") or char_value in _MINUS_SIGN_CHARS:
|
||||
return 7
|
||||
# General punctuation ------------------------------------------------------
|
||||
if cat.startswith("P"):
|
||||
return 8
|
||||
# Number -------------------------------------------------------------------
|
||||
if cat.startswith("N"):
|
||||
return 1
|
||||
# Mark ---------------------------------------------------------------------
|
||||
if cat.startswith("M"):
|
||||
return 5
|
||||
# Math symbol --------------------------------------------------------------
|
||||
if cat == "Sm":
|
||||
return 9
|
||||
return 11
|
||||
|
||||
|
||||
def is_word_category(number: int) -> bool:
|
||||
"""is c a 'word-y' category (letter / digit / mark)?"""
|
||||
return number == 3 or number == 2 or number == 1 or number == 5
|
||||
|
||||
|
||||
def is_punct_category(number: int) -> bool:
|
||||
"""is c a punctuation-y category (dash / punct / sentence)?"""
|
||||
return number == 7 or number == 8 or number == 6
|
||||
|
||||
|
||||
# Unicode trim strips the package whitespace set used by text parsing.
|
||||
# Python str.strip uses a DIFFERENT set: it ALSO strips U+001C-001F and U+0085
|
||||
# Trim keeps U+001C..U+001F and strips U+FEFF to match the intended whitespace set.
|
||||
# (Same set as parser_pdfium_charlevel._UNICODE_WHITESPACE; defined here to avoid a
|
||||
# circular import -- parser imports from model, not vice-versa.)
|
||||
_UNICODE_WHITESPACE_CHARS = (
|
||||
"\t\n\x0b\x0c\r \xa0 "
|
||||
" "
|
||||
"
"
|
||||
)
|
||||
|
||||
|
||||
def _trim_unicode_ws(text: str) -> str:
|
||||
"""Strip the package whitespace set, not Python's broader ``str.strip`` set."""
|
||||
return text.strip(_UNICODE_WHITESPACE_CHARS)
|
||||
|
||||
|
||||
# Unicode-compatible ``\s`` = WhiteSpace + LineTerminator = the same 25-cp set as
|
||||
# _UNICODE_WHITESPACE_CHARS. Bare Python ``\s`` differs: stdlib ``re`` ``\s`` ALSO matches
|
||||
# U+001C-U+001F and U+0085, the ``regex`` module ``\s`` matches U+0085, and
|
||||
# NEITHER matches U+FEFF (which does). Splice this char-class BODY into
|
||||
# regex definitions ("[" + _UNICODE_WHITESPACE_CLASS + "]") instead of a bare ``\s``.
|
||||
_UNICODE_WHITESPACE_CLASS = r"\t\n\x0b\x0c\r\x20\xa0 -
"
|
||||
|
||||
|
||||
def _round_half_up_to_int(value: float) -> int:
|
||||
"""Round a non-negative finite number to an integer using exact half-up semantics. The ``floor(x + 0.5)`` idiom is not equivalent at the single double ``0.49999999999999994``: adding 0.5 rounds up to ``1.0`` so floor gives 1. Compute the fractional part directly (exact for x >= 0 by Sterbenz) and compare to 0.5."""
|
||||
score_value = math.floor(value)
|
||||
frac = value - score_value
|
||||
if frac < 0.5:
|
||||
return score_value
|
||||
return score_value + 1 # frac > 0.5, or an exact 0.5 tie
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Per-string character-category accumulator #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class CharStats:
|
||||
"""Collect first/last character category, per-category counts, and total character count."""
|
||||
|
||||
__slots__ = ("secondary_slot", "tertiary_slot", "primary_slot", "auxiliary_slot")
|
||||
|
||||
def __init__(self, other_text: str):
|
||||
self.secondary_slot = 0
|
||||
self.tertiary_slot = 0
|
||||
self.primary_slot = [0] * 12
|
||||
self.auxiliary_slot = 0
|
||||
for secondary_item in other_text:
|
||||
cat = char_category(secondary_item)
|
||||
if self.secondary_slot == 0:
|
||||
self.secondary_slot = cat
|
||||
self.tertiary_slot = cat
|
||||
self.primary_slot[cat] += 1
|
||||
self.auxiliary_slot += 1
|
||||
|
||||
|
||||
def merge_char_stats(char_stats: CharStats, other_char_stats: CharStats) -> None:
|
||||
"""merge b into a in place."""
|
||||
if char_stats.secondary_slot == 0:
|
||||
char_stats.secondary_slot = other_char_stats.secondary_slot
|
||||
if other_char_stats.tertiary_slot != 0:
|
||||
char_stats.tertiary_slot = other_char_stats.tertiary_slot
|
||||
for candidate_item in range(12):
|
||||
char_stats.primary_slot[candidate_item] += other_char_stats.primary_slot[candidate_item]
|
||||
char_stats.auxiliary_slot += other_char_stats.auxiliary_slot
|
||||
|
||||
|
||||
def letter_count(char_stats: CharStats) -> int:
|
||||
"""count of letter-like chars (uppercase + lowercase + other-letter)."""
|
||||
return char_stats.primary_slot[3] + char_stats.primary_slot[2] + char_stats.primary_slot[4]
|
||||
|
||||
|
||||
def punct_count(char_stats: CharStats) -> int:
|
||||
"""count of sentence-punctuation chars (6 + 7 + 8)."""
|
||||
return char_stats.primary_slot[6] + char_stats.primary_slot[7] + char_stats.primary_slot[8]
|
||||
|
||||
|
||||
def info_weight(char_stats: CharStats) -> float:
|
||||
"""'informational' weight. ``letters + 2*other_letter + 0.5*(non-letter)`` -- biases towards alphabetic content; non-letter chars contribute half. """
|
||||
return char_stats.primary_slot[3] + char_stats.primary_slot[2] + 2 * char_stats.primary_slot[4] + 0.5 * (char_stats.auxiliary_slot - letter_count(char_stats))
|
||||
|
||||
|
||||
def is_upper_dominant(char_stats: CharStats) -> bool:
|
||||
"""uppercase-dominant string detector. True iff (uppercase chars) > max(letters*3/4, letters-4) and (uppercase chars) > max(3, total/3). """
|
||||
secondary_item = char_stats.primary_slot[2]
|
||||
candidate_item = letter_count(char_stats)
|
||||
return secondary_item > max(candidate_item * 3 / 4, candidate_item - 4) and secondary_item > max(3, char_stats.auxiliary_slot / 3)
|
||||
@@ -0,0 +1,131 @@
|
||||
"""Numbering-prefix detection and numeric parsing."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import re
|
||||
import unicodedata
|
||||
|
||||
import regex as regex_module # supports Unicode \p{...} property classes
|
||||
|
||||
from .char_stats import (
|
||||
_trim_unicode_ws,
|
||||
_UNICODE_WHITESPACE_CLASS,
|
||||
)
|
||||
from .span_line import (
|
||||
Line,
|
||||
raw_text_of_line,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Numbering detection #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
# Uses Unicode property classes (\p{Number} / \P{Number}), compiled with the
|
||||
# ``regex`` module (stdlib ``re`` can't express them). Matches:
|
||||
# - leading roman or digit (group 1)
|
||||
# - dotted lowercase a-h (group 2)
|
||||
# - dotted lowercase ivx (group 3)
|
||||
_NUMBERING_PREFIX_RE = regex_module.compile(
|
||||
r"^(?:"
|
||||
r"([IVX]+|[1-91-9]\p{Number}?)(?:[..。。):]|-\P{Number}|-$|[" + _UNICODE_WHITESPACE_CLASS + r"]|$)"
|
||||
r"|(?:([A-Ha-h])|([ivx]))[..。。)]"
|
||||
r")"
|
||||
)
|
||||
|
||||
# Bracketed numeric labels such as "[1]" or "(1)".
|
||||
_BRACKETED_NUM_RE = re.compile(r"^[\[\(] *([1-9][0-9]?) *[\)\]]")
|
||||
|
||||
|
||||
# string grammar (ToNumber). ASCII digits ONLY: Python's
|
||||
# ``\d`` and ``float`` both accept Unicode decimal digits (e.g. Arabic-Indic
|
||||
# ٢) and ``float`` also accepts ``1_000`` / ``inf`` / ``nan``, none of which
|
||||
# ``Number`` accepts -- hence the explicit ``[0-9]`` classes.
|
||||
_TO_NUMBER_DEC = re.compile(r"^[+-]?(?:[0-9]+\.?[0-9]*|\.[0-9]+)(?:[eE][+-]?[0-9]+)?$")
|
||||
_TO_NUMBER_INF = re.compile(r"^[+-]?Infinity$")
|
||||
_TO_NUMBER_HEX = re.compile(r"^0[xX][0-9a-fA-F]+$")
|
||||
_TO_NUMBER_OCT = re.compile(r"^0[oO][0-7]+$")
|
||||
_TO_NUMBER_BIN = re.compile(r"^0[bB][01]+$")
|
||||
|
||||
|
||||
def to_number(text: str) -> float:
|
||||
"""NFKC-normalized numeric conversion with decimal, exponent, hex, octal, binary, and Infinity forms."""
|
||||
if text is None:
|
||||
return math.nan
|
||||
token_value = _trim_unicode_ws(unicodedata.normalize("NFKC", text))
|
||||
if token_value == "":
|
||||
return 0.0
|
||||
if _TO_NUMBER_INF.match(token_value):
|
||||
return -math.inf if token_value[0] == "-" else math.inf
|
||||
if _TO_NUMBER_HEX.match(token_value):
|
||||
return float(int(token_value[2:], 16))
|
||||
if _TO_NUMBER_OCT.match(token_value):
|
||||
return float(int(token_value[2:], 8))
|
||||
if _TO_NUMBER_BIN.match(token_value):
|
||||
return float(int(token_value[2:], 2))
|
||||
if _TO_NUMBER_DEC.match(token_value):
|
||||
return float(token_value)
|
||||
return math.nan
|
||||
|
||||
|
||||
def _detect_numbering(line: Line) -> None:
|
||||
"""Detect leading section numbering and cache the numbering kind and text on the line."""
|
||||
if line.state_slot != -1:
|
||||
return # already computed
|
||||
line.state_slot = 0
|
||||
if line.char_count() <= 0:
|
||||
return
|
||||
# Drop-capital / large-first-char detection (layout branch).
|
||||
# If first span is smaller, sits above the next non-empty span, and is
|
||||
# numeric -> use that span's text as the numbering.
|
||||
if len(line.primary_slot) > 1:
|
||||
secondary_item = line.primary_slot[0]
|
||||
candidate_item = line.primary_slot[2] if (line.primary_slot[1].char_count() <= 0 and len(line.primary_slot) > 2) else line.primary_slot[1]
|
||||
if (
|
||||
secondary_item.bbox_height() < candidate_item.bbox_height()
|
||||
and secondary_item.bottom_edge() > candidate_item.bottom_edge() + 0.05 * candidate_item.bbox_height()
|
||||
and not math.isnan(to_number(secondary_item.text))
|
||||
):
|
||||
line.state_slot = 1
|
||||
line.style_slot = secondary_item.text
|
||||
return
|
||||
text = raw_text_of_line(line)
|
||||
measure_item = _NUMBERING_PREFIX_RE.match(text)
|
||||
if measure_item and measure_item.group(1) and "1" <= measure_item.group(1)[0] <= "9":
|
||||
line.state_slot = 1
|
||||
line.style_slot = measure_item.group(1)
|
||||
return
|
||||
if measure_item and (measure_item.group(1) or measure_item.group(3)):
|
||||
# Roman uppercase (group 1) or other -- both uppercase-ish
|
||||
line.state_slot = 2
|
||||
line.style_slot = measure_item.group(1) or measure_item.group(3)
|
||||
return
|
||||
if measure_item and measure_item.group(2):
|
||||
line.state_slot = 3
|
||||
line.style_slot = measure_item.group(2)
|
||||
return
|
||||
second_matrix = _BRACKETED_NUM_RE.match(text)
|
||||
if second_matrix:
|
||||
line.state_slot = 1
|
||||
line.style_slot = second_matrix.group(1)
|
||||
return
|
||||
|
||||
|
||||
def numbering_text(line: Line) -> str:
|
||||
"""get the cached numbering string."""
|
||||
_detect_numbering(line)
|
||||
return line.style_slot
|
||||
|
||||
|
||||
def numbering_value(line: Line) -> float:
|
||||
"""get numbering as a number, NaN if non-digit numbering."""
|
||||
text = numbering_text(line)
|
||||
return to_number(text) if line.state_slot == 1 else math.nan
|
||||
|
||||
|
||||
def numbering_kind(line: Line) -> int:
|
||||
"""get numbering type (0 none, 1 digit, 2 upper, 3 lower)."""
|
||||
_detect_numbering(line)
|
||||
return line.state_slot
|
||||
@@ -0,0 +1,229 @@
|
||||
"""Rectangle types, geometry predicates, and ordering comparators."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
|
||||
from .char_stats import (
|
||||
_max_nan_propagating,
|
||||
_min_nan_propagating,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Rectangle model #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class RectLike:
|
||||
"""Empty base for objects that expose bbox accessors."""
|
||||
|
||||
pass
|
||||
|
||||
|
||||
class Rect(RectLike):
|
||||
"""Axis-aligned bbox. PDF coordinates: top > bottom (y increases upward). """
|
||||
|
||||
__slots__ = ("left", "right", "top", "primary_slot")
|
||||
|
||||
def __init__(self, other_item: float, candidate_item: float, reference_item: float, next_item: float):
|
||||
self.left = other_item
|
||||
self.right = candidate_item
|
||||
self.top = reference_item
|
||||
self.primary_slot = next_item # bottom
|
||||
|
||||
# --- geometry accessors ----------------------------------
|
||||
|
||||
def left_edge(self) -> float: return self.left
|
||||
def right_edge(self) -> float: return self.right
|
||||
def top_edge(self) -> float: return self.top
|
||||
def bottom_edge(self) -> float: return self.primary_slot # bottom
|
||||
def bbox_width(self) -> float: return _max_nan_propagating(0.0, self.right - self.left) # width
|
||||
def bbox_height(self) -> float: return _max_nan_propagating(0.0, self.top - self.primary_slot) # height
|
||||
def area(self) -> float: return self.bbox_width() * self.bbox_height() # area
|
||||
def center_x(self) -> float: return (self.left + self.right) / 2 # x-center
|
||||
def center_y(self) -> float: return (self.top + self.primary_slot) / 2 # y-center
|
||||
|
||||
def contains(self, other_rect: "Rect") -> bool:
|
||||
return (
|
||||
self.left <= other_rect.left
|
||||
and self.right >= other_rect.right
|
||||
and self.top >= other_rect.top
|
||||
and self.primary_slot <= other_rect.primary_slot
|
||||
)
|
||||
|
||||
|
||||
# Shared empty / inverted rectangle used to initialize accumulators.
|
||||
EMPTY_RECT = Rect(math.inf, -math.inf, -math.inf, math.inf)
|
||||
|
||||
|
||||
class Bounded(RectLike):
|
||||
"""Mixin-style wrapper around an owned ``Rect``."""
|
||||
|
||||
__slots__ = ("secondary_slot",)
|
||||
|
||||
def __init__(self, other_rect: Rect):
|
||||
self.secondary_slot = other_rect
|
||||
|
||||
def left_edge(self) -> float: return self.secondary_slot.left
|
||||
def right_edge(self) -> float: return self.secondary_slot.right
|
||||
def top_edge(self) -> float: return self.secondary_slot.top
|
||||
def bottom_edge(self) -> float: return self.secondary_slot.primary_slot
|
||||
def bbox_width(self) -> float: return self.secondary_slot.bbox_width()
|
||||
def bbox_height(self) -> float: return self.secondary_slot.bbox_height()
|
||||
def area(self) -> float: return self.secondary_slot.area()
|
||||
def center_x(self) -> float: return self.secondary_slot.center_x()
|
||||
def center_y(self) -> float: return self.secondary_slot.center_y()
|
||||
|
||||
|
||||
def rect_union(rect: Rect, other_rect: Rect) -> Rect:
|
||||
"""bbox union."""
|
||||
return Rect(
|
||||
_min_nan_propagating(rect.left, other_rect.left),
|
||||
_max_nan_propagating(rect.right, other_rect.right),
|
||||
_max_nan_propagating(rect.top, other_rect.top),
|
||||
_min_nan_propagating(rect.primary_slot, other_rect.primary_slot),
|
||||
)
|
||||
|
||||
|
||||
def rect_intersection(rect: Rect, other_rect: Rect) -> Rect:
|
||||
"""bbox intersection; disjoint boxes may have inverted horizontal or vertical edges."""
|
||||
return Rect(
|
||||
_max_nan_propagating(rect.left, other_rect.left),
|
||||
_min_nan_propagating(rect.right, other_rect.right),
|
||||
_min_nan_propagating(rect.top, other_rect.top),
|
||||
_max_nan_propagating(rect.primary_slot, other_rect.primary_slot),
|
||||
)
|
||||
|
||||
|
||||
def extend_top_to(rect: Rect, other_item: float) -> Rect:
|
||||
"""Clip the rectangle top to be at least ``other_value``."""
|
||||
return Rect(rect.left, rect.right, _max_nan_propagating(rect.top, other_item), rect.primary_slot)
|
||||
|
||||
|
||||
def extend_bottom_to(rect: Rect, other_item: float) -> Rect:
|
||||
"""Clip the rectangle bottom to be at most ``other_value``."""
|
||||
return Rect(rect.left, rect.right, rect.top, _min_nan_propagating(rect.primary_slot, other_item))
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Sort comparators #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def cmp_left_edge(left_value: Bounded, right_value: Bounded) -> float:
|
||||
"""Order by (left asc, right asc, top desc, bottom desc). Returns the raw delta, not a normalised -1/0/1, because callers only consume the sign."""
|
||||
if left_value.left_edge() != right_value.left_edge():
|
||||
return left_value.left_edge() - right_value.left_edge()
|
||||
if left_value.right_edge() != right_value.right_edge():
|
||||
return left_value.right_edge() - right_value.right_edge()
|
||||
if left_value.top_edge() != right_value.top_edge():
|
||||
return right_value.top_edge() - left_value.top_edge()
|
||||
return right_value.bottom_edge() - left_value.bottom_edge()
|
||||
|
||||
|
||||
# Python's ``sorted`` accepts a key, not a cmp. Provide key functions too.
|
||||
def left_edge_key(primary_item: Bounded) -> tuple:
|
||||
return (primary_item.left_edge(), primary_item.right_edge(), -primary_item.top_edge(), -primary_item.bottom_edge())
|
||||
|
||||
|
||||
def cmp_reading_order(left_value: Bounded, right_value: Bounded) -> float:
|
||||
"""Order by (top desc, bottom desc, left asc, right asc). Top-of-page rows come first; within a row, leftmost first. Returns the raw delta because callers only consume the sign."""
|
||||
if left_value.top_edge() != right_value.top_edge():
|
||||
return right_value.top_edge() - left_value.top_edge()
|
||||
if left_value.bottom_edge() != right_value.bottom_edge():
|
||||
return right_value.bottom_edge() - left_value.bottom_edge()
|
||||
if left_value.left_edge() != right_value.left_edge():
|
||||
return left_value.left_edge() - right_value.left_edge()
|
||||
return left_value.right_edge() - right_value.right_edge()
|
||||
|
||||
|
||||
def reading_order_key(primary_item: Bounded) -> tuple:
|
||||
return (-primary_item.top_edge(), -primary_item.bottom_edge(), primary_item.left_edge(), primary_item.right_edge())
|
||||
|
||||
|
||||
def cmp_bottom_edge(left_value: Bounded, right_value: Bounded) -> float:
|
||||
"""Order by (bottom asc, top asc, left asc, right asc). Returns the raw delta because callers only consume the sign."""
|
||||
if left_value.bottom_edge() != right_value.bottom_edge():
|
||||
return left_value.bottom_edge() - right_value.bottom_edge()
|
||||
if left_value.top_edge() != right_value.top_edge():
|
||||
return left_value.top_edge() - right_value.top_edge()
|
||||
if left_value.left_edge() != right_value.left_edge():
|
||||
return left_value.left_edge() - right_value.left_edge()
|
||||
return left_value.right_edge() - right_value.right_edge()
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Alignment / overlap predicates #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def magnitude_ratio(value: float, other_item: float) -> float:
|
||||
"""Return the larger-magnitude-over-smaller-magnitude ratio with IEEE-754 division semantics. Division by zero yields +/-Infinity for a nonzero non-NaN numerator and NaN for +/-0 over +/-0 and NaN over +/-0. Downstream threshold tests rely on signed infinity, so divide-by-zero must not be collapsed to NaN. """
|
||||
# NaN comparisons take the false arm, which selects ``other_value / value``.
|
||||
if abs(value) > abs(other_item):
|
||||
num, den = value, other_item
|
||||
else:
|
||||
num, den = other_item, value
|
||||
# raw `num/den`. Python raises ZeroDivisionError on den == +/-0, so the
|
||||
# IEEE cases are spelled out: x/±0 = ±Infinity with sign(x) XOR sign(±0)
|
||||
# 5/-0 = -Infinity, ±0/±0 = NaN, NaN/±0 = NaN. A NaN denominator passes
|
||||
# `den != 0` and divides through to NaN.
|
||||
if den != 0:
|
||||
return num / den
|
||||
if num == 0 or math.isnan(num):
|
||||
return math.nan
|
||||
return math.copysign(math.inf, num) * math.copysign(1.0, den)
|
||||
|
||||
|
||||
def same_x_extent(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
|
||||
"""Return whether both horizontal edges are within the tolerance."""
|
||||
return abs(primary_item.left_edge() - secondary_item.left_edge()) <= candidate_item and abs(primary_item.right_edge() - secondary_item.right_edge()) <= candidate_item
|
||||
|
||||
|
||||
def same_y_extent(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
|
||||
"""Return whether both vertical edges are within the tolerance."""
|
||||
return abs(primary_item.top_edge() - secondary_item.top_edge()) <= candidate_item and abs(primary_item.bottom_edge() - secondary_item.bottom_edge()) <= candidate_item
|
||||
|
||||
|
||||
def intervals_overlap(value: float, other_item: float, candidate_item: float, reference_item: float) -> bool:
|
||||
"""Return whether the two closed ranges overlap by either endpoint."""
|
||||
return (value <= candidate_item and candidate_item <= other_item) or (candidate_item <= value and value <= reference_item)
|
||||
|
||||
|
||||
def y_overlaps(primary_item: Bounded, secondary_item: Bounded) -> bool:
|
||||
"""Return whether the vertical intervals of two boxes overlap."""
|
||||
return intervals_overlap(primary_item.bottom_edge(), primary_item.top_edge(), secondary_item.bottom_edge(), secondary_item.top_edge())
|
||||
|
||||
|
||||
def left_aligned(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
|
||||
"""Return whether left edges match within the tolerance."""
|
||||
return abs(primary_item.left_edge() - secondary_item.left_edge()) <= candidate_item
|
||||
|
||||
|
||||
def right_aligned(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
|
||||
"""Return whether right edges match within the tolerance."""
|
||||
return abs(primary_item.right_edge() - secondary_item.right_edge()) <= candidate_item
|
||||
|
||||
|
||||
def center_aligned(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
|
||||
"""Return whether two boxes are center-aligned within the tolerance. Their left and right edge offsets must have opposite signs, then pass the center-distance tolerance."""
|
||||
reference_item = primary_item.left_edge() - secondary_item.left_edge()
|
||||
entry_item = primary_item.right_edge() - secondary_item.right_edge()
|
||||
def sign(signed_delta):
|
||||
if signed_delta > 0: return 1
|
||||
if signed_delta < 0: return -1
|
||||
return 0
|
||||
if sign(reference_item) != -sign(entry_item):
|
||||
return False
|
||||
return abs(primary_item.center_x() - secondary_item.center_x()) <= max(candidate_item, min(abs(reference_item), abs(entry_item)) / 2)
|
||||
|
||||
|
||||
def x_aligned(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
|
||||
"""any of left / right / center aligned."""
|
||||
return left_aligned(primary_item, secondary_item, candidate_item) or right_aligned(primary_item, secondary_item, candidate_item) or center_aligned(primary_item, secondary_item, candidate_item)
|
||||
|
||||
|
||||
def x_centers_close(primary_item: Bounded, secondary_item: Bounded) -> bool:
|
||||
"""Return whether x-centers match within the secondary box width tolerance."""
|
||||
return abs(secondary_item.center_x() - primary_item.center_x()) <= max(1, secondary_item.bbox_width() / 10)
|
||||
@@ -0,0 +1,228 @@
|
||||
"""Span and Line types with text and style helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from decimal import Decimal, ROUND_HALF_UP
|
||||
from typing import Any, Iterator, Optional, Protocol
|
||||
|
||||
from .char_stats import (
|
||||
_trim_unicode_ws,
|
||||
CharStats,
|
||||
merge_char_stats,
|
||||
letter_count,
|
||||
info_weight,
|
||||
)
|
||||
from .rects import (
|
||||
Rect,
|
||||
EMPTY_RECT,
|
||||
Bounded,
|
||||
rect_union,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Text span #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
# Font-style detectors. Neither pattern is multiline or Unicode-aware: the end
|
||||
# anchor binds at end of INPUT (Python's `$` would also match before a trailing
|
||||
# newline, hence `\Z`), and case folding stays ASCII-only, so U+017F, U+0130
|
||||
# and U+0131 do not fold onto "s"/"i". The digit classes are spelled out, so
|
||||
# the ASCII flag touches nothing else here.
|
||||
_bold_font_re = re.compile(r"(bold|timesb)", re.IGNORECASE | re.ASCII)
|
||||
_italic_font_re = re.compile(r"(ital|it\Z|i[1-9][0-9]*\Z|obliq)", re.IGNORECASE | re.ASCII)
|
||||
# Font-name canonicalization map.
|
||||
_font_name_aliases = {
|
||||
"timesnewroman": "Times",
|
||||
"times-new-roman": "Times",
|
||||
"timesroman": "Times",
|
||||
"times-roman": "Times",
|
||||
"timesnew": "Times",
|
||||
"times-new": "Times",
|
||||
}
|
||||
|
||||
# subset prefix regex: 6 uppercase letters + plus sign
|
||||
_subset_prefix_re = re.compile(r"^[A-Z]{6}\+")
|
||||
|
||||
|
||||
class Span(Bounded):
|
||||
"""Span emitted by one text-showing item. Stores raw and trimmed text, character statistics, skew, font family/name, font size, bold/italic flags, and bbox helpers."""
|
||||
|
||||
__slots__ = (
|
||||
"text", "state_slot", "char_stats", "previous_slot", "font_family", "font_name", "font_size", "primary_slot", "measure_slot",
|
||||
)
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
bbox: Rect,
|
||||
text: str,
|
||||
font_name_raw: str,
|
||||
font_size: float,
|
||||
bold: bool,
|
||||
italic: bool,
|
||||
skew: float = 0.0,
|
||||
font_family: str = "",
|
||||
):
|
||||
"""Create a span from parser-normalized text, font, style, skew, and bounding-box fields."""
|
||||
super().__init__(bbox)
|
||||
self.text = text
|
||||
self.state_slot = _trim_unicode_ws(text)
|
||||
self.char_stats = CharStats(self.state_slot)
|
||||
self.previous_slot = skew
|
||||
self.font_family = font_family
|
||||
|
||||
# ---- font name normalisation -----------
|
||||
reference_item = font_name_raw
|
||||
if _subset_prefix_re.match(reference_item):
|
||||
reference_item = reference_item[7:]
|
||||
reference_item = _font_name_aliases.get(reference_item.lower(), reference_item)
|
||||
|
||||
self.font_name = reference_item
|
||||
|
||||
self.font_size = font_size
|
||||
|
||||
# Bold comes from the adapter flag or from the normalized font name.
|
||||
self.primary_slot = bool(bold) or bool(_bold_font_re.search(self.font_name))
|
||||
# Italic is name-derived only. The ``italic`` parameter is accepted
|
||||
# for adapter compatibility but is not consulted.
|
||||
del italic # noqa: F841 -- explicitly drop the arg
|
||||
self.measure_slot = bool(_italic_font_re.search(self.font_name))
|
||||
|
||||
def char_count(self) -> int: # type: ignore[override]
|
||||
"""Span char count."""
|
||||
return self.char_stats.auxiliary_slot
|
||||
|
||||
def font_style(self) -> str:
|
||||
""""<fontName> B" or "<fontName> R"."""
|
||||
return f"{self.font_name} {'B' if self.primary_slot else 'R'}"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Text line #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class Line(Bounded):
|
||||
"""A list of spans on roughly the same baseline, with line-wide character statistics, first letter-bearing span, weighted bold/italic/skew/font-size aggregates, cached text, numbering state, column index, and span list."""
|
||||
|
||||
__slots__ = (
|
||||
"primary_slot", "char_stats", "alignment_slot", "weighted_ratio_primary", "weighted_ratio_secondary", "weighted_ratio_tertiary", "metric_slot", "previous_slot", "measure_slot", "marker_slot", "state_slot", "style_slot", "cache_slot",
|
||||
)
|
||||
|
||||
def __init__(self):
|
||||
super().__init__(EMPTY_RECT)
|
||||
self.primary_slot: list[Span] = []
|
||||
self.char_stats: CharStats = CharStats("")
|
||||
self.alignment_slot: Optional[Span] = None
|
||||
self.weighted_ratio_primary: float = 0.0
|
||||
self.weighted_ratio_secondary: float = 0.0
|
||||
|
||||
self.weighted_ratio_tertiary: float = 0.0
|
||||
self.metric_slot: float = 0.0
|
||||
self.previous_slot: float = 0.0
|
||||
# Column index assigned by the column pass; -1 means unassigned.
|
||||
self.measure_slot: int = -1
|
||||
self.marker_slot: Optional[str] = None
|
||||
self.state_slot: int = -1
|
||||
self.style_slot: str = ""
|
||||
self.cache_slot: float = 0.0
|
||||
|
||||
def __iter__(self) -> Iterator[Span]:
|
||||
return iter(self.primary_slot)
|
||||
|
||||
def char_count(self) -> int: # type: ignore[override]
|
||||
return self.char_stats.auxiliary_slot
|
||||
|
||||
def avg_font_size(self) -> float:
|
||||
return self.metric_slot
|
||||
|
||||
def bold_frac(self) -> float:
|
||||
return self.weighted_ratio_primary
|
||||
|
||||
def skew_frac(self) -> float:
|
||||
return self.weighted_ratio_tertiary
|
||||
|
||||
|
||||
def append_span(line: Line, other_span: Span) -> Line:
|
||||
"""append span b into line a, updating weighted fields. Every aggregate field is updated in one pass so downstream line scoring sees the same weighted style, size, and geometry summaries. """
|
||||
line.primary_slot.append(other_span)
|
||||
span = info_weight(line.char_stats)
|
||||
added_weight = info_weight(other_span.char_stats)
|
||||
total_weight = span + added_weight
|
||||
if total_weight > 0:
|
||||
line.weighted_ratio_primary = (line.weighted_ratio_primary * span + (1 if other_span.primary_slot else 0) * added_weight) / total_weight
|
||||
line.weighted_ratio_secondary = (line.weighted_ratio_secondary * span + (1 if other_span.measure_slot else 0) * added_weight) / total_weight
|
||||
line.weighted_ratio_tertiary = (line.weighted_ratio_tertiary * span + other_span.previous_slot * added_weight) / total_weight
|
||||
line.metric_slot = (line.metric_slot * span + other_span.font_size * added_weight) / total_weight
|
||||
merge_char_stats(line.char_stats, other_span.char_stats)
|
||||
if line.alignment_slot is None and letter_count(other_span.char_stats) > 0:
|
||||
line.alignment_slot = other_span
|
||||
if other_span.char_count() <= 0:
|
||||
return line
|
||||
span = line.area()
|
||||
line.previous_slot = max(line.previous_slot, other_span.bbox_height())
|
||||
line.secondary_slot = rect_union(line.secondary_slot, other_span.secondary_slot)
|
||||
line.cache_slot = min(1.0, (line.cache_slot * span + other_span.area()) / max(1.0, line.area()))
|
||||
line.marker_slot = None
|
||||
line.state_slot = -1
|
||||
line.style_slot = ""
|
||||
return line
|
||||
|
||||
|
||||
def last_span(line: Line) -> Span:
|
||||
"""last span of line."""
|
||||
return line.primary_slot[-1]
|
||||
|
||||
|
||||
class _HasCharCount(Protocol):
|
||||
"""Anything with K (char count), A (width), N (height)."""
|
||||
|
||||
def char_count(self) -> int: ...
|
||||
def bbox_width(self) -> float: ...
|
||||
def bbox_height(self) -> float: ...
|
||||
|
||||
|
||||
def avg_char_width(primary_item: _HasCharCount) -> float:
|
||||
"""Width per character. Returns 0 if there are no characters."""
|
||||
return 0.0 if primary_item.char_count() <= 0 else primary_item.bbox_width() / primary_item.char_count()
|
||||
|
||||
|
||||
def raw_text_of_line(line: Line) -> str:
|
||||
"""concatenate raw text of all spans (no trimming)."""
|
||||
parts = []
|
||||
for span in line.primary_slot:
|
||||
parts.append(span.text)
|
||||
return "".join(parts)
|
||||
|
||||
|
||||
def text_of_line(line: Line) -> str:
|
||||
"""cached trimmed line text."""
|
||||
if line.marker_slot is not None:
|
||||
return line.marker_slot
|
||||
line.marker_slot = _trim_unicode_ws(raw_text_of_line(line))
|
||||
return line.marker_slot
|
||||
|
||||
|
||||
def avg_char_width2(primary_item: _HasCharCount) -> float:
|
||||
"""Width per character. Returns 0 for empty text."""
|
||||
return 0.0 if primary_item.char_count() <= 0 else primary_item.bbox_width() / primary_item.char_count()
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Text block #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
_ONE_DECIMAL_QUANTUM = Decimal("0.1")
|
||||
|
||||
|
||||
def _format_half_up_one_decimal(value: float) -> str:
|
||||
"""Round the exact double half-away-from-zero; the stats module uses the same helper."""
|
||||
return str(Decimal(value).quantize(_ONE_DECIMAL_QUANTUM, rounding=ROUND_HALF_UP))
|
||||
|
||||
|
||||
def style_key(span: "Span") -> str:
|
||||
"""Style hash ``"<fontName> <B|R> <size rounded to 0.1>"`` using shared half-up rounding."""
|
||||
return f"{span.font_style()} {_format_half_up_one_decimal(span.font_size)}"
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Heading predicates and section-keyword helpers. The full outline tree is assembled in ``outline_assembly``. This module keeps
|
||||
the lower-level heading checks that decide whether a block is a plausible
|
||||
outline heading based on numbering, style, geometry, and section-keyword tries.
|
||||
"""
|
||||
|
||||
import re
|
||||
from collections import defaultdict
|
||||
from typing import Optional
|
||||
|
||||
from ..labels import extract_structural_number
|
||||
from ..model import numbering_text, numbering_kind, block_text, is_caps_heavy, Block
|
||||
from ..stats import column_index_of
|
||||
from ..tokens import set_case_fold, TrieConfig, build_trie, tokenize_block, trie_full_match
|
||||
|
||||
from .filtering import (
|
||||
SECTION_KEYWORD_TRIE,
|
||||
heading_order_key,
|
||||
_NON_HEADING_TYPES,
|
||||
_DOT_LEADER_RE,
|
||||
_CAPTION_LABEL_RE,
|
||||
_EQUATION_LABEL_RE,
|
||||
_PAREN_FRAGMENT_RE,
|
||||
_looks_like_pseudo_code,
|
||||
is_heading_candidate,
|
||||
_style_key,
|
||||
_numbering_depth,
|
||||
collect_headings,
|
||||
_heading_signature,
|
||||
_matches_section_keywords,
|
||||
_PSEUDO_CODE_PATTERNS,
|
||||
_AUTHOR_PATTERNS,
|
||||
_BULLET_LIST_RE,
|
||||
filter_by_clique,
|
||||
)
|
||||
from .tree import (
|
||||
extract_top_level_headings,
|
||||
assign_levels,
|
||||
_heading_title,
|
||||
_heading_page_num,
|
||||
build_tree,
|
||||
validate,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"is_heading_candidate",
|
||||
"collect_headings",
|
||||
"filter_by_clique",
|
||||
"assign_levels",
|
||||
"build_tree",
|
||||
"validate",
|
||||
"extract_top_level_headings",
|
||||
"SECTION_KEYWORD_TRIE",
|
||||
"heading_order_key",
|
||||
]
|
||||
@@ -0,0 +1,241 @@
|
||||
"""Heading candidate collection, filtering, and keyword screening."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import defaultdict
|
||||
from typing import Optional
|
||||
|
||||
from ..labels import extract_structural_number
|
||||
from ..model import numbering_text, numbering_kind, block_text, is_caps_heavy, Block
|
||||
from ..stats import column_index_of
|
||||
from ..tokens import set_case_fold, TrieConfig, build_trie, tokenize_block, trie_full_match
|
||||
|
||||
|
||||
# English section keywords loaded into a case-folded trie matching tokenized
|
||||
# block text exactly.
|
||||
SECTION_KEYWORD_TRIE = build_trie(
|
||||
[
|
||||
"acknowledgements", "acknowledgments",
|
||||
"background",
|
||||
"conclusion", "conclusions",
|
||||
"discussion",
|
||||
"introduction",
|
||||
"materials and methods",
|
||||
"method", "methods",
|
||||
"results",
|
||||
],
|
||||
set_case_fold(TrieConfig(), True),
|
||||
)
|
||||
|
||||
|
||||
def heading_order_key(block: Block, page_lookup: dict[int, int]) -> tuple:
|
||||
"""Sort by page, then column index and reading position."""
|
||||
return (
|
||||
page_lookup.get(id(block), 1),
|
||||
column_index_of(block),
|
||||
-block.top_edge(),
|
||||
-block.bottom_edge(),
|
||||
block.left_edge(),
|
||||
block.right_edge(),
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Heading candidate gates #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
# Block types excluded from heading candidacy:
|
||||
# 1 header, 2 footer, 3 references body, 9 TOC page content,
|
||||
# 12 watermark/caption, 13 claimed labeled-section body, 99 title.
|
||||
_NON_HEADING_TYPES = frozenset({1, 2, 3, 9, 12, 13, 99})
|
||||
|
||||
_DOT_LEADER_RE = re.compile(r"\.{4,}\s*\d+\s*$")
|
||||
_CAPTION_LABEL_RE = re.compile(
|
||||
r"^\s*(?:figure|fig\.?|table|tab\.?|algorithm|alg\.?|equation|eq\.?|listing)\s+\d",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
# Equation labels like "(1)", "(2.3)", "(a)", "(i)", "(*)" -- parenthesised
|
||||
# short labels that the numbering detector mistakes for "1." section starts.
|
||||
_EQUATION_LABEL_RE = re.compile(
|
||||
r"^\s*[\[\(]\s*(?:[0-9]+(?:\.\d+)?[a-z]?|[a-z]|[ivx]+)\s*[\)\]]\s*$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
# Parenthesised body fragments like "(current cost)", "(estimated cost)".
|
||||
_PAREN_FRAGMENT_RE = re.compile(r"^\s*[\[\(][^\]\)]{1,40}[\]\)]\s*$")
|
||||
|
||||
|
||||
def _looks_like_pseudo_code(text: str) -> bool:
|
||||
"""Reject pseudo-code and math fragments that can resemble numbered headings."""
|
||||
if any(pat.search(text) for pat in _PSEUDO_CODE_PATTERNS):
|
||||
return True
|
||||
# No alphabetic word of >= 3 letters? Reject.
|
||||
if not re.search(r"[A-Za-zÀ-ÿ一-鿿가-]{3,}", text):
|
||||
return True
|
||||
# Bullet-list item: "1. long flowing prose..."
|
||||
if _BULLET_LIST_RE.match(text) and len(text) > 80:
|
||||
return True
|
||||
# Parenthesised fragment: "(current cost)", "(maximum flow)"
|
||||
if _PAREN_FRAGMENT_RE.match(text):
|
||||
return True
|
||||
# Author-block heuristics
|
||||
if any(pat.search(text) for pat in _AUTHOR_PATTERNS):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def is_heading_candidate(block: Block, body_size: float, body_bold: bool) -> bool:
|
||||
"""Return whether a block has the visual and textual shape of a heading."""
|
||||
if block.type in _NON_HEADING_TYPES:
|
||||
return False
|
||||
if block.line_count() > 6:
|
||||
return False
|
||||
text = block_text(block).strip()
|
||||
if len(text) < 2 or len(text) > 200:
|
||||
return False
|
||||
if _DOT_LEADER_RE.search(text):
|
||||
return False
|
||||
if _CAPTION_LABEL_RE.match(text):
|
||||
return False
|
||||
if _EQUATION_LABEL_RE.match(text):
|
||||
return False
|
||||
if _looks_like_pseudo_code(text):
|
||||
return False
|
||||
marker_type = getattr(block, "marker_slot", 0)
|
||||
if marker_type == 4:
|
||||
return True
|
||||
block_size = block.avg_font_size() or body_size
|
||||
size_gain = block_size / max(body_size, 1e-3)
|
||||
if size_gain >= 1.08:
|
||||
return True
|
||||
if block.bold_frac() > 0.5 and not body_bold and size_gain >= 0.95:
|
||||
return True
|
||||
if block.line_count() >= 1 and numbering_kind(block.line()) != 0 and len(text) <= 120 and (
|
||||
block.bold_frac() > 0.3 or size_gain >= 1.0
|
||||
):
|
||||
return True
|
||||
if is_caps_heavy(block) and len(text) <= 80 and size_gain >= 1.0:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Style buckets + level assignment #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _style_key(block: Block) -> tuple[str, float, bool]:
|
||||
"""Hashable signature for grouping headings into hierarchy levels."""
|
||||
first_span = block.line().primary_slot[0] if block.line().primary_slot else None
|
||||
font = first_span.font_name if first_span else ""
|
||||
return (font, round(block.avg_font_size(), 1), block.bold_frac() > 0.5)
|
||||
|
||||
|
||||
def _numbering_depth(block: Block) -> Optional[int]:
|
||||
"""Return the section-numbering depth, e.g. ``1.2.3-> 3. None if the block doesn't start with a digit-style number (only digit chains use ``.``-separated depth; Roman / letter labels return 1). """
|
||||
if numbering_kind(block.line()) != 1:
|
||||
return None
|
||||
text = numbering_text(block.line())
|
||||
if not text:
|
||||
return None
|
||||
if re.match(r"^\d+(?:\.\d+)*$", text):
|
||||
return text.count(".") + 1
|
||||
return 1
|
||||
|
||||
|
||||
def collect_headings(doc) -> list[Block]:
|
||||
"""Walk all pages, gather heading-candidate blocks in reading order."""
|
||||
body_size = doc.secondary_slot.primary_slot
|
||||
body_bold = doc.secondary_slot.tertiary_slot == 0
|
||||
out: list[Block] = []
|
||||
for page in doc.primary_slot:
|
||||
for block in (page.secondary_slot or []):
|
||||
if is_heading_candidate(block, body_size, body_bold):
|
||||
out.append(block)
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Clique selection #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _heading_signature(block: Block) -> str:
|
||||
"""Return the first visible span's font style for a heading block."""
|
||||
if not block.primary_slot or not block.primary_slot[0].primary_slot:
|
||||
return ""
|
||||
return block.primary_slot[0].primary_slot[0].font_style()
|
||||
|
||||
|
||||
def _matches_section_keywords(block: Block) -> bool:
|
||||
"""Full-match canonical English section names after stripping a leading structural number. For example, "1 Introduction" tokenizes as ["1", "Introduction"], and the numeric prefix must be removed before keyword matching."""
|
||||
tokens = tokenize_block(block)
|
||||
prefix = extract_structural_number(tokens)
|
||||
if prefix is not None:
|
||||
tokens = tokens.slice(prefix.length)
|
||||
return trie_full_match(SECTION_KEYWORD_TRIE, tokens)
|
||||
|
||||
|
||||
_PSEUDO_CODE_PATTERNS = (
|
||||
re.compile(r"^\s*\d+\s*:"), # "2:" / "10:" lead -> pseudo-code step
|
||||
re.compile(r"[∀-⋿←-⇿≤≥≠∈∉∂∇∑∏√]"), # math operators
|
||||
re.compile(r"^\s*\d+[a-z]"), # "9else", "15return" (no space)
|
||||
re.compile(r"^[\d.\s/]+$"), # pure numbers / decimals
|
||||
re.compile(r"^\s*\d+\s*[+\-*/=]\s*\d"), # arithmetic
|
||||
re.compile(r"^\s*[a-z]+\s*[+\-*/=]\s*"), # variable assignments
|
||||
re.compile(r"\bwhile\b|\bif\b|\belse\b|\bfor\b|\breturn\b|\bdo\b", re.IGNORECASE), # code keywords
|
||||
# Subfigure captions like "(a) RETINA", "(b) IRMA", "(i) plot"
|
||||
re.compile(r"^\s*[\(\[]\s*[a-zivx]+\s*[\)\]]\s+\w"),
|
||||
)
|
||||
|
||||
# Author block / affiliation patterns:
|
||||
# * "Yu Tang†, Leong Hou U‡, ..." -- multiple comma-separated names with
|
||||
# affiliation markers
|
||||
# * "Kimi Team" -- short "X Team" / "X Lab" / "X Group" naming
|
||||
# * "†The University of ..." -- starts with affiliation marker
|
||||
# * "{user, another}@domain" -- email block
|
||||
_AUTHOR_PATTERNS = (
|
||||
re.compile(r"[†‡§¶∗*]"), # affiliation markers
|
||||
re.compile(r"@\S+\."), # contains email
|
||||
re.compile(r"^\s*\S+\s+(?:Team|Group|Lab|Labs|Inc\.|Corp\.|Co\.)\s*$"),
|
||||
)
|
||||
# Bullet-list items: "1. " followed by long flowing text (>80 chars total)
|
||||
_BULLET_LIST_RE = re.compile(r"^\s*\d+\.\s+\w")
|
||||
|
||||
|
||||
def filter_by_clique(headings: list[Block]) -> list[Block]:
|
||||
"""Keep headings that match the dominant section-keyword style group."""
|
||||
if not headings:
|
||||
return headings
|
||||
anchor_groups: dict[str, list[Block]] = defaultdict(list)
|
||||
for state_item in headings:
|
||||
if _matches_section_keywords(state_item):
|
||||
anchor_groups[_heading_signature(state_item)].append(state_item)
|
||||
if not anchor_groups:
|
||||
return headings
|
||||
winner_sig, winner_group = max(anchor_groups.items(), key=lambda item_pair: len(item_pair[1]))
|
||||
if len(winner_group) <= 1:
|
||||
return headings
|
||||
out: list[Block] = []
|
||||
for state_item in headings:
|
||||
if _heading_signature(state_item) == winner_sig:
|
||||
out.append(state_item)
|
||||
continue
|
||||
# Different font from heading clique. Only keep if it's a clearly
|
||||
# numbered heading that doesn't smell of pseudo-code / math.
|
||||
if numbering_kind(state_item.line()) != 1:
|
||||
continue
|
||||
numbering = numbering_text(state_item.line())
|
||||
if not re.match(r"^\d+(?:\.\d+){0,2}$", numbering):
|
||||
continue
|
||||
text = block_text(state_item)
|
||||
if any(pat.search(text) for pat in _PSEUDO_CODE_PATTERNS):
|
||||
continue
|
||||
# Also require: at least one alphabetic word AFTER the number
|
||||
# ("2 Introduction" yes, "9else" no, "1.804 1.737 1.692" no)
|
||||
after_num = re.sub(r"^\s*\d+(?:\.\d+){0,2}\s*[.:)]?\s*", "", text)
|
||||
if not re.search(r"[A-Za-zÀ-ÿ一-鿿가-]{3,}", after_num):
|
||||
continue
|
||||
out.append(state_item)
|
||||
return out
|
||||
@@ -0,0 +1,132 @@
|
||||
"""Level assignment and outline tree construction."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import defaultdict
|
||||
from ..model import numbering_text, numbering_kind, block_text, is_caps_heavy, Block
|
||||
|
||||
from .filtering import (
|
||||
_style_key,
|
||||
_numbering_depth,
|
||||
)
|
||||
|
||||
|
||||
def extract_top_level_headings(headings: list[Block], levels: dict[int, int]) -> list[Block]:
|
||||
"""Flatten the heading tree, returning only top-level headings."""
|
||||
return [heading for heading in headings if levels.get(id(heading), 6) <= 1]
|
||||
|
||||
|
||||
def assign_levels(headings: list[Block]) -> dict[int, int]:
|
||||
"""Return ``{id(block) -> level}``. 1. Bucket by style key. 2. Rank styles by (size DESC, bold DESC) and assign level 1..6 in that order (anything below the 6th distinct style is clamped to 6). 3. If a heading has digit-numbering, its level is overridden to min(numbering_depth, style_level) -- numbering wins for deeper grouping but never promotes a heading above its style rank. """
|
||||
buckets: dict[tuple[str, float, bool], list[Block]] = defaultdict(list)
|
||||
for state_item in headings:
|
||||
buckets[_style_key(state_item)].append(state_item)
|
||||
ranked = sorted(buckets.keys(), key=lambda key_value: (-key_value[1], not key_value[2]))
|
||||
style_level = {key_value: min(index_value + 1, 6) for index_value, key_value in enumerate(ranked)}
|
||||
out: dict[int, int] = {}
|
||||
for state_item in headings:
|
||||
lvl = style_level.get(_style_key(state_item), 6)
|
||||
depth = _numbering_depth(state_item)
|
||||
if depth is not None:
|
||||
lvl = max(1, min(lvl, depth))
|
||||
out[id(state_item)] = lvl
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Tree assembly #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _heading_title(block: Block) -> str:
|
||||
"""Cleaned title text for output (no dot leaders, single-line)."""
|
||||
text = block_text(block).strip()
|
||||
text = re.sub(r"\s+", " ", text)
|
||||
return text
|
||||
|
||||
|
||||
def _heading_page_num(block: Block, page_lookup) -> int:
|
||||
"""Find the 1-based page number that owns this block. ``page_lookup`` is a dict ``{id(block) -> page.u}`` precomputed by the caller for O(1) lookup. """
|
||||
return page_lookup.get(id(block), 1)
|
||||
|
||||
|
||||
def build_tree(headings: list[Block], levels: dict[int, int], page_lookup, total_pages: int) -> list[dict]:
|
||||
"""Assemble nested ``{title, start_index, end_index, nodes}`` tree."""
|
||||
if not headings:
|
||||
return []
|
||||
|
||||
root: list[dict] = []
|
||||
stack: list[tuple[int, dict]] = []
|
||||
for state_item in headings:
|
||||
title = _heading_title(state_item)
|
||||
if not title:
|
||||
continue
|
||||
node = {
|
||||
"title": title,
|
||||
"start_index": _heading_page_num(state_item, page_lookup),
|
||||
"end_index": _heading_page_num(state_item, page_lookup),
|
||||
"nodes": [],
|
||||
}
|
||||
lvl = levels.get(id(state_item), 6)
|
||||
while stack and stack[-1][0] >= lvl:
|
||||
stack.pop()
|
||||
if not stack:
|
||||
root.append(node)
|
||||
else:
|
||||
stack[-1][1]["nodes"].append(node)
|
||||
stack.append((lvl, node))
|
||||
|
||||
# Fill end_index in DFS order.
|
||||
flat: list[dict] = []
|
||||
|
||||
def _walk_nodes(nodes: list[dict]) -> None:
|
||||
for count_item in nodes:
|
||||
flat.append(count_item)
|
||||
_walk_nodes(count_item["nodes"])
|
||||
|
||||
_walk_nodes(root)
|
||||
for index_value, count_item in enumerate(flat):
|
||||
next_start = flat[index_value + 1]["start_index"] if index_value + 1 < len(flat) else total_pages
|
||||
count_item["end_index"] = max(count_item["start_index"], next_start - 1 if next_start > count_item["start_index"] else count_item["start_index"])
|
||||
if flat:
|
||||
flat[-1]["end_index"] = max(flat[-1]["start_index"], total_pages)
|
||||
|
||||
# Drop empty children so the JSON matches the shape the rest of PageIndex emits.
|
||||
def _drop_empty_children(nodes: list[dict]) -> list[dict]:
|
||||
for count_item in nodes:
|
||||
if count_item["nodes"]:
|
||||
_drop_empty_children(count_item["nodes"])
|
||||
else:
|
||||
del count_item["nodes"]
|
||||
return nodes
|
||||
|
||||
return _drop_empty_children(root)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Outline validation #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def validate(headings: list[Block], levels: dict[int, int], doc) -> bool:
|
||||
"""Return whether the outline has enough top-level headings spanning a meaningful fraction of the document."""
|
||||
top = [state_item for state_item in headings if levels.get(id(state_item), 6) <= 2]
|
||||
if len(top) < 3:
|
||||
return False
|
||||
if len(top) >= 5:
|
||||
return True
|
||||
last_page = 1
|
||||
for state_item in top:
|
||||
# Direct page lookup would need a page back-reference; we use the document
|
||||
# order proxy (top is already in reading order).
|
||||
# Find by scanning document pages for the page containing the block.
|
||||
page_num = 1
|
||||
for page in doc.primary_slot:
|
||||
if state_item in (page.secondary_slot or []):
|
||||
page_num = page.page_index
|
||||
break
|
||||
if page_num - last_page > 0.5 * len(doc.primary_slot):
|
||||
return False
|
||||
last_page = page_num
|
||||
return True
|
||||
@@ -0,0 +1,110 @@
|
||||
"""Outline assembly chain. This module turns heading candidates and labeled section regions into the final
|
||||
nested outline tree. It groups candidates by numbering depth, style signature,
|
||||
script compatibility, document order, and local clusters, then serializes the
|
||||
tree into the public PageIndex JSON shape.
|
||||
"""
|
||||
|
||||
import math
|
||||
from typing import Any, Callable, Optional
|
||||
|
||||
from sortedcontainers import SortedKeyList
|
||||
from ..model import (
|
||||
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
|
||||
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
|
||||
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
|
||||
)
|
||||
from ..stats import style_key as style_key_fn, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
|
||||
from ..tokens import (
|
||||
Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, avg_char_width as avg_char_width_fn, trie_full_match, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Numbering-pattern clique selection.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
# Section-keyword trie shared with outline filtering.
|
||||
from ..outline import SECTION_KEYWORD_TRIE
|
||||
|
||||
from .candidates import (
|
||||
_viewport_y_fraction,
|
||||
HeadingCandidate,
|
||||
OutlineNode,
|
||||
compare_heading_order,
|
||||
_compare_block_order,
|
||||
heading_order_key,
|
||||
is_script_compatible,
|
||||
heading_signature,
|
||||
parent_signature,
|
||||
cached_signature,
|
||||
is_in_oo_range,
|
||||
has_style_neighbor,
|
||||
)
|
||||
from .style_context import (
|
||||
StyleCluster,
|
||||
pick_style_bucket,
|
||||
has_conflict_in_context,
|
||||
is_compatible_with_context,
|
||||
OutlineContext,
|
||||
NumberingTrie,
|
||||
insert_numbering,
|
||||
count_sibling_numberings,
|
||||
OutlineState,
|
||||
_apply_heading_to_state,
|
||||
compare_heading_depth,
|
||||
)
|
||||
from .cliques import (
|
||||
find_keyword_clique,
|
||||
CliqueTreeNode,
|
||||
find_ancestor_next_sibling,
|
||||
descend_to_deepest_last,
|
||||
append_tree_child,
|
||||
CliqueTreeBuilder,
|
||||
block_style_signature,
|
||||
is_member_of_tree,
|
||||
can_share_heading_style,
|
||||
compare_block_order,
|
||||
heading_precedes_line,
|
||||
CliqueFilterContext,
|
||||
detect_body_headings,
|
||||
partition_candidates,
|
||||
interleave_clusters,
|
||||
)
|
||||
from .selection import (
|
||||
min_font_distance,
|
||||
should_reject_heading,
|
||||
push_heading_to_state,
|
||||
HierarchyStack,
|
||||
find_parent_heading,
|
||||
is_appendix_nesting_ok,
|
||||
extract_sub_headings,
|
||||
extract_top_level_headings,
|
||||
is_outline_valid,
|
||||
is_chapter_outline_valid,
|
||||
)
|
||||
from .assembly import (
|
||||
mark_outline_block_types,
|
||||
compute_max_heading_gap,
|
||||
has_table_or_prominent,
|
||||
is_landscape_or_empty,
|
||||
build_heading_from_block,
|
||||
assemble_outline,
|
||||
_flatten_outline_nodes,
|
||||
_heading_appears_at_page_top,
|
||||
outline_to_dict_tree,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"HeadingCandidate", "OutlineNode",
|
||||
"compare_heading_order", "heading_order_key", "compare_heading_depth",
|
||||
"is_script_compatible", "heading_signature", "parent_signature", "cached_signature", "is_in_oo_range", "has_style_neighbor", "pick_style_bucket", "has_conflict_in_context", "is_compatible_with_context",
|
||||
"StyleCluster", "OutlineContext", "NumberingTrie", "insert_numbering", "count_sibling_numberings",
|
||||
"OutlineState",
|
||||
"find_keyword_clique", "detect_body_headings", "CliqueFilterContext",
|
||||
"partition_candidates", "interleave_clusters", "push_heading_to_state", "should_reject_heading", "find_parent_heading", "HierarchyStack", "extract_sub_headings", "min_font_distance",
|
||||
"extract_top_level_headings", "is_outline_valid", "is_chapter_outline_valid", "mark_outline_block_types", "compute_max_heading_gap", "has_table_or_prominent",
|
||||
"build_heading_from_block",
|
||||
"assemble_outline",
|
||||
"outline_to_dict_tree",
|
||||
]
|
||||
@@ -0,0 +1,345 @@
|
||||
"""Final outline assembly and conversion to the output dict tree."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Callable, Optional
|
||||
from ..model import (
|
||||
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
|
||||
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
|
||||
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, avg_char_width as avg_char_width_fn, trie_full_match, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
from .candidates import (
|
||||
HeadingCandidate,
|
||||
OutlineNode,
|
||||
heading_order_key,
|
||||
)
|
||||
from .style_context import (
|
||||
OutlineState,
|
||||
compare_heading_depth,
|
||||
)
|
||||
from .cliques import (
|
||||
find_keyword_clique,
|
||||
CliqueFilterContext,
|
||||
detect_body_headings,
|
||||
partition_candidates,
|
||||
interleave_clusters,
|
||||
)
|
||||
from .selection import (
|
||||
should_reject_heading,
|
||||
push_heading_to_state,
|
||||
HierarchyStack,
|
||||
find_parent_heading,
|
||||
extract_sub_headings,
|
||||
)
|
||||
|
||||
|
||||
def mark_outline_block_types(item_list: list[OutlineNode]) -> None:
|
||||
"""Mark outline blocks as numbered or unnumbered headings."""
|
||||
for block in item_list:
|
||||
block.heading.group_slot.type = 8 if block.heading.has_numbering else 7
|
||||
mark_outline_block_types(block.child_nodes)
|
||||
|
||||
|
||||
def compute_max_heading_gap(outline_nodes: list[OutlineNode], other_number: int) -> dict:
|
||||
"""Compute the maximum page-position gap between outline nodes."""
|
||||
if not outline_nodes:
|
||||
return {"max_gap": 0, "last_page_position": other_number}
|
||||
heading = 0
|
||||
for stack_outline_node in outline_nodes:
|
||||
page_pos = stack_outline_node.heading.page.page_index + stack_outline_node.heading.auxiliary_slot
|
||||
heading = max(heading, page_pos - other_number)
|
||||
other_number = page_pos
|
||||
rec = compute_max_heading_gap(stack_outline_node.child_nodes, other_number)
|
||||
heading = max(heading, rec["max_gap"])
|
||||
other_number = rec["last_page_position"]
|
||||
return {"max_gap": heading, "last_page_position": other_number}
|
||||
|
||||
|
||||
def has_table_or_prominent(outline_nodes: list[OutlineNode]) -> bool:
|
||||
"""Return True if any heading is a table-like or prominent entry."""
|
||||
return any(secondary_item.heading.type == 5 or secondary_item.heading.is_prominent for secondary_item in outline_nodes)
|
||||
|
||||
|
||||
def is_landscape_or_empty(doc) -> bool:
|
||||
"""Return True for mostly-landscape or near-empty documents with little outline text."""
|
||||
if doc.secondary_slot.secondary_slot >= 1e3:
|
||||
return False
|
||||
secondary_item = 0
|
||||
candidate_item = 0.0
|
||||
for page in doc.primary_slot:
|
||||
if page.bounds.bbox_width() > page.bounds.bbox_height() and page.primary_slot.secondary_slot < 1e3:
|
||||
secondary_item += 1
|
||||
candidate_item += page.primary_slot.secondary_slot
|
||||
count_item = len(doc.primary_slot)
|
||||
return secondary_item >= 0.9 * count_item or (secondary_item >= 0.7 * count_item and candidate_item >= 0.5 * doc.secondary_slot.state_slot)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Build a heading candidate from a block #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def build_heading_from_block(block: Block, page, anchor: Optional[Block] = None) -> HeadingCandidate:
|
||||
"""Build a heading candidate wrapper for a heading block."""
|
||||
tokens = tokenize_block(block)
|
||||
# Extract structural numbering from the leading line.
|
||||
item_list: list[int] = []
|
||||
has_numbering = False
|
||||
prefix: Optional[TokenView] = None
|
||||
title: TokenView = tokens
|
||||
if numbering_kind(block.line()) == 1:
|
||||
num_str = numbering_text(block.line())
|
||||
if num_str:
|
||||
try:
|
||||
parts = [int(number_part) for number_part in num_str.replace(".", ".").split(".") if number_part.strip()]
|
||||
if all(0 <= number_part < 1000 for number_part in parts):
|
||||
item_list = parts
|
||||
has_numbering = True
|
||||
# Strip the leading number tokens from g
|
||||
skip = 0
|
||||
while skip < tokens.length:
|
||||
tok = tokens.token_at(skip)
|
||||
if tok is None:
|
||||
break
|
||||
if tok.type == 1 or tok.str in "..":
|
||||
skip += 1
|
||||
else:
|
||||
break
|
||||
title = tokens.slice(skip)
|
||||
except (ValueError, AttributeError):
|
||||
pass
|
||||
|
||||
# Type from labeled-section classification or from numbering.
|
||||
marker_type = getattr(block, "marker_slot", 0) or 0
|
||||
if marker_type == 4:
|
||||
type_ = 4
|
||||
elif marker_type == 5:
|
||||
type_ = 5
|
||||
elif marker_type == 11:
|
||||
type_ = 11
|
||||
elif item_list:
|
||||
type_ = 1
|
||||
elif is_caps_heavy(block) and block.line_count() == 1:
|
||||
type_ = 2 # uppercase short heading
|
||||
else:
|
||||
type_ = 0
|
||||
|
||||
# Prominence flag: big font / bold-and-prominent.
|
||||
body_size_threshold = page.primary_slot.primary_slot + 0.5 if page.primary_slot else 0
|
||||
ja_flag = (
|
||||
block.avg_font_size() > body_size_threshold + 1.5
|
||||
or (block.bold_frac() > 0.5 and block.avg_font_size() >= body_size_threshold)
|
||||
)
|
||||
|
||||
return HeadingCandidate(
|
||||
type_=type_,
|
||||
page=page,
|
||||
group_value=block,
|
||||
anchor=anchor,
|
||||
numbering_value=item_list,
|
||||
tokens=prefix,
|
||||
title_tokens=title,
|
||||
has_numbering_flag=has_numbering,
|
||||
prominent_flag=ja_flag,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Main outline assembler #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def assemble_outline(doc, labeled: list[OutlineNode]) -> list[OutlineNode]:
|
||||
"""Produce the outline tree as a list of outline nodes. Arguments: ``doc`` is the document state; ``labeled`` is the list of outline nodes wrapping labeled headings. Output is a list of root outline nodes. Each node contains child nodes recursively. """
|
||||
# ----- Stage 1: collect general headings.
|
||||
from ..heading_detection import build_doc_heading_candidates
|
||||
# Labeled headings prime the type gates used by general heading filtering.
|
||||
general: list[HeadingCandidate] = build_doc_heading_candidates(doc, labeled)
|
||||
|
||||
# ----- Stage 2: merge with labeled
|
||||
if len(labeled) + len(general) > 0:
|
||||
combined = list(general)
|
||||
for labeled_region_node in labeled:
|
||||
combined.append(labeled_region_node.heading)
|
||||
combined.sort(key=heading_order_key)
|
||||
# Build the keyword clique before body-heading filtering so the filter
|
||||
# can test whether a block is already represented in the candidate tree.
|
||||
clique = find_keyword_clique(combined)
|
||||
filtered = detect_body_headings(CliqueFilterContext(doc, combined, lambda line, other_line: compare_heading_depth(line, other_line, clique)))
|
||||
general.extend(filtered)
|
||||
# No dedup here: duplicate candidates that wrap the same block are
|
||||
# collapsed downstream by partitioning and already-placed-block checks.
|
||||
general = sorted(general, key=heading_order_key)
|
||||
|
||||
# ----- Stage 3: partition + cluster
|
||||
if labeled:
|
||||
# No pre-filter: partitioning re-separates labeled vs general, so any
|
||||
# labeled block backfilled into the general list is handled there.
|
||||
result = partition_candidates(general, labeled)
|
||||
general = result["remaining"]
|
||||
labeled = result["labeled"]
|
||||
clusters = interleave_clusters(general, labeled)
|
||||
else:
|
||||
clusters = [{"labeled_anchor": None, "cluster_candidates": general}]
|
||||
|
||||
# ----- Stage 4: assemble tree
|
||||
state = OutlineState(clusters)
|
||||
if not state.measure_slot and state.option_slot <= state.previous_slot:
|
||||
return []
|
||||
|
||||
out: list[OutlineNode] = []
|
||||
for cluster in clusters:
|
||||
cluster_anchor = cluster.get("labeled_anchor")
|
||||
cluster_candidates = cluster.get("cluster_candidates", [])
|
||||
if cluster_anchor is not None:
|
||||
push_heading_to_state(state, cluster_anchor.heading)
|
||||
out.append(cluster_anchor)
|
||||
sub = extract_sub_headings(doc, state, cluster_anchor, cluster_candidates)
|
||||
target = cluster_anchor.child_nodes if cluster_anchor is not None else out
|
||||
target.extend(sub)
|
||||
|
||||
sub_clique = find_keyword_clique(cluster_candidates) if cluster_candidates else None
|
||||
stack = HierarchyStack(sub_clique)
|
||||
for insertion_candidate in cluster_candidates:
|
||||
if should_reject_heading(state, insertion_candidate):
|
||||
continue
|
||||
push_heading_to_state(state, insertion_candidate)
|
||||
insertion_candidate.group_slot.used_as_heading = True
|
||||
stack_outline_node = OutlineNode(insertion_candidate)
|
||||
parent = find_parent_heading(stack, insertion_candidate)
|
||||
if parent is not None:
|
||||
parent.child_nodes.append(stack_outline_node)
|
||||
elif cluster_anchor is not None:
|
||||
cluster_anchor.child_nodes.append(stack_outline_node)
|
||||
else:
|
||||
out.append(stack_outline_node)
|
||||
stack.push(stack_outline_node)
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Outline tree -> PageIndex dict tree #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _flatten_outline_nodes(outline_node_list: list[OutlineNode]) -> list[OutlineNode]:
|
||||
"""Walk an outline tree DFS to a flat list, preserving order."""
|
||||
out: list[OutlineNode] = []
|
||||
|
||||
def _walk_nodes(items: list[OutlineNode]) -> None:
|
||||
for item in items:
|
||||
out.append(item)
|
||||
if item.child_nodes:
|
||||
_walk_nodes(item.child_nodes)
|
||||
|
||||
_walk_nodes(outline_node_list)
|
||||
return out
|
||||
|
||||
|
||||
def _heading_appears_at_page_top(heading: HeadingCandidate) -> bool:
|
||||
"""Return whether a heading begins its page with no flowing content above it."""
|
||||
top_heading = heading.group_slot
|
||||
page = heading.page
|
||||
if top_heading is None or page is None:
|
||||
return True
|
||||
group_index = getattr(top_heading, "reading_order_index", 0)
|
||||
for block in (page.secondary_slot or []):
|
||||
if block is top_heading or getattr(block, "reading_order_index", 0) >= group_index:
|
||||
continue # only blocks before the heading
|
||||
if block.char_count() <= 0:
|
||||
continue # no text
|
||||
if block.type in (1, 2, 12): # header / footer / watermark
|
||||
continue
|
||||
return False # real content precedes the heading
|
||||
return True
|
||||
|
||||
|
||||
def outline_to_dict_tree(outline_node_list: list[OutlineNode], total_pages: int) -> list[dict]:
|
||||
"""Convert the outline tree directly to PageIndex JSON shape. Preserves the natural outline nesting without font-overlay rewriting. """
|
||||
flat_nodes: list[dict] = []
|
||||
|
||||
def _walk_nodes(items: list[OutlineNode]) -> list[dict]:
|
||||
result: list[dict] = []
|
||||
for item in items:
|
||||
# Title text is the numbering prefix plus the heading tokens, but
|
||||
# the two are carried as separate fields and trimmed one by one,
|
||||
# then rejoined with a single space and only for a non-empty
|
||||
# prefix. A prefix's string form ends in a space after every
|
||||
# space-flagged token, so trimming the parts separately is what
|
||||
# keeps that space out of the join.
|
||||
# Trim with the Unicode WhiteSpace+LineTerminator set, not Python's
|
||||
# str.strip set: they differ on U+FEFF, U+0085, and U+001C-1F.
|
||||
prefix_tokens = item.heading.secondary_slot
|
||||
token = item.heading.primary_slot
|
||||
child = _trim_unicode_ws(str(prefix_tokens)) if prefix_tokens is not None else ""
|
||||
node = _trim_unicode_ws(str(token)) if token is not None else ""
|
||||
title = (child + " " if child else "") + node
|
||||
if not title:
|
||||
if item.child_nodes:
|
||||
result.extend(_walk_nodes(item.child_nodes))
|
||||
continue
|
||||
node = {
|
||||
"title": title,
|
||||
"node_id": "",
|
||||
"start_index": item.heading.page.page_index,
|
||||
"end_index": item.heading.page.page_index,
|
||||
"nodes": _walk_nodes(item.child_nodes) if item.child_nodes else [],
|
||||
"_appear_start": _heading_appears_at_page_top(item.heading),
|
||||
}
|
||||
flat_nodes.append(node)
|
||||
result.append(node)
|
||||
return result
|
||||
|
||||
root = _walk_nodes(outline_node_list)
|
||||
|
||||
# Fill end_index via DFS-order next-start - 1; last node extends to doc end.
|
||||
flat: list[dict] = []
|
||||
|
||||
def _collect(nodes: list[dict]) -> None:
|
||||
for count_item in nodes:
|
||||
flat.append(count_item)
|
||||
_collect(count_item["nodes"])
|
||||
|
||||
_collect(root)
|
||||
for line, outline_entry in enumerate(flat):
|
||||
if line + 1 < len(flat):
|
||||
nxt = flat[line + 1]
|
||||
# page_index post_processing (utils.post_processing): if the next
|
||||
# heading starts at the top of its page, this section ends the page
|
||||
# before it; otherwise the next heading sits below this section's
|
||||
# tail, so the two share that boundary page and the end extends onto
|
||||
# it.
|
||||
boundary = (
|
||||
nxt["start_index"] - 1
|
||||
if nxt["_appear_start"]
|
||||
else nxt["start_index"]
|
||||
)
|
||||
else:
|
||||
boundary = total_pages
|
||||
outline_entry["end_index"] = max(
|
||||
outline_entry["start_index"],
|
||||
boundary if boundary > outline_entry["start_index"] else outline_entry["start_index"],
|
||||
)
|
||||
if flat:
|
||||
flat[-1]["end_index"] = max(flat[-1]["start_index"], total_pages)
|
||||
|
||||
# Stable DFS pre-order node ids, zero-padded to 4 (PageIndex convention;
|
||||
# uses zero-padded depth-first ids). Drop the
|
||||
# transient appear_start marker now that end_index is settled.
|
||||
for line, outline_entry in enumerate(flat):
|
||||
outline_entry["node_id"] = str(line).zfill(4)
|
||||
del outline_entry["_appear_start"]
|
||||
|
||||
def _drop_empty_children(nodes: list[dict]) -> list[dict]:
|
||||
for count_item in nodes:
|
||||
if count_item["nodes"]:
|
||||
_drop_empty_children(count_item["nodes"])
|
||||
else:
|
||||
del count_item["nodes"]
|
||||
return nodes
|
||||
|
||||
return _drop_empty_children(root)
|
||||
@@ -0,0 +1,255 @@
|
||||
"""Heading candidate and outline node types plus ordering and signature helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ..model import (
|
||||
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
|
||||
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
|
||||
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
|
||||
)
|
||||
from ..stats import style_key as style_key_fn, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
|
||||
from ..tokens import (
|
||||
Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, avg_char_width as avg_char_width_fn, trie_full_match, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Heading candidate wrapper #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _viewport_y_fraction(viewport_box, rot: int, user_x: float, user_y: float) -> float:
|
||||
"""Return viewport-normalized y coordinate for a PDF user-space point. Applies the same page ``/Rotate`` and the unrotated view box to a user-space point, then normalises the y component by the viewport height."""
|
||||
x_min, y_min, x_max, y_max = viewport_box
|
||||
center_x = (x_max + x_min) / 2.0
|
||||
center_y = (y_max + y_min) / 2.0
|
||||
rotation = rot % 360
|
||||
if rotation < 0:
|
||||
rotation += 360
|
||||
if rotation == 90:
|
||||
x_axis_scale, y_axis_scale = 1, 0
|
||||
x_axis_sign = 0
|
||||
elif rotation == 180:
|
||||
x_axis_scale, y_axis_scale = 0, 1
|
||||
x_axis_sign = -1
|
||||
elif rotation == 270:
|
||||
x_axis_scale, y_axis_scale = -1, 0
|
||||
x_axis_sign = 0
|
||||
else:
|
||||
x_axis_scale, y_axis_scale = 0, -1
|
||||
x_axis_sign = 1
|
||||
if x_axis_sign == 0:
|
||||
viewport_offset = abs(center_x - x_min)
|
||||
height = abs(x_max - x_min)
|
||||
else:
|
||||
viewport_offset = abs(center_y - y_min)
|
||||
height = abs(y_max - y_min)
|
||||
# transform[1]=b, transform[3]=d, transform[5]=off_y - b*cx - d*cy;
|
||||
# the viewport y-coordinate = b*x + d*y + transform[5].
|
||||
viewport_y = x_axis_scale * user_x + y_axis_scale * user_y + (viewport_offset - x_axis_scale * center_x - y_axis_scale * center_y)
|
||||
return viewport_y / (height or 1.0)
|
||||
|
||||
|
||||
class HeadingCandidate:
|
||||
"""One heading candidate. It stores the candidate type, page, underlying block, optional anchor block, numbering array, optional prefix tokens, title tokens, structural-numbering flag, prominence flag, dominant script family, and vertical page position."""
|
||||
|
||||
__slots__ = ("type", "page", "group_slot", "tertiary_slot", "numbering", "secondary_slot", "primary_slot", "has_numbering", "is_prominent", "state_slot", "auxiliary_slot")
|
||||
|
||||
def __init__(self, type_, page, group_value, anchor, numbering_value, tokens, title_tokens, has_numbering_flag, prominent_flag):
|
||||
self.type = type_
|
||||
self.page = page
|
||||
self.group_slot = group_value
|
||||
self.tertiary_slot = anchor
|
||||
self.numbering = numbering_value or []
|
||||
self.secondary_slot = tokens
|
||||
self.primary_slot = title_tokens
|
||||
self.has_numbering = has_numbering_flag
|
||||
self.is_prominent = prominent_flag
|
||||
# Compute the dominant script family over prefix and title tokens.
|
||||
acc = ScriptHistogram()
|
||||
if tokens is not None:
|
||||
for token_value in tokens:
|
||||
tally_scripts(acc, token_value.str)
|
||||
if title_tokens is not None:
|
||||
for token_value in title_tokens:
|
||||
tally_scripts(acc, token_value.str)
|
||||
self.state_slot = dominant_script_family(acc)
|
||||
# Compute vertical fraction on page. The viewport applies the page
|
||||
# /Rotate and view box; when that metadata is absent, fall back to the
|
||||
# origin-0 upright shortcut.
|
||||
viewport_box_value = getattr(page, "viewport_box", None)
|
||||
if viewport_box_value is not None:
|
||||
self.auxiliary_slot = _viewport_y_fraction(viewport_box_value, getattr(page, "rot", 0) or 0, group_value.left_edge(), group_value.top_edge())
|
||||
else:
|
||||
page_height = page.bounds.bbox_height() or 1.0
|
||||
self.auxiliary_slot = (page.bounds.top_edge() - group_value.top_edge()) / page_height
|
||||
|
||||
def __repr__(self) -> str: # diagnostic
|
||||
return f"<HeadingCandidate t={self.type} M={self.numbering} G={block_text(self.group_slot)[:30]!r}>"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Outline node #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class OutlineNode:
|
||||
"""Heading plus child outline nodes."""
|
||||
|
||||
__slots__ = ("heading", "child_nodes")
|
||||
|
||||
def __init__(self, heading: HeadingCandidate):
|
||||
self.heading = heading
|
||||
self.child_nodes: list["OutlineNode"] = []
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Page and reading-position comparator.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def compare_heading_order(heading_candidate: HeadingCandidate, other_heading_candidate: HeadingCandidate) -> float:
|
||||
"""Order by page, then block reading position."""
|
||||
if heading_candidate.page.page_index != other_heading_candidate.page.page_index:
|
||||
return heading_candidate.page.page_index - other_heading_candidate.page.page_index
|
||||
return _compare_block_order(heading_candidate.group_slot, other_heading_candidate.group_slot)
|
||||
|
||||
|
||||
def _compare_block_order(block: Block, other_block: Block) -> float:
|
||||
"""Compare by column index first, then by reading position."""
|
||||
from ..model import cmp_reading_order
|
||||
from ..stats import column_index_of as _column_index
|
||||
heading_anchor = _column_index(block)
|
||||
other_column_index = _column_index(other_block)
|
||||
if heading_anchor != other_column_index:
|
||||
return heading_anchor - other_column_index
|
||||
return cmp_reading_order(block, other_block)
|
||||
|
||||
|
||||
def heading_order_key(heading_candidate: HeadingCandidate) -> tuple:
|
||||
from ..stats import column_index_of as _column_index
|
||||
return (heading_candidate.page.page_index, _column_index(heading_candidate.group_slot), -heading_candidate.group_slot.top_edge(), -heading_candidate.group_slot.bottom_edge(), heading_candidate.group_slot.left_edge(), heading_candidate.group_slot.right_edge())
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Candidate compatibility and style-cluster helpers #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def is_script_compatible(number: int, other_heading_candidate: HeadingCandidate) -> bool:
|
||||
"""Return whether candidate script/type is compatible with prior context. Args: a: integer previous script/type context b: heading candidate """
|
||||
candidate_item = other_heading_candidate.state_slot
|
||||
if candidate_item == 0 or candidate_item == 2 or candidate_item == 10:
|
||||
return True
|
||||
if number == candidate_item:
|
||||
return False
|
||||
if other_heading_candidate.type == 5:
|
||||
return False
|
||||
if other_heading_candidate.is_prominent:
|
||||
return False
|
||||
if len(other_heading_candidate.numbering) > 0:
|
||||
return False
|
||||
if number == 3 and candidate_item == 9:
|
||||
return False
|
||||
if number == 9 and candidate_item == 3:
|
||||
return False
|
||||
if number == 7 and candidate_item == 5:
|
||||
return False
|
||||
if number == 6 and candidate_item == 3:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def heading_signature(heading_candidate: HeadingCandidate) -> str:
|
||||
"""Return a full heading signature including numbering or text."""
|
||||
if len(heading_candidate.numbering) > 0:
|
||||
# Numbering arrays are serialized as comma-joined values, not Python
|
||||
# list representations.
|
||||
return f"{heading_candidate.type}|{','.join(map(str, heading_candidate.numbering))}"
|
||||
heading = f"{heading_candidate.type}|"
|
||||
if heading_candidate.primary_slot is not None:
|
||||
for token in heading_candidate.primary_slot:
|
||||
if is_char_token(token):
|
||||
heading += token.str.lower()
|
||||
return heading
|
||||
|
||||
|
||||
def parent_signature(heading_candidate: HeadingCandidate) -> str:
|
||||
"""Return the signature of the candidate's parent numbering prefix."""
|
||||
secondary_item = f"{heading_candidate.type}|"
|
||||
for candidate_item in range(len(heading_candidate.numbering) - 1):
|
||||
if candidate_item > 0:
|
||||
secondary_item += ","
|
||||
secondary_item += str(heading_candidate.numbering[candidate_item])
|
||||
return secondary_item
|
||||
|
||||
|
||||
def cached_signature(primary_item: "StyleCluster", other_heading_candidate: HeadingCandidate) -> str:
|
||||
"""Cached heading-signature lookup. Keyed by the candidate object itself, not by object id, because addresses can be reused after a discarded object is collected."""
|
||||
candidate_item = primary_item.auxiliary_slot.get(other_heading_candidate)
|
||||
if candidate_item is not None:
|
||||
return candidate_item
|
||||
candidate_item = heading_signature(other_heading_candidate)
|
||||
primary_item.auxiliary_slot[other_heading_candidate] = candidate_item
|
||||
return candidate_item
|
||||
|
||||
|
||||
def is_in_oo_range(primary_item: "StyleCluster", other_heading_candidate: HeadingCandidate) -> bool:
|
||||
"""Return True if the candidate lies within a style cluster's order range."""
|
||||
if primary_item.primary_slot is None or primary_item.tertiary_slot is None:
|
||||
return False
|
||||
return compare_heading_order(other_heading_candidate, primary_item.primary_slot) >= 0 and compare_heading_order(other_heading_candidate, primary_item.tertiary_slot) <= 0
|
||||
|
||||
|
||||
def has_style_neighbor(style: "StyleCluster", other_heading_candidate: HeadingCandidate, candidate_item: float) -> bool:
|
||||
"""Return True if a candidate is close to a compatible style neighbor."""
|
||||
candidate_score = heading_score(other_heading_candidate.group_slot)
|
||||
|
||||
def cmp_target():
|
||||
return {"z": candidate_score, "HeadingCandidate": other_heading_candidate}
|
||||
|
||||
matched = [False]
|
||||
|
||||
def fcheck(item):
|
||||
if abs(candidate_score - heading_score(item.group_slot)) >= candidate_item:
|
||||
return True
|
||||
# Within tolerance, check signature match:
|
||||
measure_item = other_heading_candidate.group_slot
|
||||
line_value = item.group_slot
|
||||
if abs(heading_score(measure_item) - heading_score(line_value)) >= candidate_item:
|
||||
state_item = False
|
||||
elif len(other_heading_candidate.numbering) > 0 and len(item.numbering) > 0:
|
||||
state_item = (other_heading_candidate.type == item.type and len(other_heading_candidate.numbering) == len(item.numbering))
|
||||
elif (len(other_heading_candidate.numbering) <= 0 and len(item.numbering) > 1) or (len(item.numbering) <= 0 and len(other_heading_candidate.numbering) > 1):
|
||||
state_item = False
|
||||
else:
|
||||
block = measure_item.isolated_centered
|
||||
other_centered = line_value.isolated_centered
|
||||
if block or other_centered:
|
||||
state_item = (block == other_centered)
|
||||
elif dominant_style_of(measure_item) == dominant_style_of(line_value):
|
||||
state_item = True
|
||||
else:
|
||||
if first_span_of(measure_item).font_style() != first_span_of(line_value).font_style():
|
||||
state_item = False
|
||||
else:
|
||||
state_item = abs(dominant_font_size(measure_item) - dominant_font_size(line_value)) < candidate_item
|
||||
if state_item:
|
||||
matched[0] = True
|
||||
return True
|
||||
return False
|
||||
|
||||
# Walk sibling candidates in both directions from the candidate's page position
|
||||
if style.secondary_slot is None:
|
||||
return False
|
||||
# SortedKeyList walk
|
||||
target_key = (candidate_score, heading_order_key(other_heading_candidate))
|
||||
idx = style.secondary_slot.bisect_right(other_heading_candidate)
|
||||
for scan_index in range(idx, len(style.secondary_slot)):
|
||||
if fcheck(style.secondary_slot[scan_index]):
|
||||
break
|
||||
if not matched[0]:
|
||||
for scan_index in range(idx - 1, -1, -1):
|
||||
if fcheck(style.secondary_slot[scan_index]):
|
||||
break
|
||||
return matched[0]
|
||||
@@ -0,0 +1,403 @@
|
||||
"""Keyword cliques, clique trees, body-heading detection, and candidate partitioning."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Callable, Optional
|
||||
from ..model import (
|
||||
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
|
||||
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
|
||||
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
|
||||
)
|
||||
from ..stats import style_key as style_key_fn, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
|
||||
from ..tokens import (
|
||||
Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, avg_char_width as avg_char_width_fn, trie_full_match, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Numbering-pattern clique selection.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
# Section-keyword trie shared with outline filtering.
|
||||
from ..outline import SECTION_KEYWORD_TRIE
|
||||
|
||||
from .candidates import (
|
||||
HeadingCandidate,
|
||||
OutlineNode,
|
||||
compare_heading_order,
|
||||
heading_order_key,
|
||||
has_style_neighbor,
|
||||
)
|
||||
from .style_context import (
|
||||
StyleCluster,
|
||||
is_compatible_with_context,
|
||||
OutlineContext,
|
||||
)
|
||||
|
||||
|
||||
def find_keyword_clique(heading_candidates: list[HeadingCandidate]) -> Optional[StyleCluster]:
|
||||
"""Find the largest clique of section-keyword headings sharing a font signature."""
|
||||
buckets: dict[str, StyleCluster] = {}
|
||||
for candidate_item in heading_candidates:
|
||||
if candidate_item.primary_slot is None:
|
||||
continue
|
||||
if not trie_full_match(SECTION_KEYWORD_TRIE, candidate_item.primary_slot):
|
||||
continue
|
||||
first = first_token(candidate_item.primary_slot)
|
||||
if first is None or not first.anchor_ranges:
|
||||
continue
|
||||
font_size = first_anchor_span(first).font_style()
|
||||
style_cluster = buckets.get(font_size)
|
||||
if style_cluster is not None:
|
||||
if style_cluster.has_nearby_duplicate(candidate_item):
|
||||
return None # conflict -> abort
|
||||
if has_style_neighbor(style_cluster, candidate_item, 2.0):
|
||||
style_cluster.add(candidate_item)
|
||||
else:
|
||||
style_cluster = StyleCluster()
|
||||
buckets[font_size] = style_cluster
|
||||
style_cluster.add(candidate_item)
|
||||
winner: Optional[StyleCluster] = None
|
||||
max_size = 0
|
||||
for style_cluster in buckets.values():
|
||||
if style_cluster.size() > max_size:
|
||||
winner = style_cluster
|
||||
max_size = style_cluster.size()
|
||||
if winner is None or max_size <= 1:
|
||||
return None
|
||||
for entry_item in heading_candidates:
|
||||
if winner.contains(entry_item):
|
||||
continue
|
||||
if has_style_neighbor(winner, entry_item, 0.5):
|
||||
winner.add(entry_item)
|
||||
return winner
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Clique-based clusters #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class CliqueTreeNode:
|
||||
"""Tree node used by clique-based heading filtering. Each node holds a heading candidate, parent pointer, child list, and sibling links. ``next`` walks the in-order successor."""
|
||||
|
||||
__slots__ = ("heading", "parent", "primary_slot", "secondary_slot", "tertiary_slot")
|
||||
|
||||
def __init__(self, heading, parent):
|
||||
self.heading = heading
|
||||
self.parent = parent if parent is not None else self
|
||||
self.primary_slot: list = []
|
||||
self.secondary_slot = None
|
||||
self.tertiary_slot = None
|
||||
|
||||
def next(self):
|
||||
if self.primary_slot:
|
||||
return self.primary_slot[0]
|
||||
if self.secondary_slot is not None:
|
||||
return self.secondary_slot
|
||||
return find_ancestor_next_sibling(self.parent)
|
||||
|
||||
|
||||
def find_ancestor_next_sibling(primary_item: CliqueTreeNode):
|
||||
"""walk up parents until we find one with a next sibling."""
|
||||
if primary_item.parent is primary_item:
|
||||
return None
|
||||
return primary_item.secondary_slot or find_ancestor_next_sibling(primary_item.parent)
|
||||
|
||||
|
||||
def descend_to_deepest_last(primary_item: CliqueTreeNode) -> CliqueTreeNode:
|
||||
"""descend to deepest last-child."""
|
||||
while primary_item.primary_slot:
|
||||
primary_item = primary_item.primary_slot[-1]
|
||||
return primary_item
|
||||
|
||||
|
||||
def append_tree_child(ao_tree, parent_node: CliqueTreeNode, heading) -> None:
|
||||
"""Append a new clique-tree child and advance the builder cursor."""
|
||||
new_node = CliqueTreeNode(heading, parent_node)
|
||||
last = parent_node.primary_slot[-1] if parent_node.primary_slot else None
|
||||
if last is not None:
|
||||
last.secondary_slot = new_node
|
||||
new_node.tertiary_slot = last
|
||||
parent_node.primary_slot.append(new_node)
|
||||
ao_tree.primary_slot = new_node
|
||||
|
||||
|
||||
class CliqueTreeBuilder:
|
||||
"""(class at table entry). Builds a clique-tree from a heading list using a comparator. Each heading is placed by walking the cursor up/down based on comparator result. Depth capped at 8. """
|
||||
|
||||
__slots__ = ("root", "primary_slot")
|
||||
|
||||
def __init__(self, headings: list[HeadingCandidate], compare):
|
||||
self.root = CliqueTreeNode(None, None)
|
||||
self.primary_slot = self.root
|
||||
depth = 0
|
||||
for height in headings:
|
||||
while True:
|
||||
if self.primary_slot is self.root:
|
||||
append_tree_child(self, self.primary_slot, height)
|
||||
depth += 1
|
||||
break
|
||||
comparison = compare(self.primary_slot.heading, height)
|
||||
if comparison < 0:
|
||||
self.primary_slot = self.primary_slot.parent
|
||||
depth -= 1
|
||||
else:
|
||||
if comparison > 0 and depth < 8:
|
||||
append_tree_child(self, self.primary_slot, height)
|
||||
depth += 1
|
||||
else:
|
||||
append_tree_child(self, self.primary_slot.parent, height)
|
||||
break
|
||||
|
||||
|
||||
def block_style_signature(block) -> str:
|
||||
"""Return a block-style signature combining dominant style and caps-heavy state."""
|
||||
from ..model import dominant_style_of, is_caps_heavy
|
||||
# The boolean portion is lower-case because the signature is used as an
|
||||
# opaque stable key.
|
||||
return f"{dominant_style_of(block)} {'true' if is_caps_heavy(block) else 'false'}"
|
||||
|
||||
|
||||
def is_member_of_tree(doc, block, target_sig: str, sentence_like: bool, node: CliqueTreeNode) -> bool:
|
||||
"""Return whether the target block is already represented by an ancestor in the candidate tree, using heading signature, body-text weight, and recursive parent traversal."""
|
||||
from ..model import is_sentence_like
|
||||
from ..stats import info_weight
|
||||
if node is None or node.parent is node:
|
||||
return False
|
||||
tree_parent_candidate = node.heading
|
||||
if tree_parent_candidate is None or tree_parent_candidate.type == 5 or tree_parent_candidate.is_prominent:
|
||||
return False
|
||||
if len(tree_parent_candidate.numbering) > 0:
|
||||
return is_member_of_tree(doc, block, target_sig, sentence_like, node.parent)
|
||||
parent_block = tree_parent_candidate.group_slot
|
||||
if target_sig != block_style_signature(parent_block) or (sentence_like and is_sentence_like(parent_block)):
|
||||
return is_member_of_tree(doc, block, target_sig, sentence_like, node.parent)
|
||||
if info_weight(block.char_stats) >= max(100, 4 * info_weight(parent_block.char_stats)):
|
||||
return is_member_of_tree(doc, block, target_sig, sentence_like, node.parent)
|
||||
return True
|
||||
|
||||
|
||||
def can_share_heading_style(heading, other_heading, neighbor_map) -> bool:
|
||||
"""Return whether two blocks can share a heading-style assignment after checking overlap, style signature, neighboring ambiguity, and predecessor consistency."""
|
||||
from ..model import y_overlaps, dominant_style_of
|
||||
from ..heading_detection import neighbor_right, neighbor_above
|
||||
if other_heading is None or not y_overlaps(heading, other_heading) or dominant_style_of(heading) != dominant_style_of(other_heading):
|
||||
return False
|
||||
heading_above = neighbor_above(neighbor_map, heading)
|
||||
other_above = neighbor_above(neighbor_map, other_heading)
|
||||
heading_right = neighbor_right(neighbor_map, heading)
|
||||
other_right = neighbor_right(neighbor_map, other_heading)
|
||||
if (heading_above is not None and heading_above.marker_slot != 0
|
||||
or other_above is not None and other_above.marker_slot != 0
|
||||
or heading_right is not None and heading_right.marker_slot != 0
|
||||
or other_right is not None and other_right.marker_slot != 0):
|
||||
return True
|
||||
if (heading_right is not other_right
|
||||
and (heading_right is not None and heading_right.is_body_paragraph)
|
||||
and (other_right is not None and other_right.is_body_paragraph)):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def compare_block_order(left_value, right_value) -> float:
|
||||
"""Compare blocks or lines by column index first, then reading position."""
|
||||
from ..model import cmp_reading_order
|
||||
from ..stats import column_index_of
|
||||
left_column_index = column_index_of(left_value)
|
||||
right_column_index = column_index_of(right_value)
|
||||
if left_column_index != right_column_index:
|
||||
return left_column_index - right_column_index
|
||||
return cmp_reading_order(left_value, right_value)
|
||||
|
||||
|
||||
def heading_precedes_line(line_heading_candidate: HeadingCandidate, page, line) -> bool:
|
||||
"""Return whether the heading candidate sorts before the given page/line position."""
|
||||
if line_heading_candidate.page.page_index < page.page_index:
|
||||
return True
|
||||
if line_heading_candidate.page.page_index != page.page_index:
|
||||
return False
|
||||
return compare_block_order(line_heading_candidate.group_slot, line) < 0
|
||||
|
||||
|
||||
class CliqueFilterContext:
|
||||
"""State for clique-based body-heading discovery."""
|
||||
|
||||
__slots__ = ("auxiliary_slot", "state_slot", "tertiary_slot", "measure_slot", "secondary_slot", "option_slot", "primary_slot", "candidates", "compare")
|
||||
|
||||
def __init__(self, doc, candidates: list[HeadingCandidate], compare):
|
||||
self.auxiliary_slot = doc
|
||||
self.state_slot: set = set()
|
||||
self.tertiary_slot: dict = {}
|
||||
for reference_item in candidates:
|
||||
self.state_slot.add(reference_item.group_slot)
|
||||
if reference_item.has_numbering:
|
||||
continue
|
||||
if len(reference_item.numbering) > 0:
|
||||
continue
|
||||
sig = block_style_signature(reference_item.group_slot)
|
||||
self.tertiary_slot[sig] = self.tertiary_slot.get(sig, 0) + 1
|
||||
self.measure_slot = CliqueTreeBuilder(candidates, compare)
|
||||
self.secondary_slot = self.measure_slot.root
|
||||
self.option_slot = CliqueTreeBuilder(list(reversed(candidates)), compare)
|
||||
self.primary_slot = self.option_slot.primary_slot
|
||||
self.candidates = candidates
|
||||
self.compare = compare
|
||||
|
||||
|
||||
def detect_body_headings(filter_context: CliqueFilterContext) -> list[HeadingCandidate]:
|
||||
"""Discover body headings by comparing unvisited blocks against clique trees."""
|
||||
from ..model import style_key, dominant_style_of, last_span, last_line_of, first_span_of
|
||||
from ..heading_detection import neighbor_right, neighbor_above, closest_body_neighbor_above, PageNeighborMap as _bo_class, is_cover_page
|
||||
from ..tokens import first_token, tokenize_block
|
||||
|
||||
out: list[HeadingCandidate] = []
|
||||
if not filter_context.candidates:
|
||||
return out
|
||||
|
||||
# Reset cursors to root of forward tree / deepest of reversed tree.
|
||||
filter_context.secondary_slot = filter_context.measure_slot.root
|
||||
filter_context.primary_slot = filter_context.option_slot.primary_slot
|
||||
|
||||
for page in filter_context.auxiliary_slot.primary_slot:
|
||||
if is_cover_page(filter_context.auxiliary_slot, page):
|
||||
continue
|
||||
all_blocks = page.output_slot
|
||||
if len(all_blocks) <= 0:
|
||||
continue
|
||||
neighbor_cache = _bo_class(page)
|
||||
for block in page.secondary_slot:
|
||||
# Advance the forward tree cursor while the next node is before
|
||||
# the current page and block in reading order.
|
||||
while True:
|
||||
next_item = filter_context.secondary_slot.next()
|
||||
if (next_item is None
|
||||
or next_item.heading is None
|
||||
or not heading_precedes_line(next_item.heading, page, block)):
|
||||
break
|
||||
filter_context.secondary_slot = next_item
|
||||
# Advance the reverse tree cursor while the predecessor is before
|
||||
# cursor's heading is still before the current page and block.
|
||||
while filter_context.primary_slot.heading is not None and heading_precedes_line(filter_context.primary_slot.heading, page, block):
|
||||
left_sib = filter_context.primary_slot.tertiary_slot
|
||||
filter_context.primary_slot = descend_to_deepest_last(left_sib) if left_sib is not None else filter_context.primary_slot.parent
|
||||
if filter_context.primary_slot is filter_context.option_slot.root:
|
||||
break
|
||||
|
||||
if block in filter_context.state_slot:
|
||||
continue
|
||||
if filter_context.secondary_slot.heading is None:
|
||||
continue
|
||||
# Body-heading filters.
|
||||
if (block.char_count() <= 0 or block.skew_frac() > 1
|
||||
or (block.char_count() <= 1 and block.char_stats.secondary_slot != 4)
|
||||
or block.line_count() >= 5
|
||||
or block.type != 0
|
||||
or block.marker_slot != 0
|
||||
or (block.char_stats.primary_slot[2] <= 0 and block.char_stats.primary_slot[4] <= 0)):
|
||||
continue
|
||||
if block.measure_slot:
|
||||
continue
|
||||
value = block.bold_frac()
|
||||
if 0.1 < value < 0.9:
|
||||
continue
|
||||
block_style = dominant_style_of(block)
|
||||
first_tok = first_token(tokenize_block(block))
|
||||
# Compare against the dominant style, first span, last token
|
||||
# anchor, and last span. The last anchor matters for wrapped tokens.
|
||||
anchor = first_tok.anchor_ranges[-1].anchor_span if (first_tok is not None and first_tok.anchor_ranges) else None
|
||||
if (block_style != style_key(first_span_of(block))
|
||||
and (anchor is None or block_style != style_key(anchor))
|
||||
and block_style != style_key(last_span(last_line_of(block)))):
|
||||
continue
|
||||
if block_style == page.primary_slot.auxiliary_slot:
|
||||
continue
|
||||
above = neighbor_above(neighbor_cache, block)
|
||||
if (above is not None
|
||||
and above.bottom_edge() - block.top_edge() < 0.3 * block.avg_font_size()
|
||||
and block.line_count() > 1):
|
||||
continue
|
||||
if above is not None and above.type == 3:
|
||||
continue
|
||||
sig = block_style_signature(block)
|
||||
pred_neigh = neighbor_right(neighbor_cache, block)
|
||||
# Reject when the block repeats the style signature of a close
|
||||
# vertical or right-side neighbor.
|
||||
if above is not None and sig == block_style_signature(above):
|
||||
continue
|
||||
if pred_neigh is not None and sig == block_style_signature(pred_neigh):
|
||||
continue
|
||||
previous_block = all_blocks[block.orig_index - 1] if 0 <= block.orig_index - 1 < len(all_blocks) else None
|
||||
next_block = all_blocks[block.orig_index + 1] if 0 <= block.orig_index + 1 < len(all_blocks) else None
|
||||
if can_share_heading_style(block, previous_block, neighbor_cache):
|
||||
continue
|
||||
if can_share_heading_style(block, next_block, neighbor_cache):
|
||||
continue
|
||||
if filter_context.tertiary_slot.get(sig, 0) < 3:
|
||||
continue
|
||||
# Sentence-like flag: enough long lowercase-leading word tokens make
|
||||
# a block look like body text rather than a heading.
|
||||
tok_total = 0
|
||||
tok_g3 = 0
|
||||
for token in tokenize_block(block):
|
||||
if token.type != 2 or len(token.str) < 5:
|
||||
continue
|
||||
tok_total += 1
|
||||
if token.primary_slot == 3:
|
||||
tok_g3 += 1
|
||||
sentence_like = tok_g3 >= max(2, tok_total / 2)
|
||||
# A block must fit either the forward or reverse clique cursor.
|
||||
if not (is_member_of_tree(filter_context, block, sig, sentence_like, filter_context.secondary_slot)
|
||||
or is_member_of_tree(filter_context, block, sig, sentence_like, filter_context.primary_slot)):
|
||||
continue
|
||||
body_heading_candidate = HeadingCandidate(
|
||||
0, page, block,
|
||||
closest_body_neighbor_above(neighbor_cache, block),
|
||||
[], None, tokenize_block(block),
|
||||
False, False,
|
||||
)
|
||||
out.append(body_heading_candidate)
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Partition candidates and interleave clusters #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def partition_candidates(heading_candidates: list[HeadingCandidate], other_outline_nodes: list[OutlineNode]) -> dict:
|
||||
"""Partition candidates into labeled-compatible and remaining groups."""
|
||||
labeled_headings = [entry_item.heading for entry_item in other_outline_nodes]
|
||||
accepted_context = OutlineContext(labeled_headings)
|
||||
remaining: list[HeadingCandidate] = []
|
||||
for entry_item in heading_candidates:
|
||||
if is_compatible_with_context(accepted_context, entry_item):
|
||||
other_outline_nodes.append(OutlineNode(entry_item))
|
||||
accepted_context.add(entry_item)
|
||||
else:
|
||||
remaining.append(entry_item)
|
||||
other_outline_nodes.sort(key=lambda sort_node: heading_order_key(sort_node.heading))
|
||||
return {"remaining": remaining, "labeled": other_outline_nodes}
|
||||
|
||||
|
||||
def interleave_clusters(heading_candidates: list[HeadingCandidate], other_outline_nodes: list[OutlineNode]) -> list[dict]:
|
||||
"""Interleave general candidates between successive labeled headings. Returns clusters with the labeled heading and intervening candidates. """
|
||||
out: list[dict] = []
|
||||
index = 0
|
||||
previous: Optional[OutlineNode] = None
|
||||
acc: list[HeadingCandidate] = []
|
||||
for labeled_outline_node in other_outline_nodes:
|
||||
boundary_candidate = labeled_outline_node.heading
|
||||
while index < len(heading_candidates) and compare_heading_order(heading_candidates[index], boundary_candidate) < 0:
|
||||
acc.append(heading_candidates[index])
|
||||
index += 1
|
||||
if acc or previous is not None:
|
||||
out.append({"labeled_anchor": previous, "cluster_candidates": acc})
|
||||
acc = []
|
||||
previous = labeled_outline_node
|
||||
while index < len(heading_candidates):
|
||||
acc.append(heading_candidates[index])
|
||||
index += 1
|
||||
out.append({"labeled_anchor": previous, "cluster_candidates": acc})
|
||||
return out
|
||||
@@ -0,0 +1,385 @@
|
||||
"""Heading rejection rules, hierarchy stack, and sub/top-level heading extraction."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Any, Callable, Optional
|
||||
from ..model import (
|
||||
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
|
||||
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
|
||||
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
|
||||
)
|
||||
from ..tokens import (
|
||||
Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, avg_char_width as avg_char_width_fn, trie_full_match, first_anchor_span, is_char_token, is_word_token,
|
||||
)
|
||||
|
||||
from .candidates import (
|
||||
HeadingCandidate,
|
||||
OutlineNode,
|
||||
heading_signature,
|
||||
parent_signature,
|
||||
is_in_oo_range,
|
||||
has_style_neighbor,
|
||||
)
|
||||
from .style_context import (
|
||||
StyleCluster,
|
||||
count_sibling_numberings,
|
||||
OutlineState,
|
||||
compare_heading_depth,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# cp / bp -- state mutators (,) #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def min_font_distance(state: OutlineState, other_heading_candidate: HeadingCandidate) -> float:
|
||||
"""minimum font-distance between b and any other heading in the same fontStyle bucket within b's line."""
|
||||
min_value = math.inf
|
||||
line = other_heading_candidate.group_slot.line()
|
||||
for token_list in (other_heading_candidate.secondary_slot, other_heading_candidate.primary_slot):
|
||||
if token_list is None:
|
||||
continue
|
||||
for token in token_list:
|
||||
if token.type != 2:
|
||||
continue
|
||||
for anchor in token.anchor_ranges:
|
||||
if anchor.line is not line:
|
||||
return min_value
|
||||
span = anchor.anchor_span
|
||||
tree = state.state_slot.get(span.font_style())
|
||||
if tree is None:
|
||||
continue
|
||||
for entry in tree:
|
||||
if entry["heading"] is other_heading_candidate:
|
||||
continue
|
||||
diff = abs(span.font_size - entry["size"])
|
||||
if diff < min_value:
|
||||
min_value = diff
|
||||
if diff <= 0:
|
||||
return 0
|
||||
return min_value
|
||||
|
||||
|
||||
def should_reject_heading(state: OutlineState, other_heading_candidate: HeadingCandidate) -> bool:
|
||||
"""should we REJECT heading b given current state? True = reject."""
|
||||
if other_heading_candidate.type == 0:
|
||||
for previous in state.style_slot:
|
||||
if previous is None:
|
||||
continue
|
||||
if compare_heading_depth(previous, other_heading_candidate) == 1:
|
||||
continue
|
||||
style_cluster = state.marker_slot.get(parent_signature(previous))
|
||||
if style_cluster is not None and is_in_oo_range(style_cluster, other_heading_candidate):
|
||||
return True
|
||||
if state.primary_slot is not None and state.primary_slot.is_prominent and other_heading_candidate.type == 0:
|
||||
count = 0
|
||||
for candidate_token in tokenize_block(other_heading_candidate.group_slot):
|
||||
if is_word_token(candidate_token) or candidate_token.type == 1:
|
||||
count += 1
|
||||
if count >= 3:
|
||||
break
|
||||
if count >= 3:
|
||||
return True
|
||||
if (
|
||||
other_heading_candidate.type == 1 and len(other_heading_candidate.numbering) <= 1
|
||||
and (
|
||||
(0 if (other_heading_candidate.type != 1 or len(other_heading_candidate.numbering) <= 0) else count_sibling_numberings(state.cache_slot, other_heading_candidate, 0)) <= 1
|
||||
)
|
||||
):
|
||||
return True
|
||||
if other_heading_candidate.type in (1, 5, 9, 10, 7):
|
||||
reject = False
|
||||
else:
|
||||
distance = min_font_distance(state, other_heading_candidate)
|
||||
if distance <= 0.9:
|
||||
reject = False
|
||||
elif distance >= math.inf:
|
||||
reject = True
|
||||
else:
|
||||
reject = not (is_caps_heavy(other_heading_candidate.group_slot) and other_heading_candidate.tertiary_slot is not None and other_heading_candidate.group_slot.bottom_edge() - other_heading_candidate.tertiary_slot.top_edge() < 5 * other_heading_candidate.group_slot.bbox_height())
|
||||
if reject:
|
||||
return True
|
||||
if other_heading_candidate.type == 1:
|
||||
first = other_heading_candidate.numbering[0]
|
||||
if (first < state.secondary_slot and first < state.tertiary_slot) or (state.secondary_slot > 0 and first > state.secondary_slot + 2):
|
||||
return True
|
||||
if len(other_heading_candidate.numbering) == 1 and state.auxiliary_slot is not None:
|
||||
if first == state.tertiary_slot:
|
||||
return True
|
||||
existing = state.auxiliary_slot.group_slot
|
||||
candidate_style = style_key(first_anchor_span(first_token(other_heading_candidate.primary_slot))) if other_heading_candidate.primary_slot is not None and first_token(other_heading_candidate.primary_slot) is not None else ""
|
||||
state_style = style_key(first_anchor_span(first_token(state.auxiliary_slot.primary_slot))) if state.auxiliary_slot.primary_slot is not None and first_token(state.auxiliary_slot.primary_slot) is not None else ""
|
||||
if candidate_style != state_style:
|
||||
# Bold-fraction comparison uses exact half-up integer rounding;
|
||||
# Python f-string rounding is half-even.
|
||||
if abs(dominant_font_size(other_heading_candidate.group_slot) - dominant_font_size(existing)) > 0.5 or _round_half_up_to_int(other_heading_candidate.group_slot.bold_frac()) != _round_half_up_to_int(existing.bold_frac()):
|
||||
return True
|
||||
if (
|
||||
state.primary_slot is not None
|
||||
and other_heading_candidate.type == 4 and state.primary_slot.type == 4
|
||||
and len(state.primary_slot.numbering) > 0 and len(other_heading_candidate.numbering) > 0
|
||||
and (state.primary_slot.numbering[0] > other_heading_candidate.numbering[0] or (len(other_heading_candidate.numbering) == 1 and state.primary_slot.numbering[0] == other_heading_candidate.numbering[0]))
|
||||
):
|
||||
return True
|
||||
if (state.primary_slot is not None and state.primary_slot.type == 8 and len(other_heading_candidate.numbering) <= 0):
|
||||
from ..model import _strip_diacritics
|
||||
candidate_tokens = other_heading_candidate.primary_slot or []
|
||||
tokens = state.primary_slot.primary_slot or []
|
||||
if len(candidate_tokens) == len(tokens):
|
||||
same = True
|
||||
for heading in range(len(candidate_tokens)):
|
||||
token = candidate_tokens[heading] if heading < len(candidate_tokens) else None
|
||||
state_token = tokens[heading] if heading < len(tokens) else None
|
||||
if token is None or state_token is None:
|
||||
same = False
|
||||
break
|
||||
if _strip_diacritics(token.str.lower()) != _strip_diacritics(state_token.str.lower()):
|
||||
same = False
|
||||
break
|
||||
if same:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def push_heading_to_state(state: OutlineState, other_heading_candidate: HeadingCandidate) -> None:
|
||||
"""Push a heading into the outline state and update level trackers."""
|
||||
if len(other_heading_candidate.numbering) > 0:
|
||||
# Ensure S is long enough
|
||||
while len(state.style_slot) < len(other_heading_candidate.numbering):
|
||||
state.style_slot.append(None)
|
||||
state.style_slot[len(other_heading_candidate.numbering) - 1] = other_heading_candidate
|
||||
if other_heading_candidate.type == 1:
|
||||
first = other_heading_candidate.numbering[0]
|
||||
state.secondary_slot = max(state.secondary_slot, first)
|
||||
state.tertiary_slot = max(state.tertiary_slot, first)
|
||||
if len(other_heading_candidate.numbering) == 1:
|
||||
state.auxiliary_slot = other_heading_candidate
|
||||
elif other_heading_candidate.type in (8, 9):
|
||||
state.tertiary_slot = 0
|
||||
state.primary_slot = other_heading_candidate
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Hierarchy-walk stack #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class HierarchyStack:
|
||||
"""Tree-walk stack of currently open outline nodes."""
|
||||
|
||||
__slots__ = ("auxiliary_slot", "primary_slot", "secondary_slot", "tertiary_slot")
|
||||
|
||||
def __init__(self, anchor):
|
||||
self.auxiliary_slot = anchor
|
||||
self.primary_slot: list[OutlineNode] = []
|
||||
self.secondary_slot = False
|
||||
self.tertiary_slot = False
|
||||
|
||||
def pop(self) -> Optional[OutlineNode]:
|
||||
return self.primary_slot.pop() if self.primary_slot else None
|
||||
|
||||
def push(self, other_outline_node: OutlineNode) -> None:
|
||||
self.primary_slot.append(other_outline_node)
|
||||
self.secondary_slot = self.secondary_slot or other_outline_node.heading.type == 4
|
||||
self.tertiary_slot = self.tertiary_slot or other_outline_node.heading.is_prominent
|
||||
|
||||
|
||||
def find_parent_heading(stack: HierarchyStack, other_heading_candidate: HeadingCandidate) -> Optional[OutlineNode]:
|
||||
"""Pop entries from the stack until a parent for the candidate is found."""
|
||||
heading: Optional[HeadingCandidate] = None
|
||||
while stack.primary_slot:
|
||||
stack_outline_node = stack.primary_slot[-1]
|
||||
state_candidate = stack_outline_node.heading
|
||||
if other_heading_candidate.is_prominent and len(other_heading_candidate.numbering) <= 1 and state_candidate.type != 8:
|
||||
stack.pop()
|
||||
heading = state_candidate
|
||||
continue
|
||||
if state_candidate.is_prominent and other_heading_candidate.type == 5:
|
||||
stack.pop()
|
||||
heading = state_candidate
|
||||
continue
|
||||
cmp = compare_heading_depth(state_candidate, other_heading_candidate, stack.auxiliary_slot)
|
||||
if cmp != -1:
|
||||
if cmp == 1:
|
||||
return stack_outline_node
|
||||
# Appendix and Roman/letter headings can nest under the current
|
||||
# parent only when the numbering sequence remains coherent.
|
||||
if (state_candidate.type != other_heading_candidate.type and other_heading_candidate.type in (4, 2) and not stack.tertiary_slot
|
||||
and is_appendix_nesting_ok(stack, other_heading_candidate, heading)):
|
||||
first_number = other_heading_candidate.numbering[0] if other_heading_candidate.numbering else 0
|
||||
if heading is None:
|
||||
if first_number == 1:
|
||||
return stack_outline_node
|
||||
else:
|
||||
# Empty numbering on the previous heading cannot establish
|
||||
# an increasing appendix sequence.
|
||||
if other_heading_candidate.type == heading.type and other_heading_candidate.numbering and heading.numbering and first_number > heading.numbering[0]:
|
||||
return stack_outline_node
|
||||
stack.pop()
|
||||
heading = state_candidate
|
||||
return None
|
||||
|
||||
|
||||
def is_appendix_nesting_ok(stack: HierarchyStack, other_heading_candidate: HeadingCandidate, candidate_heading_candidate: Optional[HeadingCandidate]) -> bool:
|
||||
"""Return whether an appendix candidate may be nested under the current stack state. Non-appendix headings always pass; appendix headings pass when the stack is already in appendix mode, has no numbering context, or starts at appendix depth 1..3."""
|
||||
if other_heading_candidate.type != 4:
|
||||
return True
|
||||
if stack.secondary_slot:
|
||||
return True
|
||||
# Last heading info
|
||||
if not stack.primary_slot:
|
||||
return True
|
||||
entry_item = stack.primary_slot[-1].heading
|
||||
if len(entry_item.numbering) <= 0:
|
||||
return True
|
||||
return entry_item.numbering[0] <= 3
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Sub-headings within a cluster #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def extract_sub_headings(doc, state: OutlineState, parent_node: Optional[OutlineNode], cluster_candidates: list[HeadingCandidate]) -> list[OutlineNode]:
|
||||
"""Walk a cluster's candidate list and emit subheadings. The input list is consumed in place so later passes do not reprocess headings already assigned to this cluster."""
|
||||
if not cluster_candidates:
|
||||
return []
|
||||
# Content cap: walk from the parent page to the first candidate page and
|
||||
# abort the cluster if accumulated body-block text exceeds 1000.
|
||||
from ..stats import info_weight as _info_weight
|
||||
first = cluster_candidates[0]
|
||||
page_index = (parent_node.heading.page.page_index - 1) if parent_node is not None else 0
|
||||
acc = 0
|
||||
end_pg = min(first.page.page_index, len(doc.primary_slot))
|
||||
while page_index < end_pg:
|
||||
heading_page = doc.primary_slot[page_index]
|
||||
if getattr(heading_page, "state_slot", False):
|
||||
for block in heading_page.output_slot:
|
||||
if page_index >= first.page.page_index - 1 and block.reading_order_index >= first.group_slot.reading_order_index:
|
||||
break
|
||||
if getattr(block, "is_body_paragraph", None):
|
||||
acc += _info_weight(block.char_stats)
|
||||
if acc >= 1000:
|
||||
return []
|
||||
page_index += 1
|
||||
out: list[OutlineNode] = []
|
||||
parent_anchor = parent_node if (parent_node is not None and parent_node.heading.type == 5) else None
|
||||
seen_signatures: set[str] = set()
|
||||
style_cluster = StyleCluster()
|
||||
saw_numbered = False
|
||||
index = 0
|
||||
while index < len(cluster_candidates):
|
||||
cluster_candidate = cluster_candidates[index]
|
||||
if not (
|
||||
cluster_candidate.type == 5
|
||||
or cluster_candidate.type == 6
|
||||
or (cluster_candidate.type == 11 and cluster_candidate.has_numbering and parent_anchor is not None and index <= 1)
|
||||
):
|
||||
next_item = cluster_candidates[index + 1] if index + 1 < len(cluster_candidates) else None
|
||||
if next_item and next_item.type == 5 and next_item.page is cluster_candidate.page and next_item.tertiary_slot is cluster_candidate.tertiary_slot:
|
||||
index += 1
|
||||
continue
|
||||
break
|
||||
candidate_signature = heading_signature(cluster_candidate)
|
||||
if candidate_signature in seen_signatures:
|
||||
index += 1
|
||||
continue
|
||||
if should_reject_heading(state, cluster_candidate):
|
||||
index += 1
|
||||
continue
|
||||
push_heading_to_state(state, cluster_candidate)
|
||||
seen_signatures.add(candidate_signature)
|
||||
if cluster_candidate.has_numbering:
|
||||
saw_numbered = True
|
||||
elif saw_numbered:
|
||||
break
|
||||
if parent_anchor is None:
|
||||
parent_anchor = OutlineNode(cluster_candidate)
|
||||
out.append(parent_anchor)
|
||||
style_cluster.add(cluster_candidate)
|
||||
index += 1
|
||||
continue
|
||||
anchor_heading_candidate = parent_anchor.heading
|
||||
if cluster_candidate.page.page_index > anchor_heading_candidate.page.page_index:
|
||||
break
|
||||
cmp = compare_heading_depth(anchor_heading_candidate, cluster_candidate)
|
||||
if cmp != 1:
|
||||
if not has_style_neighbor(style_cluster, cluster_candidate, 1.0):
|
||||
break
|
||||
parent_anchor = OutlineNode(cluster_candidate)
|
||||
out.append(parent_anchor)
|
||||
style_cluster.add(cluster_candidate)
|
||||
index += 1
|
||||
# Remove processed items so the outline loop does not reprocess them.
|
||||
del cluster_candidates[:index]
|
||||
if (
|
||||
len(out) >= 3
|
||||
or (len(out) == 2 and out[0].heading.has_numbering and out[1].heading.has_numbering)
|
||||
) and out[0].heading.type != 5:
|
||||
return []
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Flatten outline to top-level headings #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def extract_top_level_headings(item_list: list[OutlineNode]) -> list[OutlineNode]:
|
||||
"""Walk the outline and emit top-level prominent headings."""
|
||||
out: list[OutlineNode] = []
|
||||
saw_prominent = False
|
||||
for heading in item_list:
|
||||
if heading.heading.is_prominent:
|
||||
if not saw_prominent:
|
||||
out.append(heading)
|
||||
saw_prominent = True
|
||||
else:
|
||||
saw_prominent = False
|
||||
out.extend(extract_top_level_headings(heading.child_nodes))
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Outline validation.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def is_outline_valid(doc, item_list: list[OutlineNode]) -> bool:
|
||||
"""Return True when top-level headings span a meaningful fraction of the document."""
|
||||
top = extract_top_level_headings(item_list)
|
||||
if len(top) < 3:
|
||||
return False
|
||||
if len(top) >= 5:
|
||||
return True
|
||||
last_page = 1
|
||||
for top_node in top:
|
||||
line = top_node.heading.page.page_index
|
||||
if line - last_page > 0.5 * len(doc.primary_slot):
|
||||
return False
|
||||
last_page = line
|
||||
return True
|
||||
|
||||
|
||||
def is_chapter_outline_valid(doc, item_list: list[OutlineNode]) -> bool:
|
||||
"""Secondary validity check based on chapter count and inter-chapter span."""
|
||||
chapters = 0
|
||||
span = 0
|
||||
previous = -1
|
||||
for chapter_outline_node in item_list:
|
||||
chapter_page = chapter_outline_node.heading.page.page_index
|
||||
if previous >= 0:
|
||||
span += chapter_page - previous
|
||||
previous = -1
|
||||
if chapter_outline_node.heading.type == 8:
|
||||
chapters += 1
|
||||
previous = chapter_page
|
||||
if previous >= 0:
|
||||
span += len(doc.primary_slot) - previous + 1
|
||||
return (
|
||||
chapters >= 3
|
||||
and span >= 0.7 * len(doc.primary_slot)
|
||||
and span / max(1, chapters) < 100
|
||||
)
|
||||
@@ -0,0 +1,365 @@
|
||||
"""Style clusters, outline context/state, numbering trie, and depth comparison."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Any, Callable, Optional
|
||||
|
||||
from sortedcontainers import SortedKeyList
|
||||
from ..model import (
|
||||
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
|
||||
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
|
||||
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
|
||||
)
|
||||
|
||||
from .candidates import (
|
||||
HeadingCandidate,
|
||||
compare_heading_order,
|
||||
heading_order_key,
|
||||
parent_signature,
|
||||
cached_signature,
|
||||
is_in_oo_range,
|
||||
has_style_neighbor,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Font/style-clustered heading group #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class StyleCluster:
|
||||
"""Group of headings sharing a font/style signature."""
|
||||
|
||||
__slots__ = ("auxiliary_slot", "state_slot", "secondary_slot", "primary_slot", "tertiary_slot")
|
||||
|
||||
def __init__(self):
|
||||
self.auxiliary_slot: dict[HeadingCandidate, str] = {} # candidate -> signature
|
||||
self.state_slot: dict[str, HeadingCandidate] = {} # signature -> candidate
|
||||
self.secondary_slot: SortedKeyList = SortedKeyList(
|
||||
key=lambda sort_node: (heading_score(sort_node.group_slot), heading_order_key(sort_node))
|
||||
)
|
||||
self.primary_slot: Optional[HeadingCandidate] = None # min by heading order
|
||||
self.tertiary_slot: Optional[HeadingCandidate] = None # max by heading order
|
||||
|
||||
def size(self) -> int:
|
||||
return len(self.secondary_slot)
|
||||
|
||||
def contains(self, other_heading_candidate: HeadingCandidate) -> bool:
|
||||
# Containment is key-based, not object identity. SortedKeyList's ``in``
|
||||
# tests identity among equal-key elements, so compare the sort keys.
|
||||
idx = self.secondary_slot.bisect_left(other_heading_candidate)
|
||||
return idx < len(self.secondary_slot) and self.secondary_slot.key(self.secondary_slot[idx]) == self.secondary_slot.key(other_heading_candidate)
|
||||
|
||||
def add(self, other_heading_candidate: HeadingCandidate) -> None:
|
||||
self.state_slot[cached_signature(self, other_heading_candidate)] = other_heading_candidate
|
||||
# Keep set semantics over the sort key: equal-key elements are dropped,
|
||||
# while the signature and min/max heading-order state still update.
|
||||
idx = self.secondary_slot.bisect_left(other_heading_candidate)
|
||||
if idx >= len(self.secondary_slot) or self.secondary_slot.key(self.secondary_slot[idx]) != self.secondary_slot.key(other_heading_candidate):
|
||||
self.secondary_slot.add(other_heading_candidate)
|
||||
if self.primary_slot is None or compare_heading_order(other_heading_candidate, self.primary_slot) < 0:
|
||||
self.primary_slot = other_heading_candidate
|
||||
if self.tertiary_slot is None or compare_heading_order(other_heading_candidate, self.tertiary_slot) > 0:
|
||||
self.tertiary_slot = other_heading_candidate
|
||||
|
||||
def has_nearby_duplicate(self, other_heading_candidate: HeadingCandidate) -> bool:
|
||||
""""have we seen a nearby matching signature within +/- 20 pages?"."""
|
||||
existing = self.state_slot.get(cached_signature(self, other_heading_candidate))
|
||||
return existing is not None and abs(other_heading_candidate.page.page_index - existing.page.page_index) < 20
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Outline-context style-bucket operations #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def pick_style_bucket(outline_context: "OutlineContext", other_heading_candidate: HeadingCandidate) -> StyleCluster:
|
||||
"""Pick the right style bucket for a candidate."""
|
||||
if other_heading_candidate.type == 10:
|
||||
return outline_context.secondary_slot
|
||||
if other_heading_candidate.type == 8:
|
||||
return outline_context.primary_slot
|
||||
if len(other_heading_candidate.numbering) > 0:
|
||||
return outline_context.auxiliary_slot
|
||||
return outline_context.tertiary_slot
|
||||
|
||||
|
||||
def has_conflict_in_context(outline_context: "OutlineContext", other_heading_candidate: HeadingCandidate) -> bool:
|
||||
"""Return True if a candidate conflicts with the existing outline context."""
|
||||
if other_heading_candidate.type != 10 and is_in_oo_range(outline_context.secondary_slot, other_heading_candidate):
|
||||
return True
|
||||
if other_heading_candidate.type != 8 and is_in_oo_range(outline_context.primary_slot, other_heading_candidate):
|
||||
return True
|
||||
if len(other_heading_candidate.numbering) <= 0 and is_in_oo_range(outline_context.auxiliary_slot, other_heading_candidate):
|
||||
return True
|
||||
if other_heading_candidate.type != 8 and len(other_heading_candidate.numbering) > 0 and outline_context.primary_slot.size() > 0:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def is_compatible_with_context(outline_context: "OutlineContext", other_heading_candidate: HeadingCandidate) -> bool:
|
||||
"""Return True iff a candidate can be added to the outline context."""
|
||||
if has_conflict_in_context(outline_context, other_heading_candidate):
|
||||
return False
|
||||
# Find the nearest predecessor by heading order.
|
||||
text: Optional[HeadingCandidate] = None
|
||||
for item in outline_context.state_slot:
|
||||
if compare_heading_order(item, other_heading_candidate) <= 0:
|
||||
if text is None or compare_heading_order(item, text) > 0:
|
||||
text = item
|
||||
else:
|
||||
break
|
||||
if text is not None:
|
||||
candidate_block = other_heading_candidate.group_slot
|
||||
previous_block = text.group_slot
|
||||
if previous_block.isolated_centered and not candidate_block.isolated_centered:
|
||||
return False
|
||||
if not other_heading_candidate.is_prominent and heading_score(previous_block) > heading_score(candidate_block) + 0.5:
|
||||
return False
|
||||
if text.is_prominent and not other_heading_candidate.is_prominent and heading_score(previous_block) > heading_score(candidate_block) - 0.5:
|
||||
return False
|
||||
style_cluster = pick_style_bucket(outline_context, other_heading_candidate)
|
||||
if not style_cluster.has_nearby_duplicate(other_heading_candidate) and has_style_neighbor(style_cluster, other_heading_candidate, 1.0):
|
||||
return True
|
||||
if len(other_heading_candidate.numbering) == 1 and has_style_neighbor(outline_context.tertiary_slot, other_heading_candidate, 1.0):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Outline-context group.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class OutlineContext:
|
||||
"""Bundles style clusters for chapter, appendix, numbered, and general headings."""
|
||||
|
||||
__slots__ = ("secondary_slot", "primary_slot", "auxiliary_slot", "tertiary_slot", "state_slot")
|
||||
|
||||
def __init__(self, headings: list[HeadingCandidate]):
|
||||
self.secondary_slot = StyleCluster() # type == 10
|
||||
|
||||
self.primary_slot = StyleCluster() # type == 8
|
||||
|
||||
self.auxiliary_slot = StyleCluster() # has M (numbered)
|
||||
self.tertiary_slot = StyleCluster() # everything else
|
||||
# The ordered heading list is set-like by heading-order key, with the
|
||||
# first equal-key candidate retained.
|
||||
self.state_slot: list[HeadingCandidate] = []
|
||||
seen_keys: set = set()
|
||||
for secondary_item in headings:
|
||||
self.add(secondary_item)
|
||||
key_value = heading_order_key(secondary_item)
|
||||
if key_value not in seen_keys:
|
||||
seen_keys.add(key_value)
|
||||
self.state_slot.append(secondary_item)
|
||||
self.state_slot.sort(key=heading_order_key)
|
||||
|
||||
def add(self, other_heading_candidate: HeadingCandidate) -> None:
|
||||
pick_style_bucket(self, other_heading_candidate).add(other_heading_candidate)
|
||||
|
||||
def has_nearby_duplicate(self, other_heading_candidate: HeadingCandidate) -> bool:
|
||||
return pick_style_bucket(self, other_heading_candidate).has_nearby_duplicate(other_heading_candidate)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Numbering-prefix tree.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class NumberingTrie:
|
||||
"""a recursive map for numbering prefixes."""
|
||||
|
||||
__slots__ = ("primary_slot", "secondary_slot")
|
||||
|
||||
def __init__(self):
|
||||
self.primary_slot: dict[int, "NumberingTrie"] = {}
|
||||
self.secondary_slot = 0
|
||||
|
||||
|
||||
def insert_numbering(primary_item: NumberingTrie, other_heading_candidate: HeadingCandidate, index: int) -> None:
|
||||
"""Insert the candidate numbering suffix into the trie."""
|
||||
if index == len(other_heading_candidate.numbering):
|
||||
primary_item.secondary_slot += 1
|
||||
return
|
||||
reference_item = primary_item.primary_slot.get(other_heading_candidate.numbering[index])
|
||||
if reference_item is None:
|
||||
reference_item = NumberingTrie()
|
||||
primary_item.primary_slot[other_heading_candidate.numbering[index]] = reference_item
|
||||
insert_numbering(reference_item, other_heading_candidate, index + 1)
|
||||
|
||||
|
||||
def count_sibling_numberings(primary_item: NumberingTrie, other_heading_candidate: HeadingCandidate, index: int) -> int:
|
||||
"""Count sibling numbering branches at the target depth."""
|
||||
if index >= len(other_heading_candidate.numbering) - 1:
|
||||
count = 0
|
||||
for reference_item in primary_item.primary_slot.values():
|
||||
if reference_item.secondary_slot > 0:
|
||||
count += 1
|
||||
return count
|
||||
reference_item = primary_item.primary_slot.get(other_heading_candidate.numbering[index])
|
||||
if reference_item is None:
|
||||
return 0
|
||||
return count_sibling_numberings(reference_item, other_heading_candidate, index + 1)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Global outline state.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class OutlineState:
|
||||
"""Global state for outline assembly walks."""
|
||||
|
||||
__slots__ = ("state_slot", "cache_slot", "marker_slot", "previous_slot", "option_slot", "measure_slot", "style_slot", "secondary_slot", "auxiliary_slot", "tertiary_slot", "primary_slot")
|
||||
|
||||
def __init__(self, clusters: list[dict]):
|
||||
self.state_slot: dict = {}
|
||||
self.cache_slot = NumberingTrie()
|
||||
self.marker_slot: dict[str, StyleCluster] = {}
|
||||
self.previous_slot = math.inf
|
||||
self.option_slot = -math.inf
|
||||
self.measure_slot = False
|
||||
clusters_by_parent_signature: dict[str, list[StyleCluster]] = {}
|
||||
for cluster in clusters:
|
||||
heading_branch = cluster.get("labeled_anchor")
|
||||
cluster_candidates = cluster.get("cluster_candidates", [])
|
||||
if heading_branch is not None:
|
||||
_apply_heading_to_state(self, heading_branch.heading)
|
||||
for state_candidate in cluster_candidates:
|
||||
_apply_heading_to_state(self, state_candidate)
|
||||
if len(state_candidate.numbering) <= 0:
|
||||
continue
|
||||
key = parent_signature(state_candidate)
|
||||
item_list = clusters_by_parent_signature.get(key)
|
||||
if item_list is None:
|
||||
item_list = []
|
||||
clusters_by_parent_signature[key] = item_list
|
||||
placed = None
|
||||
for style_cluster in item_list:
|
||||
if not style_cluster.has_nearby_duplicate(state_candidate) and has_style_neighbor(style_cluster, state_candidate, 1.0):
|
||||
placed = style_cluster
|
||||
break
|
||||
if placed is None and len(item_list) < 3:
|
||||
placed = StyleCluster()
|
||||
item_list.append(placed)
|
||||
if placed is not None:
|
||||
placed.add(state_candidate)
|
||||
for key, group in clusters_by_parent_signature.items():
|
||||
group.sort(key=lambda bucket_group: -bucket_group.size())
|
||||
best = group[0]
|
||||
if best.size() <= 2:
|
||||
continue
|
||||
self.marker_slot[key] = best
|
||||
self.style_slot: list[Optional[HeadingCandidate]] = []
|
||||
self.secondary_slot = 0
|
||||
self.auxiliary_slot: Optional[HeadingCandidate] = None
|
||||
self.tertiary_slot = 0
|
||||
self.primary_slot: Optional[HeadingCandidate] = None
|
||||
|
||||
|
||||
def _apply_heading_to_state(state: OutlineState, other_heading_candidate: HeadingCandidate) -> None:
|
||||
"""Add a heading to the per-font tree and update document-level outline state."""
|
||||
seen: set[str] = set()
|
||||
line = other_heading_candidate.group_slot.line()
|
||||
for token_list in (other_heading_candidate.secondary_slot, other_heading_candidate.primary_slot):
|
||||
if token_list is None:
|
||||
continue
|
||||
for token in token_list:
|
||||
if token.line() is not line:
|
||||
break
|
||||
if token.type != 2:
|
||||
continue
|
||||
for anchor in token.anchor_ranges:
|
||||
if anchor.line is not line:
|
||||
break
|
||||
span = anchor.anchor_span
|
||||
style = style_key(span)
|
||||
if style in seen:
|
||||
continue
|
||||
seen.add(style)
|
||||
font_size = span.font_style()
|
||||
tree = state.state_slot.get(font_size)
|
||||
if tree is None:
|
||||
tree = SortedKeyList(
|
||||
key=lambda heading: (heading["size"], heading_order_key(heading["heading"]))
|
||||
)
|
||||
state.state_slot[font_size] = tree
|
||||
# Each per-font-size bucket is set-like by (size, heading-order)
|
||||
# key, retaining the first equal-key entry.
|
||||
entry = {"size": span.font_size, "heading": other_heading_candidate}
|
||||
idx = tree.bisect_left(entry)
|
||||
if idx >= len(tree) or tree.key(tree[idx]) != tree.key(entry):
|
||||
tree.add(entry)
|
||||
if other_heading_candidate.type == 1 and len(other_heading_candidate.numbering) > 0:
|
||||
insert_numbering(state.cache_slot, other_heading_candidate, 0)
|
||||
state.previous_slot = min(state.previous_slot, other_heading_candidate.page.page_index)
|
||||
state.option_slot = max(state.option_slot, other_heading_candidate.page.page_index)
|
||||
if not state.measure_slot:
|
||||
state.measure_slot = other_heading_candidate.is_prominent
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Pairwise heading-depth comparator #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def compare_heading_depth(heading_candidate: HeadingCandidate, other_heading_candidate: HeadingCandidate, clique: Optional[StyleCluster] = None) -> int:
|
||||
"""Compare two heading candidates for relative nesting depth. Returns ``-1`` when the first candidate should be shallower, ``1`` when it should be deeper, and ``0`` when both candidates should share a level. The decision combines special heading types, numbering depth, structural numbering, style prominence, centered layout, clique membership, and bold weight. """
|
||||
special = heading_candidate.type in (8, 9, 10)
|
||||
other_special = other_heading_candidate.type in (8, 9, 10)
|
||||
if special and other_special:
|
||||
return 0
|
||||
if special or other_special:
|
||||
return 1 if special else -1
|
||||
if (heading_candidate.type == 1 and other_heading_candidate.type == 1) or (heading_candidate.type == 4 and other_heading_candidate.type == 4):
|
||||
left_length = len(heading_candidate.numbering)
|
||||
right_length = len(other_heading_candidate.numbering)
|
||||
if left_length == right_length:
|
||||
return 0
|
||||
return 1 if left_length < right_length else -1
|
||||
if heading_candidate.type == 2 and other_heading_candidate.type == 2:
|
||||
return 0
|
||||
if heading_candidate.type == 11 and len(other_heading_candidate.numbering) == 1:
|
||||
return -1
|
||||
heading_block = heading_candidate.group_slot
|
||||
block = other_heading_candidate.group_slot
|
||||
if heading_candidate.has_numbering != other_heading_candidate.has_numbering:
|
||||
return -1 if heading_candidate.has_numbering else 1
|
||||
if heading_candidate.has_numbering and other_heading_candidate.has_numbering and abs(first_span_of(heading_block).font_size - first_span_of(block).font_size) < 0.9:
|
||||
return 0
|
||||
score = heading_score(heading_block)
|
||||
other_score = heading_score(block)
|
||||
heading_in_clique = clique is not None and clique.contains(heading_candidate)
|
||||
in_value = clique is not None and clique.contains(other_heading_candidate)
|
||||
# Z-based major-gap return
|
||||
if abs(score - other_score) > 1.9 or (abs(score - other_score) > 0.9 and (not heading_in_clique or not in_value)):
|
||||
return 1 if score > other_score else -1
|
||||
# uppercase-dominant comparison
|
||||
heading_caps_heavy = is_caps_heavy(heading_block)
|
||||
caps_heavy = is_caps_heavy(block)
|
||||
if heading_caps_heavy != caps_heavy:
|
||||
return 1 if heading_caps_heavy else -1
|
||||
# paragraph-end / isolated-centered comparison
|
||||
centered_flag = heading_block.isolated_centered
|
||||
if centered_flag != block.isolated_centered:
|
||||
return 1 if centered_flag else -1
|
||||
# skew (rotation) comparison -- skipped if both type 5
|
||||
if not (heading_candidate.type == 5 and other_heading_candidate.type == 5):
|
||||
heading_skewed = heading_block.previous_slot > 0.99
|
||||
skew = block.previous_slot > 0.99
|
||||
if heading_skewed != skew:
|
||||
return -1 if heading_skewed else 1
|
||||
# uppercase or both in clique -> tied
|
||||
if heading_caps_heavy or (heading_in_clique and in_value):
|
||||
return 0
|
||||
# clique containment asymmetric
|
||||
if (heading_in_clique and other_heading_candidate.type != 4) or (in_value and heading_candidate.type != 4):
|
||||
return 1 if heading_in_clique else -1
|
||||
# Bold comparison.
|
||||
bold = heading_block.bold_frac() > 0.5
|
||||
other_bold = block.bold_frac() > 0.5
|
||||
if bold != other_bold:
|
||||
return 1 if bold else -1
|
||||
return 0
|
||||
@@ -0,0 +1,171 @@
|
||||
"""PDFium-backed text-item reconstruction via textpage chars and bbox-mapped font handles.
|
||||
|
||||
The parser reconstructs content-stream text items from rendered characters while
|
||||
preserving the geometry needed by downstream line clustering and heading
|
||||
detection. The merge thresholds operate on glyph advance, font size, text
|
||||
matrix scale, and spacing introduced by char spacing, text-position operators,
|
||||
and ``TJ`` adjustments.
|
||||
|
||||
Per page, the reconstruction uses rendered character origins, glyph widths,
|
||||
font bbox containment, effective font size, text-item merging, baseline-anchored
|
||||
character boxes, and the minimum font size derived in each emitted chunk. Those
|
||||
calibrations keep small caps, math glyphs, ligatures, Type 3 fonts, rotated
|
||||
text, and vertical writing stable enough for layout statistics.
|
||||
"""
|
||||
|
||||
import bisect
|
||||
import ctypes
|
||||
import difflib
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
import unicodedata
|
||||
from collections import Counter
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from typing import Union
|
||||
|
||||
import pypdfium2 as pdfium
|
||||
import pypdfium2.raw as pdfium_c
|
||||
|
||||
# Raw PDF object access (ToUnicode CMaps, content streams, font dicts, /WMode)
|
||||
# that PDFium does not expose, read via PyPDF2 -- already a project dependency and
|
||||
# permissively licensed. A thin adapter exposes the small raw-object API the
|
||||
# helpers below need, so their calibrated logic stays unchanged.
|
||||
import PyPDF2 as _pypdf2 # declared dependency (also imported by pageindex.utils/client)
|
||||
from PyPDF2.generic import (
|
||||
IndirectObject as PdfIndirectRef, NameObject as PdfName, NumberObject as PdfNumber,
|
||||
FloatObject as PdfFloat, BooleanObject as PdfBoolean,
|
||||
DictionaryObject as PdfDictionary, ArrayObject as PdfArray,
|
||||
)
|
||||
|
||||
from ..model import Span, Rect
|
||||
|
||||
from .pdf_objects import (
|
||||
_pdf_tok,
|
||||
_pdf_obj_str,
|
||||
_pdf_typed,
|
||||
_PdfPage,
|
||||
_PdfDoc,
|
||||
_PDF_WHITESPACE_BYTES,
|
||||
_PDF_DELIMITER_BYTES,
|
||||
_PDF_STRING_ESCAPE_BYTES,
|
||||
_decode_pdf_name,
|
||||
)
|
||||
from .text_normalize import (
|
||||
_DROP_CHARS,
|
||||
_NORMALIZED_UNICODES,
|
||||
_normalize_unicodes,
|
||||
TRACKING_SPACE_FACTOR,
|
||||
NON_SPACE_GAP_FACTOR,
|
||||
NEGATIVE_SPACE_FACTOR,
|
||||
SPACE_IN_FLOW_MIN_FACTOR,
|
||||
SPACE_IN_FLOW_MAX_FACTOR,
|
||||
_WHITESPACE_CODEPOINTS,
|
||||
_is_whitespace,
|
||||
_is_zero_width_diacritic,
|
||||
_is_invisible_format_mark,
|
||||
_BIDI_BASE_TYPES,
|
||||
_BIDI_ARABIC_TYPES,
|
||||
_apply_bidi_reordering,
|
||||
_rtl_sign,
|
||||
_reverse_if_rtl,
|
||||
_read_end,
|
||||
_read_gap,
|
||||
)
|
||||
from .content_stream import (
|
||||
_FLUSH_OPS,
|
||||
_SHOW_OPS,
|
||||
_OP_LEX_PREFIX,
|
||||
_OP_OPERAND_COUNTS,
|
||||
_tokenize_show_operators,
|
||||
_assign_vertical_tags,
|
||||
_assign_show_tz,
|
||||
_page_vertical_resource_names,
|
||||
)
|
||||
from .glyph_tables import (
|
||||
_GLYPHLIST_PATH,
|
||||
_cached_glyphs,
|
||||
_cached_encodings,
|
||||
_load_glyph_tables,
|
||||
_get_unicode_for_glyph,
|
||||
_from_char_code,
|
||||
)
|
||||
from .cmap_parse import (
|
||||
_utf16be_units_to_str,
|
||||
_NUM_DECIMAL_RE,
|
||||
_NUM_INFINITY_RE,
|
||||
_NUM_HEX_RE,
|
||||
_NUM_OCTAL_RE,
|
||||
_NUM_BINARY_RE,
|
||||
_WHITESPACE_STRIP,
|
||||
_ieee_div,
|
||||
_compute_skew,
|
||||
_to_number,
|
||||
_parse_int,
|
||||
_cmap_str_to_int,
|
||||
_parse_tounicode_cmap,
|
||||
)
|
||||
from .font_unicode import (
|
||||
_TYPE1_SPECIAL_BYTES,
|
||||
_TYPE1_WHITESPACE_BYTES,
|
||||
_type1_builtin_encoding,
|
||||
_simple_font_to_unicode,
|
||||
_font_unicode_map,
|
||||
)
|
||||
from .code_walk import (
|
||||
_resource_dict_xrefs,
|
||||
_page_show_codes,
|
||||
_char_category,
|
||||
_walk_codes,
|
||||
)
|
||||
from .unicode_apply import (
|
||||
_apply_font_unicode,
|
||||
_synthesize_dropped_glyphs,
|
||||
)
|
||||
from .geometry import (
|
||||
_obj_rotation,
|
||||
_xf_point,
|
||||
_compose_mtx,
|
||||
_IDENT_MTX,
|
||||
_collect_text_objs,
|
||||
_build_obj_index,
|
||||
_char_render_fs,
|
||||
_find_obj_for_char,
|
||||
)
|
||||
from .char_extract import (
|
||||
_extract_raw_chars,
|
||||
_accumulate_type3_extents,
|
||||
_type3_size_by_font,
|
||||
_apply_type3_sizes,
|
||||
_finalize_chars,
|
||||
_inherited_box,
|
||||
_page_view_rect,
|
||||
_off_page,
|
||||
)
|
||||
from .merge import _merge_text_items
|
||||
from .remerge import (
|
||||
_start_rot_span,
|
||||
_grow_rot_span,
|
||||
_merge_rotated_one,
|
||||
_remerge_rotated,
|
||||
_new_oblique_span,
|
||||
_close_oblique,
|
||||
_oblique_space,
|
||||
_merge_oblique_one,
|
||||
_remerge_oblique,
|
||||
_start_vert_span,
|
||||
_close_vert_span,
|
||||
_merge_vertical_one,
|
||||
_grow_vert_span,
|
||||
_remerge_vertical,
|
||||
)
|
||||
from .pipeline import (
|
||||
_page_pass1,
|
||||
_page_pass2,
|
||||
_page_spans,
|
||||
parse_charlevel_meta,
|
||||
parse_charlevel,
|
||||
)
|
||||
|
||||
__all__ = ["parse_charlevel", "parse_charlevel_meta"]
|
||||
@@ -0,0 +1,393 @@
|
||||
"""Raw textpage char extraction, Type3 sizing, and page viewport handling."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ctypes
|
||||
import pypdfium2.raw as pdfium_c
|
||||
|
||||
from .text_normalize import (
|
||||
_is_whitespace,
|
||||
_is_zero_width_diacritic,
|
||||
_is_invisible_format_mark,
|
||||
)
|
||||
from .geometry import (
|
||||
_collect_text_objs,
|
||||
_build_obj_index,
|
||||
_find_obj_for_char,
|
||||
)
|
||||
|
||||
|
||||
def _extract_raw_chars(page, text_page) -> tuple[list[dict], list[dict]]:
|
||||
"""First pass: walk textpage chars and attach font info via the bbox-containing text-object lookup. Returns ``(raw_chars, objects)``; glyph widths and the identity-matrix Type-3 size override are applied later, after document-wide Type-3 extents are known."""
|
||||
objects = _collect_text_objs(page, text_page)
|
||||
if not objects:
|
||||
return [], []
|
||||
obj_index = _build_obj_index(objects)
|
||||
|
||||
# First pass: collect raw textpage chars with their host obj.
|
||||
count_item = pdfium_c.FPDFText_CountChars(text_page)
|
||||
font_name_buffer = (ctypes.c_char * 256)()
|
||||
flags = ctypes.c_int(0)
|
||||
field = ctypes.c_float(0)
|
||||
# Per-char FFI out-buffers and entry points, hoisted: each is overwritten
|
||||
# by its call (buffers whose call result is unchecked are re-zeroed below,
|
||||
# so a failed call reads back 0 exactly as a fresh buffer would).
|
||||
char_origin_x = ctypes.c_double(0); char_origin_y = ctypes.c_double(0)
|
||||
char_left_box = ctypes.c_double(0); char_right_box = ctypes.c_double(0)
|
||||
value = ctypes.c_double(0); char_top_box = ctypes.c_double(0)
|
||||
loose_box = pdfium_c.FS_RECTF(0, 0, 0, 0)
|
||||
u32 = ctypes.c_uint32(0)
|
||||
fs32 = ctypes.c_float(0)
|
||||
byref = ctypes.byref
|
||||
ox_ref = byref(char_origin_x); oy_ref = byref(char_origin_y)
|
||||
l_ref = byref(char_left_box); r_ref = byref(char_right_box)
|
||||
b_ref = byref(value); t_ref = byref(char_top_box)
|
||||
loose_ref = byref(loose_box)
|
||||
w_ref = byref(field)
|
||||
flags_ref = byref(flags)
|
||||
get_unicode = pdfium_c.FPDFText_GetUnicode
|
||||
is_generated = pdfium_c.FPDFText_IsGenerated
|
||||
get_char_origin = pdfium_c.FPDFText_GetCharOrigin
|
||||
get_char_box = pdfium_c.FPDFText_GetCharBox
|
||||
get_loose_box = pdfium_c.FPDFText_GetLooseCharBox
|
||||
get_font_info = pdfium_c.FPDFText_GetFontInfo
|
||||
get_glyph_width = pdfium_c.FPDFFont_GetGlyphWidth
|
||||
js_is_ws = _is_whitespace
|
||||
name_cache: dict[bytes, str] = {}
|
||||
raw_chars: list[dict] = []
|
||||
last_obj: dict | None = None
|
||||
for index_value in range(count_item):
|
||||
codepoint = get_unicode(text_page, index_value)
|
||||
if codepoint < 0:
|
||||
continue
|
||||
# u == 0 (PDFium found no unicode for the glyph) is KEPT as '\x00':
|
||||
# text extraction emits the raw charcode for unmapped codes, so its items
|
||||
# really contain chr(0) for extension-font pieces at code 0, and the
|
||||
# textpage char carries normal geometry. Skipping it lost the char AND desynced
|
||||
# the unicode walk's object pairing around it.
|
||||
ch_str = chr(codepoint)
|
||||
is_ws = js_is_ws(codepoint)
|
||||
# FPDFText_IsGenerated returns a c_int: 1 generated, 0 real, -1 error.
|
||||
# Only a POSITIVE 1 may mark a char generated -- the -1 has to read the
|
||||
# same way here as it does in the page-mode unicode walk, or the two
|
||||
# char sets disagree and that walk desyncs.
|
||||
is_gen = is_generated(text_page, index_value) == 1
|
||||
# PDFium inserts is_generated chars as layout placeholders for
|
||||
# Td/Tm jumps with no literal content-stream char (typically
|
||||
# " ", "\r", "\n"). Dropping them outright leaves an
|
||||
# unexplained advance gap that the merger then turns into a
|
||||
# fake-space chunk, splitting e.g. "2.1 Computing the EMD"
|
||||
# into three spans (2.1, " ", Computing the EMD) that pipeline
|
||||
# treats as a numeric prefix alone (not a heading). Keep
|
||||
# generated whitespace so the merger's whitespace branch fires
|
||||
# save_last_char without emitting, letting the next visible
|
||||
# glyph compute a tracking-size in-flow advance. Drop only
|
||||
# non-whitespace generated chars (very rare).
|
||||
if is_gen and not is_ws:
|
||||
continue
|
||||
char_origin_x.value = 0.0; char_origin_y.value = 0.0
|
||||
get_char_origin(text_page, index_value, ox_ref, oy_ref)
|
||||
ox_v = char_origin_x.value; oy_v = char_origin_y.value
|
||||
|
||||
# Fetch char bbox first so we can use its center for the obj
|
||||
# lookup — origin alone fails when adjacent obj bboxes nearly
|
||||
# touch (e.g. math-heavy page "(", math italic font \x01, ")" all on the same line
|
||||
# with sub-pt gaps, where origin x falls inside the wrong obj's
|
||||
# tolerance window). Using bbox center gives unambiguous
|
||||
# containment.
|
||||
char_left_box.value = 0.0; char_right_box.value = 0.0; value.value = 0.0; char_top_box.value = 0.0
|
||||
get_char_box(text_page, index_value, l_ref, r_ref, b_ref, t_ref)
|
||||
char_left, char_right, char_top, char_bottom = char_left_box.value, char_right_box.value, char_top_box.value, value.value
|
||||
# Tight (ink) box center -> font-object disambiguation only.
|
||||
center_x = (char_left + char_right) / 2 if char_right > char_left else ox_v
|
||||
center_y = (char_top + char_bottom) / 2 if char_top > char_bottom else oy_v
|
||||
# Horizontal extent for the SPAN comes from the LOOSE char box (the
|
||||
# glyph's full advance cell), not the tight ink box. the PDF text-item
|
||||
# widths are advance-based; the ink box undershoots each glyph's right
|
||||
# edge by its side bearing (e.g. "]" ink-right 274.0 vs advance 275.2,
|
||||
# as expected for advance-based text items). Using the ink box cumulatively under-fills
|
||||
# display-math gaps so the column detector mis-reads them as gutters
|
||||
# and splits a line ("E[x] = μ" -> "E[x]" fragment). Fall back to the
|
||||
# ink box if the loose box is unavailable/degenerate.
|
||||
# (_loose is only READ when the call succeeded, so the hoisted struct
|
||||
# never leaks a previous char's values.)
|
||||
if (get_loose_box(text_page, index_value, loose_ref)
|
||||
and loose_box.right > loose_box.left):
|
||||
loose_left, loose_right = loose_box.left, loose_box.right
|
||||
# Vertical edges of the loose (advance-cell) box. For vertical-
|
||||
# writing (Identity-V / WMode 1) text PDFium builds this cell by
|
||||
# advancing -y from the PEN, so its upper edge IS the pen y and
|
||||
# its extent IS the per-char vertical advance (W2/DW2 applied by
|
||||
# PDFium itself). PDFium fills top/bottom in flow order here, so
|
||||
# they arrive inverted (top < bottom); keep both raw edges.
|
||||
cell_top, cell_bottom = loose_box.top, loose_box.bottom
|
||||
else:
|
||||
loose_left, loose_right = char_left, char_right
|
||||
cell_top, cell_bottom = char_top, char_bottom
|
||||
|
||||
# Character font size disambiguates overlapping objects, such as large
|
||||
# figure labels sharing a y range with smaller heading text.
|
||||
# text-page and character-index lookup read the true per-char rendered size
|
||||
# (FPDFText_GetMatrix) and the reported font size (FPDFText_GetFontSize)
|
||||
# lazily, only to break a multi-object containment tie — see
|
||||
# _find_obj_for_char.
|
||||
obj = _find_obj_for_char(
|
||||
obj_index, center_x, center_y, tol=1.0, char_fs=None, text_page=text_page, char_idx=index_value
|
||||
)
|
||||
if obj is None:
|
||||
obj = (
|
||||
_find_obj_for_char(obj_index, ox_v, oy_v, tol=1.0,
|
||||
char_fs=None, text_page=text_page, char_idx=index_value)
|
||||
or _find_obj_for_char(obj_index, ox_v, oy_v, tol=5.0,
|
||||
char_fs=None, text_page=text_page, char_idx=index_value)
|
||||
or last_obj
|
||||
)
|
||||
if obj is None:
|
||||
continue
|
||||
last_obj = obj
|
||||
|
||||
name = get_font_info(text_page, index_value, font_name_buffer, 256, flags_ref)
|
||||
if name > 1:
|
||||
raw_name = font_name_buffer[:name]
|
||||
char_font_name = name_cache.get(raw_name)
|
||||
if char_font_name is None:
|
||||
char_font_name = raw_name.decode(
|
||||
"latin-1", errors="replace").rstrip("\x00")
|
||||
name_cache[raw_name] = char_font_name
|
||||
else:
|
||||
char_font_name = obj["font_name"]
|
||||
|
||||
# Use baseline (oy) as bbox bottom and baseline + fs_eff as top.
|
||||
# the span anchoring rule uses matrix.f (= baseline y) for both top/
|
||||
# bottom anchors of its span, so chars of the same line all
|
||||
# land at the same bottom even when their ink extends below
|
||||
# baseline ("(", "g", "y" with descenders) or above ("\x01"
|
||||
# math glyphs). This is what the heading heuristics' tokenizer assumes when
|
||||
# checking |c1.C - c2.C| < 1 to decide whether two spans are on
|
||||
# the same line.
|
||||
baseline_y = oy_v
|
||||
char_top = baseline_y + obj["fs_eff"]
|
||||
# Capture the raw glyph advance now, while this page (and thus the
|
||||
# font handle) is alive. The fs_eff-dependent scaling happens later
|
||||
# in _finalize_chars, after the document-wide Type-3 size is known,
|
||||
# so deferring the call would require keeping every page open just to
|
||||
# keep font handles valid (PDFium frees the font when the page is
|
||||
# closed -> dangling handle).
|
||||
u32.value = codepoint
|
||||
fs32.value = obj["fs_raw"]
|
||||
get_glyph_width(obj["font"], u32, fs32, w_ref)
|
||||
raw_chars.append({
|
||||
"i": index_value, "ch": ch_str, "u": codepoint,
|
||||
"is_gen": is_gen,
|
||||
"is_ws": is_ws,
|
||||
"is_mn": _is_zero_width_diacritic(codepoint),
|
||||
"is_cf": _is_invisible_format_mark(codepoint),
|
||||
"ox": ox_v, "oy": oy_v,
|
||||
"left": loose_left, "right": loose_right,
|
||||
"top": char_top, "bottom": baseline_y,
|
||||
"box_top": char_top,
|
||||
"box_bottom": char_bottom,
|
||||
"cell_top": cell_top, "cell_bot": cell_bottom,
|
||||
"w_raw": field.value,
|
||||
"obj": obj, "font_name": char_font_name,
|
||||
})
|
||||
|
||||
return raw_chars, objects
|
||||
|
||||
|
||||
def _accumulate_type3_extents(raw_chars: list[dict], acc: dict) -> None:
|
||||
"""Accumulate document-wide per-font glyph-bbox extents for identity-matrix Type-3 fonts. These fonts use a synthesized font bbox from the union of CharProc glyph boxes and render every glyph at that uniform height. PDFium reports a constant font size and identity CTM for these fonts, but its char box returns each glyph's declared bounds exactly, so box-top/bottom relative to the baseline reveal the rendered glyph extents. Aggregating across the whole document makes the font sizing coverage-independent; a per-page union would drift with sparse page content. Scoped to the identity-matrix Type-3 branch so normal and scaled-matrix fonts are untouched."""
|
||||
for candidate_item in raw_chars:
|
||||
item_value = candidate_item["obj"]
|
||||
if item_value["fs_raw"] >= 1.5 or item_value["scale_y"] >= 1.5 or candidate_item["is_ws"]:
|
||||
continue
|
||||
top = candidate_item["box_top"] - candidate_item["oy"]
|
||||
bot = candidate_item["box_bottom"] - candidate_item["oy"]
|
||||
if top <= bot: # degenerate glyph box (text extraction skips d1 i==0)
|
||||
continue
|
||||
_xref_key = item_value["font_key"]
|
||||
entry_item = acc.get(_xref_key)
|
||||
if entry_item is None:
|
||||
acc[_xref_key] = [top, bot]
|
||||
else:
|
||||
if top > entry_item[0]:
|
||||
entry_item[0] = top
|
||||
if bot < entry_item[1]:
|
||||
entry_item[1] = bot
|
||||
|
||||
|
||||
def _type3_size_by_font(acc: dict) -> dict:
|
||||
"""font handle -> rendered font.bbox height = max ascent - min descent, i.e. span merger ``a = font.bbox[3] - font.bbox[1]`` in page units. Snap to the shortest decimal (PDFium float32 vs span merger float64) for clean knife-edge size comparisons downstream (the page-median gate)."""
|
||||
out: dict = {}
|
||||
for _xref_key, (top, bot) in acc.items():
|
||||
if top > bot:
|
||||
out[_xref_key] = float(f"{top - bot:.6g}")
|
||||
return out
|
||||
|
||||
|
||||
def _apply_type3_sizes(raw_chars: list[dict], size_by_font: dict) -> None:
|
||||
"""Override fs_eff with the document-wide Type-3 size and reset each char's span top to baseline + that size."""
|
||||
if not size_by_font:
|
||||
return
|
||||
for candidate_item in raw_chars:
|
||||
item_value = candidate_item["obj"]
|
||||
if item_value["fs_raw"] >= 1.5 or item_value["scale_y"] >= 1.5:
|
||||
continue
|
||||
font_size_value = size_by_font.get(item_value["font_key"])
|
||||
if font_size_value:
|
||||
item_value["fs_eff"] = font_size_value
|
||||
candidate_item["top"] = candidate_item["oy"] + font_size_value
|
||||
|
||||
|
||||
def _finalize_chars(raw_chars: list[dict]) -> list[dict]:
|
||||
"""Second pass: compute glyph_w per char and emit the merged-ready dicts. The right glyph width definition depends on how PDFium reports the font's metrics: (a) Normal Type 1 fonts (fs_raw >= 1.5, scale.a ~= 1): FPDFFont_GetGlyphWidth(font, code, fs_raw) returns the advance in page units. Use as-is x matrix.a. (b) Scaled-matrix Type 3 (fs_raw < 1.5 but matrix scale >= 1.5, e.g. vector-heavy page's a scaled Type-3 subset with scale=36.49): GetGlyphWidth at fs_raw=0.19 gives font-natural-unit width; x matrix scale recovers page units. (c) Identity-matrix Type 3 (fs_raw < 1.5, matrix.a ~= 1, e.g. identity-matrix Type-3 sample an identity-matrix Type-3 font): GetGlyphWidth's output is wrong by an unknown FontMatrix factor (PDFium doesn't fold this for these fonts). Fall back to neighbor-step fallback (next_char.ox - this_char.ox within same obj). """
|
||||
out: list[dict] = []
|
||||
for key_value, candidate_item in enumerate(raw_chars):
|
||||
if candidate_item.get("drop"):
|
||||
# Folded into the previous char by _apply_font_unicode (PDFium's
|
||||
# decomposition of a glyph text extraction emits as ONE precomposed char).
|
||||
continue
|
||||
obj = candidate_item["obj"]
|
||||
# w_raw = FPDFFont_GetGlyphWidth(font, code, fs_raw), captured in the
|
||||
# first pass while the page/font handle was alive.
|
||||
raw = candidate_item["w_raw"]
|
||||
if "w_synth" in candidate_item:
|
||||
# Synthesized glyph (PDFium font-layer drop): the advance was
|
||||
# computed from the surviving neighbors' pen gap.
|
||||
glyph_w = candidate_item["w_synth"]
|
||||
elif obj["fs_raw"] >= 1.5 or obj["scale_y"] >= 1.5:
|
||||
# Cases (a) and (b): GetGlyphWidth + matrix scaling works.
|
||||
glyph_w = raw * obj["scale_x"]
|
||||
else:
|
||||
# Case (c): Identity-matrix Type 3 — derive from neighbor.
|
||||
nxt = raw_chars[key_value + 1] if key_value + 1 < len(raw_chars) else None
|
||||
if (
|
||||
nxt is not None
|
||||
and nxt["obj"] is obj
|
||||
and abs(nxt["oy"] - candidate_item["oy"]) < 0.5
|
||||
and nxt["ox"] > candidate_item["ox"]
|
||||
):
|
||||
glyph_w = nxt["ox"] - candidate_item["ox"]
|
||||
else:
|
||||
# Last char in obj or new line — scale by fs_eff/fs_raw.
|
||||
scale = (obj["fs_eff"] / obj["fs_raw"]) if obj["fs_raw"] > 0 else 1.0
|
||||
glyph_w = raw * scale
|
||||
|
||||
# NOTE: glyph_w is PDFium's FPDFFont_GetGlyphWidth, used by the extraction advance model
|
||||
# the font's glyph width; this is the advance model. For some RTL
|
||||
# (Hebrew/Arabic) fonts PDFium's GetGlyphWidth does not match the actual
|
||||
# rendered char spacing, which leaves spurious intra-word spaces; that is
|
||||
# a PDF backend DATA LIMITATION (PDFium's hmtx/advance reporting),
|
||||
# not a condition to compensate for here (any positional override conflates
|
||||
# glyph advance with TJ word-gaps and breaks shaped Arabic). Left as-is.
|
||||
|
||||
reference_item = {
|
||||
"ch": candidate_item["ch"],
|
||||
"is_ws": candidate_item["is_ws"],
|
||||
"is_mn": candidate_item["is_mn"],
|
||||
"is_cf": candidate_item["is_cf"],
|
||||
"ox": candidate_item["ox"], "oy": candidate_item["oy"],
|
||||
"glyph_w": glyph_w,
|
||||
"fs": obj["fs_eff"],
|
||||
"fs_x": obj["fs_raw"] * obj["scale_x"] if obj["scale_x"] > 0 else obj["fs_eff"],
|
||||
# Left edge from the text-positioning pen origin (ox), matching
|
||||
# span merger, not the glyph ink box: the ink-box left drifts ~0.1pt by
|
||||
# first-glyph side bearing, which trips the column-alignment gate
|
||||
# gate (tol 0.1) and over-splits double-spaced blocks. Right stays
|
||||
# ink-box (pen-right via glyph_w is unreliable for Type-3 fonts).
|
||||
"left": candidate_item["ox"], "right": candidate_item["right"],
|
||||
"top": candidate_item["top"], "bottom": candidate_item["bottom"],
|
||||
"font_name": candidate_item["font_name"],
|
||||
# Unique per-font identity (the PDFium font handle, == span merger'
|
||||
# loaded font identity). The merger splits chunks on this, not on font_name:
|
||||
# identity-matrix Type-3 fonts (an identity-matrix Type-3 font) all report an
|
||||
# empty name, so a name-based split can't separate a 12pt body run
|
||||
# from an inline 11pt code word ("...of expressions..."). span merger
|
||||
# emits a separate text item per font, so the body keeps fs=12 and
|
||||
# the code word fs=11.16 instead of the whole run collapsing to the
|
||||
# smaller fs_min.
|
||||
"font_key": obj["font_key"],
|
||||
"weight": obj["weight"],
|
||||
"obj": obj, # host text object (Tj/show-text)
|
||||
}
|
||||
if obj.get("vertical"):
|
||||
# Vertical-writing pen model, from the loose advance cell (probe-
|
||||
# for Identity-V: cell upper edge == pen y, cell extent ==
|
||||
# the per-char vertical advance with W2/DW2 applied by PDFium, and
|
||||
# the cell is horizontally centred on the pen x because the default
|
||||
# vertical origin vx is w/2 -- the default vertical-origin convention when the
|
||||
# font has no per-char vmetric).
|
||||
pen_y = max(candidate_item["cell_top"], candidate_item["cell_bot"])
|
||||
reference_item["v_pen_x"] = (candidate_item["left"] + candidate_item["right"]) / 2.0
|
||||
reference_item["v_pen_y"] = pen_y
|
||||
# pen y after this glyph's advance (text extraction previous glyph transform[5])
|
||||
reference_item["v_after"] = min(candidate_item["cell_top"], candidate_item["cell_bot"])
|
||||
out.append(reference_item)
|
||||
return out
|
||||
|
||||
|
||||
def _inherited_box(pdf_doc, page_idx: int, name: str):
|
||||
"""span merger ``inherited page-box lookup`` definition: MediaBox/CropBox resolved through the page-tree ``/Parent`` chain (page-tree inheritance lookup). PDFium's FPDFPage_Get*Box does NOT inherit (pdfium bug 1786), so inherited boxes must come from the PyPDF2 channel. Returns a raw 4-tuple or None (absent / not a 4-number array, matching span merger length gate)."""
|
||||
try:
|
||||
xref_cursor = pdf_doc.page_xref(page_idx)
|
||||
for _ in range(32):
|
||||
token_value, value = pdf_doc.xref_get_key(xref_cursor, name)
|
||||
if token_value != "null":
|
||||
if token_value != "array":
|
||||
return None
|
||||
box_tokens = value.strip().lstrip("[").rstrip("]").split()
|
||||
if len(box_tokens) != 4:
|
||||
return None # span merger: array check and length == 4
|
||||
|
||||
try:
|
||||
return tuple(float(box_token) for box_token in box_tokens)
|
||||
except ValueError:
|
||||
return None
|
||||
parent_key_type, position_value = pdf_doc.xref_get_key(xref_cursor, "Parent")
|
||||
if parent_key_type != "xref":
|
||||
return None
|
||||
xref_cursor = int(position_value.split()[0])
|
||||
except Exception:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _page_view_rect(page, med_raw=None, crop_raw=None) -> tuple[float, float, float, float] | None:
|
||||
"""span merger ``normalized page view`` : rectangle normalization'd CropBox clamped to the rectangle normalization'd MediaBox. Differing boxes are intersected (rectangle intersection); an empty or zero-area intersection, and a degenerate CropBox, fall back to the MediaBox (a degenerate MediaBox falls back to US-Letter, text extraction US-Letter fallback media box). ``med_raw``/``crop_raw`` are the INHERITED boxes from ``_inherited_box`` (None = absent/no reader); the PDFium getters below are the non-inheriting fallback."""
|
||||
def norm(secondary_item):
|
||||
if secondary_item is None:
|
||||
return None
|
||||
box_x_min, box_y_min, box_x_max, box_y_max = secondary_item
|
||||
count_item = (min(box_x_min, box_x_max), min(box_y_min, box_y_max), max(box_x_min, box_x_max), max(box_y_min, box_y_max))
|
||||
return count_item if (count_item[2] - count_item[0] > 0 and count_item[3] - count_item[1] > 0) else None
|
||||
|
||||
med = norm(med_raw)
|
||||
if med is None:
|
||||
try:
|
||||
med = norm(tuple(page.get_mediabox()))
|
||||
except Exception:
|
||||
med = None
|
||||
if med is None:
|
||||
med = (0.0, 0.0, 612.0, 792.0)
|
||||
crop = norm(crop_raw)
|
||||
if crop is None:
|
||||
try:
|
||||
crop = norm(tuple(page.get_cropbox()))
|
||||
except Exception:
|
||||
crop = None
|
||||
if crop is None or crop == med:
|
||||
return med
|
||||
x_min, y_min = max(crop[0], med[0]), max(crop[1], med[1])
|
||||
x_max, y_max = min(crop[2], med[2]), min(crop[3], med[3])
|
||||
if x_max - x_min <= 0 or y_max - y_min <= 0:
|
||||
return med
|
||||
return (x_min, y_min, x_max, y_max)
|
||||
|
||||
|
||||
def _off_page(mapping: dict, view_box) -> bool:
|
||||
"""Position-comparison view box test: a non-diacritic glyph whose text origin is outside the page view box is dropped. The check compares ``pos - view box origin`` against the raw x1/y1 upper bounds, not width/height. ``view_box`` is the normalized page view as ``(x0, y0, x1, y1)``; None disables the test."""
|
||||
if view_box is None:
|
||||
return False
|
||||
origin_offset_x = mapping["ox"] - view_box[0]
|
||||
origin_offset_y = mapping["oy"] - view_box[1]
|
||||
return origin_offset_x < 0 or origin_offset_x > view_box[2] or origin_offset_y < 0 or origin_offset_y > view_box[3]
|
||||
@@ -0,0 +1,341 @@
|
||||
"""PostScript number parsing and ToUnicode CMap interpretation."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import re
|
||||
|
||||
from .pdf_objects import (
|
||||
_PDF_WHITESPACE_BYTES,
|
||||
_PDF_DELIMITER_BYTES,
|
||||
_PDF_STRING_ESCAPE_BYTES,
|
||||
)
|
||||
from .text_normalize import _WHITESPACE_CODEPOINTS
|
||||
|
||||
|
||||
def _utf16be_units_to_str(units: list[int]) -> str:
|
||||
"""Decode UTF-16BE token bytes into text. Odd trailing bytes pair with 0. A unit can exceed 0xFF during range carry, and no byte mask is applied before surrogate handling, so a composed value may exceed 0xFFFF and become an astral character."""
|
||||
if len(units) % 2:
|
||||
units = units + [0]
|
||||
out: list[int] = []
|
||||
key_value = 0
|
||||
while key_value < len(units):
|
||||
width_one = (units[key_value] << 8) | units[key_value + 1]
|
||||
key_value += 2
|
||||
if (width_one & 0xF800) != 0xD800:
|
||||
out.append(width_one)
|
||||
continue
|
||||
width_two = 0
|
||||
if key_value < len(units):
|
||||
width_two = (units[key_value] << 8) | units[key_value + 1]
|
||||
key_value += 2
|
||||
out.append(((width_one & 0x3FF) << 10) + (width_two & 0x3FF) + 0x10000)
|
||||
return "".join(chr(candidate_item) for candidate_item in out)
|
||||
|
||||
|
||||
# ASCII-only numeric grammar used for PDF numeric-name heuristics. It uses the
|
||||
# same decimal grammar as model.to_number but without NFKC normalization. Trim set is the
|
||||
# Unicode WhiteSpace + LineTerminator set, not Python's str.strip set.
|
||||
_NUM_DECIMAL_RE = re.compile(r"^[+-]?(?:[0-9]+\.?[0-9]*|\.[0-9]+)(?:[eE][+-]?[0-9]+)?$")
|
||||
_NUM_INFINITY_RE = re.compile(r"^[+-]?Infinity$")
|
||||
_NUM_HEX_RE = re.compile(r"^0[xX][0-9a-fA-F]+$")
|
||||
_NUM_OCTAL_RE = re.compile(r"^0[oO][0-7]+$")
|
||||
_NUM_BINARY_RE = re.compile(r"^0[bB][01]+$")
|
||||
_WHITESPACE_STRIP = "".join(chr(unit_value) for unit_value in _WHITESPACE_CODEPOINTS)
|
||||
|
||||
|
||||
def _ieee_div(value: float, other_item: float) -> float:
|
||||
"""IEEE-754 division, no ZeroDivisionError (``0/0-> NaN, ``x/±0-> ±Inf with the usual sign rules)."""
|
||||
if other_item != 0.0:
|
||||
return value / other_item
|
||||
if value == 0.0 or value != value:
|
||||
return math.nan
|
||||
return math.inf if (value > 0.0) == (math.copysign(1.0, other_item) > 0.0) else -math.inf
|
||||
|
||||
|
||||
def _compute_skew(mtx: tuple) -> float:
|
||||
"""Return the text matrix skew score for an item. transform's rotation/shear ratios, no zero guard (cardinal rotation -> Inf, upright -> 0). Degenerate case: the matrix-size path folds font size into the transform, so ``Tf 0`` text gives 0/0 = NaN there; the PDFium object matrix keeps font size separate and yields finite ratios (degenerate invisible text only)."""
|
||||
primary_item, secondary_item, candidate_item, reference_item = mtx
|
||||
quad_one = _ieee_div(secondary_item, primary_item)
|
||||
quad_two = _ieee_div(candidate_item, reference_item)
|
||||
return quad_one * quad_one + quad_two * quad_two
|
||||
|
||||
|
||||
def _to_number(text: str) -> float:
|
||||
"""/ ``numeric conversion`` (no NFKC): trim parser whitespace, ``""-> 0, then the numeric literal grammar (decimal/exponent, ``0x``/``0o``/``0b``, ``+-Infinity``); anything else -> NaN."""
|
||||
token_value = text.strip(_WHITESPACE_STRIP)
|
||||
if token_value == "":
|
||||
return 0.0
|
||||
if _NUM_INFINITY_RE.match(token_value):
|
||||
return -math.inf if token_value[0] == "-" else math.inf
|
||||
if _NUM_HEX_RE.match(token_value):
|
||||
return float(int(token_value[2:], 16))
|
||||
if _NUM_OCTAL_RE.match(token_value):
|
||||
return float(int(token_value[2:], 8))
|
||||
if _NUM_BINARY_RE.match(token_value):
|
||||
return float(int(token_value[2:], 2))
|
||||
if _NUM_DECIMAL_RE.match(token_value):
|
||||
return float(token_value)
|
||||
return math.nan
|
||||
|
||||
|
||||
def _parse_int(text: str, radix: int) -> float:
|
||||
"""skip leading parser whitespace, an optional sign, an optional ``0x`` prefix when ``radix == 16``, then the leading run of radix digits. Returns ``NaN`` (as in the heading heuristics) when no digit is consumed."""
|
||||
token_value = text.lstrip(_WHITESPACE_STRIP)
|
||||
index_value = 0
|
||||
neg = False
|
||||
if index_value < len(token_value) and token_value[index_value] in "+-":
|
||||
neg = token_value[index_value] == "-"
|
||||
index_value += 1
|
||||
if radix == 16 and token_value[index_value:index_value + 2] in ("0x", "0X"):
|
||||
index_value += 2
|
||||
digits = "0123456789abcdefghijklmnopqrstuvwxyz"[:radix]
|
||||
start = index_value
|
||||
val = 0
|
||||
while index_value < len(token_value) and token_value[index_value].lower() in digits:
|
||||
val = val * radix + digits.index(token_value[index_value].lower())
|
||||
index_value += 1
|
||||
if index_value == start:
|
||||
return math.nan
|
||||
return float(-val if neg else val)
|
||||
|
||||
|
||||
def _cmap_str_to_int(seq) -> int:
|
||||
"""Accumulate CMap definition-code bytes with 32-bit unsigned wrap."""
|
||||
primary_item = 0
|
||||
for codepoint in seq:
|
||||
primary_item = ((primary_item << 8) | codepoint) & 0xFFFFFFFF
|
||||
return primary_item
|
||||
|
||||
|
||||
def _parse_tounicode_cmap(data: bytes) -> dict[int, str]:
|
||||
"""CMap reader for ToUnicode streams, following text extraction CMap parsing + ToUnicode parsing: bfchar/bfrange with hex, literal-string, and (bfrange dst / array elements) integer tokens, plus cidchar/cidrange (numeric entries -> code-point conversion, the numeric-CID class). Structural junk is contained per block like CMap parsing's warn-and-continue catch (the block is dropped, the map survives); only decode-level errors (chr on a code-point conversion-invalid value) propagate so the caller reaches span merger ToUnicode parsing rejection path (-> no included map)."""
|
||||
tokens: list = []
|
||||
index_value, count_item = 0, len(data)
|
||||
while index_value < count_item:
|
||||
candidate_item = data[index_value]
|
||||
if candidate_item in _PDF_WHITESPACE_BYTES:
|
||||
index_value += 1
|
||||
elif candidate_item == 0x25: # comment
|
||||
while index_value < count_item and data[index_value] not in b"\r\n":
|
||||
index_value += 1
|
||||
elif candidate_item == 0x3C: # << dict-open (skip) or <hex>
|
||||
if index_value + 1 < count_item and data[index_value + 1] == 0x3C:
|
||||
index_value += 2
|
||||
continue
|
||||
state_item = data.find(b">", index_value)
|
||||
if state_item < 0:
|
||||
break # unterminated hex string: stop and keep tokens already read
|
||||
hex_values = "".join(chr(secondary_item) for secondary_item in data[index_value + 1:state_item]
|
||||
if chr(secondary_item) in "0123456789abcdefABCDEF")
|
||||
if len(hex_values) % 2:
|
||||
hex_values = hex_values[:-1] # drop a lone trailing hex digit
|
||||
tokens.append(("hex", tuple(bytes.fromhex(hex_values))))
|
||||
index_value = state_item + 1
|
||||
elif candidate_item == 0x3E: # >> dict-close (skip)
|
||||
index_value += 2 if (index_value + 1 < count_item and data[index_value + 1] == 0x3E) else 1
|
||||
elif candidate_item in b"[]":
|
||||
tokens.append(("delim", chr(candidate_item)))
|
||||
index_value += 1
|
||||
elif candidate_item == 0x2F: # /name
|
||||
state_item = index_value + 1
|
||||
while state_item < count_item and data[state_item] not in _PDF_WHITESPACE_BYTES and data[state_item] not in _PDF_DELIMITER_BYTES:
|
||||
state_item += 1
|
||||
tokens.append(("name", data[index_value + 1:state_item].decode("latin-1")))
|
||||
index_value = state_item
|
||||
elif candidate_item == 0x28: # (string) -- literal-string lexer code units (dst values)
|
||||
depth = 0
|
||||
unicode_scalar: list[int] = []
|
||||
while index_value < count_item:
|
||||
byte_value = data[index_value]
|
||||
if byte_value == 0x5C:
|
||||
if index_value + 1 >= count_item:
|
||||
index_value += 1
|
||||
break
|
||||
entry_item = data[index_value + 1]
|
||||
if entry_item in _PDF_STRING_ESCAPE_BYTES:
|
||||
unicode_scalar.append(_PDF_STRING_ESCAPE_BYTES[entry_item])
|
||||
index_value += 2
|
||||
elif 0x30 <= entry_item <= 0x37:
|
||||
state_item = index_value + 1
|
||||
val = 0
|
||||
while state_item < count_item and state_item - index_value <= 3 and 0x30 <= data[state_item] <= 0x37:
|
||||
val = (val << 3) | (data[state_item] - 0x30)
|
||||
state_item += 1
|
||||
unicode_scalar.append(val)
|
||||
index_value = state_item
|
||||
elif entry_item in (0x0D, 0x0A):
|
||||
index_value += 2
|
||||
if entry_item == 0x0D and index_value < count_item and data[index_value] == 0x0A:
|
||||
index_value += 1
|
||||
else:
|
||||
unicode_scalar.append(entry_item)
|
||||
index_value += 2
|
||||
continue
|
||||
if byte_value == 0x28:
|
||||
if depth:
|
||||
unicode_scalar.append(byte_value)
|
||||
depth += 1
|
||||
elif byte_value == 0x29:
|
||||
depth -= 1
|
||||
if depth == 0:
|
||||
index_value += 1
|
||||
break
|
||||
unicode_scalar.append(byte_value)
|
||||
else:
|
||||
unicode_scalar.append(byte_value)
|
||||
index_value += 1
|
||||
tokens.append(("hex", tuple(unicode_scalar)))
|
||||
else:
|
||||
state_item = index_value
|
||||
while state_item < count_item and data[state_item] not in _PDF_WHITESPACE_BYTES and data[state_item] not in _PDF_DELIMITER_BYTES:
|
||||
state_item += 1
|
||||
word = data[index_value:state_item].decode("latin-1")
|
||||
if (0x30 <= data[index_value] <= 0x39) or data[index_value] in b"+-.":
|
||||
try:
|
||||
numeric_value = float(word)
|
||||
except ValueError:
|
||||
numeric_value = 0.0
|
||||
tokens.append(("num", numeric_value))
|
||||
else:
|
||||
tokens.append(("op", word))
|
||||
index_value = state_item
|
||||
|
||||
out: dict[int, str] = {}
|
||||
|
||||
def codepoint_to_string(numeric_value: float) -> str:
|
||||
# ToUnicode parsing numeric entry: code-point conversion(token) -- its
|
||||
# RangeError (non-integer / out of range) kills the whole map, so
|
||||
# chr's ValueError propagate.
|
||||
codepoint = int(numeric_value)
|
||||
if codepoint != numeric_value:
|
||||
raise ValueError("code-point conversion non-integer")
|
||||
return chr(codepoint)
|
||||
|
||||
def is_int(numeric_value: float) -> bool:
|
||||
# The integer test that guards the numeric-entry check and selects the
|
||||
# destination branch rejects +-Infinity, NaN AND any fractional value.
|
||||
return math.isfinite(numeric_value) and numeric_value == int(numeric_value)
|
||||
|
||||
def map_range_units(range_start: int, range_end: int, units: list[int]) -> None:
|
||||
# text extraction CMap.bf-range mapping : ``last byte`` is FIXED to
|
||||
# the ORIGINAL dst length-1; only THAT byte index is incremented. On
|
||||
# 0xFF overflow it carries into byte last byte-1 (byte-to-character conversion ToUint16
|
||||
# == the & 0xFFFF) and sets the tail to 0x00; the next non-overflow
|
||||
# step is substring(0,last byte)+chr(next), so a 1-byte dst collapses
|
||||
# back to ONE byte. A 1-byte 0xFF overflow gives "\x00\x00"
|
||||
# Empty destinations yield "" for the first code and "\x00" for each
|
||||
# subsequent code after carry.
|
||||
last_byte = len(units) - 1
|
||||
for code in range(range_start, range_end + 1):
|
||||
out[code] = _utf16be_units_to_str(units)
|
||||
if last_byte < 0:
|
||||
units = [0x00]
|
||||
continue
|
||||
cur = units[last_byte] if last_byte < len(units) else 0
|
||||
nxt = cur + 1
|
||||
if nxt > 0xFF:
|
||||
if last_byte - 1 >= 0:
|
||||
units = (units[:last_byte - 1]
|
||||
+ [(units[last_byte - 1] + 1) & 0xFFFF, 0x00])
|
||||
else:
|
||||
units = [0x00, 0x00]
|
||||
else:
|
||||
units = units[:last_byte] + [nxt]
|
||||
|
||||
key_value = 0
|
||||
while key_value < len(tokens):
|
||||
kind, val = tokens[key_value]
|
||||
if kind == "op" and val == "beginbfchar":
|
||||
key_value += 1
|
||||
while key_value + 1 < len(tokens) and tokens[key_value][0] == "hex":
|
||||
src = _cmap_str_to_int(tokens[key_value][1])
|
||||
if tokens[key_value + 1][0] != "hex":
|
||||
# the heading heuristics string-operand check throws -> CMap parsing catch drops the
|
||||
# rest of the block, map survives.
|
||||
key_value += 2
|
||||
break
|
||||
out[src] = _utf16be_units_to_str(list(tokens[key_value + 1][1]))
|
||||
key_value += 2
|
||||
elif kind == "op" and val == "beginbfrange":
|
||||
key_value += 1
|
||||
while (key_value + 1 < len(tokens) and tokens[key_value][0] == "hex"
|
||||
and tokens[key_value + 1][0] == "hex"):
|
||||
src_start = _cmap_str_to_int(tokens[key_value][1])
|
||||
src_end = _cmap_str_to_int(tokens[key_value + 1][1])
|
||||
key_value += 2
|
||||
if src_end - src_start > 0xFFFFFF:
|
||||
# The range-limit throw is raised from INSIDE the bf-range
|
||||
# mapping itself, i.e. from inside the call that CMap
|
||||
# parsing wraps, so the rest of the block goes with it (the
|
||||
# destination has already been lexed -- for an array, up to
|
||||
# and including the "]").
|
||||
if key_value < len(tokens) and tokens[key_value] == ("delim", "["):
|
||||
while key_value < len(tokens) and tokens[key_value] != ("delim", "]"):
|
||||
key_value += 1
|
||||
key_value += 1
|
||||
elif key_value < len(tokens) and tokens[key_value][0] in ("hex", "num"):
|
||||
key_value += 1
|
||||
break
|
||||
if key_value < len(tokens) and tokens[key_value] == ("delim", "["):
|
||||
key_value += 1
|
||||
code = src_start
|
||||
# The array form stores EVERY lexed object up to "]" or end
|
||||
# of input; the UTF-16BE walk over a value that has no
|
||||
# length (a name, an operator) runs zero times and yields
|
||||
# the empty string.
|
||||
while key_value < len(tokens) and tokens[key_value] != ("delim", "]"):
|
||||
if code <= src_end:
|
||||
dst_token = tokens[key_value]
|
||||
if dst_token[0] == "hex":
|
||||
out[code] = _utf16be_units_to_str(list(dst_token[1]))
|
||||
elif dst_token[0] == "num":
|
||||
out[code] = codepoint_to_string(dst_token[1])
|
||||
else:
|
||||
out[code] = ""
|
||||
code += 1
|
||||
key_value += 1
|
||||
if key_value < len(tokens):
|
||||
key_value += 1
|
||||
elif key_value < len(tokens) and tokens[key_value][0] == "hex":
|
||||
units = list(tokens[key_value][1])
|
||||
key_value += 1
|
||||
map_range_units(src_start, src_end, units)
|
||||
elif key_value < len(tokens) and tokens[key_value][0] == "num" and is_int(tokens[key_value][1]):
|
||||
# Integer destinations are one UTF-16 unit, then the normal
|
||||
# increment walk applies. A non-integer number is neither an
|
||||
# integer nor a string nor "[", so it falls through to the
|
||||
# `else` arm below.
|
||||
units = [int(tokens[key_value][1]) & 0xFFFF]
|
||||
key_value += 1
|
||||
map_range_units(src_start, src_end, units)
|
||||
else:
|
||||
break # parse error -> contained: drop the block
|
||||
elif kind == "op" and val == "begincidchar":
|
||||
key_value += 1
|
||||
while (key_value + 1 < len(tokens) and tokens[key_value][0] == "hex"
|
||||
and tokens[key_value + 1][0] == "num"):
|
||||
if not is_int(tokens[key_value + 1][1]):
|
||||
# The integer check throws -> the CMap parsing catch drops
|
||||
# the rest of the block, map survives.
|
||||
key_value += 2
|
||||
break
|
||||
out[_cmap_str_to_int(tokens[key_value][1])] = codepoint_to_string(tokens[key_value + 1][1])
|
||||
key_value += 2
|
||||
elif kind == "op" and val == "begincidrange":
|
||||
key_value += 1
|
||||
while (key_value + 2 < len(tokens) and tokens[key_value][0] == "hex"
|
||||
and tokens[key_value + 1][0] == "hex" and tokens[key_value + 2][0] == "num"):
|
||||
src_start = _cmap_str_to_int(tokens[key_value][1])
|
||||
src_end = _cmap_str_to_int(tokens[key_value + 1][1])
|
||||
start = tokens[key_value + 2][1]
|
||||
key_value += 3
|
||||
if not is_int(start):
|
||||
break # the integer check precedes CID-range mapping: block dropped
|
||||
if src_end - src_start > 0xFFFFFF:
|
||||
break # CID-range range-limit: the block is dropped too
|
||||
for code in range(src_start, src_end + 1):
|
||||
out[code] = codepoint_to_string(start + (code - src_start))
|
||||
else:
|
||||
key_value += 1
|
||||
return out
|
||||
@@ -0,0 +1,249 @@
|
||||
"""Resource-dictionary xref walking and per-page show-code enumeration."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import unicodedata
|
||||
from PyPDF2.generic import (
|
||||
IndirectObject as PdfIndirectRef, NameObject as PdfName, NumberObject as PdfNumber,
|
||||
FloatObject as PdfFloat, BooleanObject as PdfBoolean,
|
||||
DictionaryObject as PdfDictionary, ArrayObject as PdfArray,
|
||||
)
|
||||
|
||||
from .pdf_objects import _decode_pdf_name
|
||||
from .text_normalize import (
|
||||
_normalize_unicodes,
|
||||
_WHITESPACE_CODEPOINTS,
|
||||
_is_whitespace,
|
||||
)
|
||||
from .content_stream import _tokenize_show_operators
|
||||
|
||||
|
||||
def _resource_dict_xrefs(pdf_doc, owner_xref: int, sub: str) -> dict[bytes, int]:
|
||||
"""{canonical resname bytes: xref} for /Redefinitions/<sub> of a page or Form XObject dict, following indirection; for pages, /Redefinitions may be inherited through the /Parent chain."""
|
||||
val = ("null", "null")
|
||||
xref_cursor = owner_xref
|
||||
for _ in range(32): # /Parent chain (pages); XObjects never recurse here
|
||||
val = pdf_doc.xref_get_key(xref_cursor, f"Resources/{sub}")
|
||||
if val[0] != "null":
|
||||
break
|
||||
if pdf_doc.xref_get_key(xref_cursor, "Resources")[0] != "null":
|
||||
break # Redefinitions exists but lacks <sub>
|
||||
parent_key_type, parent_xref_value = pdf_doc.xref_get_key(xref_cursor, "Parent")
|
||||
if parent_key_type != "xref":
|
||||
break
|
||||
xref_cursor = int(parent_xref_value.split()[0])
|
||||
if val[0] == "xref":
|
||||
body = pdf_doc.xref_object(int(val[1].split()[0]), compressed=True)
|
||||
elif val[0] == "dict":
|
||||
body = val[1]
|
||||
else:
|
||||
return {}
|
||||
out: dict[bytes, int] = {}
|
||||
for measure_item in re.finditer(r"/([^\s/\[\]<>()]+)\s+(\d+)\s+\d+\s+R", body):
|
||||
out[_decode_pdf_name(measure_item.group(1).encode("latin-1"))] = int(measure_item.group(2))
|
||||
# DIRECT (inline) sub-dict entries carry no `N G R` for the regex; span merger
|
||||
# reference resolution resolves them all the same, so register each as a virtual
|
||||
# pseudo-xref and the normal integer-keyed pipeline address it.
|
||||
try:
|
||||
node = pdf_doc._resolve_object(xref_cursor)
|
||||
for part in ("Resources", sub):
|
||||
if isinstance(node, PdfIndirectRef):
|
||||
node = node.get_object()
|
||||
node = node["/" + part] if (node is not None and "/" + part in node) else None
|
||||
if node is not None:
|
||||
if isinstance(node, PdfIndirectRef):
|
||||
node = node.get_object()
|
||||
for key_value in node.keys():
|
||||
raw = node.raw_get(key_value)
|
||||
if isinstance(raw, PdfIndirectRef):
|
||||
continue # indirect: the regex pass covered it
|
||||
if not hasattr(raw, "raw_get"):
|
||||
continue # not a dict (malformed entry)
|
||||
name = _decode_pdf_name(key_value.lstrip("/").encode("latin-1"))
|
||||
if name not in out:
|
||||
out[name] = pdf_doc.register_virtual(raw)
|
||||
except Exception:
|
||||
pass
|
||||
return out
|
||||
|
||||
|
||||
def _page_show_codes(
|
||||
pdf_doc, page_idx: int,
|
||||
) -> list[tuple[int | None, tuple[int, ...], float]] | None:
|
||||
"""Every show op the page paints, in paint order, as ``(font_xref | None, charcode units, horizontal scale)-- including text inside Form XObjects, spliced at their ``Do`` position with the XObject's own font redefinitions (span merger text-content extraction recurses the same way; PDFium's textpage flattens them inline). The recursion runs on a CLONE of the live text state, so a form inherits both the active font and the horizontal scale; a Tz inside the form REPLACES it and never leaks back out. None when the page can't be read."""
|
||||
page_xref = pdf_doc.page_xref(page_idx)
|
||||
|
||||
def walk(stream: bytes, fonts_res: dict[bytes, int],
|
||||
xobjs_res: dict[bytes, int], cur_font: int | None, cur_tz: float,
|
||||
visited: frozenset, depth: int,
|
||||
out: list[tuple[int | None, tuple[int, ...], float]]) -> None:
|
||||
if depth > 8:
|
||||
return
|
||||
flush_ids, fonts, show_text_units, horizontal_scales, xobject_paints = _tokenize_show_operators(stream, cur_tz)
|
||||
dict_index = 0
|
||||
for key_value in range(len(show_text_units) + 1):
|
||||
while dict_index < len(xobject_paints) and xobject_paints[dict_index][0] == key_value:
|
||||
paint_position, xname, font_at_do, tz_at_do = xobject_paints[dict_index]
|
||||
dict_index += 1
|
||||
xobject_ref = xobjs_res.get(xname) # lexer names arrive #XX-parsed
|
||||
if xobject_ref is None or xobject_ref in visited:
|
||||
continue
|
||||
state_values, string_value = pdf_doc.xref_get_key(xobject_ref, "Subtype")
|
||||
if state_values != "name" or string_value.lstrip("/") != "Form":
|
||||
continue
|
||||
sub_fonts = _resource_dict_xrefs(pdf_doc, xobject_ref, "Font") or fonts_res
|
||||
sub_xobjs = _resource_dict_xrefs(pdf_doc, xobject_ref, "XObject") or xobjs_res
|
||||
inherited = (fonts_res.get(font_at_do)
|
||||
if font_at_do is not None else None)
|
||||
try:
|
||||
sub_stream = pdf_doc.xref_stream(xobject_ref)
|
||||
except Exception:
|
||||
# span merger: "XObject should be a stream" -> recovery mode skips
|
||||
# THIS Do and keeps walking the page (a direct dict posing
|
||||
# as /Form has no stream; must not kill the whole page).
|
||||
continue
|
||||
walk(sub_stream, sub_fonts, sub_xobjs,
|
||||
inherited, tz_at_do, visited | {xobject_ref}, depth + 1, out)
|
||||
if key_value < len(show_text_units):
|
||||
resource_font_name = fonts[key_value]
|
||||
resource_font_index = (fonts_res.get(resource_font_name)
|
||||
if resource_font_name is not None else cur_font)
|
||||
out.append((resource_font_index, show_text_units[key_value], horizontal_scales[key_value]))
|
||||
|
||||
try:
|
||||
out: list[tuple[int | None, tuple[int, ...], float]] = []
|
||||
walk(
|
||||
pdf_doc[page_idx].read_contents(),
|
||||
_resource_dict_xrefs(pdf_doc, page_xref, "Font"),
|
||||
_resource_dict_xrefs(pdf_doc, page_xref, "XObject"),
|
||||
None, 1.0, frozenset(), 0, out, # the initial text state starts at scale 1
|
||||
)
|
||||
return out
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _char_category(text: str) -> tuple[bool, bool, bool]:
|
||||
"""text extraction glyph Unicode category classification over a (possibly multi-char) glyph Unicode string: first match of /^(\\s)|(\\p{Mn})|(\\p{Cf})$/u decides (isWhitespace, zero-width diacritic classification, invisible format-mark classification)."""
|
||||
for pos, char in enumerate(text):
|
||||
codepoint = ord(char)
|
||||
if pos == 0 and codepoint in _WHITESPACE_CODEPOINTS:
|
||||
return True, False, False
|
||||
cat = unicodedata.category(char)
|
||||
if cat == "Mn":
|
||||
return False, True, False
|
||||
if cat == "Cf" and pos == len(text) - 1:
|
||||
return False, False, True
|
||||
return False, False, False
|
||||
|
||||
|
||||
def _walk_codes(
|
||||
chars: list[tuple[int, str]],
|
||||
targets: list[str],
|
||||
allow_skips: bool = False,
|
||||
) -> tuple[list[tuple[int, str]], list[int], list[tuple[int, int]],
|
||||
list[tuple[int, int]]] | None:
|
||||
"""Walk one run of font-resolved per-code targets against the PDFium chars emitted for the same run; return (patches, drops, consumed, skips) or None on desync. ``consumed`` maps each consumed char's textpage index to the target index that consumed it, which lets the group re-walk repair char-to-object attribution. ``skips`` records each skipped target as (target index, char position it belongs before) for glyph re-synthesis. PDFium's emission per code is unknowable a priori: it may match the font target, fall back to the raw code, or expand a glyph into several chars. Consumption is resolved per code by candidate match: the font target, then its normalized expansions (fixed unicode substitution table, NFKC, NFKD, NFD). On an expansion match where the final span text still converges, the chars are left alone; otherwise the first char is patched to the target unicode and the rest of the run is dropped so the single glyph still carries the advance. The run is valid only if both streams end in sync. """
|
||||
text = "".join(target_char for _, target_char in chars)
|
||||
line_value = len(text)
|
||||
pos = 0
|
||||
patches: list[tuple[int, str]] = []
|
||||
drops: list[int] = []
|
||||
consumed: list[tuple[int, int]] = [] # (char textpage index, target index)
|
||||
skips: list[tuple[int, int]] = [] # (target index, char position)
|
||||
skip_until = -1
|
||||
shift_run = 0
|
||||
for index_value, token_value in enumerate(targets):
|
||||
if index_value <= skip_until:
|
||||
continue # part of an anchored skip run recorded below
|
||||
if pos >= line_value:
|
||||
if allow_skips:
|
||||
# Chars exhausted with targets left: the walk arrived here in
|
||||
# sync, so every remaining target is a glyph PDFium never
|
||||
# emitted (the both-exhaust gate in reverse).
|
||||
skips.append((index_value, pos))
|
||||
continue
|
||||
return None # codes left over: desync
|
||||
target_len = len(token_value)
|
||||
if text[pos:pos + target_len] == token_value:
|
||||
consumed.extend((chars[query_value][0], index_value) for query_value in range(pos, pos + target_len))
|
||||
pos += target_len
|
||||
shift_run = 0
|
||||
continue
|
||||
matched = False
|
||||
for normalized_text in (_normalize_unicodes(token_value),
|
||||
unicodedata.normalize("NFKC", token_value),
|
||||
unicodedata.normalize("NFKD", token_value),
|
||||
unicodedata.normalize("NFD", token_value)):
|
||||
if normalized_text != token_value and text[pos:pos + len(normalized_text)] == normalized_text:
|
||||
if _normalize_unicodes(token_value) != normalized_text:
|
||||
patches.append((chars[pos][0], token_value))
|
||||
drops.extend(chars[query_value][0] for query_value in range(pos + 1, pos + len(normalized_text)))
|
||||
consumed.extend((chars[query_value][0], index_value) for query_value in range(pos, pos + len(normalized_text)))
|
||||
pos += len(normalized_text)
|
||||
matched = True
|
||||
break
|
||||
if matched:
|
||||
shift_run = 0
|
||||
continue
|
||||
# Anchored drop-skip (LAST-RESORT mode only: the window re-walk has
|
||||
# already ruled out the stolen-edge-glyph hypothesis): when PDFium
|
||||
# genuinely never emitted the glyph (font-layer drop -- dense math-heavy page's
|
||||
# 4 α, math-heavy page's scanned-page '~', both absent from the textpage AND
|
||||
# FPDFTextObj_GetText), the target has no char anywhere. Skip it
|
||||
# WITHOUT consuming, but only when the next two unskipped targets
|
||||
# literally anchor on the upcoming chars, so a mis-decode (which
|
||||
# needs the 1-char patch below instead) can't be eaten as a skip.
|
||||
# Drops can be CONSECUTIVE (OCR pages drop runs of glyphs), so scan
|
||||
# forward for the smallest run i..i+m-1 whose following pair
|
||||
# anchors; a wrong run leaves chars unconsumed and the exhaust gate
|
||||
# below still rolls everything back.
|
||||
if allow_skips:
|
||||
skip_run_length = 0
|
||||
for skip_len in range(1, len(targets) - index_value + 1):
|
||||
anchor = targets[index_value + skip_len:index_value + skip_len + 2]
|
||||
if not anchor or not all(len(primary_item) == 1 for primary_item in anchor):
|
||||
break
|
||||
str_value = "".join(anchor)
|
||||
if text[pos:pos + len(str_value)] == str_value:
|
||||
skip_run_length = skip_len
|
||||
break
|
||||
if skip_run_length:
|
||||
skips.extend((query_value, pos) for query_value in range(index_value, index_value + skip_run_length))
|
||||
skip_until = index_value + skip_run_length - 1
|
||||
shift_run = 0
|
||||
continue
|
||||
# PDFium's textpage COLLAPSES space runs: a whitespace target facing
|
||||
# a non-whitespace char means the space's char simply does not exist
|
||||
# in the textpage (it can never be a re-decode of the current char).
|
||||
# Desync rather than mis-patch the neighbouring glyph into a space;
|
||||
# table rows can contain real star glyphs adjacent to synthetic spaces.
|
||||
if (all(_is_whitespace(ord(unit_char)) for unit_char in token_value)
|
||||
and not _is_whitespace(ord(text[pos]))):
|
||||
return None
|
||||
# Off-by-one guard for the 1-char assumption below: when an edge
|
||||
# glyph was mis-attributed to a neighbouring object, every pair
|
||||
# mismatches with the streams shifted by one, and a ligature
|
||||
# expansion elsewhere can re-balance the counts so the exhaust gate
|
||||
# alone would COMMIT the shifted alignment and attach punctuation to
|
||||
# the wrong run. The shift has a literal signature --
|
||||
# the NEXT target equals the current char(s), or the current target
|
||||
# equals the NEXT char(s) -- which legitimate decode mismatches
|
||||
# (text extraction symbol vs PDFium control char) never produce. Two
|
||||
# consecutive hits = systematic shift -> desync, letting the
|
||||
# adjacent-run group re-walk re-align both objects cleanly.
|
||||
nxt = targets[index_value + 1] if index_value + 1 < len(targets) else None
|
||||
if ((nxt is not None and text[pos:pos + len(nxt)] == nxt)
|
||||
or text[pos + 1:pos + 1 + target_len] == token_value):
|
||||
shift_run += 1
|
||||
if shift_run >= 2:
|
||||
return None
|
||||
else:
|
||||
shift_run = 0
|
||||
patches.append((chars[pos][0], token_value))
|
||||
consumed.append((chars[pos][0], index_value))
|
||||
pos += 1
|
||||
if pos != line_value:
|
||||
return None # chars left over: desync
|
||||
return patches, drops, consumed, skips
|
||||
@@ -0,0 +1,424 @@
|
||||
"""Content-stream show-operator tokenization and per-page operator tagging."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .pdf_objects import (
|
||||
_PDF_WHITESPACE_BYTES,
|
||||
_PDF_DELIMITER_BYTES,
|
||||
_PDF_STRING_ESCAPE_BYTES,
|
||||
_decode_pdf_name,
|
||||
)
|
||||
|
||||
|
||||
# Text items start at font/size changes, positional line breaks or gaps, and
|
||||
# content-stream flush operators (q/Q, Do, gs-/Font, marked content). PDFium's
|
||||
# flattened FPDF_PAGEOBJ_TEXT objects can be one-per-glyph for per-glyph Tj
|
||||
# streams, so this merger keeps a strict per-object split unless a later
|
||||
# whitespace-aware rule proves a prose continuation.
|
||||
#
|
||||
# q/Q flush grouping is intentionally disabled. It requires fragile ordinal
|
||||
# alignment between flattened PDFium objects and content-stream show operators,
|
||||
# while the merger only needs the content stream for per-show-op font names
|
||||
# (vertical-font flags) and Unicode-map reconstruction. setFont/gs-font are not
|
||||
# flush scopes here; font_key/fs changes carry the style-boundary split.
|
||||
_FLUSH_OPS = frozenset({b"q", b"Q", b"Do", b"BDC", b"BMC", b"EMC"})
|
||||
_SHOW_OPS = frozenset({b"Tj", b"TJ", b"'", b'"'})
|
||||
# content stream tokenizer content operator table: {op: (operand count, variable operand count)}.
|
||||
# Commands NOT in this table are span merger "Unknown command" -- warned and skipped
|
||||
# with the accumulated args PRESERVED (not cleared).
|
||||
# content stream tokenizer operator table's null-value entries: pure lexer aids so object parser's
|
||||
# longest-known-command walk can pass through prefixes of longer commands
|
||||
# (B -> BM -> BMC, f -> false, n -> null). Not operators.
|
||||
_OP_LEX_PREFIX = frozenset({
|
||||
b"BM", b"BD", b"true", b"fa", b"fal", b"fals", b"false",
|
||||
b"nu", b"nul", b"null",
|
||||
})
|
||||
_OP_OPERAND_COUNTS: dict[bytes, tuple[int, bool]] = {
|
||||
b"w": (1, False), b"J": (1, False), b"j": (1, False), b"M": (1, False),
|
||||
b"d": (2, False), b"ri": (1, False), b"i": (1, False), b"gs": (1, False),
|
||||
b"q": (0, False), b"Q": (0, False), b"cm": (6, False), b"m": (2, False),
|
||||
b"l": (2, False), b"c": (6, False), b"v": (4, False), b"y": (4, False),
|
||||
b"h": (0, False), b"re": (4, False), b"S": (0, False), b"s": (0, False),
|
||||
b"f": (0, False), b"F": (0, False), b"f*": (0, False), b"B": (0, False),
|
||||
b"B*": (0, False), b"b": (0, False), b"b*": (0, False), b"n": (0, False),
|
||||
b"W": (0, False), b"W*": (0, False), b"BT": (0, False), b"ET": (0, False),
|
||||
b"Tc": (1, False), b"Tw": (1, False), b"Tz": (1, False), b"TL": (1, False),
|
||||
b"Tf": (2, False), b"Tr": (1, False), b"Ts": (1, False), b"Td": (2, False),
|
||||
b"TD": (2, False), b"Tm": (6, False), b"T*": (0, False), b"Tj": (1, False),
|
||||
b"TJ": (1, False), b"'": (1, False), b'"': (3, False), b"d0": (2, False),
|
||||
b"d1": (6, False), b"CS": (1, False), b"cs": (1, False), b"SC": (4, True),
|
||||
b"SCN": (33, True), b"sc": (4, True), b"scn": (33, True), b"G": (1, False),
|
||||
b"g": (1, False), b"RG": (3, False), b"rg": (3, False), b"K": (4, False),
|
||||
b"k": (4, False), b"sh": (1, False), b"BI": (0, False), b"ID": (0, False),
|
||||
b"EI": (1, False), b"Do": (1, False), b"MP": (1, False), b"DP": (2, False),
|
||||
b"BMC": (1, False), b"BDC": (2, False), b"EMC": (0, False),
|
||||
b"BX": (0, False), b"EX": (0, False),
|
||||
}
|
||||
|
||||
|
||||
def _tokenize_show_operators(
|
||||
content_bytes: bytes, init_tz: float = 1.0,
|
||||
) -> tuple[list[int], list[bytes | None], list[tuple[int, ...]], list[float],
|
||||
list[tuple[int, bytes, bytes | None, float]]]:
|
||||
"""Tokenize a PDF page content stream. For each text-showing operator, records the active flush scope, font redefinition name, raw charcode units, and horizontal scaling (starting at ``init_tz`` on stream entry, saved and restored by q/Q), plus every Form XObject paint position with the font and horizontal scaling live at that paint. The tokenizer is deliberately tolerant of malformed operators: it skips bad or short operands, preserves unknown-command operands, and emits an empty string for a show operator with the wrong string operand type."""
|
||||
flush_ids: list[int] = []
|
||||
fonts: list[bytes | None] = []
|
||||
show_text_units: list[tuple[int, ...]] = []
|
||||
xobject_paints: list[tuple[int, bytes, bytes | None, float]] = []
|
||||
horizontal_scales: list[float] = []
|
||||
flush_id = 0
|
||||
cur_font: bytes | None = None
|
||||
font_stack: list[bytes | None] = []
|
||||
cur_tz = init_tz
|
||||
tz_stack: list[float] = []
|
||||
opnds: list[tuple[str, object]] = []
|
||||
frames: list[tuple[str, list]] = [] # open [ / << collectors
|
||||
non_processed: list[tuple[str, object]] = []
|
||||
bi_mark: int | None = None
|
||||
|
||||
def push(kind: str, val: object) -> None:
|
||||
(frames[-1][1] if frames else opnds).append((kind, val))
|
||||
|
||||
index_value = 0
|
||||
count_item = len(content_bytes)
|
||||
while index_value < count_item:
|
||||
byte_value = content_bytes[index_value]
|
||||
if byte_value in _PDF_WHITESPACE_BYTES:
|
||||
index_value += 1
|
||||
elif byte_value == 0x25: # % comment -> end of line
|
||||
while index_value < count_item and content_bytes[index_value] not in b"\r\n":
|
||||
index_value += 1
|
||||
elif byte_value == 0x28: # ( literal string: decode per PDF 7.3.4.2
|
||||
depth = 0
|
||||
out: list[int] = []
|
||||
while index_value < count_item:
|
||||
literal_byte = content_bytes[index_value]
|
||||
if literal_byte == 0x5c: # backslash escape
|
||||
if index_value + 1 >= count_item:
|
||||
index_value += 1
|
||||
break
|
||||
escape_byte = content_bytes[index_value + 1]
|
||||
if escape_byte in _PDF_STRING_ESCAPE_BYTES:
|
||||
out.append(_PDF_STRING_ESCAPE_BYTES[escape_byte])
|
||||
index_value += 2
|
||||
elif 0x30 <= escape_byte <= 0x37: # \ddd octal, 1-3 digits
|
||||
token_end = index_value + 1
|
||||
val = 0
|
||||
while token_end < count_item and token_end - index_value <= 3 and 0x30 <= content_bytes[token_end] <= 0x37:
|
||||
val = (val << 3) | (content_bytes[token_end] - 0x30)
|
||||
token_end += 1
|
||||
# text extraction literal-string lexer pushes byte-to-character conversion with NO
|
||||
# byte mask: \400..\777 stay 256..511 (the PDF-spec
|
||||
# high-order-overflow mask is deliberately absent).
|
||||
out.append(val)
|
||||
index_value = token_end
|
||||
elif escape_byte in (0x0D, 0x0A): # \<EOL> line continuation
|
||||
index_value += 2
|
||||
if escape_byte == 0x0D and index_value < count_item and content_bytes[index_value] == 0x0A:
|
||||
index_value += 1
|
||||
else: # \x -> x
|
||||
out.append(escape_byte)
|
||||
index_value += 2
|
||||
continue
|
||||
# NOTE: bare CR/LF inside a literal string fall through to the
|
||||
# raw push below: literal strings keep bare CR/LF as-is here
|
||||
# rather than applying PDF-spec "treat as 0x0A" normalization
|
||||
# is deliberately absent there; no CRLF collapsing either).
|
||||
if literal_byte == 0x28:
|
||||
if depth:
|
||||
out.append(literal_byte)
|
||||
depth += 1
|
||||
elif literal_byte == 0x29:
|
||||
depth -= 1
|
||||
if depth == 0:
|
||||
index_value += 1
|
||||
break
|
||||
out.append(literal_byte)
|
||||
else:
|
||||
out.append(literal_byte)
|
||||
index_value += 1
|
||||
push("str", tuple(out))
|
||||
elif byte_value == 0x3c: # < : << dict-open, else <hex>
|
||||
if index_value + 1 < count_item and content_bytes[index_value + 1] == 0x3c:
|
||||
frames.append(("dict", []))
|
||||
index_value += 2
|
||||
else:
|
||||
index_value += 1
|
||||
nib: list[int] = []
|
||||
while index_value < count_item and content_bytes[index_value] != 0x3e:
|
||||
hex_byte = content_bytes[index_value]
|
||||
if 0x30 <= hex_byte <= 0x39:
|
||||
nib.append(hex_byte - 0x30)
|
||||
elif 0x41 <= hex_byte <= 0x46:
|
||||
nib.append(hex_byte - 0x37)
|
||||
elif 0x61 <= hex_byte <= 0x66:
|
||||
nib.append(hex_byte - 0x57)
|
||||
index_value += 1
|
||||
index_value += 1
|
||||
if len(nib) % 2:
|
||||
nib.pop() # drop a lone trailing hex digit
|
||||
push("str", tuple(
|
||||
(nib[key_value] << 4) | nib[key_value + 1] for key_value in range(0, len(nib), 2)
|
||||
))
|
||||
elif byte_value == 0x3e: # >> dict-close (or stray >)
|
||||
if index_value + 1 < count_item and content_bytes[index_value + 1] == 0x3e:
|
||||
index_value += 2
|
||||
if frames and frames[-1][0] == "dict":
|
||||
frames.pop()
|
||||
push("dict", None)
|
||||
# stray >> : text extraction command token -> unknown command -> tally preserved
|
||||
else:
|
||||
index_value += 1
|
||||
elif byte_value == 0x5b: # [ -- one array operand (text extraction parser builds an array)
|
||||
frames.append(("arr", []))
|
||||
index_value += 1
|
||||
elif byte_value == 0x5d: # ]
|
||||
index_value += 1
|
||||
if frames and frames[-1][0] == "arr":
|
||||
items = frames.pop()[1]
|
||||
# TJ semantics: only direct string elements show; numbers are
|
||||
# kern adjustments and nested non-strings are ignored.
|
||||
push("arr", tuple(codepoint for kerning_delta, vertical_value in items if kerning_delta == "str"
|
||||
for codepoint in vertical_value)) # type: ignore[union-attr]
|
||||
# stray ] : text extraction command token(']') -> unknown command -> tally preserved
|
||||
elif byte_value in b"{}":
|
||||
index_value += 1 # text extraction command token -> not in operator table -> "Unknown command", preserved
|
||||
elif byte_value == 0x2f: # /name operand
|
||||
index_value += 1
|
||||
token_end = index_value
|
||||
while token_end < count_item and content_bytes[token_end] not in _PDF_WHITESPACE_BYTES and content_bytes[token_end] not in _PDF_DELIMITER_BYTES:
|
||||
token_end += 1
|
||||
# Decode #XX escapes while lexing names so consumers receive the
|
||||
# canonical name and do not re-decode downstream.
|
||||
push("name", _decode_pdf_name(content_bytes[index_value:token_end]))
|
||||
index_value = token_end
|
||||
else: # number, keyword operand, or operator
|
||||
first_char = content_bytes[index_value]
|
||||
if (0x30 <= first_char <= 0x39) or first_char in b"+-.":
|
||||
# Number token: consume the whole run to whitespace/delimiter.
|
||||
# Malformed numeric runs are zeroed by the downstream parser.
|
||||
token_end = index_value
|
||||
while token_end < count_item and content_bytes[token_end] not in _PDF_WHITESPACE_BYTES and content_bytes[token_end] not in _PDF_DELIMITER_BYTES:
|
||||
token_end += 1
|
||||
else:
|
||||
# text extraction command lexing: once the accumulated run IS a
|
||||
# known command, stop extending as soon as the next char
|
||||
# would break that -- 'q1' lexes as command token 'q' + number 1
|
||||
# (real-world PDFs; text extraction built known commands for them).
|
||||
token_end = index_value
|
||||
known = False
|
||||
while token_end < count_item and content_bytes[token_end] not in _PDF_WHITESPACE_BYTES and content_bytes[token_end] not in _PDF_DELIMITER_BYTES:
|
||||
cand = content_bytes[index_value:token_end + 1]
|
||||
if (known and cand not in _OP_OPERAND_COUNTS
|
||||
and cand not in _OP_LEX_PREFIX):
|
||||
break
|
||||
token_end += 1
|
||||
known = (content_bytes[index_value:token_end] in _OP_OPERAND_COUNTS
|
||||
or content_bytes[index_value:token_end] in _OP_LEX_PREFIX)
|
||||
operator_token = content_bytes[index_value:token_end]
|
||||
index_value = token_end
|
||||
if not operator_token:
|
||||
index_value += 1
|
||||
continue
|
||||
if (0x30 <= first_char <= 0x39) or first_char in b"+-.":
|
||||
# text extraction number lexer accepts only digit/sign/dot/exponent
|
||||
# runs; Python float would also take "-inf"/"nan" tokens,
|
||||
# which must not poison the operand (or the Tz state).
|
||||
try:
|
||||
num_val = float(operator_token)
|
||||
except ValueError:
|
||||
num_val = 0.0
|
||||
if num_val != num_val or num_val in (float("inf"), -float("inf")):
|
||||
num_val = 0.0
|
||||
push("num", num_val)
|
||||
continue
|
||||
if operator_token in (b"true", b"false"):
|
||||
push("other", None) # text extraction booleans -> operands
|
||||
continue
|
||||
if operator_token == b"null":
|
||||
continue # content operator evaluator: `if (obj != null) args append`
|
||||
|
||||
if frames:
|
||||
# text extraction builds arrays/dicts by recursive object parser: a command
|
||||
# token inside an open [ / << becomes an ELEMENT, never an op.
|
||||
frames[-1][1].append(("other", None))
|
||||
continue
|
||||
if operator_token == b"BI":
|
||||
# text extraction object parser intercepts BI (inline-image parser): the
|
||||
# image never reaches the operator table protocol; it becomes ONE arg.
|
||||
bi_mark = len(opnds)
|
||||
continue
|
||||
spec = _OP_OPERAND_COUNTS.get(operator_token)
|
||||
if spec is None:
|
||||
continue # span merger: warn "Unknown command", tally PRESERVED
|
||||
if operator_token == b"ID": # inline image data (text extraction inline-image parser)
|
||||
# Filter-specific ender first (content stream tokenizer dispatch): DCT scans
|
||||
# for the FFD9 EOI, ASCII85 for '~>', ASCIIHex for '>'; then
|
||||
# the 'EI' marker whose FOLLOWING byte is SPACE/LF/CR
|
||||
# (inline-image end search -- there is NO whitespace
|
||||
# requirement BEFORE the marker: inline-image data can touch it).
|
||||
# span merger extra 10-byte-lookahead / lookahead false-EI checks
|
||||
# are not reproduced (light version).
|
||||
filt = b""
|
||||
if bi_mark is not None:
|
||||
for kerning_delta, vertical_value in opnds[bi_mark:]:
|
||||
if kerning_delta == "name" and vertical_value in (
|
||||
b"DCTDecode", b"DCT", b"ASCII85Decode",
|
||||
b"A85", b"ASCIIHexDecode", b"AHx"):
|
||||
filt = vertical_value
|
||||
break
|
||||
key_value = index_value + 1
|
||||
if filt in (b"DCTDecode", b"DCT"):
|
||||
measure_item = content_bytes.find(b"\xff\xd9", key_value)
|
||||
if measure_item >= 0:
|
||||
key_value = measure_item + 2
|
||||
elif filt in (b"ASCII85Decode", b"A85"):
|
||||
measure_item = content_bytes.find(b"~>", key_value)
|
||||
if measure_item >= 0:
|
||||
key_value = measure_item + 2
|
||||
elif filt in (b"ASCIIHexDecode", b"AHx"):
|
||||
measure_item = content_bytes.find(b">", key_value)
|
||||
if measure_item >= 0:
|
||||
key_value = measure_item + 1
|
||||
while key_value < count_item - 1:
|
||||
if (content_bytes[key_value] == 0x45 and content_bytes[key_value + 1] == 0x49
|
||||
and (key_value + 2 >= count_item or content_bytes[key_value + 2] in b" \n\r")):
|
||||
index_value = key_value + 2
|
||||
break
|
||||
key_value += 1
|
||||
else:
|
||||
index_value = count_item # EOF recovery (text extraction inline-image end recovery)
|
||||
if bi_mark is not None:
|
||||
del opnds[bi_mark:] # the BI..ID dict guts
|
||||
bi_mark = None
|
||||
opnds.append(("other", None)) # the InlineImage operand
|
||||
# text extraction then executes a synthetic command token EI (operand count 1):
|
||||
# pre-BI dangles shift into deferred-operand, the image
|
||||
# operand is consumed, args end empty.
|
||||
while len(opnds) > 1:
|
||||
non_processed.append(opnds.pop(0))
|
||||
opnds.clear()
|
||||
else:
|
||||
# Stray ID without BI: text extraction dispatches it via operator table
|
||||
# (operand count 0), shifting every pending arg into the
|
||||
# deferred-operand stack before the no-op executes.
|
||||
non_processed.extend(opnds)
|
||||
opnds.clear()
|
||||
continue
|
||||
need, variable = spec
|
||||
if not variable and len(opnds) != need:
|
||||
while len(opnds) > need:
|
||||
non_processed.append(opnds.pop(0))
|
||||
while len(opnds) < need and non_processed:
|
||||
opnds.insert(0, non_processed.pop())
|
||||
if len(opnds) < need:
|
||||
# Detail: "Skipping command ...: expected N args" + args
|
||||
|
||||
# cleared; the op has NO side effect (no flush, no Tf).
|
||||
opnds.clear()
|
||||
continue
|
||||
if operator_token in _FLUSH_OPS:
|
||||
flush_id += 1
|
||||
if operator_token == b"q":
|
||||
font_stack.append(cur_font)
|
||||
tz_stack.append(cur_tz)
|
||||
elif operator_token == b"Q":
|
||||
if font_stack:
|
||||
cur_font = font_stack.pop()
|
||||
if tz_stack:
|
||||
cur_tz = tz_stack.pop()
|
||||
elif operator_token == b"Do" and opnds[0][0] == "name":
|
||||
xobject_paints.append((len(flush_ids), opnds[0][1], cur_font, cur_tz)) # type: ignore[arg-type]
|
||||
elif operator_token == b"Tf":
|
||||
if opnds[0][0] == "name":
|
||||
cur_font = opnds[0][1] # type: ignore[assignment]
|
||||
else:
|
||||
# text extraction REPLACES the font either way: a non-name slot
|
||||
# loads undefined -> fallback/fallback font, so the previous
|
||||
# font is gone. None = "no usable resname" here.
|
||||
cur_font = None
|
||||
elif operator_token == b"Tz":
|
||||
# content stream tokenizer horizontal-scale operator: the text state's horizontal scale = args[0]/100
|
||||
# (any type, ToNumber-coerced). We track only numeric operands:
|
||||
# the divisor must account for what PDFIUM folded into the object
|
||||
# matrix, and PDFium's own parser rejects non-numeric Tz --
|
||||
# following span merger coercion here would break consistency with the horizontal font scale.
|
||||
if opnds[0][0] == "num":
|
||||
cur_tz = opnds[0][1] / 100.0 # type: ignore[operator]
|
||||
elif operator_token in _SHOW_OPS:
|
||||
if operator_token == b"TJ":
|
||||
primary_item = opnds[0]
|
||||
# the spaced-text show operator iterates elements by .length/.at, which a
|
||||
# plain STRING also satisfies -- its chars all show.
|
||||
units = primary_item[1] if primary_item[0] in ("arr", "str") else ()
|
||||
elif operator_token == b'"':
|
||||
primary_item = opnds[2]
|
||||
units = primary_item[1] if primary_item[0] == "str" else ()
|
||||
else: # Tj, '
|
||||
primary_item = opnds[0]
|
||||
units = primary_item[1] if primary_item[0] == "str" else ()
|
||||
# A wrong-typed slot or an empty string yields ZERO glyphs in
|
||||
# span merger (glyph conversion -> no item pushed) and no PDFium text
|
||||
# object either -- emit no show entry, so both ordinal
|
||||
# alignments (objects <-> show ops) stay tight.
|
||||
if units:
|
||||
flush_ids.append(flush_id)
|
||||
fonts.append(cur_font)
|
||||
horizontal_scales.append(cur_tz)
|
||||
show_text_units.append(units) # type: ignore[arg-type]
|
||||
opnds.clear() # executed op consumes its args (caller resets)
|
||||
return flush_ids, fonts, show_text_units, horizontal_scales, xobject_paints
|
||||
|
||||
|
||||
def _assign_vertical_tags(
|
||||
objects: list[dict],
|
||||
show_fonts: list[bytes | None] | None = None,
|
||||
vertical_resnames: set[bytes] | None = None,
|
||||
) -> None:
|
||||
"""Tag each text object (paint order) with the vertical-font flag from its matching show-text operator. Ordinal alignment is valid when object and show-op counts agree, such as ligature-free pages and per-glyph CJK Tj streams. On a count mismatch the tag stays False and vertical runs fall back to per-glyph handling. Objects keep ``flush_id=None`` so the merger always splits per object."""
|
||||
if not objects or not vertical_resnames or not show_fonts:
|
||||
return
|
||||
if len(show_fonts) != len(objects):
|
||||
return
|
||||
for item_value, font_name_value in zip(objects, show_fonts):
|
||||
if font_name_value is not None and font_name_value in vertical_resnames:
|
||||
item_value["vertical"] = True
|
||||
|
||||
|
||||
def _assign_show_tz(objects: list[dict], show_tzs: list[float]) -> None:
|
||||
"""Tag each text object with its show-op's text horizontal scale (Tz/100) by the same ordinal alignment as ``_assign_vertical_tags``; on a count mismatch every object keeps tz=1.0 (thresholds behave as before)."""
|
||||
if not objects or not show_tzs or len(show_tzs) != len(objects):
|
||||
return
|
||||
for item_value, timezone_value in zip(objects, show_tzs):
|
||||
item_value["tz"] = timezone_value
|
||||
|
||||
|
||||
def _page_vertical_resource_names(pdf_doc, page_idx: int) -> set[bytes]:
|
||||
"""Font resource names (``F4`` of ``/F4 14 Tf``) on this page whose encoding is a vertical CMap: a predefined ``*-V`` name (Identity-V, UniJIS-UCS2-V, ...) or an embedded CMap stream with ``/WMode 1``. This derives the vertical-font flag used by the item merger, read from the same PyPDF2 document already opened for content streams. Returns an empty set on any failure, which leaves vertical handling disabled for that page."""
|
||||
names: set[bytes] = set()
|
||||
try:
|
||||
for rec in pdf_doc[page_idx].get_fonts(full=True):
|
||||
xref, font_extension, font_type, _basefont, resname, enc = rec[:6]
|
||||
# Predefined vertical CMaps: every shipped vertical bcmap ends in
|
||||
# "-V" EXCEPT the bare Adobe-Japan1 "V" (bcmaps/V.bcmap, header
|
||||
# bit 1 set -- content stream tokenizer reads verticality from that bit).
|
||||
if isinstance(enc, str) and (enc == "V" or enc.endswith("-V")):
|
||||
names.add(resname.encode("latin-1", "replace"))
|
||||
continue
|
||||
# Embedded CMap: /Encoding is an indirect stream; vertical iff its
|
||||
# dict carries /WMode 1.
|
||||
page, resource_names = pdf_doc.xref_get_key(xref, "Encoding")
|
||||
if page == "xref":
|
||||
width_type, width_value_local = pdf_doc.xref_get_key(int(resource_names.split()[0]), "WMode")
|
||||
if width_type in ("int", "real"):
|
||||
try:
|
||||
wmode_number = float(width_value_local.split()[0])
|
||||
except ValueError:
|
||||
wmode_number = float("nan")
|
||||
# Only integer-valued nonzero numbers enable the vertical
|
||||
# font flag, so ``/WMode 1.0`` still counts.
|
||||
if wmode_number.is_integer() and wmode_number != 0:
|
||||
names.add(resname.encode("latin-1", "replace"))
|
||||
except Exception:
|
||||
return set()
|
||||
return names
|
||||
@@ -0,0 +1,394 @@
|
||||
"""Simple-font encoding resolution and per-font Unicode map construction."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from .glyph_tables import (
|
||||
_load_glyph_tables,
|
||||
_get_unicode_for_glyph,
|
||||
_from_char_code,
|
||||
)
|
||||
from .cmap_parse import (
|
||||
_to_number,
|
||||
_parse_int,
|
||||
_parse_tounicode_cmap,
|
||||
)
|
||||
|
||||
|
||||
_TYPE1_SPECIAL_BYTES = b"/[]{}()"
|
||||
# content stream tokenizer tokenises with PDF parser whitespace = {SP, TAB, CR, LF}
|
||||
# ONLY -- narrower than the content-stream/CMap lexer's whitespace-byte set (no 0x0C, no 0x00).
|
||||
_TYPE1_WHITESPACE_BYTES = frozenset(b" \t\r\n")
|
||||
|
||||
|
||||
def _type1_builtin_encoding(font_file: bytes):
|
||||
"""content stream tokenizer font-header extraction's /Encoding case, run over the cleartext segment of an embedded Type1 font file. Returns ("named", encoding-name) | ("array", {code: glyphname}) | None."""
|
||||
end = font_file.find(b"eexec")
|
||||
head = font_file[: end if end >= 0 else len(font_file)]
|
||||
|
||||
header_tokens: list[bytes] = []
|
||||
index_value, count_item = 0, len(head)
|
||||
while index_value < count_item:
|
||||
candidate_item = head[index_value]
|
||||
if candidate_item in _TYPE1_WHITESPACE_BYTES:
|
||||
index_value += 1
|
||||
elif candidate_item == 0x25: # % comment runs to EOL (PDF token reader's comment eater)
|
||||
while index_value < count_item and head[index_value] not in b"\r\n":
|
||||
index_value += 1
|
||||
elif candidate_item in _TYPE1_SPECIAL_BYTES:
|
||||
header_tokens.append(head[index_value:index_value + 1])
|
||||
index_value += 1
|
||||
else:
|
||||
state_item = index_value
|
||||
while state_item < count_item and head[state_item] not in _TYPE1_WHITESPACE_BYTES and head[state_item] not in _TYPE1_SPECIAL_BYTES:
|
||||
state_item += 1
|
||||
header_tokens.append(head[index_value:state_item])
|
||||
index_value = state_item
|
||||
|
||||
def _header_token(number: int) -> bytes | None:
|
||||
return header_tokens[number] if number < len(header_tokens) else None
|
||||
|
||||
# font-header extraction consumes "/"+name PAIRS and keeps scanning after each
|
||||
# case, so a LATER /Encoding overwrites an earlier one (last wins), and a
|
||||
# "//Encoding" pair is consumed whole (its bare "Encoding" never matches).
|
||||
result: tuple | None = None
|
||||
page_value = 0
|
||||
while page_value < len(header_tokens):
|
||||
if header_tokens[page_value] != b"/":
|
||||
page_value += 1
|
||||
continue
|
||||
name_tok = _header_token(page_value + 1)
|
||||
page_value += 2 # the name scanner advances past the slash unconditionally after a '/'
|
||||
if name_tok != b"Encoding":
|
||||
continue
|
||||
arg = _header_token(page_value)
|
||||
if arg is None:
|
||||
# Detail: encoding lookup(null) -> null assigned to built-in encoding.
|
||||
|
||||
result = None
|
||||
break
|
||||
if not arg.isdigit():
|
||||
# named encoding: encoding lookup(name) -- null when unknown
|
||||
# Overwrite any previous result; later encoding declarations win.
|
||||
glyph_name, encs = _load_glyph_tables()
|
||||
name = arg.decode("latin-1")
|
||||
result = ("named", name) if name in encs else None
|
||||
page_value += 1
|
||||
continue
|
||||
# Decimal integer count is parsed through float64, then coerced to int32.
|
||||
# Huge digit strings may round or overflow to Infinity before coercion.
|
||||
array_size_float = float(arg)
|
||||
size = 0 if array_size_float == float("inf") else ((int(array_size_float) + 2**31) % 2**32) - 2**31
|
||||
page_value += 1 # at 'array'
|
||||
enc: dict[int, str] = {}
|
||||
for _ in range(size):
|
||||
token_value = _header_token(page_value)
|
||||
while token_value is not None and token_value not in (b"dup", b"def"):
|
||||
page_value += 1
|
||||
token_value = _header_token(page_value)
|
||||
if token_value is None:
|
||||
# Invalid headers abort the scan and keep any previous encoding.
|
||||
return result
|
||||
if token_value == b"def":
|
||||
break
|
||||
page_value += 1 # past 'dup'
|
||||
# Malformed integer tokens coerce to 0 and do not abort the entry.
|
||||
token_value = _header_token(page_value)
|
||||
try:
|
||||
value = _parse_int(token_value.decode("latin-1"), 10) if token_value is not None else 0.0
|
||||
except OverflowError:
|
||||
value = float("inf") # huge digit run
|
||||
if value != value or value == float("inf") or value == -float("inf"):
|
||||
value = 0.0 # ToInt32(NaN / ±Infinity) = 0
|
||||
idx = ((int(value) + 2**31) % 2**32) - 2**31
|
||||
page_value += 1
|
||||
page_value += 1 # '/' slot consumed blindly
|
||||
group_value = _header_token(page_value)
|
||||
page_value += 1
|
||||
if group_value is not None:
|
||||
enc[idx] = group_value.decode("latin-1")
|
||||
page_value += 1 # 'put' slot consumed blindly
|
||||
result = ("array", enc) # keep scanning: a later /Encoding wins
|
||||
return result
|
||||
|
||||
|
||||
def _simple_font_to_unicode(
|
||||
default_enc: list[str],
|
||||
base_encoding_name: str | None,
|
||||
differences: dict[int, str],
|
||||
force_glyphs: bool = False,
|
||||
) -> dict[int, str]:
|
||||
"""content stream tokenizer simple-font Unicode-map construction, detailed behavior (including the byte-to-character conversion 16-bit truncation on glyphlist hits, the Gxx/g00xx/Cdd/cdd/u heuristics, the base encoding correction branch, and the forced glyph-name pass re-parse when a Cdd name turns out hexadecimal)."""
|
||||
glyphs, encs = _load_glyph_tables()
|
||||
encoding: dict[int, str] = {font: glyph_name_value for font, glyph_name_value in enumerate(default_enc)}
|
||||
for font, glyph_name_value in differences.items():
|
||||
if glyph_name_value == ".notdef":
|
||||
continue # text extraction skips .notdef (.notdef entries)
|
||||
encoding[font] = glyph_name_value
|
||||
|
||||
to_unicode: dict[int, str] = {}
|
||||
for charcode in sorted(encoding):
|
||||
glyph_name = encoding[charcode]
|
||||
if glyph_name == "":
|
||||
continue
|
||||
codepoint = glyphs.get(glyph_name)
|
||||
if codepoint is not None:
|
||||
to_unicode[charcode] = _from_char_code(codepoint)
|
||||
continue
|
||||
code = 0
|
||||
glyph_prefix = glyph_name[0]
|
||||
if glyph_prefix == "G": # Gxx
|
||||
if len(glyph_name) == 3:
|
||||
parsed_integer = _parse_int(glyph_name[1:], 16)
|
||||
code = int(parsed_integer) if parsed_integer == parsed_integer else 0 # pi==pi: not NaN
|
||||
elif glyph_prefix == "g": # g00xx
|
||||
if len(glyph_name) == 5:
|
||||
parsed_integer = _parse_int(glyph_name[1:], 16)
|
||||
code = int(parsed_integer) if parsed_integer == parsed_integer else 0
|
||||
elif glyph_prefix in ("C", "c"): # Cdd{d} / cdd{d}
|
||||
if 3 <= len(glyph_name) <= 4:
|
||||
code_str = glyph_name[1:]
|
||||
if force_glyphs:
|
||||
parsed_integer = _parse_int(code_str, 16)
|
||||
code = int(parsed_integer) if parsed_integer == parsed_integer else 0
|
||||
else:
|
||||
# First try the full numeric grammar. Only when that is NaN
|
||||
# and tolerant base-16 parsing succeeds do we re-parse the
|
||||
# whole encoding as base-16. Non-integer numeric values pass
|
||||
# through and then fail the integer gate below.
|
||||
num = _to_number(code_str)
|
||||
if num != num: # NaN
|
||||
parsed_integer = _parse_int(code_str, 16)
|
||||
if parsed_integer == parsed_integer:
|
||||
return _simple_font_to_unicode(
|
||||
default_enc, base_encoding_name,
|
||||
differences, force_glyphs=True)
|
||||
code = 0
|
||||
elif num.is_integer():
|
||||
code = int(num)
|
||||
else:
|
||||
code = 0
|
||||
elif glyph_prefix == "u":
|
||||
unicode_unit = _get_unicode_for_glyph(glyph_name, glyphs)
|
||||
if unicode_unit != -1:
|
||||
code = unicode_unit
|
||||
if 0 < code <= 0x10FFFF:
|
||||
# Prefer the base encoding glyph when code == charcode
|
||||
if base_encoding_name and code == charcode:
|
||||
base = encs.get(base_encoding_name)
|
||||
# the heading heuristics base encoding[charcode] for charcode > 255 is undefined
|
||||
# (falsy) -- fall through instead of IndexError.
|
||||
if base and 0 <= charcode < len(base) and base[charcode]:
|
||||
to_unicode[charcode] = _from_char_code(
|
||||
glyphs.get(base[charcode], 0))
|
||||
continue
|
||||
to_unicode[charcode] = chr(code) # code-point conversion
|
||||
return to_unicode
|
||||
|
||||
|
||||
def _font_unicode_map(pdf_doc, xref: int) -> tuple[int, dict[int, str]] | None:
|
||||
"""Return the final per-charcode glyph-unicode map for one font as ``(bytes_per_code, {charcode: unicode})``. Simple fonts use 1-byte codes; Identity-H/V composite fonts use 2-byte codes with the included ToUnicode map. ``None`` means uncovered input such as non-Identity composite CMaps or unreadable dictionaries; callers then skip the page patch walk and keep PDFium's output."""
|
||||
glyphs, encs = _load_glyph_tables()
|
||||
|
||||
def _xref_key(number: int, other_text: str) -> tuple[str, str]:
|
||||
return pdf_doc.xref_get_key(number, other_text)
|
||||
|
||||
pdf_value_type, pdf_value = _xref_key(xref, "Subtype")
|
||||
subtype = pdf_value.lstrip("/") if pdf_value_type == "name" else ""
|
||||
if subtype == "Type0":
|
||||
pdf_value_type, pdf_value = _xref_key(xref, "Encoding")
|
||||
if pdf_value_type != "name" or pdf_value.lstrip("/") not in ("Identity-H", "Identity-V"):
|
||||
return None
|
||||
# text extraction reads ToUnicode from the DESCENDANT dict first, then the
|
||||
# Type0 dict (the composite-font prepass uses the descendant for composites).
|
||||
desc_xref = 0
|
||||
delta_top, delta_value = _xref_key(xref, "DescendantFonts")
|
||||
if delta_top == "xref":
|
||||
delta_value = pdf_doc.xref_object(int(delta_value.split()[0]), compressed=True)
|
||||
delta_top = "array"
|
||||
if delta_top == "array":
|
||||
delta_matrix = re.search(r"(\d+)\s+\d+\s+R", delta_value)
|
||||
if delta_matrix:
|
||||
desc_xref = int(delta_matrix.group(1))
|
||||
pdf_value_type, pdf_value = ("null", "null")
|
||||
if desc_xref:
|
||||
pdf_value_type, pdf_value = _xref_key(desc_xref, "ToUnicode")
|
||||
if pdf_value_type != "xref":
|
||||
pdf_value_type, pdf_value = _xref_key(xref, "ToUnicode")
|
||||
tu_map: dict[int, str] | None = None
|
||||
if pdf_value_type == "xref":
|
||||
try:
|
||||
tu_map = _parse_tounicode_cmap(
|
||||
pdf_doc.xref_stream(int(pdf_value.split()[0])))
|
||||
except Exception:
|
||||
tu_map = None # ToUnicode parsing rejects -> no ToUnicode map
|
||||
# The "font carries a ToUnicode map" flag is set only for a present,
|
||||
# accepted and NON-EMPTY map. A missing, rejected or empty ToUnicode all
|
||||
# leave it false, so all three take the composite branch below.
|
||||
if tu_map:
|
||||
return 2, tu_map
|
||||
# No usable ToUnicode: predefined-collection Unicode-map construction
|
||||
# maps Adobe-{GB1,CNS1,Japan1,Korea1} CIDSystemInfo through the shipped
|
||||
# Adobe-XX-UCS2 bcmap (real unicode per cid) -- not implemented.
|
||||
# Returning identity chr(cid) would actively CORRUPT PDFium's
|
||||
# table-driven decode for that class, so keep the guarded None (PDFium
|
||||
# output). Every other registry/ordering IS the identity fallback.
|
||||
if desc_xref:
|
||||
right_type, right_value_local = _xref_key(desc_xref, "CIDSystemInfo/Registry")
|
||||
other_type, other_value_local = _xref_key(desc_xref, "CIDSystemInfo/Ordering")
|
||||
reg = re.sub(r"[()\s]", "", right_value_local) if right_type != "null" else ""
|
||||
ordering = re.sub(r"[()\s]", "", other_value_local) if other_type != "null" else ""
|
||||
if reg == "Adobe" and ordering in ("GB1", "CNS1", "Japan1", "Korea1"):
|
||||
return None
|
||||
return 2, {} # identity Unicode map: unicode == chr(cid)
|
||||
pdf_value_type, pdf_value = _xref_key(xref, "BaseFont")
|
||||
base_font = pdf_value.lstrip("/") if pdf_value_type == "name" else ""
|
||||
|
||||
flags = 0
|
||||
fd_xref = 0
|
||||
has_descriptor = False
|
||||
pdf_value_type, pdf_value = _xref_key(xref, "FontDescriptor")
|
||||
if pdf_value_type == "xref":
|
||||
fd_xref = int(pdf_value.split()[0])
|
||||
has_descriptor = True
|
||||
font_token, font_value = _xref_key(fd_xref, "Flags")
|
||||
if font_token == "int":
|
||||
flags = int(font_value)
|
||||
elif pdf_value_type == "dict":
|
||||
has_descriptor = True
|
||||
flags_match = re.search(r"/Flags\s+([+-]?\d+)", pdf_value)
|
||||
if flags_match:
|
||||
flags = int(flags_match.group(1))
|
||||
if not has_descriptor and subtype != "Type3":
|
||||
# font loading's simulated descriptor (span merger `if (!descriptor)`,
|
||||
# non-Type3 branch): flags come from the BaseFont name with the style
|
||||
# suffix stripped -- Symbol/Dingbats/ZapfDingbats get Symbolic, all
|
||||
# else Nonsymbolic. (the heading heuristics also sets Serif/FixedPitch there; nothing in
|
||||
# this implementation consults those bits, so they are not simulated.) A missing
|
||||
# BaseFont makes the heading heuristics throw parse error -> fallback font, i.e. text extraction DROPS
|
||||
# that font's text entirely; returning None keeps PDFium's decode
|
||||
# instead -- the implementation's conservative boundary, not the same branch. Type3
|
||||
# takes the OTHER the heading heuristics arm:
|
||||
# a barebones descriptor with NO flags and NO BaseFont requirement
|
||||
# (dvips bitmap fonts have neither), so flags stay 0 there.
|
||||
if not base_font:
|
||||
return None
|
||||
base_wo_style = re.sub(r"[,_]", "-", base_font).split("-")[0]
|
||||
flags = 4 if base_wo_style in ("Symbol", "Dingbats", "ZapfDingbats") else 32
|
||||
|
||||
file_key = None
|
||||
if fd_xref:
|
||||
for char_code in ("FontFile", "FontFile2", "FontFile3"):
|
||||
font_token, font_value = _xref_key(fd_xref, char_code)
|
||||
if font_token == "xref":
|
||||
file_key = (char_code, int(font_value.split()[0]))
|
||||
break
|
||||
|
||||
# --- encoding and Differences extraction: /Encoding -> base encodingName + differences
|
||||
differences: dict[int, str] = {}
|
||||
base_encoding_name: str | None = None
|
||||
pdf_value_type, pdf_value = _xref_key(xref, "Encoding")
|
||||
enc_obj: str | None = None
|
||||
if pdf_value_type == "name":
|
||||
base_encoding_name = pdf_value.lstrip("/")
|
||||
elif pdf_value_type == "xref":
|
||||
enc_obj = pdf_doc.xref_object(int(pdf_value.split()[0]), compressed=True)
|
||||
elif pdf_value_type == "dict":
|
||||
enc_obj = pdf_value
|
||||
if enc_obj is not None:
|
||||
flags_match = re.search(r"/BaseEncoding\s*/([^\s/\[\]<>()]+)", enc_obj)
|
||||
if flags_match:
|
||||
base_encoding_name = flags_match.group(1)
|
||||
flags_match = re.search(r"/Differences\s*\[", enc_obj)
|
||||
if flags_match:
|
||||
depth = 1
|
||||
scan_index = flags_match.end()
|
||||
while scan_index < len(enc_obj) and depth:
|
||||
if enc_obj[scan_index] == "[":
|
||||
depth += 1
|
||||
elif enc_obj[scan_index] == "]":
|
||||
depth -= 1
|
||||
scan_index += 1
|
||||
idx = 0
|
||||
for token_match in re.findall(r"/([^\s/\[\]<>()]+)|(\d+)", enc_obj[flags_match.end():scan_index - 1]):
|
||||
if token_match[1]:
|
||||
idx = int(token_match[1])
|
||||
else:
|
||||
name_value = re.sub(
|
||||
r"#([0-9a-fA-F]{2})",
|
||||
lambda encoding_key: chr(int(encoding_key.group(1), 16)), token_match[0])
|
||||
differences[idx] = name_value
|
||||
idx += 1
|
||||
# Table 114: a named base encoding must be one of these three.
|
||||
if base_encoding_name not in ("MacRomanEncoding", "MacExpertEncoding",
|
||||
"WinAnsiEncoding"):
|
||||
base_encoding_name = None
|
||||
|
||||
if base_encoding_name:
|
||||
default_name = base_encoding_name
|
||||
else:
|
||||
symbolic = bool(flags & 4)
|
||||
nonsymbolic = bool(flags & 32)
|
||||
default_name = "StandardEncoding"
|
||||
if subtype == "TrueType" and not nonsymbolic:
|
||||
default_name = "WinAnsiEncoding"
|
||||
if symbolic:
|
||||
default_name = "MacRomanEncoding"
|
||||
if file_key is None:
|
||||
if re.search(r"Symbol", base_font, re.IGNORECASE):
|
||||
default_name = "SymbolSetEncoding"
|
||||
elif re.search(r"Dingbats|Wingdings", base_font, re.IGNORECASE):
|
||||
default_name = "ZapfDingbatsEncoding"
|
||||
default_enc = encs[default_name]
|
||||
has_encoding = bool(base_encoding_name) or bool(differences)
|
||||
|
||||
included: dict[int, str] | None = None
|
||||
pdf_value_type, pdf_value = _xref_key(xref, "ToUnicode")
|
||||
if pdf_value_type == "xref":
|
||||
try:
|
||||
included = _parse_tounicode_cmap(pdf_doc.xref_stream(int(pdf_value.split()[0])))
|
||||
except Exception:
|
||||
included = None # ToUnicode parsing error path: treat as absent
|
||||
|
||||
# Detail: included ToUnicode-map flag = !!toUnicode and toUnicode.length > 0. An
|
||||
|
||||
# empty-but-valid ToUnicode (parsed to {}) is treated as ABSENT, so fall
|
||||
# through to _simple_font_to_unicode + the Type1 builtin amend below
|
||||
# (Type 1 Unicode-map repair), while preserving the existing item-boundary semantics.
|
||||
if included:
|
||||
final = dict(included)
|
||||
if has_encoding: # predefined collection Unicode-map construction -> fallback Unicode map gap fill
|
||||
for font, glyph_name in _simple_font_to_unicode(
|
||||
default_enc, base_encoding_name, differences).items():
|
||||
if font not in final:
|
||||
final[font] = glyph_name
|
||||
return 1, final
|
||||
|
||||
final = _simple_font_to_unicode(default_enc, base_encoding_name, differences)
|
||||
# Type 1 Unicode-map repair: amend from the embedded Type1 program's builtin
|
||||
# encoding (codes not already fixed by the dict's Encoding entry).
|
||||
if file_key is not None and file_key[0] == "FontFile" and subtype in (
|
||||
"Type1", "MMType1"):
|
||||
try:
|
||||
builtin = _type1_builtin_encoding(pdf_doc.xref_stream(file_key[1]))
|
||||
except Exception:
|
||||
builtin = None
|
||||
if builtin is not None:
|
||||
kind, payload = builtin
|
||||
# `built-in encoding == properties.defaultEncoding` (same module
|
||||
|
||||
# array object) -- true iff both name the same predefined encoding.
|
||||
if not (kind == "named" and payload == default_name):
|
||||
items: list[tuple[int, str]] = (
|
||||
list(enumerate(encs[payload])) if isinstance(payload, str)
|
||||
else sorted(payload.items()))
|
||||
for font, name_value in items:
|
||||
if has_encoding and (base_encoding_name or font in differences):
|
||||
continue
|
||||
if not name_value:
|
||||
continue
|
||||
codepoint = _get_unicode_for_glyph(name_value, glyphs)
|
||||
if codepoint != -1:
|
||||
final[font] = _from_char_code(codepoint) # amend overwrites
|
||||
return 1, final
|
||||
@@ -0,0 +1,236 @@
|
||||
"""Transform matrices, text-object collection, and char-to-object mapping."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ctypes
|
||||
import math
|
||||
import pypdfium2.raw as pdfium_c
|
||||
|
||||
|
||||
def _obj_rotation(value: float, other_item: float, candidate_item: float, reference_item: float) -> int:
|
||||
"""Classify a text-object matrix as upright, cardinal rotation, or oblique. Near-cardinal matrices snap to the cardinal bucket; genuinely oblique matrices use the baseline remerge path."""
|
||||
x_scale = math.hypot(value, other_item)
|
||||
y_scale = math.hypot(candidate_item, reference_item)
|
||||
if x_scale < 1e-9 or y_scale < 1e-9:
|
||||
return 0
|
||||
eps = 1e-3
|
||||
if abs(other_item) < eps * x_scale and abs(candidate_item) < eps * y_scale:
|
||||
return 0 if value >= 0 else 180
|
||||
if abs(value) < eps * x_scale and abs(reference_item) < eps * y_scale:
|
||||
return 90 if other_item > 0 else 270
|
||||
return -1
|
||||
|
||||
|
||||
def _xf_point(items: tuple, other_item: float, candidate_item: float) -> tuple[float, float]:
|
||||
"""Apply an (a,b,c,d,e,f) PDF matrix to a point (row-vector convention)."""
|
||||
return (items[0] * other_item + items[2] * candidate_item + items[4], items[1] * other_item + items[3] * candidate_item + items[5])
|
||||
|
||||
|
||||
def _compose_mtx(first_matrix: tuple, second_matrix: tuple) -> tuple:
|
||||
"""Matrix product applying ``m1`` first, then ``m2``."""
|
||||
return (
|
||||
first_matrix[0] * second_matrix[0] + first_matrix[1] * second_matrix[2],
|
||||
first_matrix[0] * second_matrix[1] + first_matrix[1] * second_matrix[3],
|
||||
first_matrix[2] * second_matrix[0] + first_matrix[3] * second_matrix[2],
|
||||
first_matrix[2] * second_matrix[1] + first_matrix[3] * second_matrix[3],
|
||||
first_matrix[4] * second_matrix[0] + first_matrix[5] * second_matrix[2] + second_matrix[4],
|
||||
first_matrix[4] * second_matrix[1] + first_matrix[5] * second_matrix[3] + second_matrix[5],
|
||||
)
|
||||
|
||||
|
||||
_IDENT_MTX = (1.0, 0.0, 0.0, 1.0, 0.0, 0.0)
|
||||
|
||||
|
||||
def _collect_text_objs(page, text_page) -> list[dict]:
|
||||
"""Per-page list of (font_handle, fs_raw, matrix_scale_*, bbox, ...) for each text object. Used for bbox-containment lookup. Walks Form XObjects manually in stream order, composing each ancestor form's matrix. Without the composition a scaled or shifted chart's text objects land at the wrong page position and every chart glyph fails the bbox-containment lookup."""
|
||||
objects: list[dict] = []
|
||||
sz_field = ctypes.c_float(0)
|
||||
matrix = pdfium_c.FS_MATRIX()
|
||||
font_name_buffer = (ctypes.c_char * 256)()
|
||||
bounds_left = ctypes.c_float(0)
|
||||
value = ctypes.c_float(0)
|
||||
bounds_right = ctypes.c_float(0)
|
||||
bounds_top = ctypes.c_float(0)
|
||||
|
||||
def iter_text_objs(parent, anc_mtx, depth):
|
||||
"""Yield (raw_text_obj, ancestor_matrix) in stream order."""
|
||||
object_count = (pdfium_c.FPDFFormObj_CountObjects(parent) if parent is not None
|
||||
else pdfium_c.FPDFPage_CountObjects(page.raw))
|
||||
for text in range(object_count):
|
||||
raw = (pdfium_c.FPDFFormObj_GetObject(parent, text) if parent is not None
|
||||
else pdfium_c.FPDFPage_GetObject(page.raw, text))
|
||||
if not raw:
|
||||
continue
|
||||
typ = pdfium_c.FPDFPageObj_GetType(raw)
|
||||
if typ == pdfium_c.FPDF_PAGEOBJ_TEXT:
|
||||
yield raw, anc_mtx
|
||||
elif typ == pdfium_c.FPDF_PAGEOBJ_FORM and depth < 10:
|
||||
pdfium_c.FPDFPageObj_GetMatrix(raw, matrix)
|
||||
font_matrix = (matrix.a, matrix.b, matrix.c, matrix.d, matrix.e, matrix.f)
|
||||
yield from iter_text_objs(raw, _compose_mtx(font_matrix, anc_mtx), depth + 1)
|
||||
|
||||
for raw_obj, anc_mtx in iter_text_objs(None, _IDENT_MTX, 0):
|
||||
font = pdfium_c.FPDFTextObj_GetFont(raw_obj)
|
||||
if not font:
|
||||
continue
|
||||
pdfium_c.FPDFTextObj_GetFontSize(raw_obj, ctypes.byref(sz_field))
|
||||
fs_raw = sz_field.value
|
||||
pdfium_c.FPDFPageObj_GetMatrix(raw_obj, matrix)
|
||||
# Effective (page-space) matrix: the object's own matrix composed with
|
||||
# its ancestor forms' -- text extraction folds that ancestor chain into the text matrix.
|
||||
matrix_a, matrix_b, matrix_c, matrix_d, _, _ = _compose_mtx(
|
||||
(matrix.a, matrix.b, matrix.c, matrix.d, matrix.e, matrix.f), anc_mtx)
|
||||
scale_x = math.sqrt(matrix_a * matrix_a + matrix_b * matrix_b) or 1.0
|
||||
scale_y = math.sqrt(matrix_c * matrix_c + matrix_d * matrix_d) or 1.0
|
||||
|
||||
if not pdfium_c.FPDFPageObj_GetBounds(
|
||||
raw_obj, ctypes.byref(bounds_left), ctypes.byref(value),
|
||||
ctypes.byref(bounds_right), ctypes.byref(bounds_top)):
|
||||
continue
|
||||
# Bounds include the object's own matrix but not its ancestors'; map
|
||||
# the four corners into page space.
|
||||
x00, y00 = _xf_point(anc_mtx, bounds_left.value, value.value)
|
||||
x01, y01 = _xf_point(anc_mtx, bounds_left.value, bounds_top.value)
|
||||
x10, y10 = _xf_point(anc_mtx, bounds_right.value, value.value)
|
||||
x11, y11 = _xf_point(anc_mtx, bounds_right.value, bounds_top.value)
|
||||
object_left = min(x00, x01, x10, x11)
|
||||
object_right = max(x00, x01, x10, x11)
|
||||
text = min(y00, y01, y10, y11)
|
||||
object_top = max(y00, y01, y10, y11)
|
||||
ink_height = max(0.0, object_top - text)
|
||||
# text extraction folds Tfs (text font size) + FontMatrix into the text transform
|
||||
# so ``hypot(transform[2], transform[3])`` always gives the
|
||||
# rendered font size. PDFium splits these and doesn't fold non-identity
|
||||
# FontMatrix back. Rendered-font-size fallback chain:
|
||||
# raw >= 1.5 and scale > 0 -> raw * scale (normal text)
|
||||
# scale >= 1.5 -> scale (Type 3: raw=0.1, ctm=N)
|
||||
# raw >= 1.5 -> raw (no scale info)
|
||||
# else -> ink_h (Type 3 inside identity ctm)
|
||||
if anc_mtx is not _IDENT_MTX and fs_raw > 0 and scale_y > 0:
|
||||
# Inside a Form XObject, span merger font size = hypot(trm[2],trm[3])
|
||||
# with the form CTM folded in = Tfs * composed scale, exactly
|
||||
# (scaled vector-figure case: Tf 0.167 * 72 * form 0.5722 =
|
||||
# 6.88 == the heading heuristics' item height; the placeholder chain below
|
||||
# would misread it as Type-3-with-fs-in-ctm and emit 41pt boxes
|
||||
# that swallow the neighbouring "2.2" heading). The chain stays
|
||||
# for top-level objects, for top-level objects.
|
||||
fs_eff = fs_raw * scale_y
|
||||
elif fs_raw >= 1.5 and scale_y > 0:
|
||||
fs_eff = fs_raw * scale_y
|
||||
elif scale_y >= 1.5:
|
||||
fs_eff = scale_y
|
||||
elif fs_raw >= 1.5:
|
||||
fs_eff = fs_raw
|
||||
else:
|
||||
fs_eff = max(1.0, ink_height)
|
||||
# PDFium's FS_MATRIX is float32, so a size authored as 9.9pt arrives as
|
||||
# 9.89999962; text extraction parses the content stream in float64 and keeps 9.9.
|
||||
# Snap back to the shortest decimal so knife-edge font-size comparisons
|
||||
# match the content-stream value.
|
||||
fs_eff = float(f"{fs_eff:.6g}")
|
||||
name = pdfium_c.FPDFFont_GetFontName(font, font_name_buffer, 256)
|
||||
font_name = (
|
||||
bytes(font_name_buffer[:name]).decode("latin-1", errors="replace").rstrip("\x00")
|
||||
if name > 1 else ""
|
||||
)
|
||||
weight = int(pdfium_c.FPDFFont_GetWeight(font))
|
||||
|
||||
objects.append({
|
||||
"font": font,
|
||||
# Handle address as a hashable per-document font identity; computed
|
||||
# once here so per-char consumers never re-cast.
|
||||
"font_key": ctypes.cast(font, ctypes.c_void_p).value,
|
||||
"fs_raw": fs_raw,
|
||||
"scale_x": scale_x,
|
||||
"scale_y": scale_y,
|
||||
"fs_eff": fs_eff,
|
||||
"l": object_left, "r": object_right, "b": text, "t": object_top,
|
||||
"area": max(0.0, (object_right - object_left) * (object_top - text)),
|
||||
"font_name": font_name,
|
||||
"weight": weight,
|
||||
# Rotation class of this text object (0/90/180/270, or -1 oblique).
|
||||
# text extraction normalises it inside position comparison; the charlevel
|
||||
# merger is horizontal-only, so cardinal runs (rotated-sidebar sidebar stamp,
|
||||
# chart axis labels) shatter per-glyph and are re-merged by
|
||||
# _remerge_rotated; oblique objects go to _remerge_oblique (needs the
|
||||
# matrix below for the inverse-rotation projection baseline projection).
|
||||
"rot": _obj_rotation(matrix_a, matrix_b, matrix_c, matrix_d),
|
||||
"mtx": (matrix_a, matrix_b, matrix_c, matrix_d),
|
||||
# Paint (content-stream) order. text extraction emits items in stream order but
|
||||
# PDFium's textpage reorders vertical-writing chars page-wide, so
|
||||
# _remerge_vertical needs this to restore text extraction item order.
|
||||
"page_order": len(objects),
|
||||
# True iff this object's show-op used a vertical-CMap (-V / WMode 1)
|
||||
# font -- span merger vertical-font flag. Set by _assign_vertical_tags.
|
||||
"vertical": False,
|
||||
# Show-op text horizontal scale (Tz/100). text extraction keeps Tz OUT of the space
|
||||
# thresholds (base = raw font size) while PDFium folds it into the
|
||||
# object matrix (hence into fs_x); open_chunk divides it back out.
|
||||
# Set by _assign_show_tz via the same ordinal alignment as
|
||||
# ``vertical``; stays 1.0 on a count mismatch.
|
||||
"tz": 1.0,
|
||||
})
|
||||
return objects
|
||||
|
||||
|
||||
def _build_obj_index(objects: list[dict]) -> dict[int, list[dict]]:
|
||||
"""Bucket text objects by integer y so per-character lookup scans only nearby baselines. Each object is inserted into padded y-buckets that form a superset for the exact containment check."""
|
||||
index: dict[int, list[dict]] = {}
|
||||
for item_value in objects:
|
||||
lower_bound = int(math.floor(item_value["b"])) - 6
|
||||
upper_bound = int(math.ceil(item_value["t"])) + 6
|
||||
for text_key in range(lower_bound, upper_bound + 1):
|
||||
index.setdefault(text_key, []).append(item_value)
|
||||
return index
|
||||
|
||||
|
||||
def _char_render_fs(text_page, char_idx: int) -> float:
|
||||
"""True per-char rendered size: ``FPDFText_GetMatrix`` folds Tfs and FontMatrix into the rendered text matrix, so ``sqrt(c^2+d^2)`` is the text-item height. Returns 0.0 when the call is unavailable. Read lazily, only when a char is contained by more than one object, since the FFI call is expensive and most chars have a single, unambiguous host object."""
|
||||
current_matrix = pdfium_c.FS_MATRIX()
|
||||
if pdfium_c.FPDFText_GetMatrix(text_page, char_idx, ctypes.byref(current_matrix)):
|
||||
return math.sqrt(current_matrix.c * current_matrix.c + current_matrix.d * current_matrix.d)
|
||||
return 0.0
|
||||
|
||||
|
||||
def _find_obj_for_char(
|
||||
obj_index: dict[int, list[dict]], query_origin_x: float, query_origin_y: float, tol: float = 1.0,
|
||||
char_fs: float | None = None, text_page=None, char_idx: int | None = None,
|
||||
) -> dict | None:
|
||||
"""Bbox containment lookup. When a char falls inside more than one text object, pick the candidate whose effective rendered size matches the char's true per-char matrix size from ``FPDFText_GetMatrix``. That folds Tfs and FontMatrix into the same glyph-to-font attribution used by the text-item reconstruction. This disambiguates overlapping objects such as a large figure-axis label drawn over a smaller heading, and avoids selecting tiny ghost objects that share the same raw textpage font size. Falls back to the PDFium ``fs_raw`` textpage font size and finally to smallest area."""
|
||||
first: dict | None = None
|
||||
cands: list[dict] | None = None
|
||||
for item_value in obj_index.get(int(round(query_origin_y)), ()):
|
||||
if (item_value["l"] - tol) <= query_origin_x <= (item_value["r"] + tol) and\
|
||||
(item_value["b"] - tol) <= query_origin_y <= (item_value["t"] + tol):
|
||||
if first is None:
|
||||
first = item_value
|
||||
elif cands is None:
|
||||
cands = [first, item_value]
|
||||
else:
|
||||
cands.append(item_value)
|
||||
if first is None:
|
||||
return None
|
||||
if cands is None:
|
||||
return first
|
||||
char_render = (
|
||||
_char_render_fs(text_page, char_idx) if text_page is not None and char_idx is not None
|
||||
else 0.0
|
||||
)
|
||||
if char_render > 0:
|
||||
# Match the per-char rendered size (== text extraction font size); area tiebreak.
|
||||
return min(
|
||||
cands,
|
||||
key=lambda item_value: (abs(item_value["fs_eff"] - char_render), item_value["area"]),
|
||||
)
|
||||
if char_fs is None and text_page is not None and char_idx is not None:
|
||||
# Deferred FPDFText_GetFontSize: only this rare branch (multi-candidate
|
||||
# AND no per-char matrix) consumes it, so the caller no longer pays the
|
||||
# FFI call on every char.
|
||||
char_fs = pdfium_c.FPDFText_GetFontSize(text_page, char_idx)
|
||||
if char_fs is not None and char_fs > 0:
|
||||
# Sort by absolute fs diff first, then smallest area as tiebreak.
|
||||
return min(
|
||||
cands,
|
||||
key=lambda item_value: (abs(item_value["fs_raw"] - char_fs) / max(char_fs, 0.01), item_value["area"]),
|
||||
)
|
||||
return min(cands, key=lambda item_value: item_value["area"])
|
||||
@@ -0,0 +1,82 @@
|
||||
"""Bundled glyph-name and encoding tables with cached lazy loading."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from .cmap_parse import _parse_int
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Font Unicode-map construction.
|
||||
#
|
||||
# PDFium's per-character Unicode can diverge when a simple font's ToUnicode CMap
|
||||
# is missing or incomplete. The repair path resolves the charcode through the
|
||||
# font's encoding (dictionary /Encoding BaseEncoding+Differences, or an embedded
|
||||
# Type1 program's builtin encoding) to a glyph name, maps that name through the
|
||||
# bundled glyph table, and otherwise falls back to the raw charcode. The map is
|
||||
# rebuilt from the PDF's own font dictionaries via the PyPDF2 xref channel
|
||||
# (font metadata only, no text decode), then applied where PDFium's output
|
||||
# disagrees.
|
||||
#
|
||||
# Covered rules: encoding and Differences extraction, simple-font Unicode-map
|
||||
# construction, predefined collection Unicode-map construction, ToUnicode
|
||||
# parsing, fallback Unicode-map repair, Type 1 Unicode-map repair, and glyph
|
||||
# mapping as the included ToUnicode value when present, otherwise the raw charcode.
|
||||
|
||||
# /Encoding extraction from an embedded Type1 file
|
||||
# Glyph-name Unicode lookup
|
||||
# glyph names and standard encodings are bundled in data/glyph_name_table.json
|
||||
# (kept deliberately conservative
|
||||
#
|
||||
# Boundaries (documented, all conservative -- no map entry means no patch):
|
||||
# - composite (Type0) fonts: separate path, never patched here;
|
||||
# - CFF (FontFile3) builtin encodings: not parsed here; dict-encoding-based
|
||||
# mapping still applies;
|
||||
# - symbolic-TrueType WinAnsi inference (content stream tokenizer TrueType Unicode-map repair):
|
||||
# needs the TTF name records, not implemented.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_GLYPHLIST_PATH = Path(__file__).parent.parent / "data" / "glyph_name_table.json"
|
||||
_cached_glyphs: dict[str, int] | None = None
|
||||
_cached_encodings: dict[str, list[str]] | None = None
|
||||
|
||||
|
||||
def _load_glyph_tables() -> tuple[dict[str, int], dict[str, list[str]]]:
|
||||
global _cached_glyphs, _cached_encodings
|
||||
glyphs, encodings = _cached_glyphs, _cached_encodings
|
||||
if glyphs is None or encodings is None:
|
||||
data = json.loads(_GLYPHLIST_PATH.read_text(encoding="utf-8"))
|
||||
glyphs = _cached_glyphs = data["glyphs"]
|
||||
encodings = _cached_encodings = data["encodings"]
|
||||
return glyphs, encodings
|
||||
|
||||
|
||||
def _get_unicode_for_glyph(name: str, glyphs: dict[str, int]) -> int:
|
||||
"""Resolve a glyph name through glyphlist lookup and uppercase-hex recovery patterns."""
|
||||
codepoint = glyphs.get(name)
|
||||
if codepoint is not None:
|
||||
return codepoint
|
||||
if not name:
|
||||
return -1
|
||||
if name[0] == "u":
|
||||
glyph_name_length = len(name)
|
||||
if glyph_name_length == 7 and name[1] == "n" and name[2] == "i":
|
||||
hex_str = name[3:]
|
||||
elif 5 <= glyph_name_length <= 7:
|
||||
hex_str = name[1:]
|
||||
else:
|
||||
return -1
|
||||
if hex_str == hex_str.upper():
|
||||
# Tolerant base-16 parsing trims Unicode whitespace and accepts an
|
||||
# optional sign / 0X prefix. NaN fails the >= 0 gate; "-0" passes it.
|
||||
u16 = _parse_int(hex_str, 16)
|
||||
if u16 >= 0:
|
||||
return int(u16)
|
||||
return -1
|
||||
|
||||
|
||||
def _from_char_code(number: int) -> str:
|
||||
"""Return the UTF-16 code unit after ToUint16 truncation."""
|
||||
return chr(number & 0xFFFF)
|
||||
@@ -0,0 +1,526 @@
|
||||
"""Joins page glyphs into text runs with spacing and style thresholds."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .text_normalize import (
|
||||
TRACKING_SPACE_FACTOR,
|
||||
NON_SPACE_GAP_FACTOR,
|
||||
NEGATIVE_SPACE_FACTOR,
|
||||
SPACE_IN_FLOW_MIN_FACTOR,
|
||||
SPACE_IN_FLOW_MAX_FACTOR,
|
||||
_rtl_sign,
|
||||
_read_end,
|
||||
_read_gap,
|
||||
)
|
||||
from .char_extract import _off_page
|
||||
|
||||
|
||||
def _merge_text_items(chars: list[dict], view_box=None) -> list[dict]:
|
||||
"""exact text extraction position comparison + synthetic-space insertion + last-character buffer."""
|
||||
items: list[dict] = []
|
||||
chunk: dict | None = None
|
||||
two_last = [" ", " "]
|
||||
two_last_pos = [0]
|
||||
# text extraction active text item.previous glyph transform: set only by a glyph with a real
|
||||
# advance (`if (scaled advance)`), NEVER reset by text-item flush/setFont
|
||||
# -- it survives across item flushes for the whole page. (None, None)
|
||||
# until the first real glyph.
|
||||
last_ref: tuple = (None, None)
|
||||
|
||||
def _object_merge_id(mapping: dict):
|
||||
# Per-object merge id: the boundary test below hard-splits between
|
||||
# different objects (the per-object split). q/Q grouping would require
|
||||
# fragile object/show-op ordinal alignment, so this is just the object's
|
||||
# identity.
|
||||
return id(mapping["obj"])
|
||||
|
||||
def reset_last_chars() -> None:
|
||||
two_last[0] = " "
|
||||
two_last[1] = " "
|
||||
two_last_pos[0] = 0
|
||||
|
||||
def save_last_char(char: str) -> bool:
|
||||
next_pos = (two_last_pos[0] + 1) % 2
|
||||
ret = (two_last[two_last_pos[0]] != " " and two_last[next_pos] == " ")
|
||||
two_last[two_last_pos[0]] = char
|
||||
two_last_pos[0] = next_pos
|
||||
return ret
|
||||
|
||||
def flush() -> None:
|
||||
nonlocal chunk
|
||||
if chunk is not None and chunk["str"]:
|
||||
items.append(chunk)
|
||||
chunk = None
|
||||
|
||||
def open_chunk(mapping: dict) -> None:
|
||||
nonlocal chunk, last_ref
|
||||
sign = _rtl_sign(mapping["ch"])
|
||||
# |text horizontal scale|: fs_x = |matrix scale| carries |Tz|, so the divisor is
|
||||
# the magnitude (a negative Tz uses the same text but scales it by |Tz|).
|
||||
# Tz == 0 keeps 1.0 (moot: PDFium emits no textpage chars for a
|
||||
# degenerate x-column). Boundary conditions:
|
||||
# PDFium's SYNTHESIZED layout spaces derive from the unscaled
|
||||
# text-space gap (~0.135em), so compressed Tz < ~75 can inject
|
||||
# spaces text extraction would not; negative-Tz runs re-merge via the
|
||||
# 180-degree pass with different item structure than span merger'
|
||||
# text orientation=-1 model; anisotropic CTM x rotated Tm differs
|
||||
# (norm-of-product vs span merger product-of-norms text advance scale).
|
||||
horizontal_scale_factor = abs(mapping["obj"].get("tz", 1.0))
|
||||
if not (horizontal_scale_factor > 0):
|
||||
horizontal_scale_factor = 1.0
|
||||
fs_x_tz = mapping["fs_x"] / horizontal_scale_factor
|
||||
chunk = {
|
||||
"str": [],
|
||||
"sign": sign, # +1 LTR, -1 RTL (signed x-axis)
|
||||
"obj": mapping["obj"], # host text object (Tj/show-text)
|
||||
# text extraction fixes item transform at the item's FIRST glyph
|
||||
# (item initialization) and never updates it mid-item, while
|
||||
# chunk["obj"] re-points to the LAST appended glyph's object (the
|
||||
# flush/prose bookkeeping needs that). Snapshot the opening
|
||||
# object's matrix so the emitted skew reads first-glyph geometry.
|
||||
"mtx0": mapping["obj"]["mtx"],
|
||||
"flush_id": _object_merge_id(mapping), # per-object merge id (was q/Q flush scope)
|
||||
"left": mapping["left"], "right": mapping["right"],
|
||||
"top": mapping["top"], "bottom": mapping["bottom"],
|
||||
"fs": mapping["fs"],
|
||||
"fs_min": mapping["fs"],
|
||||
# Glyph advance (FPDFFont_GetGlyphWidth*scale). Unused by the
|
||||
# horizontal merger (it reads prev_text_x); carried only so
|
||||
# _remerge_rotated can run a direct 1-D position comparison
|
||||
# (gap = next_origin - (cur_origin + glyph_w)) along the rotation axis.
|
||||
"glyph_w": mapping.get("glyph_w", 0.0),
|
||||
"font_name": mapping["font_name"],
|
||||
"font_key": mapping["font_key"],
|
||||
"weight": mapping["weight"],
|
||||
# Per-char style tallies for majority-vote at span emission.
|
||||
# span merger text item records only the first char's font name;
|
||||
# the heading heuristics' heading detection ends up marking paragraph
|
||||
# lead-ins like **Bold prefix.** Regular continuation as
|
||||
# "bold lines" because of that. Tally per-char so we can
|
||||
# emit the dominant font/weight instead.
|
||||
"font_tally": {mapping["font_name"]: 1},
|
||||
"weight_tally": {mapping["weight"]: 1},
|
||||
# ``prev_text_x`` tracks where the next glyph would land if
|
||||
# charSpacing=0 — i.e. text matrix.e after this glyph's emit.
|
||||
# For ligature components, PDFium reports them at the same
|
||||
# origin but with bbox spanning the full ligature, so taking
|
||||
# max(ox+glyph_w, bbox.right) makes the next non-ligature
|
||||
# char see a small positive advance instead of a big gap.
|
||||
# (_read_end uses the same this for an RTL chunk.)
|
||||
"prev_text_x": _read_end(mapping, sign),
|
||||
"prev_oy": mapping["oy"],
|
||||
# text extraction threshold base is text state.font size WITHOUT Tz
|
||||
# (item initialization: Tz enters only the pen advance, not
|
||||
# text advance scale). PDFium folds Tz into the object matrix, so
|
||||
# fs_x carries it; divide the show-op's text horizontal scale back out.
|
||||
"tracking": fs_x_tz * TRACKING_SPACE_FACTOR,
|
||||
"not_a_space": fs_x_tz * NON_SPACE_GAP_FACTOR,
|
||||
"negative": fs_x_tz * NEGATIVE_SPACE_FACTOR,
|
||||
"flow_min": fs_x_tz * SPACE_IN_FLOW_MIN_FACTOR,
|
||||
"flow_max": fs_x_tz * SPACE_IN_FLOW_MAX_FACTOR,
|
||||
"height": mapping["fs"],
|
||||
# True once a real whitespace glyph follows the last visible glyph in
|
||||
# this chunk; gates whether an object boundary is a prose word-break
|
||||
# (merge) or a layout jump (hard split). See the is_ws handler.
|
||||
"ws_pending": False,
|
||||
}
|
||||
if "v_pen_y" in mapping:
|
||||
# Vertical-writing pen state for _remerge_vertical (set only for
|
||||
# vertical-CMap objects).
|
||||
chunk["v_pen_x"] = mapping["v_pen_x"]
|
||||
chunk["v_pen_y"] = mapping["v_pen_y"]
|
||||
chunk["v_after"] = mapping["v_after"]
|
||||
chunk["v_last_x"] = mapping["v_pen_x"]
|
||||
if mapping["is_mn"]:
|
||||
# A zero-width diacritic has scaled advance == 0, so it does NOT
|
||||
# establish the advance reference; the chunk INHERITS the
|
||||
# page-surviving one (text extraction previous glyph transform persists across
|
||||
# flushes; (None, None) only until the page's first real glyph).
|
||||
chunk["prev_text_x"], chunk["prev_oy"] = last_ref
|
||||
else:
|
||||
last_ref = (chunk["prev_text_x"], chunk["prev_oy"])
|
||||
|
||||
def emit_fake_space(gap: float) -> None:
|
||||
"""Emit an out-of-flow synthetic space after the current chunk. The synthetic item uses the previous glyph transform, not the next glyph, so the space remains attached to the line it trails. Its height stays zero; otherwise vertical-alignment checks can attach the space to a neighboring line and create a spurious line merge. """
|
||||
assert chunk is not None
|
||||
reset_last_chars() # text extraction synthetic-space insertion standalone path resets first
|
||||
page_x, baseline = chunk["prev_text_x"], chunk["prev_oy"]
|
||||
# WIDTH = abs(gap). Synthetic out-of-flow spaces use ``width: abs(e)``,
|
||||
# where e is the out-of-flow advance (the gap it
|
||||
# spans), height 0 for horizontal text. The box therefore runs from the
|
||||
# previous glyph's pen end (px) forward by the gap: [px, px+gap] LTR,
|
||||
# [px-gap, px] RTL. In-flow spaces are handled by pushing " " into the
|
||||
# current item; standalone spaces use this separate geometry. Height
|
||||
# stays 0, so vertical-alignment guards are
|
||||
# untouched and the outline is unaffected.
|
||||
width_value = abs(gap)
|
||||
if chunk["sign"] >= 0:
|
||||
sp_left, sp_right = page_x, page_x + width_value
|
||||
else:
|
||||
sp_left, sp_right = page_x - width_value, page_x
|
||||
meta = (chunk["obj"], chunk["fs"], chunk["font_name"],
|
||||
chunk["font_key"], chunk["weight"])
|
||||
flush()
|
||||
items.append({
|
||||
"str": [" "], "sign": 1, "obj": meta[0],
|
||||
"left": sp_left, "right": sp_right,
|
||||
"top": baseline, "bottom": baseline, # HEIGHT 0
|
||||
"fs": meta[1], "fs_min": meta[1],
|
||||
"font_name": meta[2], "font_key": meta[3], "weight": meta[4],
|
||||
"font_tally": {meta[2]: 1}, "weight_tally": {meta[4]: 1},
|
||||
})
|
||||
|
||||
def extend_chunk(mapping: dict, leading_space: bool) -> None:
|
||||
nonlocal last_ref
|
||||
assert chunk is not None
|
||||
if leading_space:
|
||||
chunk["str"].append(" ")
|
||||
chunk["str"].append(mapping["ch"])
|
||||
chunk["left"] = min(chunk["left"], mapping["left"])
|
||||
chunk["right"] = max(chunk["right"], mapping["right"])
|
||||
# Text-item box accumulation: appending a glyph only grows the item's
|
||||
# width. The item's vertical box is
|
||||
# fixed at item creation -- transform[5] = first-glyph baseline, and
|
||||
# height = font size (== the item's em). A per-glyph baseline offset
|
||||
# within the item (e.g. a lowered character inside a mixed-baseline
|
||||
# logo, or any sub/superscript not split into its own item) is therefore
|
||||
# absorbed: it does NOT extend the item box. Do not expand top/bottom
|
||||
# here; they stay at the open glyph's [oy, oy+fs]. Expanding them would
|
||||
# let inline baseline offsets distort downstream line-height gates.
|
||||
chunk["prev_text_x"] = _read_end(mapping, chunk["sign"])
|
||||
chunk["prev_oy"] = mapping["oy"]
|
||||
last_ref = (chunk["prev_text_x"], chunk["prev_oy"])
|
||||
chunk["glyph_w"] = mapping.get("glyph_w", 0.0) # last glyph's advance (for _remerge_rotated)
|
||||
# When a real-space word-break us merge across a text-object boundary
|
||||
# (prose case), the chunk must adopt the new object so the rest of that
|
||||
# word's glyphs (same object, no space before them) don't re-trigger the
|
||||
# object hard-split mid-word. span merger line item has no per-glyph object.
|
||||
chunk["obj"] = mapping["obj"]
|
||||
chunk["flush_id"] = _object_merge_id(mapping)
|
||||
chunk["font_tally"][mapping["font_name"]] = chunk["font_tally"].get(mapping["font_name"], 0) + 1
|
||||
chunk["weight_tally"][mapping["weight"]] = chunk["weight_tally"].get(mapping["weight"], 0) + 1
|
||||
# Track min fs within the chunk so small-caps headings ("A"+
|
||||
# "BSTRACT") expose the body-text fs of the small-cap part
|
||||
# rather than the leading full-cap fs. The downstream big-font
|
||||
# check then doesn't false-positive on inline math labels like
|
||||
# "LEMMA 1" whose small-cap fs is below body size.
|
||||
if mapping["fs"] > 0:
|
||||
chunk["fs_min"] = min(chunk["fs_min"], mapping["fs"])
|
||||
if "v_pen_y" in mapping and "v_pen_y" in chunk:
|
||||
chunk["v_after"] = mapping["v_after"]
|
||||
chunk["v_last_x"] = mapping["v_pen_x"]
|
||||
|
||||
for text in chars:
|
||||
# text extraction text-item box accumulation char loop order (span merger+):
|
||||
# invisible format-mark classification is skipped entirely BEFORE the whitespace test.
|
||||
if text["is_cf"]:
|
||||
# The format-mark skip sits ahead of the scaled-advance and
|
||||
# char-spacing block, so the mark moves neither the text matrix nor
|
||||
# the previous-position reference: the reference pen never sees it.
|
||||
# PDFium's char origins DO include its advance, so carry the
|
||||
# reading-direction reference past it; otherwise that advance
|
||||
# reappears as a gap and the next glyph gets an in-flow or
|
||||
# standalone " " with no counterpart. (Char spacing, also skipped
|
||||
# here, is not separable from PDFium's origins.) With no chunk open
|
||||
# the page's previous-position reference is still unset, so the
|
||||
# position comparison is unconditionally true and there is no gap
|
||||
# to correct.
|
||||
if chunk is not None and chunk["prev_text_x"] is not None:
|
||||
chunk["prev_text_x"] += chunk["sign"] * text.get("glyph_w", 0.0)
|
||||
last_ref = (chunk["prev_text_x"], chunk["prev_oy"])
|
||||
continue
|
||||
if text["is_ws"]:
|
||||
save_last_char(" ")
|
||||
# Remember a real whitespace glyph bridged the gap. PDFium fragments a
|
||||
# flowing prose line into per-word text-objects (each with a trailing
|
||||
# space glyph); text extraction keeps the whole line as one Tj item. A real space
|
||||
# at an object boundary marks a prose word-break -> merge across it.
|
||||
# A positional (spaceless) object change marks a layout jump (table
|
||||
# cell, separate Tj) -> keep the hard object split.
|
||||
if chunk is not None:
|
||||
chunk["ws_pending"] = True
|
||||
continue
|
||||
# Zero-width diacritics append without a position-based flush: they do NOT call
|
||||
# position comparison (no position-based flush) and uses
|
||||
# scaled advance=0 (no advance) -- it just appends the mark to the
|
||||
# current item. Append it without touching prev_text_x.
|
||||
# BUT a Tf style flush is an operator-level split that already closed
|
||||
# the previous item before the glyph loop ran, so a mark arriving in a
|
||||
# different font/size (for example, a math accent over an italic letter,
|
||||
# letter, each its own Tf'd show op) opens its OWN item, with its own
|
||||
# raised origin and em height. Only
|
||||
# the position-based flush is skipped for diacritics, never the style
|
||||
# flush, so the mark passes the same font_key/fs/object boundary test
|
||||
# as any visible glyph.
|
||||
if text["is_mn"]:
|
||||
if chunk is not None and (
|
||||
chunk["font_key"] != text["font_key"]
|
||||
or abs(text["fs"] - chunk["fs"]) > 1e-6
|
||||
or (_object_merge_id(text) != chunk["flush_id"] and not chunk["ws_pending"])
|
||||
):
|
||||
flush()
|
||||
if chunk is None:
|
||||
# open_chunk inherits the page-surviving advance reference
|
||||
# (text extraction previous glyph transform persists across flushes; None only at
|
||||
# page start -- see the prev_text_x-is-None guard below).
|
||||
open_chunk(text)
|
||||
assert chunk is not None
|
||||
lead = save_last_char(text["ch"])
|
||||
# the heading heuristics pushes the last-character buffer lead into the fresh mark item
|
||||
# (a Tf flush does not reset the last-character buffer).
|
||||
if lead:
|
||||
chunk["str"].append(" ")
|
||||
chunk["str"].append(text["ch"])
|
||||
else:
|
||||
lead = save_last_char(text["ch"])
|
||||
if lead:
|
||||
chunk["str"].append(" ")
|
||||
chunk["str"].append(text["ch"])
|
||||
# span merger: a zero-width diacritic has scaled advance=0, so it neither
|
||||
# moves the text matrix NOR updates previous glyph transform (span merger
|
||||
# `if (scaled advance)` is false). The next glyph's line-break /
|
||||
# dy test therefore compares against the last VISIBLE glyph's
|
||||
# baseline -> leave BOTH prev_text_x and prev_oy untouched here
|
||||
# (the box also stays the open glyph's -- see extend_chunk note).
|
||||
continue
|
||||
|
||||
# span merger: a non-diacritic glyph whose origin is off the page view box is
|
||||
# skipped (position comparison returns false only off-page). cf/ws
|
||||
# were handled above and diacritics (is_mn) never reach here, matching
|
||||
# span merger `!zero-width diacritic classification and !position comparison`.
|
||||
|
||||
if _off_page(text, view_box):
|
||||
continue
|
||||
|
||||
if chunk is None:
|
||||
open_chunk(text)
|
||||
save_last_char(text["ch"])
|
||||
assert chunk is not None
|
||||
chunk["str"].append(text["ch"])
|
||||
continue
|
||||
|
||||
# Style boundary: split on font-identity change (font_key, the PDFium
|
||||
# font handle == span merger per-font loaded font identity) OR ANY font size change.
|
||||
# span merger emit a separate text item on every setFont
|
||||
# (Tf) operator -- i.e. on any font OR size change. Represent
|
||||
|
||||
# that with an exact effective-fs compare (the 1e-6 is only to absorb
|
||||
# float noise in the snapped fs). An earlier 10% tolerance under-split
|
||||
# small-caps runs; exact is intentional here.
|
||||
|
||||
# font_key/fs is a proxy for the Tf flush, not a literal replay of every
|
||||
# content-stream flush boundary. Text-item boundaries are driven mostly
|
||||
# by position comparison; PDFium exposes final glyph coordinates, so the
|
||||
# per-object + font_key/fs proxy gives the heading pipeline the intended
|
||||
# span structure without overfitting to partial operator state. The
|
||||
# remaining boundary cases, such as missing-glyph fallback handles or
|
||||
# rendered-size jitter under scaled Type-3 matrices, are limited to span
|
||||
# boundaries.
|
||||
# The heading heuristics' downstream heading detector then
|
||||
# treats a chunk's first-char style as the whole chunk's style:
|
||||
#
|
||||
# * font split: a paragraph lead-in like "**Inflation.**
|
||||
# Consumer price..." would otherwise be a single bold chunk
|
||||
# and false-detect as a heading on every paragraph. (font_name
|
||||
# alone can't separate identity-matrix Type-3 fonts, whose names
|
||||
# are all empty, so a 12pt body run and an inline 11pt code word
|
||||
# would merge and collapse to the smaller fs_min.)
|
||||
# * fs split: inline math labels like "LEMMA 1 (...) ..."
|
||||
# (first-cap large + small-cap rest + body) would otherwise
|
||||
# merge into a single chunk that pipeline accepts as a
|
||||
# heading; splitting forces the small-cap rest into its own
|
||||
# chunk where the heading heuristics' short-text/type checks reject it. Same
|
||||
# guard also helps math-heavy page/identity-matrix Type-3 sample exercise items
|
||||
# ("X.Y www") and section headings stay detectable —
|
||||
# without it they collapse into the surrounding body chunk.
|
||||
#
|
||||
# Trade-off: small-caps "ABSTRACT" / "ECONOMIC ANALYSIS"
|
||||
# don't merge across the cap-to-small-cap fs step. The
|
||||
# downstream tokenizer relaxation (LineTokenizer.add_line below)
|
||||
# joins them at token level instead.
|
||||
ws_bridge = chunk["ws_pending"]
|
||||
chunk["ws_pending"] = False
|
||||
if chunk["prev_text_x"] is None:
|
||||
# The item was opened by a zero-width diacritic AT PAGE START (no
|
||||
# real glyph has set the page's advance reference yet, so span merger'
|
||||
# previous glyph transform is still null): position comparison returns
|
||||
# true unconditionally -- no positional boundary, no fake space,
|
||||
# no line break. Only the style flush (the Tf proxy) still
|
||||
# applies; otherwise the glyph appends plainly and, being a real
|
||||
# advance, establishes the reference via extend_chunk.
|
||||
if (chunk["font_key"] != text["font_key"]
|
||||
or abs(text["fs"] - chunk["fs"]) > 1e-6):
|
||||
flush()
|
||||
open_chunk(text)
|
||||
lead = save_last_char(text["ch"])
|
||||
assert chunk is not None
|
||||
if lead: # Detail: no reset on this path, the lead survives
|
||||
|
||||
chunk["str"].append(" ")
|
||||
chunk["str"].append(text["ch"])
|
||||
else:
|
||||
extend_chunk(text, save_last_char(text["ch"]))
|
||||
continue
|
||||
if (
|
||||
chunk["font_key"] != text["font_key"]
|
||||
or abs(text["fs"] - chunk["fs"]) > 1e-6
|
||||
# Hard-split at a text-object boundary -- but ONLY when no real
|
||||
# whitespace glyph bridged it. the object merge id groups consecutive per-glyph show operators
|
||||
# objects (PDFium emits one FPDF_PAGEOBJ_TEXT per glyph when the PDF
|
||||
# draws glyphs individually) into one id, so a CJK title set as N
|
||||
# per-glyph Tj does NOT shatter into N single-glyph items -- the
|
||||
# positional logic below merges it / line-breaks it like the item merger.
|
||||
# Every normal (multi-glyph) object keeps its own id, so this stays
|
||||
# the per-object split for Latin text: on dense justified tables (2023
|
||||
# dense table document) each fragment is its own object -> hard split, exact
|
||||
# with the item merger. On flowing prose PDFium may split
|
||||
# per word with a real space glyph between words, where text extraction keeps
|
||||
# the whole line as one item; a real space at the boundary (ws_bridge)
|
||||
# marks the prose case -> fall through to the in-flow/out-of-flow gap
|
||||
# logic, which merges the word-objects into one line item like span merger.
|
||||
or (_object_merge_id(text) != chunk["flush_id"] and not ws_bridge)
|
||||
):
|
||||
# span merger position comparison runs synthetic-space insertion for EVERY glyph,
|
||||
# including the first glyph of a new item/Tj. So an out-of-flow gap
|
||||
# across an item boundary still gets a standalone height-0 " "
|
||||
# (this is the trailing space after math-heavy page's "...y)" before the next
|
||||
# equation-number object). An in-flow / adjacent boundary does not.
|
||||
# text extraction position comparison order: a line break (|advance-y| >
|
||||
# height -> line-break emission) or backward jump takes precedence over the
|
||||
# space logic; only a same-line gap past tracking-space threshold emits the
|
||||
# standalone " " (covering BOTH the in-flow empty-item case and
|
||||
# the out-of-flow synthetic-space insertion case, which are identical here).
|
||||
boundary_gap = _read_gap(chunk["prev_text_x"], text, chunk["sign"])
|
||||
_same_line = abs(text["oy"] - chunk["prev_oy"]) <= chunk["height"]
|
||||
if _same_line and boundary_gap > chunk["tracking"]:
|
||||
emit_fake_space(boundary_gap)
|
||||
keep_lead = False # the heading heuristics synthetic-space insertion reset the last-character buffer
|
||||
else:
|
||||
# the heading heuristics resets the two-char buffer on every positional branch
|
||||
# (line-break emission / negative / non-space) but NOT in the tracking
|
||||
# window (non-space, tracking-space threshold] -- a thin real space
|
||||
# just before a Tf-style flush survives into the new item.
|
||||
keep_lead = (_same_line
|
||||
and chunk["not_a_space"] < boundary_gap <= chunk["tracking"])
|
||||
flush()
|
||||
open_chunk(text)
|
||||
lead = save_last_char(text["ch"])
|
||||
assert chunk is not None
|
||||
if lead and keep_lead:
|
||||
chunk["str"].append(" ")
|
||||
chunk["str"].append(text["ch"])
|
||||
continue
|
||||
|
||||
advance = _read_gap(chunk["prev_text_x"], text, chunk["sign"])
|
||||
line_delta_y = text["oy"] - chunk["prev_oy"]
|
||||
height = chunk["height"]
|
||||
|
||||
# Ligature decomposition: PDFium reports consecutive ligature
|
||||
# components at the same x origin (e.g. "fi" -> 'f' and 'i' at the
|
||||
# identical origin). prev_text_x was set to prev.ox +
|
||||
# prev.glyph_w, so we see advance ≈ -prev.glyph_w. Glyph widths
|
||||
# of typical Latin chars are in [0.2*fs, 0.9*fs]. Detect this
|
||||
# case (negative advance whose magnitude is in that range) and
|
||||
# silently merge — matches span merger on same-origin
|
||||
# ligature components. ONLY within one text object: decomposition
|
||||
# is per-glyph, so both components always share the show op. A
|
||||
# cross-object negative advance is a real content-stream back-jump
|
||||
# that text extraction itself sees and breaks on (TeX standalone accents:
|
||||
# math-heavy page 'accented name stem'+'¨'+'lkopf' is three show ops, '¨' jumps back -0.41fs;
|
||||
# text extraction raw-categorizes U+00A8 as a normal glyph -- category comes
|
||||
# from glyph Unicode BEFORE the normalized Unicode expansion -- so
|
||||
# position comparison flushes and ' ̈lkopf' opens a new item).
|
||||
if (text["obj"] is chunk["obj"] and abs(line_delta_y) < 0.1 * height
|
||||
and -0.9 * chunk["fs"] <= advance < -0.2 * chunk["fs"]):
|
||||
lead = save_last_char(text["ch"])
|
||||
# Don't extend prev_text_x backwards; ligature component
|
||||
# shares position with prev, so prev_text_x stays the same.
|
||||
if lead:
|
||||
chunk["str"].append(" ")
|
||||
chunk["str"].append(text["ch"])
|
||||
chunk["left"] = min(chunk["left"], text["left"])
|
||||
chunk["right"] = max(chunk["right"], text["right"])
|
||||
# Ligature component shares the open glyph's item box; only width
|
||||
# grows (see extend_chunk note -- text extraction never expands the item's
|
||||
# vertical extent on append).
|
||||
chunk["prev_oy"] = text["oy"]
|
||||
last_ref = (chunk["prev_text_x"], chunk["prev_oy"])
|
||||
continue
|
||||
|
||||
# text extraction position comparison compares advance-x against
|
||||
# ``text orientation * threshold`` (text orientation = sign(item.width)).
|
||||
# We get the same result for HORIZONTAL text by normalising the gap into
|
||||
# READING DIRECTION up front: ``advance`` (= _read_gap with chunk["sign"]
|
||||
# from _rtl_sign) is already signed so that "forward" is positive for BOTH
|
||||
# LTR and RTL, hence the thresholds below are compared UNMULTIPLIED.
|
||||
# * LTR (sign=+1): intentional to the literal layout-classifier form.
|
||||
|
||||
# * horizontal RTL (Hebrew/Arabic, sign=-1): handled via the x-axis
|
||||
# paired logic in _read_end/_read_gap (added in 528f958).
|
||||
# VERTICAL text (span merger vertical-font flag / advance-y branch) is NOT handled
|
||||
# here: a vertical column shatters per-glyph below and is re-merged by
|
||||
# the gated _remerge_vertical post-pass (detection: vertical-CMap fonts
|
||||
# via _page_vertical_resnames; matched against the item merger's
|
||||
# vertical-text rules.
|
||||
if advance < chunk["negative"]:
|
||||
if abs(line_delta_y) > 0.5 * height:
|
||||
# the heading heuristics line-break emission calls reset the last-character buffer before flushing.
|
||||
reset_last_chars()
|
||||
flush()
|
||||
open_chunk(text)
|
||||
save_last_char(text["ch"])
|
||||
assert chunk is not None
|
||||
chunk["str"].append(text["ch"])
|
||||
else:
|
||||
reset_last_chars()
|
||||
flush()
|
||||
open_chunk(text)
|
||||
save_last_char(text["ch"])
|
||||
assert chunk is not None
|
||||
chunk["str"].append(text["ch"])
|
||||
continue
|
||||
|
||||
if abs(line_delta_y) > height:
|
||||
# the heading heuristics line-break emission calls reset the last-character buffer before flushing.
|
||||
reset_last_chars()
|
||||
flush()
|
||||
open_chunk(text)
|
||||
save_last_char(text["ch"])
|
||||
assert chunk is not None
|
||||
chunk["str"].append(text["ch"])
|
||||
continue
|
||||
|
||||
if advance <= chunk["not_a_space"]:
|
||||
reset_last_chars()
|
||||
|
||||
if advance <= chunk["tracking"]:
|
||||
lead = save_last_char(text["ch"])
|
||||
extend_chunk(text, lead)
|
||||
continue
|
||||
|
||||
if chunk["flow_min"] <= advance <= chunk["flow_max"]:
|
||||
reset_last_chars()
|
||||
chunk["str"].append(" ")
|
||||
lead = save_last_char(text["ch"])
|
||||
extend_chunk(text, lead)
|
||||
continue
|
||||
|
||||
# OUT-OF-FLOW gap (advance > in-flow space threshold): text extraction synthetic-space insertion
|
||||
# flushes the current item and pushes a
|
||||
# STANDALONE " " item with height 0, then a new item begins at this glyph.
|
||||
# The zero height is load-bearing: the heading heuristics' vertical-alignment test
|
||||
# can't align this inter-run space with a neighbouring line, so it doesn't
|
||||
# cause a spurious line merge (the math-heavy page inline math heading heading drop). The
|
||||
# standalone " " also keeps the word separator in the joined line text Yf
|
||||
# so a positionally-spaced title like "3 The section heading"
|
||||
# (Type-3 fonts, no real space glyphs) does not collapse to
|
||||
# "3TheStaticSemantics" and lose its section number to the _Tf regex.
|
||||
reset_last_chars()
|
||||
emit_fake_space(advance) # standalone height-0 " ", width=abs(gap) (exact span merger)
|
||||
open_chunk(text)
|
||||
save_last_char(text["ch"])
|
||||
assert chunk is not None
|
||||
chunk["str"].append(text["ch"])
|
||||
|
||||
flush()
|
||||
return items
|
||||
@@ -0,0 +1,191 @@
|
||||
"""Raw PDF object access (PyPDF2-backed) and PDF lexical primitives."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from PyPDF2.generic import (
|
||||
IndirectObject as PdfIndirectRef, NameObject as PdfName, NumberObject as PdfNumber,
|
||||
FloatObject as PdfFloat, BooleanObject as PdfBoolean,
|
||||
DictionaryObject as PdfDictionary, ArrayObject as PdfArray,
|
||||
)
|
||||
|
||||
|
||||
def _pdf_tok(value) -> str:
|
||||
"""Serialise one PDF value back to content-syntax (for xref_object's regex)."""
|
||||
if isinstance(value, PdfIndirectRef):
|
||||
return f"{value.idnum} {value.generation} R"
|
||||
if isinstance(value, PdfName):
|
||||
return str(value)
|
||||
if isinstance(value, PdfBoolean):
|
||||
return "true" if value.value else "false"
|
||||
if isinstance(value, PdfDictionary):
|
||||
return _pdf_obj_str(value)
|
||||
if isinstance(value, PdfArray):
|
||||
return "[ " + " ".join(_pdf_tok(array_item) for array_item in value) + " ]"
|
||||
return str(value)
|
||||
|
||||
|
||||
def _pdf_obj_str(obj) -> str:
|
||||
"""Serialize an object body as a PDF-syntax string."""
|
||||
if isinstance(obj, PdfIndirectRef):
|
||||
obj = obj.get_object()
|
||||
if isinstance(obj, PdfDictionary):
|
||||
parts = ["<<"]
|
||||
for key_value, val in obj.items():
|
||||
parts.append(str(key_value))
|
||||
parts.append(_pdf_tok(val))
|
||||
parts.append(">>")
|
||||
return " ".join(parts)
|
||||
if isinstance(obj, PdfArray):
|
||||
return "[ " + " ".join(_pdf_tok(array_item) for array_item in obj) + " ]"
|
||||
return _pdf_tok(obj)
|
||||
|
||||
|
||||
def _pdf_typed(value):
|
||||
"""Return ``(type, value-string)`` for a raw, unresolved PDF value."""
|
||||
if value is None:
|
||||
return ("null", "null")
|
||||
if isinstance(value, PdfIndirectRef):
|
||||
return ("xref", f"{value.idnum} {value.generation} R")
|
||||
if isinstance(value, PdfName):
|
||||
return ("name", str(value))
|
||||
if isinstance(value, PdfBoolean):
|
||||
return ("bool", "true" if value.value else "false")
|
||||
if isinstance(value, PdfFloat):
|
||||
return ("real", str(value))
|
||||
if isinstance(value, PdfNumber):
|
||||
return ("int", str(int(value)))
|
||||
if isinstance(value, PdfDictionary):
|
||||
return ("dict", _pdf_obj_str(value))
|
||||
if isinstance(value, PdfArray):
|
||||
return ("array", _pdf_obj_str(value))
|
||||
try:
|
||||
return ("string", str(value))
|
||||
except Exception:
|
||||
return ("null", "null")
|
||||
|
||||
|
||||
class _PdfPage:
|
||||
__slots__ = ("_page_object",)
|
||||
|
||||
def __init__(self, page):
|
||||
self._page_object = page
|
||||
|
||||
def read_contents(self) -> bytes:
|
||||
candidate_item = self._page_object.get_contents()
|
||||
if candidate_item is None:
|
||||
return b""
|
||||
if isinstance(candidate_item, PdfIndirectRef):
|
||||
candidate_item = candidate_item.get_object()
|
||||
if hasattr(candidate_item, "get_data"):
|
||||
return candidate_item.get_data()
|
||||
# /Contents is an array of streams; concatenate them with a single
|
||||
# space (intentional); join the raw decompressed data the same.
|
||||
|
||||
return b" ".join(text.get_object().get_data() for text in candidate_item)
|
||||
|
||||
def get_fonts(self, full: bool = True):
|
||||
out: list = []
|
||||
res = self._page_object.get("/Resources")
|
||||
if res is None:
|
||||
return out
|
||||
fonts = res.get_object().get("/Font")
|
||||
if fonts is None:
|
||||
return out
|
||||
for _xref_key, ref in fonts.get_object().items():
|
||||
idnum = ref.idnum if isinstance(ref, PdfIndirectRef) else 0
|
||||
filter_context = ref.get_object()
|
||||
subtype = str(filter_context.get("/Subtype", "")).lstrip("/")
|
||||
basefont = str(filter_context.get("/BaseFont", "")).lstrip("/")
|
||||
enc_raw = filter_context.raw_get("/Encoding") if "/Encoding" in filter_context else None
|
||||
enc = str(enc_raw).lstrip("/") if isinstance(enc_raw, PdfName) else ""
|
||||
out.append((idnum, "", subtype, basefont, str(_xref_key).lstrip("/"), enc))
|
||||
return out
|
||||
|
||||
|
||||
class _PdfDoc:
|
||||
"""PyPDF2-backed adapter for raw object and stream access PDFium cannot expose."""
|
||||
|
||||
__slots__ = ("_reader", "_virtual")
|
||||
|
||||
def __init__(self, reader):
|
||||
self._reader = reader
|
||||
# Negative pseudo-xrefs for DIRECT (inline) dicts that have no object
|
||||
# number -- text extraction reference resolution treats direct and indirect values alike,
|
||||
# so inline font dicts must be addressable by the same integer-keyed
|
||||
# pipeline (_redefinition_dict_xrefs registers them).
|
||||
self._virtual: dict[int, object] = {}
|
||||
|
||||
def register_virtual(self, obj) -> int:
|
||||
vid = -(len(self._virtual) + 1)
|
||||
self._virtual[vid] = obj
|
||||
return vid
|
||||
|
||||
@property
|
||||
def page_count(self) -> int:
|
||||
return len(self._reader.pages)
|
||||
|
||||
def __getitem__(self, idx):
|
||||
return _PdfPage(self._reader.pages[idx])
|
||||
|
||||
def page_xref(self, idx: int) -> int:
|
||||
return self._reader.pages[idx].indirect_reference.idnum
|
||||
|
||||
def _resolve_object(self, xref: int):
|
||||
if xref < 0:
|
||||
return self._virtual.get(xref)
|
||||
return PdfIndirectRef(xref, 0, self._reader).get_object()
|
||||
|
||||
def xref_get_key(self, xref: int, _xref_key: str):
|
||||
cur = self._resolve_object(xref)
|
||||
parts = _xref_key.split("/")
|
||||
for index_value, part in enumerate(parts):
|
||||
if cur is None:
|
||||
return ("null", "null")
|
||||
if isinstance(cur, PdfIndirectRef):
|
||||
cur = cur.get_object()
|
||||
if not hasattr(cur, "raw_get"):
|
||||
return ("null", "null")
|
||||
name = "/" + part
|
||||
if name not in cur:
|
||||
return ("null", "null")
|
||||
if index_value == len(parts) - 1:
|
||||
return _pdf_typed(cur.raw_get(name))
|
||||
cur = cur[name]
|
||||
return _pdf_typed(cur)
|
||||
|
||||
def xref_stream(self, xref: int) -> bytes:
|
||||
return self._resolve_object(xref).get_data()
|
||||
|
||||
def xref_object(self, xref: int, compressed: bool = True) -> str:
|
||||
return _pdf_obj_str(self._resolve_object(xref))
|
||||
|
||||
def close(self) -> None:
|
||||
try:
|
||||
self._reader.stream.close()
|
||||
except Exception:
|
||||
pass
|
||||
_PDF_WHITESPACE_BYTES = frozenset({0x20, 0x09, 0x0d, 0x0a, 0x0c, 0x00})
|
||||
_PDF_DELIMITER_BYTES = frozenset(b"()<>[]{}/%")
|
||||
|
||||
|
||||
_PDF_STRING_ESCAPE_BYTES = {0x6E: 0x0A, 0x72: 0x0D, 0x74: 0x09, 0x62: 0x08, 0x66: 0x0C,
|
||||
0x28: 0x28, 0x29: 0x29, 0x5C: 0x5C}
|
||||
|
||||
|
||||
def _decode_pdf_name(raw: bytes) -> bytes:
|
||||
"""Decode #XX escapes in a PDF name token to its canonical bytes."""
|
||||
if b"#" not in raw:
|
||||
return raw
|
||||
out = bytearray()
|
||||
index_value = 0
|
||||
while index_value < len(raw):
|
||||
if raw[index_value] == 0x23 and index_value + 2 < len(raw):
|
||||
try:
|
||||
out.append(int(raw[index_value + 1:index_value + 3], 16))
|
||||
index_value += 3
|
||||
continue
|
||||
except ValueError:
|
||||
pass
|
||||
out.append(raw[index_value])
|
||||
index_value += 1
|
||||
return bytes(out)
|
||||
@@ -0,0 +1,295 @@
|
||||
"""Whole-document parse drivers assembling per-page charlevel metadata."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from typing import Union
|
||||
|
||||
import pypdfium2 as pdfium
|
||||
|
||||
# Raw PDF object access (ToUnicode CMaps, content streams, font dicts, /WMode)
|
||||
# that PDFium does not expose, read via PyPDF2 -- already a project dependency and
|
||||
# permissively licensed. A thin adapter exposes the small raw-object API the
|
||||
# helpers below need, so their calibrated logic stays unchanged.
|
||||
import PyPDF2 as _pypdf2 # declared dependency (also imported by pageindex.utils/client)
|
||||
|
||||
from ..model import Span, Rect
|
||||
|
||||
from .pdf_objects import _PdfDoc
|
||||
from .text_normalize import (
|
||||
_DROP_CHARS,
|
||||
_NORMALIZED_UNICODES,
|
||||
_apply_bidi_reordering,
|
||||
_reverse_if_rtl,
|
||||
)
|
||||
from .content_stream import (
|
||||
_tokenize_show_operators,
|
||||
_assign_vertical_tags,
|
||||
_assign_show_tz,
|
||||
_page_vertical_resource_names,
|
||||
)
|
||||
from .cmap_parse import _compute_skew
|
||||
from .code_walk import _page_show_codes
|
||||
from .unicode_apply import _apply_font_unicode
|
||||
from .char_extract import (
|
||||
_extract_raw_chars,
|
||||
_accumulate_type3_extents,
|
||||
_type3_size_by_font,
|
||||
_apply_type3_sizes,
|
||||
_finalize_chars,
|
||||
_inherited_box,
|
||||
_page_view_rect,
|
||||
)
|
||||
from .merge import _merge_text_items
|
||||
from .remerge import (
|
||||
_remerge_rotated,
|
||||
_remerge_oblique,
|
||||
_remerge_vertical,
|
||||
)
|
||||
|
||||
|
||||
def _page_pass1(pdf, pdf_doc, page_idx: int, type3_ext: dict, font_map_cache: dict):
|
||||
"""Pass-1 body for ONE page: extract raw chars, tag objects, accumulate
|
||||
Type-3 extents into ``type3_ext``. Returns ``(page, raw_chars, page_vb,
|
||||
page_rot)``; the PAGE is returned still open — the caller owns closing it
|
||||
(the sequential driver must keep every page open until pass 2's Type-3
|
||||
size lookups are done; see keep_pages in ``parse_charlevel_meta``)."""
|
||||
page = pdf[page_idx]
|
||||
text_page = page.get_textpage()
|
||||
raw_chars, objects = _extract_raw_chars(page, text_page.raw)
|
||||
try:
|
||||
media_box_raw = _inherited_box(pdf_doc, page_idx, "MediaBox") if pdf_doc is not None else None
|
||||
crop_box_raw = _inherited_box(pdf_doc, page_idx, "CropBox") if pdf_doc is not None else None
|
||||
page_vb = _page_view_rect(page, media_box_raw, crop_box_raw) # (x0, y0, x1, y1) page space
|
||||
except Exception:
|
||||
page_vb = None # no box -> off-page test disabled
|
||||
try:
|
||||
page_rot = int(page.get_rotation()) # PDFium /Rotate (0/90/180/270)
|
||||
except Exception:
|
||||
page_rot = 0
|
||||
show_fonts: list[bytes | None] = []
|
||||
show_tzs: list[float] = []
|
||||
vert_names: set[bytes] = set()
|
||||
if pdf_doc is not None and page_idx < pdf_doc.page_count:
|
||||
try:
|
||||
# show-op flush ids (q/Q flush scope) are no longer used -- the merge-id
|
||||
# grouping was removed; only show_fonts (per-op font resname)
|
||||
# feeds vertical tagging.
|
||||
show_flush_ids, show_fonts, show_text_units, horizontal_scales, xobject_paints = _tokenize_show_operators(
|
||||
pdf_doc[page_idx].read_contents())
|
||||
except Exception:
|
||||
show_fonts = []
|
||||
vert_names = _page_vertical_resource_names(pdf_doc, page_idx)
|
||||
# Tz follows the text into Form XObjects (the whole text state is
|
||||
# cloned for the recursion), so the per-show-op horizontal-scale
|
||||
# list has to come from the SAME form-descending walk as the codes:
|
||||
# the page's own stream alone under-counts every form page and the
|
||||
# ordinal gate below would then drop the tag for the whole page.
|
||||
try:
|
||||
show_codes = _page_show_codes(pdf_doc, page_idx)
|
||||
if show_codes:
|
||||
show_tzs = [horizontal_scale for _fx, _s, horizontal_scale in show_codes]
|
||||
# Patch per-char unicode to span merger glyph Unicode where
|
||||
# PDFium's decode differs (guarded: any failure keeps
|
||||
# PDFium's output).
|
||||
if raw_chars:
|
||||
_apply_font_unicode(
|
||||
text_page.raw, raw_chars, objects, show_codes, pdf_doc,
|
||||
font_map_cache)
|
||||
except Exception:
|
||||
pass
|
||||
_assign_vertical_tags(objects, show_fonts, vert_names)
|
||||
_assign_show_tz(objects, show_tzs)
|
||||
_accumulate_type3_extents(raw_chars, type3_ext)
|
||||
text_page.close()
|
||||
return page, raw_chars, page_vb, page_rot
|
||||
|
||||
|
||||
def _page_pass2(raw_chars: list[dict], page_vb, size_by_font: dict) -> list[dict]:
|
||||
"""Pass-2 body for ONE page: apply the document-wide Type-3 sizes,
|
||||
restore paint order, finalize glyph widths, run the text merger."""
|
||||
_apply_type3_sizes(raw_chars, size_by_font)
|
||||
# text extraction emits glyphs in CONTENT-STREAM (paint) order; PDFium's textpage
|
||||
# reorders whole segments page-wide (math-heavy page margin labels 'margin label' /
|
||||
# 'Section N' arrive at a different point of the char stream than their
|
||||
# show ops). obj["page_order"] is the object's stream position (objects
|
||||
# parse sequentially, incl. the Form XObject walk), so sorting real
|
||||
# glyphs by it restores span merger processing order for the merger.
|
||||
# GENERATED chars (PDFium's synthetic layout whitespace -- no span merger
|
||||
# counterpart, pure merger bookkeeping) keep no position of their own:
|
||||
# their geometric obj lookup can land on the WRONG object (the
|
||||
# multi-column "4 | Super | vision" heading puts the '4'->'S' gap
|
||||
# space inside the 'vision' object, which would re-emit it mid-word as
|
||||
# "Super vision"), so each one stays glued behind the real glyph that
|
||||
# precedes it in textpage order. Character-level ordering's
|
||||
# own items on the reordered pages.
|
||||
keys: list[tuple] = [()] * len(raw_chars)
|
||||
last_key = None
|
||||
lead_gens: list[int] = []
|
||||
for key_value, candidate_item in enumerate(raw_chars):
|
||||
if candidate_item["is_gen"]:
|
||||
if last_key is None:
|
||||
lead_gens.append(key_value)
|
||||
else:
|
||||
keys[key_value] = (last_key[0], last_key[1], 1, key_value)
|
||||
else:
|
||||
last_key = (candidate_item["obj"]["page_order"], candidate_item["i"])
|
||||
keys[key_value] = (last_key[0], last_key[1], 0, key_value)
|
||||
for key_value in lead_gens:
|
||||
keys[key_value] = (-1, -1, 1, key_value)
|
||||
raw_chars[:] = [raw_chars[key_value] for key_value in sorted(range(len(raw_chars)),
|
||||
key=keys.__getitem__)]
|
||||
fin = _finalize_chars(raw_chars)
|
||||
merged = _merge_text_items(fin, page_vb)
|
||||
merged = _remerge_rotated(merged) # collapse cardinal-rotated per-glyph shards
|
||||
merged = _remerge_vertical(merged) # collapse vertical-writing per-glyph shards
|
||||
return _remerge_oblique(merged, fin) # oblique objects: inverse-rotation projection re-merge
|
||||
|
||||
|
||||
def _page_spans(raw: list[dict]) -> list[Span]:
|
||||
"""Final emission for ONE page: merged chunks -> ``Span`` objects."""
|
||||
spans: list[Span] = []
|
||||
for item in raw:
|
||||
# the heading heuristics pushes normalized glyph Unicode = the normalized-Unicode table[u] or u
|
||||
|
||||
# per glyph, a WHOLE-string lookup. Each r["str"] piece is one glyph's
|
||||
# unicode (or a synthesized space), so look up per piece -- a
|
||||
# multi-codepoint ToUnicode value is left intact when the whole-string
|
||||
# lookup misses, instead of decomposing a table-key char inside it.
|
||||
# span merger: normalized glyph Unicode = RTL ligature reversal(the normalized-Unicode table
|
||||
# [u] or u) -- the table lookup is then wrapped in RTL ligature reversal, which
|
||||
|
||||
# reverses a multi-char Arabic/Hebrew ligature value (span merger
|
||||
#). Apply per piece (each r["str"] piece is one glyph's unicode).
|
||||
joined = "".join(
|
||||
_reverse_if_rtl(_NORMALIZED_UNICODES.get(page_value, page_value)) for page_value in item["str"] # type: ignore[arg-type]
|
||||
)
|
||||
# text extraction text-item flush -> bidirectional transform: the joined item
|
||||
# text runs the bidi pass ON TOP of the per-glyph RTL ligature reversal
|
||||
# above (both layers exist in span merger). Pass-through for LTR text
|
||||
# and vertical items (dir 'ttb').
|
||||
joined = _apply_bidi_reordering(joined, -1, bool(item["obj"].get("vertical")))
|
||||
text = joined.translate(_DROP_CHARS)
|
||||
if not text:
|
||||
continue
|
||||
# font_size = hypot(text matrix[2], text matrix[3])
|
||||
# taken once at the item's open glyph, i.e. the chunk's first-char
|
||||
# fs. The merger breaks a chunk on any fs change (exact compare;
|
||||
# see the font_key/fs guard above) and never lowers fs mid-chunk, so
|
||||
# chunk["fs"] (set in open_chunk from the first char) is exactly
|
||||
# that value. Emit it rather than the per-chunk minimum.
|
||||
fs_emit = item["fs"]
|
||||
spans.append(
|
||||
Span(
|
||||
bbox=Rect(item["left"], item["right"], item["top"], item["bottom"]),
|
||||
text=text,
|
||||
font_name_raw=item["font_name"],
|
||||
font_size=fs_emit,
|
||||
# the heading heuristics bold is name-regex only (the font-name bold regex,
|
||||
# OR'd into the emitted span). span merger bold detector ignores the descriptor
|
||||
# ForceBold flag and numeric weight, so we must NOT inject a
|
||||
# weight-based bold here — that over-bolds Demi/Medium/bold math font
|
||||
# faces (weight 665-675) text extraction treats as regular.
|
||||
bold=False,
|
||||
italic=False,
|
||||
# Span skew score: P = (f[1]/f[0])² + (f[2]/f[3])² from the item
|
||||
# transform (IEEE: cardinal rotation -> Inf, upright -> 0).
|
||||
# The owning object's PDFium matrix has the same
|
||||
# rotation/shear structure as span merger item transform.
|
||||
# mtx0 = the FIRST glyph's object matrix (text extraction fixes the
|
||||
# item transform at open); standalone fake-space items
|
||||
# carry no mtx0 and fall back to their obj (= the previous
|
||||
# glyph's object == text extraction previous glyph transform for that space).
|
||||
skew=_compute_skew(item.get("mtx0") or item["obj"]["mtx"]),
|
||||
)
|
||||
)
|
||||
return spans
|
||||
|
||||
|
||||
def parse_charlevel_meta(doc_handle: Union[str, Path, BytesIO]) -> tuple[list[list[Span]], list]:
|
||||
if isinstance(doc_handle, (str, Path)):
|
||||
pdf = pdfium.PdfDocument(str(doc_handle))
|
||||
elif isinstance(doc_handle, BytesIO):
|
||||
pdf = pdfium.PdfDocument(doc_handle)
|
||||
else:
|
||||
pdf = doc_handle
|
||||
|
||||
# Open the same document in PyPDF2 (already a project dependency) to read the
|
||||
# page content streams: span merger item-flush operators (q/Q save/restore, marked
|
||||
# content, XObject) live there and PDFium's flattened object model cannot expose
|
||||
# them. Optional/guarded -- any failure leaves flush_id unset so the merger
|
||||
# keeps its per-object split (the fallback behavior). A separate bytes copy
|
||||
# avoids racing pypdfium2's read of the same BytesIO.
|
||||
pdf_doc = None
|
||||
if _pypdf2 is not None:
|
||||
try:
|
||||
if isinstance(doc_handle, (str, Path)):
|
||||
pdf_doc = _PdfDoc(_pypdf2.PdfReader(str(doc_handle)))
|
||||
elif isinstance(doc_handle, BytesIO):
|
||||
# Read a copy so we never race pdfium's read of the same buffer.
|
||||
pdf_doc = _PdfDoc(_pypdf2.PdfReader(BytesIO(doc_handle.getvalue())))
|
||||
except Exception:
|
||||
pdf_doc = None
|
||||
|
||||
# Pass 1: extract raw chars for every page (including each glyph's raw
|
||||
# advance) and accumulate per-font identity-matrix Type-3 glyph-bbox
|
||||
# extents document-wide, so each Type-3 font is sized once over every
|
||||
# glyph it renders anywhere (coverage-independent), matching span merger
|
||||
# synthesizing font.bbox once from the CharProcs. Font handles are only
|
||||
# stable per document while their pages stay open (see keep_pages below).
|
||||
per_page: list[list[dict]] = []
|
||||
page_view_boxes: list = [] # parallel to per_page: text extraction page view box per page
|
||||
page_rotations: list = [] # parallel: PDFium page /Rotate in degrees per page
|
||||
type3_ext: dict = {}
|
||||
font_map_cache: dict = {}
|
||||
# Hold every page open until pass 2's Type-3 size lookups are done.
|
||||
# type3_ext / size_by_font key on the raw FPDF_FONT pointer VALUE, and
|
||||
# PDFium frees a font once the last page using it closes -- a later
|
||||
# page's (different) font can then be allocated at the same address,
|
||||
# silently merging two fonts' extent bins. Which addresses get reused
|
||||
# depends on the process's prior malloc state, so the output could vary
|
||||
# with whatever ran earlier in the process. Keeping the pages alive makes
|
||||
# the handle a true per-document
|
||||
# font identity (PDFium's document-level font cache returns one handle
|
||||
# per font redefinition).
|
||||
keep_pages = []
|
||||
for page_idx in range(len(pdf)):
|
||||
page, raw_chars, page_vb, page_rot = _page_pass1(
|
||||
pdf, pdf_doc, page_idx, type3_ext, font_map_cache)
|
||||
keep_pages.append(page)
|
||||
per_page.append(raw_chars)
|
||||
page_view_boxes.append(page_vb)
|
||||
page_rotations.append(page_rot)
|
||||
if pdf_doc is not None and pdf_doc is not doc_handle:
|
||||
try:
|
||||
pdf_doc.close()
|
||||
except Exception:
|
||||
pass
|
||||
size_by_font = _type3_size_by_font(type3_ext)
|
||||
|
||||
# Pass 2: apply the document-wide Type-3 sizes, finalize glyph widths,
|
||||
# then run text extraction text merger.
|
||||
raw_pages: list[list[dict]] = []
|
||||
for page_view_index, raw_chars in enumerate(per_page):
|
||||
raw_pages.append(_page_pass2(raw_chars, page_view_boxes[page_view_index], size_by_font))
|
||||
for page_handle in keep_pages:
|
||||
try:
|
||||
page_handle.close()
|
||||
except Exception:
|
||||
pass
|
||||
keep_pages.clear()
|
||||
|
||||
out: list[list[Span]] = []
|
||||
for raw in raw_pages:
|
||||
out.append(_page_spans(raw))
|
||||
pdf.close()
|
||||
# Per-page viewport metadata (text extraction normalized page view = cropbox clamped to the
|
||||
# mediabox, via _page_view_rect, + /Rotate) parallel to out, so heading
|
||||
# coordinates can apply span merger viewport-coordinate transform.
|
||||
return out, list(zip(page_view_boxes, page_rotations))
|
||||
|
||||
|
||||
def parse_charlevel(doc_handle: Union[str, Path, BytesIO]) -> list[list[Span]]:
|
||||
"""Per-page span entry: per-page spans only (drops viewport meta). the high-level TOC pipeline uses ``parse_charlevel_meta`` to also get the per-page (view box, /Rotate) for heading coordinates; every other caller just wants the spans. """
|
||||
return parse_charlevel_meta(doc_handle)[0]
|
||||
@@ -0,0 +1,327 @@
|
||||
"""Re-merges rotated, oblique, and vertical spans after the first join pass."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
|
||||
from .text_normalize import (
|
||||
TRACKING_SPACE_FACTOR,
|
||||
NEGATIVE_SPACE_FACTOR,
|
||||
SPACE_IN_FLOW_MIN_FACTOR,
|
||||
SPACE_IN_FLOW_MAX_FACTOR,
|
||||
)
|
||||
|
||||
|
||||
def _start_rot_span(chunk: dict) -> dict:
|
||||
"""A fresh single-glyph rotated span = a deep-enough copy of the merger chunk (keeps fs/font/obj/sign so the downstream span conversion is unchanged)."""
|
||||
span = dict(chunk)
|
||||
span["str"] = list(chunk["str"])
|
||||
span["font_tally"] = dict(chunk.get("font_tally", {}))
|
||||
span["weight_tally"] = dict(chunk.get("weight_tally", {}))
|
||||
return span
|
||||
|
||||
|
||||
def _grow_rot_span(cur: dict, chunk: dict) -> None:
|
||||
"""Extend a rotated span with the next glyph: append text, union the page box (left/right/top/bottom stay in page coords -> output box is exact), merge the per-char style tallies."""
|
||||
cur["str"].extend(chunk["str"])
|
||||
cur["left"] = min(cur["left"], chunk["left"])
|
||||
cur["right"] = max(cur["right"], chunk["right"])
|
||||
cur["top"] = max(cur["top"], chunk["top"])
|
||||
cur["bottom"] = min(cur["bottom"], chunk["bottom"])
|
||||
for span, count in chunk.get("font_tally", {}).items():
|
||||
cur["font_tally"][span] = cur["font_tally"].get(span, 0) + count
|
||||
for span, count in chunk.get("weight_tally", {}).items():
|
||||
cur["weight_tally"][span] = cur["weight_tally"].get(span, 0) + count
|
||||
|
||||
|
||||
def _merge_rotated_one(group: list[dict], rot: int) -> list[dict]:
|
||||
"""1-D position comparison along the rotation axis for one cardinally rotated text object. ``read_origin`` is the glyph origin in reading order (90 reads up +y, 270 down -y, 180 left -x); the pen advances by glyph_w, so the inter-glyph gap is ``next_origin - (cur_origin + glyph_w)``. In-flow gaps join, larger gaps start a new item, and the box remains the page-space AABB required by downstream layout. Cardinal rotation intentionally does less than the oblique path: its box convention cannot match the oblique item-box convention, and the extra out-of-flow/cross-axis branches are not useful for these short rotated labels."""
|
||||
def read_origin(chunk: dict) -> float:
|
||||
if rot == 90:
|
||||
return chunk["bottom"]
|
||||
if rot == 270:
|
||||
return -chunk["top"]
|
||||
if rot == 180:
|
||||
return -chunk["right"]
|
||||
return chunk["left"]
|
||||
|
||||
ordered = sorted(group, key=read_origin)
|
||||
spans: list[dict] = []
|
||||
cur: dict | None = None
|
||||
pen = 0.0
|
||||
for chunk in ordered:
|
||||
font_size = chunk.get("fs", 0.0) or 0.0
|
||||
glyph_width = chunk.get("glyph_w", 0.0) or 0.0
|
||||
origin = read_origin(chunk)
|
||||
if cur is None:
|
||||
cur = _start_rot_span(chunk)
|
||||
pen = origin + glyph_width
|
||||
continue
|
||||
gap = origin - pen
|
||||
if gap <= font_size * SPACE_IN_FLOW_MAX_FACTOR:
|
||||
if gap > font_size * TRACKING_SPACE_FACTOR:
|
||||
cur["str"].append(" ")
|
||||
_grow_rot_span(cur, chunk)
|
||||
else:
|
||||
spans.append(cur)
|
||||
cur = _start_rot_span(chunk)
|
||||
pen = origin + glyph_width
|
||||
if cur is not None:
|
||||
spans.append(cur)
|
||||
return spans
|
||||
|
||||
|
||||
def _remerge_rotated(items: list[dict]) -> list[dict]:
|
||||
"""Re-merge the per-glyph chunks of each rotated text object into text items along the rotation axis. Upright text is untouched; merged spans keep the first chunk position for reading order."""
|
||||
rot_groups: dict[int, list[dict]] = {}
|
||||
for item in items:
|
||||
obj = item.get("obj")
|
||||
if isinstance(obj, dict) and obj.get("rot") in (90, 180, 270):
|
||||
rot_groups.setdefault(id(obj), []).append(item)
|
||||
if not rot_groups:
|
||||
return items
|
||||
|
||||
merged_for = {
|
||||
oid: _merge_rotated_one(group, group[0]["obj"]["rot"])
|
||||
for oid, group in rot_groups.items()
|
||||
}
|
||||
out: list[dict] = []
|
||||
emitted: set[int] = set()
|
||||
for item in items:
|
||||
obj = item.get("obj")
|
||||
if isinstance(obj, dict) and obj.get("rot") in (90, 180, 270):
|
||||
oid = id(obj)
|
||||
if oid not in emitted:
|
||||
emitted.add(oid)
|
||||
out.extend(merged_for[oid])
|
||||
else:
|
||||
out.append(item)
|
||||
return out
|
||||
|
||||
|
||||
def _new_oblique_span(glyph: dict, baseline_pos: float, cross_pos: float, glyph_width: float) -> dict:
|
||||
"""Open an oblique item at its first reading-order glyph. Records the glyph's page-space pen origin, along-baseline start, cross-axis position, and running pen so the gap logic can compare the next glyph."""
|
||||
return {
|
||||
"str": [glyph["ch"]],
|
||||
"obj": glyph["obj"],
|
||||
"fs": glyph["fs"],
|
||||
"font_name": glyph["font_name"],
|
||||
"_ox0": glyph["ox"], "_oy0": glyph["oy"],
|
||||
"_u0": baseline_pos, "_uend": baseline_pos + glyph_width, "_pen": baseline_pos + glyph_width, "_vlast": cross_pos,
|
||||
"_lox": glyph["ox"], "_loy": glyph["oy"], "_lgw": glyph_width,
|
||||
}
|
||||
|
||||
|
||||
def _close_oblique(cur: dict) -> dict:
|
||||
"""Finalize an oblique item's box. The item merger is rotation-agnostic -- it turns ANY text extraction item into a span via left=transform[4], right=+width, bottom=transform[5], top=+height -- so an oblique item's box is upright at its pen origin, with width = the along-baseline advance (text extraction item.width, NOT the diagonal x-extent the horizontal merger would compute) and height = font size."""
|
||||
width = cur["_uend"] - cur["_u0"]
|
||||
cur["left"] = cur["_ox0"]
|
||||
cur["right"] = cur["_ox0"] + width
|
||||
cur["bottom"] = cur["_oy0"]
|
||||
cur["top"] = cur["_oy0"] + cur["fs"]
|
||||
return cur
|
||||
|
||||
|
||||
def _oblique_space(cur: dict, adv: float, baseline_unit_x: float, baseline_unit_y: float, scale: float) -> dict:
|
||||
"""span merger ``synthetic-space insertion`` out-of-flow item: a STANDALONE " " at the previous glyph's pen (previous glyph transform), width=|advance-x|, height 0 (horizontal). The pen sits at the last glyph's origin advanced by its width along the baseline unit direction ``(ux,uy)``. span merger ``advance-x`` is ``(posX-lastPosX)/text advance scale``, so the width is normalised by the matrix scale (== text advance scale here); on identity CTM scale==1 so this is a no-op, but under a scaled CTM it matters. Output box = left=pen_x, right=+width, bottom=top=pen_y."""
|
||||
pen_x = cur["_lox"] + cur["_lgw"] * baseline_unit_x
|
||||
pen_y = cur["_loy"] + cur["_lgw"] * baseline_unit_y
|
||||
width_value = abs(adv) / scale
|
||||
return {
|
||||
"str": [" "], "obj": cur["obj"], "fs": cur["fs"], "font_name": cur["font_name"],
|
||||
"left": pen_x, "right": pen_x + width_value, "bottom": pen_y, "top": pen_y,
|
||||
}
|
||||
|
||||
|
||||
def _merge_oblique_one(chs: list[dict]) -> list[dict]:
|
||||
"""text extraction position comparison (inverse-rotation projection path) for ONE oblique text object's glyphs -- the explicit horizontal-branch implementation. ``inverse-rotation projection(x,y,m) = [(m0*x+m1*y)/s, (m2*x+m3*y)/s]`` (s=hypot(m0,m1)); component 0 is the reading-order (baseline) coordinate, component 1 the cross axis. Projecting each glyph's pen origin onto these gives advance-x (along, the gap beyond the prev glyph's advance) and advance-y (cross). Then apply the item split thresholds: advance-x<backward-jump threshold (back-jump) or |advance-y|>height -> split; advance-x<=tracking-space threshold -> join no space; <=in-flow space threshold -> in-flow space in str; else synthetic-space insertion -> a STANDALONE " " item then split. Items carry the item-box convention box (see _close_oblique)."""
|
||||
matrix_a, matrix_b, matrix_c, matrix_d = chs[0]["obj"]["mtx"]
|
||||
scale = math.hypot(matrix_a, matrix_b) or 1.0
|
||||
baseline_unit_x, baseline_unit_y = matrix_a / scale, matrix_b / scale # baseline unit direction (page space)
|
||||
|
||||
def along(glyph: dict) -> float:
|
||||
return (matrix_a * glyph["ox"] + matrix_b * glyph["oy"]) / scale
|
||||
|
||||
def cross(glyph: dict) -> float:
|
||||
return (matrix_c * glyph["ox"] + matrix_d * glyph["oy"]) / scale
|
||||
|
||||
ordered = sorted(chs, key=along)
|
||||
spans: list[dict] = []
|
||||
cur: dict | None = None
|
||||
for glyph in ordered:
|
||||
if glyph.get("is_ws"):
|
||||
# Skip whitespace glyphs entirely (== main span merger skips whitespace,
|
||||
# no pen update): text extraction never pushes a raw space glyph to str; the gap
|
||||
# they leave is re-synthesised by the in-flow/out-of-flow logic below
|
||||
# for the next visible glyph. This collapses runs of spaces to one and
|
||||
# trims trailing/leading spaces using the last-character buffer.
|
||||
continue
|
||||
font_size = glyph.get("fs", 0.0) or 0.0
|
||||
glyph_width = glyph.get("glyph_w", 0.0) or 0.0
|
||||
baseline_pos = along(glyph)
|
||||
cross_pos = cross(glyph)
|
||||
if cur is None:
|
||||
cur = _new_oblique_span(glyph, baseline_pos, cross_pos, glyph_width)
|
||||
continue
|
||||
baseline_gap = baseline_pos - cur["_pen"] # along-baseline gap beyond prev advance
|
||||
cross_shift = cross_pos - cur["_vlast"] # cross-axis shift
|
||||
if baseline_gap < font_size * NEGATIVE_SPACE_FACTOR or abs(cross_shift) > font_size:
|
||||
# back-jump (backward-jump threshold) or cross-axis line break: span merger
|
||||
# flush/line-break emission -- either way the item merger just starts a new item.
|
||||
spans.append(_close_oblique(cur))
|
||||
cur = _new_oblique_span(glyph, baseline_pos, cross_pos, glyph_width)
|
||||
continue
|
||||
if baseline_gap <= font_size * TRACKING_SPACE_FACTOR:
|
||||
cur["str"].append(glyph["ch"]) # join, no space
|
||||
elif baseline_gap <= font_size * SPACE_IN_FLOW_MAX_FACTOR:
|
||||
cur["str"].append(" ") # in-flow space
|
||||
cur["str"].append(glyph["ch"])
|
||||
else:
|
||||
spans.append(_close_oblique(cur)) # out-of-flow:
|
||||
spans.append(_oblique_space(cur, baseline_gap, baseline_unit_x, baseline_unit_y, scale)) # standalone " "
|
||||
cur = _new_oblique_span(glyph, baseline_pos, cross_pos, glyph_width)
|
||||
continue
|
||||
cur["_uend"] = baseline_pos + glyph_width
|
||||
cur["_pen"] = baseline_pos + glyph_width
|
||||
cur["_vlast"] = cross_pos
|
||||
cur["_lox"], cur["_loy"], cur["_lgw"] = glyph["ox"], glyph["oy"], glyph_width
|
||||
if cur is not None:
|
||||
spans.append(_close_oblique(cur))
|
||||
return spans
|
||||
|
||||
|
||||
def _remerge_oblique(items: list[dict], fin_chars: list[dict]) -> list[dict]:
|
||||
"""Rebuild oblique text objects by re-merging per-glyph chunks along the baseline and emitting item-box-convention boxes. Upright and cardinal text are untouched."""
|
||||
groups: dict[int, list[dict]] = {}
|
||||
for glyph in fin_chars:
|
||||
obj = glyph.get("obj")
|
||||
if isinstance(obj, dict) and obj.get("rot") == -1:
|
||||
groups.setdefault(id(obj), []).append(glyph)
|
||||
if not groups:
|
||||
return items
|
||||
|
||||
merged_for = {oid: _merge_oblique_one(chs) for oid, chs in groups.items()}
|
||||
out: list[dict] = []
|
||||
emitted: set[int] = set()
|
||||
for item in items:
|
||||
obj = item.get("obj")
|
||||
if isinstance(obj, dict) and obj.get("rot") == -1:
|
||||
oid = id(obj)
|
||||
if oid not in emitted:
|
||||
emitted.add(oid)
|
||||
out.extend(merged_for[oid])
|
||||
else:
|
||||
out.append(item)
|
||||
return out
|
||||
|
||||
|
||||
def _start_vert_span(chunk: dict) -> dict:
|
||||
"""Create a vertical item from its first chunk. Vertical items use the rendered font size as width, accumulate height per glyph, and keep the first glyph's pen as the item transform. The item-to-span conversion reads the style's vertical flag and flips the sign of the height offset, so a vertical item's box runs DOWN from the pen where a horizontal one runs up. The span box reproduces that convention rather than the ink AABB."""
|
||||
span = dict(chunk)
|
||||
span["str"] = list(chunk["str"])
|
||||
span["font_tally"] = dict(chunk.get("font_tally", {}))
|
||||
span["weight_tally"] = dict(chunk.get("weight_tally", {}))
|
||||
span["v_height"] = chunk["v_pen_y"] - chunk["v_after"] # first glyph's advance
|
||||
return span
|
||||
|
||||
|
||||
def _close_vert_span(mapping: dict) -> dict:
|
||||
"""Finalize the item merger-convention box of a vertical item."""
|
||||
mapping["left"] = mapping["v_pen_x"]
|
||||
mapping["right"] = mapping["v_pen_x"] + mapping["fs"]
|
||||
mapping["top"] = mapping["v_pen_y"]
|
||||
mapping["bottom"] = mapping["v_pen_y"] - abs(mapping["v_height"])
|
||||
return mapping
|
||||
|
||||
|
||||
def _merge_vertical_one(group: list[dict]) -> list[dict]:
|
||||
"""Apply vertical-writing position comparison over one text object's per-glyph chunks in stream order. The previous pen-after-advance and current pen define the along-axis gap; x shift is the cross-axis break signal. Small gaps join, in-flow gaps insert a space, out-of-flow gaps emit a standalone zero-width space item, and backward or cross-axis jumps start a new item. Whitespace glyphs are consumed by the span merger, so their advance arrives here as an in-flow gap."""
|
||||
spans: list[dict] = []
|
||||
cur: dict | None = None
|
||||
after = 0.0 # text extraction previous glyph transform[5]: pen y after the previous glyph
|
||||
last_x = 0.0 # text extraction previous glyph transform[4]
|
||||
for chunk in group:
|
||||
font_size = chunk.get("fs", 0.0) or 0.0
|
||||
if cur is None:
|
||||
cur = _start_vert_span(chunk)
|
||||
after, last_x = chunk["v_after"], chunk["v_pen_x"]
|
||||
continue
|
||||
vertical_gap = after - chunk["v_pen_y"]
|
||||
x_shift = chunk["v_pen_x"] - last_x
|
||||
direction_sign = 1.0 if cur["v_height"] >= 0 else -1.0
|
||||
width = cur["fs"]
|
||||
if vertical_gap < direction_sign * NEGATIVE_SPACE_FACTOR * font_size or abs(x_shift) > width:
|
||||
# backward jump or cross-axis break: text extraction line-break emission/flush -- both
|
||||
# end the item (we don't model line-break marker, and the item merger ignores it).
|
||||
spans.append(_close_vert_span(cur))
|
||||
cur = _start_vert_span(chunk)
|
||||
elif vertical_gap <= direction_sign * TRACKING_SPACE_FACTOR * font_size:
|
||||
cur["v_height"] += vertical_gap + (chunk["v_pen_y"] - chunk["v_after"])
|
||||
_grow_vert_span(cur, chunk)
|
||||
elif direction_sign * SPACE_IN_FLOW_MIN_FACTOR * font_size <= vertical_gap <= direction_sign * SPACE_IN_FLOW_MAX_FACTOR * font_size:
|
||||
cur["str"].append(" ")
|
||||
cur["v_height"] += vertical_gap + (chunk["v_pen_y"] - chunk["v_after"])
|
||||
_grow_vert_span(cur, chunk)
|
||||
else:
|
||||
# out-of-flow: standalone " " at previous glyph transform, width 0, height |e|
|
||||
# (vertical synthetic spaces store the gap as height and leave width at zero).
|
||||
meta = cur
|
||||
spans.append(_close_vert_span(cur))
|
||||
spans.append({
|
||||
"str": [" "], "sign": 1, "obj": meta["obj"],
|
||||
"left": last_x, "right": last_x, # WIDTH 0
|
||||
# A vertical style flips the height offset: the box runs DOWN
|
||||
# from the previous pen, like _close_vert_span's.
|
||||
"top": after, "bottom": after - abs(vertical_gap),
|
||||
"fs": meta["fs"], "fs_min": meta["fs"],
|
||||
"font_name": meta["font_name"], "font_key": meta["font_key"],
|
||||
"weight": meta["weight"],
|
||||
"font_tally": {meta["font_name"]: 1},
|
||||
"weight_tally": {meta["weight"]: 1},
|
||||
})
|
||||
cur = _start_vert_span(chunk)
|
||||
after, last_x = chunk["v_after"], chunk["v_pen_x"]
|
||||
if cur is not None:
|
||||
spans.append(_close_vert_span(cur))
|
||||
return spans
|
||||
|
||||
|
||||
def _grow_vert_span(cur: dict, chunk: dict) -> None:
|
||||
"""Append a glyph to a vertical item: text + style tallies. The box is NOT unioned here -- it is derived from the first pen + accumulated v_height in _close_vert_span, with transform fixed at the first glyph and height accumulated."""
|
||||
cur["str"].extend(chunk["str"])
|
||||
for span, count in chunk.get("font_tally", {}).items():
|
||||
cur["font_tally"][span] = cur["font_tally"].get(span, 0) + count
|
||||
for span, count in chunk.get("weight_tally", {}).items():
|
||||
cur["weight_tally"][span] = cur["weight_tally"].get(span, 0) + count
|
||||
|
||||
|
||||
def _remerge_vertical(items: list[dict]) -> list[dict]:
|
||||
"""Re-merge the per-glyph chunks of each vertical-writing (Identity-V / WMode 1) text object into PDF content tokenizer style items. Uses the same _remerge_rotated: the horizontal merger is untouched (it shatters a vertical column because the glyphs stack along its line-break axis) and this gated post-pass rewrites only vertical-object chunks. One extra wrinkle vs the rotated pass: text extraction emits items in content-stream order, but PDFium's textpage reorders vertical chars page-wide (its own column heuristic), so the merged groups are reassigned to the vertical slot positions in object paint order."""
|
||||
groups: dict[int, list[dict]] = {}
|
||||
obj_of: dict[int, dict] = {}
|
||||
for item in items:
|
||||
obj = item.get("obj")
|
||||
if (isinstance(obj, dict) and obj.get("vertical") and not obj.get("rot")
|
||||
and "v_pen_y" in item):
|
||||
oid = id(obj)
|
||||
groups.setdefault(oid, []).append(item)
|
||||
obj_of[oid] = obj
|
||||
if not groups:
|
||||
return items
|
||||
|
||||
merged_for = {oid: _merge_vertical_one(group_value) for oid, group_value in groups.items()}
|
||||
paint_order = sorted(groups, key=lambda oid: obj_of[oid]["page_order"])
|
||||
out: list[dict] = []
|
||||
slot = 0 # next paint-order group to emit at the next vertical slot
|
||||
seen: set[int] = set()
|
||||
for item in items:
|
||||
obj = item.get("obj")
|
||||
oid = id(obj) if isinstance(obj, dict) else None
|
||||
if oid in groups:
|
||||
if oid not in seen:
|
||||
seen.add(oid)
|
||||
out.extend(merged_for[paint_order[slot]])
|
||||
slot += 1
|
||||
else:
|
||||
out.append(item)
|
||||
return out
|
||||
@@ -0,0 +1,288 @@
|
||||
"""Unicode normalization tables, whitespace classes, spacing factors, and bidi reordering."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
_DROP_CHARS = str.maketrans({
|
||||
# U+FFFE is PDFium's "no unicode mapping" textpage sentinel. The
|
||||
# patch pipeline (_apply_font_unicode) replaces it with decoded text
|
||||
# wherever the map walk succeeds; a REMAINING U+FFFE means the guarded
|
||||
# walk gave up for that run, so deleting it keeps PDFium noise out of the
|
||||
# spans. A pathological ToUnicode map that intentionally emits literal
|
||||
# U+FFFE is indistinguishable from this sentinel here and is dropped.
|
||||
"": None,
|
||||
"\t": " ",
|
||||
"\n": " ",
|
||||
"\r": " ",
|
||||
# text extraction maps a glyph whose unicode lands on U+00AD to U+002D. (An
|
||||
# earlier "\x02" -> "-" entry here compensated PDFium decoding
|
||||
# re-encoded hyphens (charcode 2, ToUnicode gap) as U+0002; that decode
|
||||
# is now handled by _apply_font_unicode: mapped soft hyphens emit '-' where the
|
||||
# font's Differences name the glyph, and keeps the raw \x02 where they
|
||||
# don't, e.g. math-heavy page body ligature codes.
|
||||
"": "-",
|
||||
})
|
||||
|
||||
|
||||
# The normalized Unicode table is a fixed, sparse per-glyph
|
||||
# lookup table; a char absent from it is emitted unchanged. This is NOT Unicode
|
||||
# NFKC: NFKC over-normalises (fullwidth→ASCII, superscripts→digits, ohm→omega,
|
||||
# nbsp→space) exactly where this table leaves the glyph untouched. Apply the
|
||||
# table per code point.
|
||||
_NORMALIZED_UNICODES: dict[str, str] = json.loads(
|
||||
(Path(__file__).parent.parent / "data" / "normalized_unicodes.json")
|
||||
.read_text(encoding="utf-8")
|
||||
)
|
||||
|
||||
|
||||
def _normalize_unicodes(text: str) -> str:
|
||||
"""Apply the per-glyph normalized-Unicode substitution table to a text item. The table is keyed by single code points and never introduces table keys, so applying it to the already-joined LTR item string preserves per-glyph substitution after the text item is joined."""
|
||||
unit_count = _NORMALIZED_UNICODES
|
||||
if not any(candidate_item in unit_count for candidate_item in text):
|
||||
return text
|
||||
return "".join(unit_count.get(candidate_item, candidate_item) for candidate_item in text)
|
||||
|
||||
|
||||
# span merger
|
||||
TRACKING_SPACE_FACTOR = 0.1
|
||||
NON_SPACE_GAP_FACTOR = 0.03
|
||||
NEGATIVE_SPACE_FACTOR = -0.2
|
||||
SPACE_IN_FLOW_MIN_FACTOR = 0.1
|
||||
SPACE_IN_FLOW_MAX_FACTOR = 0.6
|
||||
|
||||
|
||||
# Whitespace classification uses the Unicode WhiteSpace + LineTerminator set.
|
||||
# Python's str.isspace is not the same set: it omits U+FEFF and adds
|
||||
# U+001C-U+001F and U+0085. Use the explicit code points so the
|
||||
# whitespace-skip branch fires on the intended glyphs.
|
||||
_WHITESPACE_CODEPOINTS = frozenset({
|
||||
0x9, 0xA, 0xB, 0xC, 0xD, 0x20, 0xA0, 0x1680,
|
||||
0x2000, 0x2001, 0x2002, 0x2003, 0x2004, 0x2005, 0x2006, 0x2007,
|
||||
0x2008, 0x2009, 0x200A, 0x2028, 0x2029, 0x202F, 0x205F, 0x3000,
|
||||
0xFEFF,
|
||||
})
|
||||
|
||||
|
||||
def _is_whitespace(number: int) -> bool:
|
||||
"""Return whether a glyph code point is classified as whitespace."""
|
||||
return number in _WHITESPACE_CODEPOINTS
|
||||
|
||||
|
||||
# Character classification checks whitespace before marks/formats, so a code
|
||||
# point such as U+FEFF that is also Cf is treated as whitespace, not as an
|
||||
# invisible format mark.
|
||||
def _is_zero_width_diacritic(number: int) -> bool:
|
||||
"""text extraction zero-width diacritic classification (group 2 = ``\\p{Mn}``)."""
|
||||
return number not in _WHITESPACE_CODEPOINTS and unicodedata.category(chr(number)) == "Mn"
|
||||
|
||||
|
||||
def _is_invisible_format_mark(number: int) -> bool:
|
||||
"""text extraction invisible format-mark classification (group 3 = ``\\p{Cf}``)."""
|
||||
return number not in _WHITESPACE_CODEPOINTS and unicodedata.category(chr(number)) == "Cf"
|
||||
|
||||
|
||||
# Bidirectional character-type tables. base bidi type table covers
|
||||
# U+0000..U+00FF; Arabic bidi type table covers U+0600..U+06FF indexed by the low byte
|
||||
# (the "" at 0x1D follows the extraction rule placeholder for nonexistent U+061D).
|
||||
|
||||
_BIDI_BASE_TYPES = (
|
||||
"BN BN BN BN BN BN BN BN BN S B S WS B BN BN BN BN BN BN BN BN BN BN BN BN "
|
||||
"BN BN B B B S WS ON ON ET ET ET ON ON ON ON ON ES CS ES CS CS EN EN EN EN "
|
||||
"EN EN EN EN EN EN CS ON ON ON ON ON ON L L L L L L L L L L L L L L L L L L "
|
||||
"L L L L L L L L ON ON ON ON ON ON L L L L L L L L L L L L L L L L L L L L "
|
||||
"L L L L L L ON ON ON ON BN BN BN BN BN BN B BN BN BN BN BN BN BN BN BN BN "
|
||||
"BN BN BN BN BN BN BN BN BN BN BN BN BN BN BN BN CS ON ET ET ET ET ON ON ON "
|
||||
"ON L ON ON BN ON ON ET ET EN EN ON L ON ON ON EN L ON ON ON ON ON L L L L "
|
||||
"L L L L L L L L L L L L L L L L L L L ON L L L L L L L L L L L L L L L L L "
|
||||
"L L L L L L L L L L L L L L ON L L L L L L L L "
|
||||
).split()
|
||||
assert len(_BIDI_BASE_TYPES) == 256
|
||||
_BIDI_ARABIC_TYPES = [
|
||||
"" if bidi_type == "~" else bidi_type for bidi_type in (
|
||||
"AN AN AN AN AN AN ON ON AL ET ET AL CS AL ON ON NSM NSM NSM NSM NSM NSM "
|
||||
"NSM NSM NSM NSM NSM AL AL ~ AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
|
||||
"AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
|
||||
"AL AL AL AL AL NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM "
|
||||
"NSM NSM NSM NSM NSM NSM AN AN AN AN AN AN AN AN AN AN ET AN AN AL AL AL "
|
||||
"NSM AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
|
||||
"AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
|
||||
"AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
|
||||
"AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
|
||||
"AL AL AL NSM NSM NSM NSM NSM NSM NSM AN ON NSM NSM NSM NSM NSM NSM AL AL "
|
||||
"NSM NSM ON NSM NSM NSM NSM AL AL EN EN EN EN EN EN EN EN EN EN AL AL AL AL "
|
||||
"AL AL "
|
||||
).split()
|
||||
]
|
||||
assert len(_BIDI_ARABIC_TYPES) == 256
|
||||
|
||||
|
||||
def _apply_bidi_reordering(text: str, start_level: int = -1, vertical: bool = False) -> str:
|
||||
"""Apply the simplified single-line UAX#9 pass used for flushed PDF text items. Empty, vertical, and purely LTR text pass through. Otherwise the pass resolves W1-W7/N1-N2/I1-I2 levels from the tables above, reverses runs, and strips literal '<'/'>'. Astral-codepoint handling follows Python strings; surrogate pairs are not corrupted because both halves classify L at equal levels and reversal spans restore the pair."""
|
||||
if not text or vertical:
|
||||
return text
|
||||
count_item = len(text)
|
||||
chars = list(text)
|
||||
types: list[str] = [""] * count_item
|
||||
num_bidi = 0
|
||||
for index_value, char in enumerate(chars):
|
||||
codepoint = ord(char)
|
||||
token_value = "L"
|
||||
if codepoint <= 0xFF:
|
||||
token_value = _BIDI_BASE_TYPES[codepoint]
|
||||
elif 0x0590 <= codepoint <= 0x05F4:
|
||||
token_value = "R"
|
||||
elif 0x0600 <= codepoint <= 0x06FF:
|
||||
token_value = _BIDI_ARABIC_TYPES[codepoint & 0xFF]
|
||||
elif 0x0700 <= codepoint <= 0x08AC:
|
||||
token_value = "AL"
|
||||
if token_value in ("R", "AL", "AN"):
|
||||
num_bidi += 1
|
||||
types[index_value] = token_value
|
||||
if num_bidi == 0:
|
||||
return text
|
||||
if start_level == -1:
|
||||
if num_bidi / count_item < 0.3 and count_item > 4:
|
||||
start_level = 0
|
||||
else:
|
||||
start_level = 1
|
||||
levels = [start_level] * count_item
|
||||
entry_item = "R" if (start_level & 1) else "L"
|
||||
sor = entry_item
|
||||
eor = sor
|
||||
# W1: NSM takes the type of the previous character (sor at run start).
|
||||
last = sor
|
||||
for index_value in range(count_item):
|
||||
if types[index_value] == "NSM":
|
||||
types[index_value] = last
|
||||
else:
|
||||
last = types[index_value]
|
||||
# W2: EN after an AL (searching back to the first strong type) becomes AN.
|
||||
last = sor
|
||||
for index_value in range(count_item):
|
||||
token_value = types[index_value]
|
||||
if token_value == "EN":
|
||||
types[index_value] = "AN" if last == "AL" else "EN"
|
||||
elif token_value in ("R", "L", "AL"):
|
||||
last = token_value
|
||||
# W3: AL -> R.
|
||||
for index_value in range(count_item):
|
||||
if types[index_value] == "AL":
|
||||
types[index_value] = "R"
|
||||
# W4: single ES between ENs -> EN; single CS between same-type numbers.
|
||||
for index_value in range(1, count_item - 1):
|
||||
if types[index_value] == "ES" and types[index_value - 1] == "EN" and types[index_value + 1] == "EN":
|
||||
types[index_value] = "EN"
|
||||
if (types[index_value] == "CS" and types[index_value - 1] in ("EN", "AN")
|
||||
and types[index_value + 1] == types[index_value - 1]):
|
||||
types[index_value] = types[index_value - 1]
|
||||
# W5: ET runs adjacent to EN -> EN.
|
||||
for index_value in range(count_item):
|
||||
if types[index_value] == "EN":
|
||||
for state_item in range(index_value - 1, -1, -1):
|
||||
if types[state_item] != "ET":
|
||||
break
|
||||
types[state_item] = "EN"
|
||||
for state_item in range(index_value + 1, count_item):
|
||||
if types[state_item] != "ET":
|
||||
break
|
||||
types[state_item] = "EN"
|
||||
# W6: WS/ES/ET/CS -> ON.
|
||||
for index_value in range(count_item):
|
||||
if types[index_value] in ("WS", "ES", "ET", "CS"):
|
||||
types[index_value] = "ON"
|
||||
# W7: EN after an L (searching back to the first strong type) -> L.
|
||||
last = sor
|
||||
for index_value in range(count_item):
|
||||
token_value = types[index_value]
|
||||
if token_value == "EN":
|
||||
types[index_value] = "L" if last == "L" else "EN"
|
||||
elif token_value in ("R", "L"):
|
||||
last = token_value
|
||||
# N1: neutrals between same-direction strongs take that direction
|
||||
# (numbers count as R); N2: leftovers take the embedding direction.
|
||||
index_value = 0
|
||||
while index_value < count_item:
|
||||
if types[index_value] == "ON":
|
||||
end = index_value + 1
|
||||
while end < count_item and types[end] == "ON":
|
||||
end += 1
|
||||
before = types[index_value - 1] if index_value > 0 else sor
|
||||
after = types[end + 1] if end + 1 < count_item else eor
|
||||
if before != "L":
|
||||
before = "R"
|
||||
if after != "L":
|
||||
after = "R"
|
||||
if before == after:
|
||||
for state_item in range(index_value, end):
|
||||
types[state_item] = before
|
||||
index_value = end - 1
|
||||
index_value += 1
|
||||
for index_value in range(count_item):
|
||||
if types[index_value] == "ON":
|
||||
types[index_value] = entry_item
|
||||
# I1/I2: level bumps.
|
||||
for index_value in range(count_item):
|
||||
token_value = types[index_value]
|
||||
if levels[index_value] % 2 == 0:
|
||||
if token_value == "R":
|
||||
levels[index_value] += 1
|
||||
elif token_value in ("AN", "EN"):
|
||||
levels[index_value] += 2
|
||||
else:
|
||||
if token_value in ("L", "AN", "EN"):
|
||||
levels[index_value] += 1
|
||||
#: reverse contiguous runs from the highest level down to the lowest
|
||||
# odd level.
|
||||
highest = -1
|
||||
lowest_odd = 99
|
||||
for layout_value in levels:
|
||||
if layout_value > highest:
|
||||
highest = layout_value
|
||||
if layout_value < lowest_odd and (layout_value & 1):
|
||||
lowest_odd = layout_value
|
||||
for level in range(highest, lowest_odd - 1, -1):
|
||||
start = -1
|
||||
for index_value in range(count_item):
|
||||
if levels[index_value] < level:
|
||||
if start >= 0:
|
||||
chars[start:index_value] = chars[start:index_value][::-1]
|
||||
start = -1
|
||||
elif start < 0:
|
||||
start = index_value
|
||||
if start >= 0:
|
||||
chars[start:count_item] = chars[start:count_item][::-1]
|
||||
# text extraction final loop: literal '<' and '>' are dropped (numBidi > 0 only).
|
||||
return "".join("" if char in "<>" else char for char in chars)
|
||||
|
||||
|
||||
def _rtl_sign(char: str) -> int:
|
||||
"""+1 for LTR runs, -1 for a strong right-to-left char (bidi class R/AL, e.g. Hebrew/Arabic). PDFium reports RTL text in logical order with decreasing char origins, so the LTR ``advance = ox - prev_text_x`` model (prev_text_x = ox+glyph_w, a right edge) yields a large negative advance. For RTL chunks the x-axis is signed with ``sign*ox`` so the reading-direction advance is positive and the existing LTR merge logic applies unchanged."""
|
||||
return -1 if unicodedata.bidirectional(char) in ("R", "AL") else 1
|
||||
|
||||
|
||||
def _reverse_if_rtl(chars: str) -> str:
|
||||
"""span merger ``RTL ligature reversal`` : reverse a multi-char (Arabic/Hebrew ligature) value when its FIRST code unit is in the Hebrew ``[0x0590,0x05ff)`` or Arabic ``[0x0600,0x06ff)`` range (Unicode range table[11]/ [13], ``right-to-left range test`` uses ``>= begin and < end``, so the range end is EXCLUSIVE). text extraction wraps every glyph's ``normalized Unicode`` in this, so a table value like "\u0626\u062c" emitted for U+FC00 is reversed to "\u062c\u0626"; a single-char value (length <= 1) is returned as-is."""
|
||||
if len(chars) <= 1:
|
||||
return chars
|
||||
first_codepoint = ord(chars[0])
|
||||
if (0x0590 <= first_codepoint < 0x05FF) or (0x0600 <= first_codepoint < 0x06FF):
|
||||
return chars[::-1]
|
||||
return chars
|
||||
|
||||
|
||||
def _read_end(mapping: dict, sign: int) -> float:
|
||||
"""The reading-direction FAR edge of a glyph (the edge facing the next char). PDFium reports the origin (ox) as the glyph's LEFT edge in both directions; the glyph extends RIGHT by glyph_w. So: * LTR (reading right): far edge = right edge = max(ox+glyph_w, ink right). * RTL (reading left): far edge = LEFT edge = ox (the origin itself). The next char's gap is then measured to its NEAR edge -- ox for LTR, ox+glyph_w for RTL -- in ``_read_gap`` below. (Earlier this added glyph_w on the RTL side too, which used the PREVIOUS glyph's width and injected spurious spaces.)"""
|
||||
if sign > 0:
|
||||
return max(mapping["ox"] + mapping["glyph_w"], mapping["right"])
|
||||
return mapping["ox"]
|
||||
|
||||
|
||||
def _read_gap(prev_far: float, other_mapping: dict, sign: int) -> float:
|
||||
"""Reading-direction gap between the previous glyph's far edge and the current glyph's NEAR edge. LTR near edge = ox (left); RTL near edge = ox+glyph_w (right). ==0 for adjacent glyphs, >0 for a word gap, <0 for a backward jump."""
|
||||
if sign > 0:
|
||||
return other_mapping["ox"] - prev_far
|
||||
return prev_far - (other_mapping["ox"] + other_mapping["glyph_w"])
|
||||
@@ -0,0 +1,375 @@
|
||||
"""Applies per-font Unicode maps to page chars and synthesizes dropped glyphs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import bisect
|
||||
import difflib
|
||||
from collections import Counter
|
||||
import pypdfium2.raw as pdfium_c
|
||||
|
||||
from .text_normalize import _is_whitespace
|
||||
from .font_unicode import _font_unicode_map
|
||||
from .code_walk import (
|
||||
_char_category,
|
||||
_walk_codes,
|
||||
)
|
||||
|
||||
|
||||
def _apply_font_unicode(
|
||||
text_page,
|
||||
raw_chars: list[dict],
|
||||
objects: list[dict],
|
||||
show_codes: list[tuple[int | None, tuple[int, ...], float]],
|
||||
pdf_doc,
|
||||
map_cache: dict,
|
||||
) -> None:
|
||||
"""Patch each char's unicode to span merger glyph Unicode (`map.get(code) or chr(code)`, content stream tokenizer glyph mapping) where PDFium's decode disagrees. Two granularities, both gated by _walk_codes' both-streams-exhaust rule: - object mode (when PDFium's text objects pair consistent with the page's show ops, the _assign_flush_ids precondition): each object's chars are walked against its own show op's codes. This is immune to PDFium's textpage segment reordering (e.g. math-heavy page margin labels emitted at a different page position than paint order) because chars keep stream order WITHIN an object; a desync rolls back only that object. - page mode (counts differ, e.g. PDFium splitting a TJ into several objects): all non-generated textpage chars are walked against all show ops' codes in paint order; any desync rolls back the whole page. """
|
||||
if not show_codes:
|
||||
return
|
||||
|
||||
def targets_for(font_xref: int | None, other_numbers: tuple[int, ...]) -> list[str] | None:
|
||||
if font_xref is None:
|
||||
return None
|
||||
if font_xref not in map_cache:
|
||||
try:
|
||||
map_cache[font_xref] = _font_unicode_map(pdf_doc, font_xref)
|
||||
except Exception:
|
||||
map_cache[font_xref] = None
|
||||
entry = map_cache[font_xref]
|
||||
if entry is None:
|
||||
return None
|
||||
next_block, measure_item = entry
|
||||
if next_block == 1:
|
||||
return [measure_item.get(code) or chr(code) for code in other_numbers]
|
||||
return [measure_item.get((other_numbers[key_value] << 8) | other_numbers[key_value + 1]) or chr((other_numbers[key_value] << 8) | other_numbers[key_value + 1])
|
||||
for key_value in range(0, len(other_numbers) - 1, 2)]
|
||||
|
||||
def apply(patches: list[tuple[int, str]], drops: list[int],
|
||||
chars_by_index: dict[int, dict]) -> None:
|
||||
for index_value, token_value in patches:
|
||||
candidate_item = chars_by_index.get(index_value)
|
||||
if candidate_item is None:
|
||||
continue # char was dropped at extraction; nothing to patch
|
||||
candidate_item["ch"] = token_value
|
||||
candidate_item["is_ws"], candidate_item["is_mn"], candidate_item["is_cf"] = _char_category(token_value)
|
||||
for index_value in drops:
|
||||
candidate_item = chars_by_index.get(index_value)
|
||||
if candidate_item is not None:
|
||||
candidate_item["drop"] = True
|
||||
|
||||
chars_by_index = {raw_char["i"]: raw_char for raw_char in raw_chars}
|
||||
|
||||
if len(objects) == len(show_codes):
|
||||
# Object mode: pair text objects with show ops ordinally (both are in
|
||||
# content-stream paint order) and walk each pair independently.
|
||||
chars_by_obj: dict[int, list[tuple[int, str]]] = {}
|
||||
for raw_char in raw_chars:
|
||||
if raw_char["is_gen"]:
|
||||
continue
|
||||
chars_by_obj.setdefault(id(raw_char["obj"]), []).append((raw_char["i"], raw_char["ch"]))
|
||||
desynced: list[int] = []
|
||||
failed_windows: list[list[int]] = []
|
||||
synth_sites: list[dict] = []
|
||||
targets_by_object_index: dict[int, list[str] | None] = {}
|
||||
for object_index, (obj, (font_index, encoded_text, _tz)) in enumerate(zip(objects, show_codes)):
|
||||
target_text_items = targets_for(font_index, encoded_text)
|
||||
targets_by_object_index[object_index] = target_text_items
|
||||
if target_text_items is None:
|
||||
continue # uncovered font: this object keeps PDFium's output
|
||||
res = _walk_codes(chars_by_obj.get(id(obj), []), target_text_items)
|
||||
if res is None:
|
||||
# Desync: often a boundary-attribution error (the geometric
|
||||
# char->object lookup parks a show op's edge glyph in the
|
||||
# NEIGHBOURING object's list: punctuation at a run boundary can
|
||||
# land in the previous object, and heavily overlapped chart
|
||||
# labels can park a leading glyph in the wrong object. Record for the
|
||||
# window re-walk below; a genuine mismatch stays rolled back
|
||||
# there too.
|
||||
desynced.append(object_index)
|
||||
continue
|
||||
apply(res[0], res[1], chars_by_index)
|
||||
# Re-walk each window of desynced objects (bridging up to 2 covered,
|
||||
# successfully-walked objects between them) as one unit: boundary-
|
||||
# attribution errors cancel inside the window (the page-mode walk
|
||||
# scoped to the ambiguous region) and the exhaust-in-sync gate still
|
||||
# rejects anything else. On commit, REASSIGN each consumed char to
|
||||
# the object whose show op consumed it -- the stream-side ownership --
|
||||
# repairing the geometric attribution for the paint-order sort, the
|
||||
# merger's font/fs identity and the Type-3 sizing alike.
|
||||
def _rewalk_window(window: list[int]) -> bool:
|
||||
char_value = sorted(
|
||||
(pair for state_item in window for pair in chars_by_obj.get(id(objects[state_item]), [])))
|
||||
text_transform: list[str] = []
|
||||
owner: list[int] = []
|
||||
for state_item in window:
|
||||
target_text_items = targets_by_object_index[state_item]
|
||||
assert target_text_items is not None
|
||||
text_transform.extend(target_text_items)
|
||||
owner.extend([state_item] * len(target_text_items))
|
||||
def _commit(res) -> bool:
|
||||
if res is None:
|
||||
return False
|
||||
apply(res[0], res[1], chars_by_index)
|
||||
for char_index, text_index in res[2]:
|
||||
candidate_item = chars_by_index.get(char_index)
|
||||
if candidate_item is not None and candidate_item["obj"] is not objects[owner[text_index]]:
|
||||
candidate_item["obj"] = objects[owner[text_index]]
|
||||
# Skipped targets are glyphs PDFium never emitted; record
|
||||
# each with its show op and surviving stream neighbours so
|
||||
# _synthesize_dropped_glyphs can re-emit it (text extraction does).
|
||||
for text_index, pos in res[3]:
|
||||
synth_sites.append({
|
||||
"t": text_transform[text_index], "owner": objects[owner[text_index]],
|
||||
"prev_i": char_value[pos - 1][0] if pos > 0 else None,
|
||||
"next_i": char_value[pos][0] if pos < len(char_value) else None,
|
||||
})
|
||||
return True
|
||||
if len(window) >= 2 and _commit(_walk_codes(char_value, text_transform)):
|
||||
return True
|
||||
# Pure-displacement fallback: PDFium's textpage can also REORDER a
|
||||
# char across the window (TeX accents again: 'accented word stem'+'´'+'es'
|
||||
# arrives as '...ilites´', and the 't' sits in the 'es' object),
|
||||
# which the linear walk above can never align. When the chars are
|
||||
# EXACTLY the targets as a multiset (no decode work left -- only
|
||||
# placement is wrong), align via SequenceMatcher and repair
|
||||
# OWNERSHIP alone: equal blocks map positionally, the few
|
||||
# displaced chars (<=4) map by literal value. Single-char targets
|
||||
# only, so target index == string position.
|
||||
def _displacement_repair() -> bool:
|
||||
if any(len(token_value) != 1 for token_value in text_transform):
|
||||
return False
|
||||
chs = "".join(candidate_item for _, candidate_item in char_value)
|
||||
tts = "".join(text_transform)
|
||||
deficit = len(tts) - len(chs)
|
||||
if (chs == tts or deficit < 0 or deficit > 8
|
||||
or (Counter(chs) - Counter(tts))):
|
||||
return False
|
||||
state_map = difflib.SequenceMatcher(None, tts, chs, autojunk=False)
|
||||
char_to_tgt: dict[int, int] = {}
|
||||
loose_target_indexes: list[int] = []
|
||||
loose_char_indexes: list[int] = []
|
||||
for tag, index_one, index_two, char_start, char_end in state_map.get_opcodes():
|
||||
if tag == "equal":
|
||||
for reference_item in range(index_two - index_one):
|
||||
char_to_tgt[char_start + reference_item] = index_one + reference_item
|
||||
else:
|
||||
loose_target_indexes.extend(range(index_one, index_two))
|
||||
loose_char_indexes.extend(range(char_start, char_end))
|
||||
if len(loose_char_indexes) > 24:
|
||||
return False
|
||||
used_targets: set[int] = set()
|
||||
for char_index in loose_char_indexes:
|
||||
cdict = chars_by_index.get(char_value[char_index][0])
|
||||
cands = [target_index for target_index in loose_target_indexes
|
||||
if target_index not in used_targets and tts[target_index] == chs[char_index]]
|
||||
if not cands:
|
||||
return False # a displaced char with no equal target
|
||||
if cdict is not None and len(cands) > 1:
|
||||
# Identical glyphs (the 21 scattered 'α' labels):
|
||||
# pick the candidate whose OBJECT box sits closest
|
||||
# to the char -- the one signal that distinguishes
|
||||
# equal-valued slots.
|
||||
origin_x, origin_y = cdict["ox"], cdict["oy"]
|
||||
def _object_distance_sq(target_index: int) -> float:
|
||||
item_value = objects[owner[target_index]]
|
||||
delta_x = max(item_value["l"] - origin_x, 0.0, origin_x - item_value["r"])
|
||||
delta_y = max(item_value["b"] - origin_y, 0.0, origin_y - item_value["t"])
|
||||
return delta_x * delta_x + delta_y * delta_y
|
||||
cands.sort(key=_object_distance_sq)
|
||||
char_to_tgt[char_index] = cands[0]
|
||||
used_targets.add(cands[0])
|
||||
# Leftover loose TARGETS = glyphs PDFium never emitted (the
|
||||
# font-layer drop class). Record each between its
|
||||
# nearest MAPPED neighbours for re-synthesis.
|
||||
leftover = [target_index for target_index in loose_target_indexes if target_index not in used_targets]
|
||||
if leftover:
|
||||
tgt_to_char = {target_index: char_index for char_index, target_index in char_to_tgt.items()}
|
||||
mapped_tis = sorted(tgt_to_char)
|
||||
for target_index in leftover:
|
||||
page_value = bisect.bisect_left(mapped_tis, target_index)
|
||||
point_value = mapped_tis[page_value - 1] if page_value > 0 else None
|
||||
normalized_token = mapped_tis[page_value] if page_value < len(mapped_tis) else None
|
||||
synth_sites.append({
|
||||
"t": tts[target_index], "owner": objects[owner[target_index]],
|
||||
"prev_i": char_value[tgt_to_char[point_value]][0] if point_value is not None else None,
|
||||
"next_i": char_value[tgt_to_char[normalized_token]][0] if normalized_token is not None else None,
|
||||
})
|
||||
for char_index, target_index in char_to_tgt.items():
|
||||
candidate_item = chars_by_index.get(char_value[char_index][0])
|
||||
if candidate_item is not None and candidate_item["obj"] is not objects[owner[target_index]]:
|
||||
candidate_item["obj"] = objects[owner[target_index]]
|
||||
return True
|
||||
if _displacement_repair():
|
||||
return True
|
||||
# Final resort: the same walk with anchored drop-skips, for
|
||||
# windows containing glyphs PDFium never emitted (font-layer
|
||||
# drops). The rest of the window still gets its patches and
|
||||
# stream-side ownership; the dropped glyphs are recorded for
|
||||
# synthesis.
|
||||
if not _commit(_walk_codes(char_value, text_transform, allow_skips=True)):
|
||||
failed_windows.append(list(window))
|
||||
return False
|
||||
return True
|
||||
# A window that resolves only by DECLARING drops (recording synth
|
||||
# sites) has trusted its local char census; when chars were stolen
|
||||
# ACROSS window boundaries that census lies (a starved window
|
||||
# "drops" a glyph whose char sits, surplus, in another failed
|
||||
# window). Track those windows so the mega pass below can supersede
|
||||
# their local verdicts.
|
||||
synth_windows: list[tuple[list[int], int, int]] = []
|
||||
def _run_window(window: list[int]) -> None:
|
||||
before = len(synth_sites)
|
||||
if _rewalk_window(window) and len(synth_sites) > before:
|
||||
synth_windows.append((list(window), before, len(synth_sites)))
|
||||
window: list[int] = []
|
||||
for object_index in desynced:
|
||||
if window:
|
||||
gap = range(window[-1] + 1, object_index)
|
||||
if (len(gap) <= 2
|
||||
and all(targets_by_object_index.get(bridge_index) is not None for bridge_index in gap)):
|
||||
window.extend(gap)
|
||||
window.append(object_index)
|
||||
continue
|
||||
_run_window(window)
|
||||
window = [object_index]
|
||||
if window:
|
||||
_run_window(window)
|
||||
# Page-scope last resort: scattered same-glyph labels (dense math-heavy page's
|
||||
# 21 'α' show ops over a vector figure) defeat per-window walks --
|
||||
# the geometric attribution piles several chars on some ops and
|
||||
# leaves others empty ACROSS window boundaries (donor ops hold a
|
||||
# stolen surplus char, starved ops none). Merge every failed AND
|
||||
# every drop-declaring window into one final window so the
|
||||
# displacement/skip repairs see the whole cluster at once: the
|
||||
# surplus cancels the deficit, stolen chars are reassigned to their
|
||||
# true ops, and only the genuine font-layer drops remain as synth
|
||||
# sites. The locally-recorded sites are dropped first (the mega
|
||||
# re-records with full context) and restored if the mega fails.
|
||||
cand = failed_windows + [window for window, _, _ in synth_windows]
|
||||
if len(cand) >= 2:
|
||||
stash = synth_sites[:]
|
||||
for _, font, window_end in reversed(synth_windows):
|
||||
del synth_sites[font:window_end]
|
||||
failed_windows = []
|
||||
mega = sorted({mega_index for window in cand for mega_index in window})
|
||||
if not _rewalk_window(mega):
|
||||
synth_sites[:] = stash # mega failed: keep local verdicts
|
||||
if synth_sites:
|
||||
# Census gate: a recorded site is a REAL font-layer drop only if
|
||||
# the PAGE-WIDE multiset still misses that value (covered ops'
|
||||
# target codepoints minus PDFium's final chars). A window-local
|
||||
# repair can otherwise declare a glyph dropped whose char simply
|
||||
# sits, mis-attributed, in an op that walked clean -- the a clipped-cell table
|
||||
# with star glyphs: 7 star codes, 7 star chars page-wide, but the
|
||||
# clip-overlapped cells starve two ops, and the donors never
|
||||
# fail so the mega pass can't see them. WHITESPACE is never
|
||||
# synthesized: a missing space char is PDFium's textpage
|
||||
# space-run normalization (text extraction runs its own space
|
||||
# normalization, already implemented in the merger), not a font-layer
|
||||
# drop.
|
||||
census: Counter = Counter()
|
||||
for text_adjustment in targets_by_object_index.values():
|
||||
if text_adjustment is not None:
|
||||
for target_text in text_adjustment:
|
||||
census.update(target_text)
|
||||
for raw_char in raw_chars:
|
||||
if not raw_char["is_gen"] and not raw_char.get("drop"):
|
||||
census.subtract(raw_char["ch"])
|
||||
kept: list[dict] = []
|
||||
for encoded_text in synth_sites:
|
||||
if all(_is_whitespace(ord(ch_)) for ch_ in encoded_text["t"]):
|
||||
continue
|
||||
if all(census[ch_] > 0 for ch_ in encoded_text["t"]):
|
||||
for ch_ in encoded_text["t"]:
|
||||
census[ch_] -= 1
|
||||
kept.append(encoded_text)
|
||||
if kept:
|
||||
_synthesize_dropped_glyphs(kept, raw_chars, chars_by_index)
|
||||
return
|
||||
|
||||
# Page mode.
|
||||
seq: list[tuple[int, str]] = []
|
||||
char_count = pdfium_c.FPDFText_CountChars(text_page)
|
||||
for char_index in range(char_count):
|
||||
if pdfium_c.FPDFText_IsGenerated(text_page, char_index) == 1:
|
||||
continue
|
||||
codepoint = pdfium_c.FPDFText_GetUnicode(text_page, char_index)
|
||||
seq.append((char_index, chr(codepoint) if codepoint > 0 else "\x00"))
|
||||
targets: list[str] = []
|
||||
for font_index, encoded_text, _tz in show_codes:
|
||||
if not encoded_text:
|
||||
continue
|
||||
text_state = targets_for(font_index, encoded_text)
|
||||
if text_state is None:
|
||||
return # uncovered font used on this page: no patch
|
||||
targets.extend(text_state)
|
||||
res = _walk_codes(seq, targets)
|
||||
if res is None:
|
||||
return
|
||||
apply(res[0], res[1], chars_by_index)
|
||||
|
||||
|
||||
def _synthesize_dropped_glyphs(
|
||||
sites: list[dict], raw_chars: list[dict], chars_by_index: dict[int, dict],
|
||||
) -> None:
|
||||
"""Re-emit glyphs PDFium's font layer never produced, even though the content stream contains them. Geometry comes from the pen model rather than a guess: PDFium still advances the pen over the missing glyph when placing surviving neighbours, so a dropped glyph starts at the previous survivor's advance-cell right edge and its advance is the gap to the next survivor's origin. With no surviving neighbour on a side, the advance is unknowable; emit zero-width there so presence and stream order are preserved without inserting a synthetic gap."""
|
||||
groups: list[list[dict]] = []
|
||||
for site in sites:
|
||||
if (groups and groups[-1][0]["prev_i"] == site["prev_i"]
|
||||
and groups[-1][0]["next_i"] == site["next_i"]
|
||||
and groups[-1][0]["owner"] is site["owner"]):
|
||||
groups[-1].append(site)
|
||||
else:
|
||||
groups.append([site])
|
||||
for group_value in groups:
|
||||
owner = group_value[0]["owner"]
|
||||
prev = chars_by_index.get(group_value[0]["prev_i"]) if group_value[0]["prev_i"] is not None else None
|
||||
nxt = chars_by_index.get(group_value[0]["next_i"]) if group_value[0]["next_i"] is not None else None
|
||||
text = "".join(site["t"] for site in group_value) # one char per target codepoint
|
||||
count_item = len(text)
|
||||
if not count_item:
|
||||
continue
|
||||
if prev is not None:
|
||||
pen, baseline_y = prev["right"], prev["oy"]
|
||||
elif nxt is not None:
|
||||
pen, baseline_y = nxt["ox"], nxt["oy"]
|
||||
else:
|
||||
# Whole show op dropped: park at the object box's pen start.
|
||||
pen, baseline_y = owner["l"], owner["b"]
|
||||
total = 0.0
|
||||
if (prev is not None and nxt is not None
|
||||
and abs(nxt["oy"] - baseline_y) < 0.5 and nxt["ox"] > pen):
|
||||
total = nxt["ox"] - pen
|
||||
adv = total / count_item
|
||||
# Textpage index: fractional, slotted against the owner's own chars
|
||||
# so the paint-order sort keys (page_order, i) place the run in
|
||||
# stream position; only order WITHIN the owner object matters.
|
||||
if prev is not None and prev["obj"] is owner:
|
||||
base, sgn = prev["i"], 1.0
|
||||
elif nxt is not None and nxt["obj"] is owner:
|
||||
base, sgn = nxt["i"], -1.0
|
||||
elif prev is not None:
|
||||
base, sgn = prev["i"], 1.0
|
||||
elif nxt is not None:
|
||||
base, sgn = nxt["i"], -1.0
|
||||
else:
|
||||
base, sgn = -1.0, 1.0
|
||||
for key_value, char in enumerate(text):
|
||||
is_ws, is_mn, is_cf = _char_category(char)
|
||||
glyph_left = pen + adv * key_value
|
||||
step = (key_value + 1) if sgn > 0 else (count_item - key_value)
|
||||
raw_chars.append({
|
||||
"i": base + sgn * step * 1e-3,
|
||||
"ch": char, "u": ord(char),
|
||||
"is_gen": False, "synth": True,
|
||||
"is_ws": is_ws, "is_mn": is_mn, "is_cf": is_cf,
|
||||
"ox": glyph_left, "oy": baseline_y,
|
||||
"left": glyph_left, "right": glyph_left + adv,
|
||||
"top": baseline_y + owner["fs_eff"], "bottom": baseline_y,
|
||||
# Degenerate ink box: PDFium reports no ink box for the glyph
|
||||
# (this also keeps it out of the Type-3 extent union).
|
||||
"box_top": baseline_y, "box_bottom": baseline_y,
|
||||
"cell_top": baseline_y, "cell_bot": baseline_y,
|
||||
"w_raw": 0.0, "w_synth": adv,
|
||||
"obj": owner, "font_name": owner["font_name"],
|
||||
})
|
||||
@@ -0,0 +1,139 @@
|
||||
"""Per-page parallel driver for the charlevel parser.
|
||||
|
||||
Wraps the UNMODIFIED per-page pipeline (``_page_pass1`` / ``_page_pass2`` /
|
||||
``_page_spans``) in a process pool. PDFium's FFI is not thread-safe and its
|
||||
handles are process-local, so parallelism uses processes, each opening its
|
||||
own copy of the document.
|
||||
|
||||
Parity contract: per-page processing depends on no cross-page state
|
||||
except the document-wide identity-matrix Type-3 extent union. An empty union
|
||||
makes ``_apply_type3_sizes`` a no-op, so per-page == whole-document exactly.
|
||||
Workers run pass 1 + pass 2 per page assuming the union stays empty and
|
||||
poison the run the moment any page accumulates an extent; the driver then
|
||||
discards the parallel attempt and reruns the document on the sequential
|
||||
path, which is the source of truth. Any other worker failure falls back the
|
||||
same way, so this entry can only ever return sequential-identical output.
|
||||
|
||||
Worker startup pays the full package import chain plus its own document
|
||||
open; ``min_pages`` routes documents too small to amortize that to the
|
||||
sequential path directly.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import multiprocessing
|
||||
import os
|
||||
from concurrent.futures import ProcessPoolExecutor
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from typing import Union
|
||||
|
||||
import pypdfium2 as pdfium
|
||||
import PyPDF2 as _pypdf2 # declared dependency (also imported by pageindex.utils/client)
|
||||
|
||||
from .model import Span
|
||||
from .parser_pdfium_charlevel import (
|
||||
parse_charlevel_meta,
|
||||
_PdfDoc,
|
||||
_page_pass1,
|
||||
_page_pass2,
|
||||
_page_spans,
|
||||
)
|
||||
|
||||
_MIN_PARALLEL_PAGES = 64
|
||||
|
||||
|
||||
class _Type3Detected(Exception):
|
||||
"""A page accumulated an identity-matrix Type-3 extent: the document
|
||||
needs the cross-page font sizing only the sequential path performs."""
|
||||
|
||||
|
||||
# Per-worker state, set once by _init_worker in each spawned process.
|
||||
_worker_pdf = None
|
||||
_worker_pdf_doc = None
|
||||
_worker_font_maps: dict = {}
|
||||
|
||||
|
||||
def _init_worker(kind: str, payload) -> None:
|
||||
global _worker_pdf, _worker_pdf_doc, _worker_font_maps
|
||||
# Open the document exactly as parse_charlevel_meta does, including
|
||||
# the guarded PyPDF2 open and its separate bytes copy.
|
||||
if kind == "path":
|
||||
_worker_pdf = pdfium.PdfDocument(payload)
|
||||
else:
|
||||
_worker_pdf = pdfium.PdfDocument(BytesIO(payload))
|
||||
_worker_pdf_doc = None
|
||||
if _pypdf2 is not None:
|
||||
try:
|
||||
if kind == "path":
|
||||
_worker_pdf_doc = _PdfDoc(_pypdf2.PdfReader(payload))
|
||||
else:
|
||||
_worker_pdf_doc = _PdfDoc(_pypdf2.PdfReader(BytesIO(payload)))
|
||||
except Exception:
|
||||
_worker_pdf_doc = None
|
||||
_worker_font_maps = {}
|
||||
|
||||
|
||||
def _run_page(page_idx: int):
|
||||
type3_ext: dict = {}
|
||||
page, raw_chars, page_vb, page_rot = _page_pass1(
|
||||
_worker_pdf, _worker_pdf_doc, page_idx, type3_ext, _worker_font_maps)
|
||||
try:
|
||||
if type3_ext:
|
||||
raise _Type3Detected(page_idx)
|
||||
merged = _page_pass2(raw_chars, page_vb, {})
|
||||
spans = _page_spans(merged)
|
||||
finally:
|
||||
page.close()
|
||||
return spans, (page_vb, page_rot)
|
||||
|
||||
|
||||
def parse_charlevel_meta_parallel(
|
||||
doc_handle: Union[str, Path, BytesIO],
|
||||
workers: int | None = None,
|
||||
min_pages: int = _MIN_PARALLEL_PAGES,
|
||||
) -> tuple[list[list[Span]], list]:
|
||||
"""Parallel-when-possible variant of ``parse_charlevel_meta``.
|
||||
|
||||
Returns the same ``(pages, page_meta)`` with identical content for
|
||||
every input. ``workers`` caps the pool size (default: CPU count - 1).
|
||||
"""
|
||||
if isinstance(doc_handle, (str, Path)):
|
||||
src = ("path", str(doc_handle))
|
||||
elif isinstance(doc_handle, BytesIO):
|
||||
src = ("bytes", doc_handle.getvalue())
|
||||
else:
|
||||
# An already-open PdfDocument cannot be reopened per worker.
|
||||
return parse_charlevel_meta(doc_handle)
|
||||
|
||||
probe = pdfium.PdfDocument(BytesIO(src[1]) if src[0] == "bytes" else src[1])
|
||||
n_pages = len(probe)
|
||||
probe.close()
|
||||
|
||||
max_w = max(1, (os.cpu_count() or 2) - 1)
|
||||
w = max(1, min(workers if workers is not None else max_w, max_w, n_pages))
|
||||
if w <= 1 or n_pages < min_pages:
|
||||
return parse_charlevel_meta(doc_handle)
|
||||
|
||||
executor = ProcessPoolExecutor(
|
||||
max_workers=w,
|
||||
mp_context=multiprocessing.get_context("spawn"),
|
||||
initializer=_init_worker,
|
||||
initargs=src,
|
||||
)
|
||||
try:
|
||||
results = list(executor.map(_run_page, range(n_pages)))
|
||||
except Exception:
|
||||
# _Type3Detected or any worker/pool failure. Cancel what is queued
|
||||
# and rerun sequentially; in-flight pages finish in their workers
|
||||
# and are discarded (separate processes, no shared PDFium state).
|
||||
executor.shutdown(wait=False, cancel_futures=True)
|
||||
return parse_charlevel_meta(doc_handle)
|
||||
executor.shutdown()
|
||||
|
||||
out = [spans for spans, _meta_entry in results]
|
||||
meta = [meta_entry for _spans, meta_entry in results]
|
||||
return out, meta
|
||||
|
||||
|
||||
__all__ = ["parse_charlevel_meta_parallel"]
|
||||
@@ -0,0 +1,50 @@
|
||||
"""Per-page pipeline orchestration. For each page, the extractor builds initial lines, computes page statistics,
|
||||
detects columns, reclusters lines with column awareness, removes line-number
|
||||
artifacts, recomputes statistics, and assigns reading order.
|
||||
"""
|
||||
|
||||
import math
|
||||
from typing import Optional
|
||||
|
||||
from sortedcontainers import SortedKeyList
|
||||
|
||||
from ..clustering import LinesContainer, cluster_lines, build_initial_lines
|
||||
from ..columns import detect_columns, ColumnDetectionContext, columns_to_x_bounds
|
||||
from ..model import (
|
||||
Span,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
x_centers_close,
|
||||
to_number,
|
||||
Rect,
|
||||
append_span,
|
||||
avg_char_width,
|
||||
Line,
|
||||
info_weight,
|
||||
)
|
||||
from ..stats import column_index_of, PageStats, compute_page_stats
|
||||
|
||||
from .page_view import (
|
||||
PageView,
|
||||
assign_reading_order,
|
||||
process_page,
|
||||
)
|
||||
from .line_numbers import (
|
||||
LineNumberCluster,
|
||||
init_line_number_cluster,
|
||||
nearest_cluster,
|
||||
validate_line_number_cluster,
|
||||
strip_line_numbers,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"assign_reading_order",
|
||||
"LineNumberCluster",
|
||||
"init_line_number_cluster",
|
||||
"nearest_cluster",
|
||||
"validate_line_number_cluster",
|
||||
"strip_line_numbers",
|
||||
"PageView",
|
||||
"process_page",
|
||||
]
|
||||
@@ -0,0 +1,162 @@
|
||||
"""Line-number column detection and stripping."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Optional
|
||||
|
||||
from sortedcontainers import SortedKeyList
|
||||
from ..model import (
|
||||
Span,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
x_centers_close,
|
||||
to_number,
|
||||
Rect,
|
||||
append_span,
|
||||
avg_char_width,
|
||||
Line,
|
||||
info_weight,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Line-number stripper #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class LineNumberCluster:
|
||||
"""Drop-cap or line-number cluster used to detect removable line numbers."""
|
||||
|
||||
__slots__ = ("lines", "left", "secondary_slot", "primary_slot", "is_valid_sequence")
|
||||
|
||||
def __init__(self, line: Line, candidate_item: float, valid_sequence_flag: bool):
|
||||
self.lines: list = [line]
|
||||
self.left: float = line.left_edge()
|
||||
self.secondary_slot: float = avg_char_width(line)
|
||||
self.primary_slot: float = candidate_item
|
||||
self.is_valid_sequence: bool = valid_sequence_flag
|
||||
|
||||
|
||||
def init_line_number_cluster(line: Line) -> LineNumberCluster:
|
||||
"""Build an initial line-number cluster for a candidate line."""
|
||||
line_number = to_number(line.primary_slot[0].state_slot)
|
||||
is_valid_integer = (line_number > 0 and line_number < 1e4 and not math.isnan(line_number) and line_number == math.floor(line_number))
|
||||
return LineNumberCluster(line, line_number, is_valid_integer)
|
||||
|
||||
|
||||
def nearest_cluster(line: LineNumberCluster, other_line: Optional[LineNumberCluster], candidate_line: Optional[LineNumberCluster]) -> Optional[LineNumberCluster]:
|
||||
"""Pick the nearer left or right cluster within two character heights."""
|
||||
distance = (line.left - other_line.left) if other_line is not None else math.inf
|
||||
candidate_distance = (candidate_line.left - line.left) if candidate_line is not None else math.inf
|
||||
tol = 2 * line.secondary_slot
|
||||
if distance > tol and candidate_distance > tol:
|
||||
return None
|
||||
return other_line if distance < candidate_distance else candidate_line
|
||||
|
||||
|
||||
def validate_line_number_cluster(rect: Rect, other_lines: list[Line], candidate_line: LineNumberCluster) -> bool:
|
||||
"""validate a candidate cluster (>= 5 lines, near left edge, bulk of body weight overlapping the cluster's vertical span)."""
|
||||
if len(candidate_line.lines) < 5:
|
||||
return False
|
||||
if candidate_line.left < 0.05 * rect.bbox_width():
|
||||
return True
|
||||
empty_line_count = 0
|
||||
flag = False
|
||||
top = -math.inf
|
||||
bot = math.inf
|
||||
min_gap = math.inf
|
||||
max_gap = -math.inf
|
||||
prev: Optional[Line] = None
|
||||
for cluster_line in candidate_line.lines:
|
||||
if cluster_line.char_count() - cluster_line.char_stats.primary_slot[1] <= 0:
|
||||
empty_line_count += 1
|
||||
first = cluster_line.alignment_slot
|
||||
if first and first.char_stats.secondary_slot == 3:
|
||||
flag = True
|
||||
top = max(top, cluster_line.top_edge())
|
||||
bot = min(bot, cluster_line.bottom_edge())
|
||||
if prev is not None:
|
||||
gap = prev.bottom_edge() - cluster_line.bottom_edge()
|
||||
min_gap = min(min_gap, gap)
|
||||
max_gap = max(max_gap, gap)
|
||||
prev = cluster_line
|
||||
# Preserve IEEE-754 division for the spacing-ratio test.
|
||||
|
||||
if min_gap != 0:
|
||||
gap_ratio = max_gap / min_gap
|
||||
elif max_gap != 0:
|
||||
gap_ratio = math.copysign(math.inf, max_gap)
|
||||
else:
|
||||
gap_ratio = math.nan
|
||||
if empty_line_count < len(candidate_line.lines) / 2 and (not flag or gap_ratio > 1.3):
|
||||
return False
|
||||
total = 0.0
|
||||
covered = 0.0
|
||||
for line in other_lines:
|
||||
block_weight = info_weight(line.char_stats)
|
||||
total += block_weight
|
||||
if line.bottom_edge() < top and line.top_edge() > bot:
|
||||
covered += block_weight
|
||||
return covered >= 0.8 * total
|
||||
|
||||
|
||||
def strip_line_numbers(rect: Rect, other_lines: list[Line]) -> list[Line]:
|
||||
"""Detect a column of line numbers and strip it. Returns the original lines if no line-numbering pattern is detected."""
|
||||
# Cluster candidates by ``left`` x-position. SortedKeyList by left.
|
||||
cluster_tree: SortedKeyList = SortedKeyList(key=lambda line_key: line_key.left)
|
||||
for line in other_lines:
|
||||
if len(line.primary_slot) == 0 or len(line.primary_slot[0].state_slot) == 0:
|
||||
continue
|
||||
if line.left_edge() > 0.15 * rect.bbox_width():
|
||||
continue
|
||||
candidate_cluster = init_line_number_cluster(line)
|
||||
if not candidate_cluster.is_valid_sequence:
|
||||
continue
|
||||
# Equal-left clusters must merge, so predecessor/successor lookup is
|
||||
# inclusive: successor = first left >= current, predecessor = last left <=
|
||||
# current. Strict bisect would fragment a fixed-x line-number column.
|
||||
idx_succ = cluster_tree.bisect_left(candidate_cluster)
|
||||
successor_cluster: Optional[LineNumberCluster] = (
|
||||
cluster_tree[idx_succ] if idx_succ < len(cluster_tree) else None
|
||||
) # type: ignore[assignment]
|
||||
idx_pred = cluster_tree.bisect_right(candidate_cluster)
|
||||
neighbor: Optional[LineNumberCluster] = (
|
||||
cluster_tree[idx_pred - 1] if idx_pred > 0 else None
|
||||
) # type: ignore[assignment]
|
||||
match = nearest_cluster(candidate_cluster, neighbor, successor_cluster)
|
||||
if match is not None:
|
||||
if match.is_valid_sequence:
|
||||
match.is_valid_sequence = (candidate_cluster.primary_slot == match.primary_slot + 1)
|
||||
match.lines.append(line)
|
||||
match.primary_slot = candidate_cluster.primary_slot
|
||||
else:
|
||||
cluster_tree.add(candidate_cluster)
|
||||
|
||||
# Find largest valid (Ua) cluster
|
||||
best: Optional[LineNumberCluster] = None
|
||||
for cluster in cluster_tree:
|
||||
if cluster.is_valid_sequence and (best is None or len(cluster.lines) > len(best.lines)):
|
||||
best = cluster
|
||||
if best is None or not validate_line_number_cluster(rect, other_lines, best):
|
||||
return other_lines
|
||||
|
||||
# Build output: for each affected line, drop its first span
|
||||
affected = set(id(line) for line in best.lines)
|
||||
out: list[Line] = []
|
||||
for source_line in other_lines:
|
||||
if id(source_line) not in affected:
|
||||
out.append(source_line)
|
||||
continue
|
||||
new_line = Line()
|
||||
first_span = source_line.primary_slot[0]
|
||||
for span in source_line:
|
||||
if span is first_span:
|
||||
continue
|
||||
append_span(new_line, span)
|
||||
if new_line.char_count() <= 0:
|
||||
continue
|
||||
new_line.measure_slot = source_line.measure_slot
|
||||
out.append(new_line)
|
||||
return out
|
||||
@@ -0,0 +1,123 @@
|
||||
"""Per-page processing driver and reading-order assignment."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Optional
|
||||
|
||||
from ..clustering import LinesContainer, cluster_lines, build_initial_lines
|
||||
from ..columns import detect_columns, ColumnDetectionContext, columns_to_x_bounds
|
||||
from ..model import (
|
||||
Span,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
x_centers_close,
|
||||
to_number,
|
||||
Rect,
|
||||
append_span,
|
||||
avg_char_width,
|
||||
Line,
|
||||
info_weight,
|
||||
)
|
||||
from ..stats import column_index_of, PageStats, compute_page_stats
|
||||
|
||||
from .line_numbers import strip_line_numbers
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Per-page reading order and paragraph-break flagging #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class PageView:
|
||||
"""Per-page mutable state carried through layout classification."""
|
||||
|
||||
__slots__ = (
|
||||
"bounds", "output_slot", "secondary_slot", "measure_slot", "page_index", "primary_slot", "tertiary_slot", "lines", "blocks",
|
||||
"text", "previous_slot", "annotations",
|
||||
"auxiliary_slot", "state_slot", "style_slot", "option_slot", "viewport_box", "rot",
|
||||
)
|
||||
|
||||
def __init__(self, page_num: int, page_bbox: Rect):
|
||||
self.bounds: Rect = page_bbox
|
||||
self.output_slot: list = []
|
||||
self.secondary_slot: list = []
|
||||
self.measure_slot: bool = False # set when a labeled section appears
|
||||
self.page_index: int = page_num
|
||||
self.primary_slot: Optional[PageStats] = None
|
||||
self.tertiary_slot: list = [] # column rects
|
||||
self.lines: list = []
|
||||
self.blocks: list = []
|
||||
self.text: Optional[list] = None # raw text items reconstructed by parser
|
||||
self.previous_slot = 0.0
|
||||
self.annotations = []
|
||||
# per-page fields used by heading detection and outline assembly:
|
||||
self.auxiliary_slot: bool = False # marked as references page
|
||||
self.state_slot: bool = False # has substantive body
|
||||
self.style_slot: set = set() # set of body-style hashes (sh)
|
||||
self.option_slot = None # reserved, unused here
|
||||
# Page viewport for heading coordinates: unrotated view box + /Rotate.
|
||||
# None -> fallback to the origin-0 upright shortcut.
|
||||
self.viewport_box: Optional[tuple] = None
|
||||
self.rot: int = 0
|
||||
|
||||
|
||||
def assign_reading_order(primary_item: PageView, other_items: list) -> None:
|
||||
"""Assign reading order and paragraph-break flags for a page. The column-aware path expects blocks, not raw lines, because the sort key reads the first child line's column index. Passing raw lines would read a different flag from the first span."""
|
||||
primary_item.output_slot = other_items
|
||||
for candidate_item in range(len(other_items)):
|
||||
setattr(other_items[candidate_item], "orig_index", candidate_item)
|
||||
|
||||
primary_item.secondary_slot = list(other_items)
|
||||
primary_item.secondary_slot.sort(key=lambda sort_block: (column_index_of(sort_block), -sort_block.top_edge(), -sort_block.bottom_edge(), sort_block.left_edge(), sort_block.right_edge()))
|
||||
|
||||
# Assign sorted index and paragraph/end-isolated flags to each item.
|
||||
for idx in range(len(primary_item.secondary_slot)):
|
||||
candidate_item = primary_item.secondary_slot[idx]
|
||||
candidate_item.reading_order_index = idx
|
||||
reference_item = primary_item.secondary_slot[idx + 1] if idx + 1 < len(primary_item.secondary_slot) else None
|
||||
# Isolated-centered is true when the item is centered on the page and
|
||||
# either has no successor, is vertically separated from it, or is not
|
||||
# left/right aligned with it. Non-page-centered items can still be
|
||||
# isolated if they are centered relative to a page-centered successor.
|
||||
if candidate_item.alignment_slot and x_centers_close(primary_item.bounds, candidate_item):
|
||||
candidate_item.isolated_centered = (not reference_item) or (reference_item.top_edge() > candidate_item.bottom_edge()) or (not left_aligned(candidate_item, reference_item, 1) and not right_aligned(candidate_item, reference_item, 1))
|
||||
else:
|
||||
candidate_item.isolated_centered = bool(
|
||||
candidate_item.alignment_slot and reference_item
|
||||
and not left_aligned(candidate_item, reference_item, 1) and not right_aligned(candidate_item, reference_item, 1)
|
||||
and center_aligned(candidate_item, reference_item, candidate_item.bbox_width() / 10) and x_centers_close(primary_item.bounds, reference_item)
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Per-page orchestrator #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def process_page(spans: list[Span], page_num: int, page_bbox: Rect) -> PageView:
|
||||
"""Run the full per-page pipeline on flat span input."""
|
||||
page = PageView(page_num, page_bbox)
|
||||
# Raw parser items are kept before clustering
|
||||
# so document statistics can accumulate the script-family histogram over them (the lines
|
||||
# below are merged + line-number-stripped, a different character multiset).
|
||||
page.text = spans
|
||||
# 1) Build initial lines.
|
||||
container = LinesContainer()
|
||||
container.primary_slot = build_initial_lines(spans, page_bbox)
|
||||
# 2) First clustering pass: no column info yet.
|
||||
cluster_lines(container, 0.75, [])
|
||||
# 3) Compute first-pass per-page stats.
|
||||
page.primary_slot = compute_page_stats(page_bbox, container.primary_slot)
|
||||
# 4) Detect column rectangles and assign each line's column index.
|
||||
column_context = ColumnDetectionContext(page_bbox, page.primary_slot, container.primary_slot)
|
||||
page.tertiary_slot = detect_columns(column_context)
|
||||
# 5) Second clustering pass: tighter tolerance with column info.
|
||||
cols = columns_to_x_bounds(page.tertiary_slot)
|
||||
cluster_lines(container, 0.5, cols)
|
||||
# 6) Strip line-number column if present.
|
||||
container.primary_slot = strip_line_numbers(page_bbox, container.primary_slot)
|
||||
# 7) Recompute stats on cleaned lines.
|
||||
page.primary_slot = compute_page_stats(page_bbox, container.primary_slot)
|
||||
page.lines = container.primary_slot
|
||||
return page
|
||||
@@ -0,0 +1,48 @@
|
||||
"""Page-level and document-level layout statistics. The statistics layer computes weighted percentiles, dominant styles, script
|
||||
families, page spacing measures, and document-wide recurrence signals used by
|
||||
classification and outline assembly.
|
||||
"""
|
||||
|
||||
import functools
|
||||
import json
|
||||
import math
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from ..model import Span, _format_half_up_one_decimal, Line, info_weight, _max_nan_propagating
|
||||
|
||||
from .scripts import (
|
||||
_SCRIPT_BUCKET_TABLE_PATH,
|
||||
SCRIPT_BUCKET_TABLE,
|
||||
char_script_bucket,
|
||||
SCRIPT_FAMILY_WEIGHTS,
|
||||
ScriptHistogram,
|
||||
tally_scripts,
|
||||
dominant_script_family,
|
||||
)
|
||||
from .aggregates import (
|
||||
_percentile_sample_cmp,
|
||||
weighted_percentile,
|
||||
style_key,
|
||||
PageStats,
|
||||
compute_page_stats,
|
||||
DocStats,
|
||||
compute_doc_stats,
|
||||
column_index_of,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"weighted_percentile",
|
||||
"style_key",
|
||||
"PageStats",
|
||||
"compute_page_stats",
|
||||
"DocStats",
|
||||
"compute_doc_stats",
|
||||
"column_index_of",
|
||||
"char_script_bucket",
|
||||
"tally_scripts",
|
||||
"dominant_script_family",
|
||||
"ScriptHistogram",
|
||||
"SCRIPT_FAMILY_WEIGHTS",
|
||||
"SCRIPT_BUCKET_TABLE",
|
||||
]
|
||||
@@ -0,0 +1,313 @@
|
||||
"""Page-level and document-level statistics aggregation."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import functools
|
||||
import math
|
||||
from typing import Optional
|
||||
|
||||
from ..model import Span, _format_half_up_one_decimal, Line, info_weight, _max_nan_propagating
|
||||
|
||||
from .scripts import (
|
||||
ScriptHistogram,
|
||||
tally_scripts,
|
||||
dominant_script_family,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Weighted percentile.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _percentile_sample_cmp(values: tuple[float, float], other_values: tuple[float, float]) -> float:
|
||||
"""Comparator for weighted percentile samples. NaN comparison results are treated as equal so insertion order is preserved for NaN-valued samples."""
|
||||
if values[0] != other_values[0]:
|
||||
return values[0] - other_values[0]
|
||||
return values[1] - other_values[1]
|
||||
|
||||
|
||||
def weighted_percentile(values: list[tuple[float, float]], other_item: float) -> float:
|
||||
"""Weighted percentile over ``(value, weight)`` samples. Returns ``NaN`` for empty input or an out-of-range percentile. Ties at the target weight return the average of current and previous values; overshoots return the current value."""
|
||||
if len(values) <= 0 or other_item < 0 or other_item > 100:
|
||||
return float("nan")
|
||||
samples = sorted(values, key=functools.cmp_to_key(_percentile_sample_cmp)) # type: ignore[arg-type]
|
||||
total = sum(page_value[1] for page_value in samples)
|
||||
target = total * other_item / 100.0
|
||||
candidate_item = 0.0
|
||||
reference_item: Optional[float] = None
|
||||
for value, weight in samples:
|
||||
if candidate_item == target:
|
||||
return value if reference_item is None else (reference_item + value) / 2.0
|
||||
reference_item = value
|
||||
candidate_item += weight
|
||||
if candidate_item > target:
|
||||
return value
|
||||
return float("nan") if reference_item is None else reference_item
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Per-span style hash #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def style_key(span: Span) -> str:
|
||||
"""Return ``"<fontStyle> <size rounded to 0.1>"`` for same-style span histograms."""
|
||||
return f"{span.font_style()} {_format_half_up_one_decimal(span.font_size)}"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Per-page statistics #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class PageStats:
|
||||
"""Per-page layout statistics used by column detection and classification."""
|
||||
|
||||
__slots__ = ("line_count", "secondary_slot", "tertiary_slot", "previous_slot", "style_slot", "cache_slot", "option_slot", "primary_slot", "measure_slot", "state_slot", "auxiliary_slot")
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
valid_line_count: int,
|
||||
total_line_weight: float,
|
||||
median_overlap_gap: float,
|
||||
median_line_width: float,
|
||||
median_char_count: float,
|
||||
median_area_metric: float,
|
||||
median_center_y: float,
|
||||
median_font_size: float,
|
||||
average_char_width: float,
|
||||
dominant_font: str,
|
||||
dominant_style: str,
|
||||
):
|
||||
self.line_count = valid_line_count
|
||||
self.secondary_slot = total_line_weight
|
||||
self.tertiary_slot = median_overlap_gap
|
||||
self.previous_slot = median_line_width
|
||||
self.style_slot = median_char_count
|
||||
self.cache_slot = median_area_metric
|
||||
self.option_slot = median_center_y
|
||||
self.primary_slot = median_font_size
|
||||
self.measure_slot = average_char_width
|
||||
self.state_slot = dominant_font
|
||||
self.auxiliary_slot = dominant_style
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Per-page statistics.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def compute_page_stats(page, other_lines: list[Line]) -> PageStats:
|
||||
"""Compute weighted medians plus dominant font/style for one page."""
|
||||
overlap_gap_samples: list[tuple[float, float]] = [] # bucket-overlap samples
|
||||
line_width_samples: list[tuple[float, float]] = [] # line-width samples
|
||||
char_count_samples: list[tuple[float, float]] = [] # line-char-count samples
|
||||
area_metric_samples: list[tuple[float, float]] = [] # line.U samples
|
||||
array: list[tuple[float, float]] = [] # y-center samples
|
||||
font_size_samples: list[tuple[float, float]] = [] # font-size samples
|
||||
|
||||
font: dict[str, float] = {} # font-name histogram (weighted)
|
||||
style: dict[str, float] = {} # style-hash histogram (weighted)
|
||||
chars = 0 # total char count across spans
|
||||
width = 0.0 # total width across spans
|
||||
total = 0.0 # total line weight (sum of tf)
|
||||
valid = 0 # valid line count
|
||||
|
||||
bucket_size = page.bbox_width() / 20.0 # page width / 20 buckets
|
||||
buckets: list[Optional[Line]] = [None] * 21 # 20 buckets, +1 guard
|
||||
|
||||
for line in other_lines:
|
||||
if line.skew_frac() > 1: # rotated/skewed line: skip
|
||||
continue
|
||||
valid += 1
|
||||
for span in line: # for each span t in line u
|
||||
span_weight = info_weight(span.char_stats)
|
||||
span_weight = span_weight * span_weight * span.bbox_height() # weight = tf^2 * height
|
||||
font[span.font_name] = font.get(span.font_name, 0.0) + span_weight
|
||||
sty = style_key(span)
|
||||
style[sty] = style.get(sty, 0.0) + span_weight
|
||||
chars += span.char_count()
|
||||
width += span.bbox_width()
|
||||
line_weight = info_weight(line.char_stats)
|
||||
total += line_weight
|
||||
sample = line_weight * line.avg_font_size() # weight = line_weight * font_size
|
||||
font_size_samples.append((line.avg_font_size(), sample))
|
||||
line_width_samples.append((line.bbox_width(), sample))
|
||||
char_count_samples.append((line.char_count(), sample))
|
||||
area_metric_samples.append((line.cache_slot, line.area())) # NB: this one is weighted by area
|
||||
array.append((line.center_y(), sample))
|
||||
|
||||
# Vertical overlap with the most recent occupant of each horizontal
|
||||
# bucket. Infinite sentinel boxes skip overlap sampling.
|
||||
|
||||
left = line.left_edge()
|
||||
right = line.right_edge()
|
||||
if not (left < float("inf") and right > float("-inf")):
|
||||
continue
|
||||
if bucket_size <= 0:
|
||||
# Degenerate zero-width pages skip overlap sampling; downstream
|
||||
# statistics still include font and line-width samples.
|
||||
continue
|
||||
# Clamp infinite sentinels before converting bucket indexes to integers.
|
||||
bucket_left = 0 if left == float("-inf") else max(0, int(left / bucket_size))
|
||||
bucket_right = 20 if right == float("inf") else min(20, math.ceil(right / bucket_size))
|
||||
best_gap = float("inf")
|
||||
best_prev: Optional[Line] = None
|
||||
idx = bucket_left
|
||||
while idx < bucket_right:
|
||||
prev_in_bucket = buckets[idx]
|
||||
buckets[idx] = line
|
||||
idx += 1
|
||||
if prev_in_bucket is None:
|
||||
continue
|
||||
gap = max(prev_in_bucket.bottom_edge(), line.top_edge()) - line.bottom_edge()
|
||||
if gap < best_gap:
|
||||
best_gap = gap
|
||||
best_prev = prev_in_bucket
|
||||
if best_gap < float("inf") and best_prev is not None:
|
||||
overlap_gap_samples.append((best_gap, info_weight(best_prev.char_stats) * line_weight))
|
||||
|
||||
# Dominant values update only on strictly greater positive weight. This
|
||||
# keeps the empty value for all-zero pages and preserves first-seen ties.
|
||||
dominant_font = ""
|
||||
dominant_font_weight = 0.0
|
||||
for font_name, weight in font.items():
|
||||
if weight > dominant_font_weight:
|
||||
dominant_font_weight = weight
|
||||
dominant_font = font_name
|
||||
dominant_style = ""
|
||||
dominant_style_weight = 0.0
|
||||
for style_name, weight in style.items():
|
||||
if weight > dominant_style_weight:
|
||||
dominant_style_weight = weight
|
||||
dominant_style = style_name
|
||||
|
||||
return PageStats(
|
||||
valid_line_count=valid,
|
||||
total_line_weight=total,
|
||||
median_overlap_gap=weighted_percentile(overlap_gap_samples, 50),
|
||||
median_line_width=weighted_percentile(line_width_samples, 50),
|
||||
median_char_count=weighted_percentile(char_count_samples, 50),
|
||||
median_area_metric=weighted_percentile(area_metric_samples, 50),
|
||||
median_center_y=weighted_percentile(array, 50),
|
||||
median_font_size=weighted_percentile(font_size_samples, 50),
|
||||
# Average char width with IEEE edge cases: no characters with positive
|
||||
# width yields +inf, and no characters with no width yields NaN.
|
||||
average_char_width=(width / chars) if chars != 0
|
||||
else (float("inf") if width > 0 else float("nan")),
|
||||
dominant_font=dominant_font,
|
||||
dominant_style=dominant_style,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Document-level statistics #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class DocStats:
|
||||
"""Document-level layout statistics: dominant script family, landscape-page count, total valid lines, total line weight, max page line weight, median page total weight, width/height percentiles, center statistic, and median body font size."""
|
||||
|
||||
__slots__ = ("tertiary_slot", "style_slot", "cache_slot", "state_slot", "previous_slot", "secondary_slot", "option_slot", "auxiliary_slot", "measure_slot", "primary_slot")
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
dominant_script: int,
|
||||
landscape_pages: int,
|
||||
total_lines: int,
|
||||
total_weight: float,
|
||||
max_page_weight: float,
|
||||
median_page_weight: float,
|
||||
median_line_width: float,
|
||||
upper_width_percentile: float,
|
||||
median_center_y: float,
|
||||
median_body_font_size: float,
|
||||
):
|
||||
self.tertiary_slot = dominant_script
|
||||
self.style_slot = landscape_pages
|
||||
self.cache_slot = total_lines
|
||||
self.state_slot = total_weight
|
||||
self.previous_slot = max_page_weight
|
||||
self.secondary_slot = median_page_weight
|
||||
self.option_slot = median_line_width
|
||||
self.auxiliary_slot = upper_width_percentile
|
||||
self.measure_slot = median_center_y
|
||||
self.primary_slot = median_body_font_size
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Document-level statistics.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def compute_doc_stats(pages: list) -> DocStats:
|
||||
"""Compute document-wide recurrence and script statistics from page records."""
|
||||
script = ScriptHistogram()
|
||||
landscape = 0
|
||||
total_lines = 0
|
||||
total_weight = 0.0
|
||||
max_weight = 0.0
|
||||
|
||||
total_weights: list[tuple[float, float]] = []
|
||||
bucket_overlaps: list[tuple[float, float]] = []
|
||||
widths: list[tuple[float, float]] = []
|
||||
char_counts: list[tuple[float, float]] = []
|
||||
upper_samples: list[tuple[float, float]] = []
|
||||
centers: list[tuple[float, float]] = []
|
||||
font_sizes: list[tuple[float, float]] = []
|
||||
|
||||
for query_value in pages:
|
||||
# Accumulate the script-family histogram over raw parser text items,
|
||||
# capped at 100k chars. Merged line text can omit line-number spans.
|
||||
if script.secondary_slot < 100_000:
|
||||
for span in (query_value.text or []):
|
||||
if script.secondary_slot >= 100_000:
|
||||
break
|
||||
tally_scripts(script, span.text)
|
||||
if query_value.bounds.bbox_width() > query_value.bounds.bbox_height():
|
||||
landscape += 1
|
||||
stats: PageStats = query_value.primary_slot
|
||||
total_lines += stats.line_count
|
||||
weight = stats.secondary_slot
|
||||
total_weight += weight
|
||||
max_weight = _max_nan_propagating(max_weight, weight)
|
||||
if stats.line_count <= 0 or weight <= 0:
|
||||
continue
|
||||
per_page = min(100.0, weight / stats.line_count)
|
||||
total_weights.append((weight, per_page))
|
||||
bucket_overlaps.append((stats.tertiary_slot, per_page))
|
||||
widths.append((stats.previous_slot, per_page))
|
||||
char_counts.append((stats.style_slot, per_page))
|
||||
upper_samples.append((stats.cache_slot, per_page))
|
||||
centers.append((stats.option_slot, per_page))
|
||||
font_sizes.append((stats.primary_slot, per_page))
|
||||
|
||||
return DocStats(
|
||||
dominant_script=dominant_script_family(script), # dominant script family
|
||||
landscape_pages=landscape,
|
||||
total_lines=total_lines,
|
||||
total_weight=total_weight,
|
||||
max_page_weight=max_weight,
|
||||
median_page_weight=weighted_percentile(total_weights, 50),
|
||||
# Width and uppercase samples are the document-wide outputs used later.
|
||||
median_line_width=weighted_percentile(widths, 50),
|
||||
upper_width_percentile=weighted_percentile(upper_samples, 80), # NB: 80th percentile, not 50
|
||||
median_center_y=weighted_percentile(centers, 50),
|
||||
median_body_font_size=weighted_percentile(font_sizes, 50),
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Column-index accessor #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def column_index_of(line: Line) -> int:
|
||||
"""Return the column index stored on the first child line/span. In normal use this receives a block, so the first child is a line and its stored column index is returned. If a raw line is passed, the same field access still succeeds but reads a different flag; the block-clustering pipeline avoids that path for column-aware ordering."""
|
||||
if not line.primary_slot:
|
||||
return -1
|
||||
first = line.primary_slot[0]
|
||||
# If ``first`` is another container (block.g[0] is a line) consult its H.
|
||||
value = getattr(first, "measure_slot", None)
|
||||
return -1 if value is None else value
|
||||
@@ -0,0 +1,85 @@
|
||||
"""Script bucket tables and script histogram helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Script-family detector and bucket table.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
_SCRIPT_BUCKET_TABLE_PATH = Path(__file__).parent.parent / "data" / "script_bucket_table.json"
|
||||
SCRIPT_BUCKET_TABLE: list[int] = json.loads(_SCRIPT_BUCKET_TABLE_PATH.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def char_script_bucket(text: str) -> int:
|
||||
"""Return the script-bucket id for a single character: empty/multi-character, ASCII punctuation/digit, control, ASCII letter, or a table-driven non-Latin script bucket."""
|
||||
if not text:
|
||||
return 0
|
||||
if len(text) != 1:
|
||||
return 10
|
||||
# Astral code points are classified as the multi-unit script bucket.
|
||||
if ord(text) > 0xFFFF:
|
||||
return 10
|
||||
if ("a" <= text <= "z") or ("A" <= text <= "Z"):
|
||||
return 3
|
||||
if "\x00" < text < " ":
|
||||
return 2
|
||||
if text < "":
|
||||
return 1
|
||||
idx = ord(text) >> 4
|
||||
if 0 <= idx < len(SCRIPT_BUCKET_TABLE):
|
||||
return SCRIPT_BUCKET_TABLE[idx]
|
||||
return 0
|
||||
|
||||
|
||||
# map: each output category -> contributing bucket weights.
|
||||
# Fixed weights for collapsing script buckets into document script families.
|
||||
SCRIPT_FAMILY_WEIGHTS: dict[int, list[tuple[int, int]]] = {
|
||||
2: [(2, 10)],
|
||||
0: [(0, 1), (2, 1)],
|
||||
3: [(3, 1), (4, -3), (5, -3), (6, -3), (7, -3), (8, -3), (9, -10)],
|
||||
4: [(4, 1)],
|
||||
5: [(5, 1), (6, -10), (7, -10)],
|
||||
6: [(6, 1)],
|
||||
7: [(7, 1)],
|
||||
8: [(8, 1)],
|
||||
9: [(9, 1)],
|
||||
10: [(10, 1)],
|
||||
}
|
||||
|
||||
|
||||
class ScriptHistogram:
|
||||
"""Script-bucket accumulator with total character count and per-bucket histogram."""
|
||||
|
||||
__slots__ = ("secondary_slot", "primary_slot")
|
||||
|
||||
def __init__(self):
|
||||
self.secondary_slot: int = 0
|
||||
self.primary_slot: list[int] = [0] * 11
|
||||
|
||||
|
||||
def tally_scripts(primary_item: ScriptHistogram, other_text: str) -> None:
|
||||
"""feed a string into the bucket accumulator."""
|
||||
for candidate_item in other_text:
|
||||
primary_item.primary_slot[char_script_bucket(candidate_item)] += 1
|
||||
primary_item.secondary_slot += 1
|
||||
|
||||
|
||||
def dominant_script_family(primary_item: ScriptHistogram) -> int:
|
||||
"""best-scoring script-family for the accumulator. Returns the output category 0..10 with the highest weighted score. """
|
||||
secondary_item = 0
|
||||
candidate_item = 0
|
||||
for reference_item in range(11):
|
||||
entry_item = SCRIPT_FAMILY_WEIGHTS.get(reference_item)
|
||||
if not entry_item:
|
||||
continue
|
||||
score_value = 0
|
||||
for (script_index, weight) in entry_item:
|
||||
score_value += weight * primary_item.primary_slot[script_index]
|
||||
if score_value > candidate_item:
|
||||
secondary_item = reference_item
|
||||
candidate_item = score_value
|
||||
return secondary_item
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Document-title detection. The scoring formula is the heart of title detection: score is a product of layout, recurrence, label, script, width, numbering, punctuation, alignment, and page-position factors. Each factor is in roughly ``[0.1, 3.0]-- the product can grow to a few
|
||||
thousand for a strong title candidate. The factors are documented in the
|
||||
scoring body. The multilingual title-keyword and institution-word sets are stored in
|
||||
``data/dictionaries.json`` as ``title`` and ``institution_words``.
|
||||
"""
|
||||
|
||||
import json
|
||||
import math
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
_trim_unicode_ws,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
Rect,
|
||||
last_span,
|
||||
heading_score,
|
||||
Line,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
block_text,
|
||||
deaccented_text,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
info_weight,
|
||||
is_upper_dominant,
|
||||
alignment_code,
|
||||
Block,
|
||||
)
|
||||
from ..stats import DocStats, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
|
||||
from ..tokens import is_superscript_adjacent, clamp_value, enumerate_tokens, jenkins_hash, trie_prefix_match, set_case_fold, TrieConfig, build_trie, tokenize_block, _de_norm, BuiltTrie, is_word_token
|
||||
|
||||
from .dicts import (
|
||||
_DICT_PATH,
|
||||
_normalize_text_key,
|
||||
_load_dicts,
|
||||
INSTITUTION_WORDS,
|
||||
TITLE_LABEL_TRIE,
|
||||
)
|
||||
from .scoring import (
|
||||
TitleCandidate,
|
||||
is_cover_like_page,
|
||||
is_title_candidate_block,
|
||||
score_title_candidate,
|
||||
)
|
||||
from .detect import (
|
||||
TitleSearchState,
|
||||
detect_title,
|
||||
)
|
||||
|
||||
__all__ = ["is_title_candidate_block", "score_title_candidate", "TitleSearchState", "TitleCandidate", "detect_title", "is_cover_like_page", "TITLE_LABEL_TRIE", "INSTITUTION_WORDS"]
|
||||
@@ -0,0 +1,140 @@
|
||||
"""Document title search over early pages."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Optional
|
||||
|
||||
from ..model import (
|
||||
_trim_unicode_ws,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
Rect,
|
||||
last_span,
|
||||
heading_score,
|
||||
Line,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
block_text,
|
||||
deaccented_text,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
info_weight,
|
||||
is_upper_dominant,
|
||||
alignment_code,
|
||||
Block,
|
||||
)
|
||||
from ..tokens import is_superscript_adjacent, clamp_value, enumerate_tokens, jenkins_hash, trie_prefix_match, set_case_fold, TrieConfig, build_trie, tokenize_block, _de_norm, BuiltTrie, is_word_token
|
||||
|
||||
from .scoring import (
|
||||
TitleCandidate,
|
||||
is_cover_like_page,
|
||||
is_title_candidate_block,
|
||||
score_title_candidate,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Title detection state.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class TitleSearchState:
|
||||
"""Title-detection state: document, visited blocks, and current best candidate."""
|
||||
|
||||
__slots__ = ("tertiary_slot", "primary_slot", "secondary_slot")
|
||||
|
||||
def __init__(self, doc):
|
||||
self.tertiary_slot = doc
|
||||
self.primary_slot: set = set()
|
||||
self.secondary_slot: Optional[TitleCandidate] = None
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Title-detection driver.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def detect_title(doc) -> Optional[TitleCandidate]:
|
||||
"""Iterate early pages, score title-like block groups, and return the best candidate."""
|
||||
state = TitleSearchState(doc)
|
||||
has_seen_da = False # "broke into body" flag
|
||||
|
||||
for page in doc.primary_slot:
|
||||
# Special branch: landscape cover document
|
||||
if (
|
||||
doc.secondary_slot.style_slot > len(doc.primary_slot) / 2
|
||||
and page.page_index <= 1
|
||||
and page.bounds.bbox_width() > page.bounds.bbox_height()
|
||||
and page.primary_slot.secondary_slot < 500
|
||||
):
|
||||
for idx, block in enumerate(page.secondary_slot):
|
||||
if (
|
||||
is_title_candidate_block(block) and id(block) not in state.primary_slot
|
||||
and heading_score(block) > page.primary_slot.primary_slot - 0.1
|
||||
):
|
||||
score_title_candidate(state, page, idx)
|
||||
break
|
||||
|
||||
if (
|
||||
is_cover_like_page(doc, page)
|
||||
or (page.page_index <= 1 and len(doc.primary_slot) >= 10 and page.primary_slot.secondary_slot < 0.8 * doc.secondary_slot.secondary_slot)
|
||||
):
|
||||
# Cover / front-matter page
|
||||
for idx, block in enumerate(page.secondary_slot):
|
||||
if not is_title_candidate_block(block) or id(block) in state.primary_slot:
|
||||
continue
|
||||
score = heading_score(block)
|
||||
if (
|
||||
(score > doc.secondary_slot.primary_slot + 0.1 and score > page.primary_slot.primary_slot + 0.1)
|
||||
or (score > doc.secondary_slot.primary_slot + 2 and score > page.primary_slot.primary_slot - 0.1)
|
||||
or (score > doc.secondary_slot.primary_slot - 0.1 and score > page.primary_slot.primary_slot - 0.1 and block.isolated_centered)
|
||||
or (score > doc.secondary_slot.primary_slot - 0.1 and score > page.primary_slot.primary_slot - 0.1
|
||||
and page.page_index <= 1 and page.primary_slot.secondary_slot < 500)
|
||||
):
|
||||
score_title_candidate(state, page, idx)
|
||||
else:
|
||||
# Body page: only consider initial blocks until we hit body text
|
||||
local_done = False
|
||||
for idx, block in enumerate(page.secondary_slot):
|
||||
if id(block) in state.primary_slot:
|
||||
continue
|
||||
score = heading_score(block)
|
||||
# Block clearly larger than body
|
||||
size_trigger = (
|
||||
is_title_candidate_block(block) and (
|
||||
(score > doc.secondary_slot.primary_slot + 0.1 and score > page.primary_slot.primary_slot + 0.1)
|
||||
or (score > doc.secondary_slot.primary_slot + 2 and score > page.primary_slot.primary_slot - 0.1)
|
||||
or (block.isolated_centered and score > page.primary_slot.primary_slot - 0.1)
|
||||
or (page.page_index == 1 and score > page.primary_slot.primary_slot + 2)
|
||||
)
|
||||
)
|
||||
if size_trigger:
|
||||
score_title_candidate(state, page, idx)
|
||||
elif block.is_body_paragraph and not block.isolated_centered:
|
||||
# Body-break flag: stop scanning once body text is reached.
|
||||
if not has_seen_da:
|
||||
if (block.bottom_edge() - page.bounds.bottom_edge() < 2 * page.bounds.bbox_height() / 3):
|
||||
has_seen_da = False
|
||||
elif block.line_count() >= 3 and alignment_code(block) == 4:
|
||||
has_seen_da = True
|
||||
else:
|
||||
digit_or_period = 0
|
||||
tokens = tokenize_block(block)
|
||||
for token in tokens:
|
||||
if is_word_token(token) or token.type == 1:
|
||||
digit_or_period += 1
|
||||
has_seen_da = digit_or_period >= len(tokens) / 3
|
||||
has_seen_da = not has_seen_da
|
||||
if has_seen_da:
|
||||
local_done = True
|
||||
break
|
||||
local_done = True
|
||||
# Once body text is seen, the flag stays sticky so a later
|
||||
# body block on this page breaks immediately.
|
||||
has_seen_da = True
|
||||
if local_done:
|
||||
break
|
||||
if doc.secondary_slot.secondary_slot < 400:
|
||||
break
|
||||
return state.secondary_slot
|
||||
@@ -0,0 +1,35 @@
|
||||
"""Dictionary tables for title detection."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from ..tokens import is_superscript_adjacent, clamp_value, enumerate_tokens, jenkins_hash, trie_prefix_match, set_case_fold, TrieConfig, build_trie, tokenize_block, _de_norm, BuiltTrie, is_word_token
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Load title-label and institution dictionaries #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
_DICT_PATH = Path(__file__).parent.parent / "data" / "dictionaries.json"
|
||||
|
||||
|
||||
def _normalize_text_key(text: str) -> str:
|
||||
"""NFKC + strip + collapse-whitespace + lowercase."""
|
||||
return " ".join(unicodedata.normalize("NFKC", text).strip().split()).lower()
|
||||
|
||||
|
||||
def _load_dicts() -> tuple[BuiltTrie, set[str]]:
|
||||
raw = json.loads(_DICT_PATH.read_text(encoding="utf-8"))
|
||||
title_label_trie = build_trie(raw.get("title", []), set_case_fold(TrieConfig(), True))
|
||||
# institution words use normalized single-token set membership.
|
||||
# The title-label dictionary stays a trie because it handles the
|
||||
|
||||
# multi-token "Title:" match; institution words are single-token only.)
|
||||
institution_words = set(raw.get("institution_words", []))
|
||||
return title_label_trie, institution_words
|
||||
|
||||
|
||||
TITLE_LABEL_TRIE, INSTITUTION_WORDS = _load_dicts()
|
||||
@@ -0,0 +1,264 @@
|
||||
"""Title-candidate scoring."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
|
||||
from ..model import (
|
||||
_trim_unicode_ws,
|
||||
left_aligned,
|
||||
right_aligned,
|
||||
center_aligned,
|
||||
Rect,
|
||||
last_span,
|
||||
heading_score,
|
||||
Line,
|
||||
last_line_of,
|
||||
first_span_of,
|
||||
block_text,
|
||||
deaccented_text,
|
||||
letter_count,
|
||||
dominant_style_of,
|
||||
info_weight,
|
||||
is_upper_dominant,
|
||||
alignment_code,
|
||||
Block,
|
||||
)
|
||||
from ..stats import DocStats, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
|
||||
from ..tokens import is_superscript_adjacent, clamp_value, enumerate_tokens, jenkins_hash, trie_prefix_match, set_case_fold, TrieConfig, build_trie, tokenize_block, _de_norm, BuiltTrie, is_word_token
|
||||
|
||||
from .dicts import (
|
||||
INSTITUTION_WORDS,
|
||||
TITLE_LABEL_TRIE,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Title candidate state container #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class TitleCandidate:
|
||||
"""Best title candidate so far: page, contributing blocks, and score."""
|
||||
|
||||
__slots__ = ("page", "output_slot", "score")
|
||||
|
||||
def __init__(self, page, blocks: list[Block], score_value: float):
|
||||
self.page = page
|
||||
self.output_slot = blocks
|
||||
self.score = score_value
|
||||
|
||||
def to_string(self) -> str:
|
||||
"""Join contributing blocks into the displayed title string, inserting one inter-block space only after the accumulator is non-empty."""
|
||||
primary_item = ""
|
||||
for block in self.output_slot:
|
||||
if primary_item:
|
||||
primary_item += " "
|
||||
primary_item += _trim_unicode_ws(tokenize_block(block).to_string())
|
||||
|
||||
return primary_item
|
||||
|
||||
def __str__(self) -> str:
|
||||
return self.to_string()
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Cover-like page predicate #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def is_cover_like_page(doc, page) -> bool:
|
||||
"""Return whether a page is sparse enough to behave like a cover page."""
|
||||
if getattr(page, "measure_slot", False):
|
||||
return False
|
||||
threshold = 0.5 * min(doc.secondary_slot.secondary_slot, 5e3)
|
||||
if page.page_index <= 1 and page.primary_slot.secondary_slot < threshold:
|
||||
return True
|
||||
early_limit = 1 + min(15, len(doc.primary_slot) / 5)
|
||||
return page.page_index < early_limit and page.primary_slot.secondary_slot < 0.8 * threshold
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# xp: candidate-block filter #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def is_title_candidate_block(block: Block) -> bool:
|
||||
"""Return whether ``block`` can be considered as a document-title candidate."""
|
||||
return (
|
||||
letter_count(block.char_stats) > 0
|
||||
and block.skew_frac() < 1
|
||||
and block.type == 0
|
||||
and block.char_count() < 400
|
||||
and block.bbox_height() < 2 * block.bbox_width()
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# yp: multiplicative scoring for a candidate group #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def score_title_candidate(zp_state, page, index: int) -> None:
|
||||
"""Score a candidate block group and update the title-search state."""
|
||||
doc = zp_state.tertiary_slot
|
||||
blocks = page.secondary_slot # sorted blocks
|
||||
title_block = blocks[index]
|
||||
title_group: list[Block] = [title_block]
|
||||
|
||||
# Try to extend with next block if alignment / style / vertical proximity match
|
||||
if index + 1 < len(blocks):
|
||||
next_item = blocks[index + 1]
|
||||
title_score = heading_score(title_block)
|
||||
height = title_block.avg_font_size()
|
||||
# Two acceptance conditions:
|
||||
if (
|
||||
(abs(title_score - heading_score(next_item)) < 0.1
|
||||
and dominant_style_of(title_block) == dominant_style_of(next_item)
|
||||
and title_block.bottom_edge() - next_item.top_edge() < height)
|
||||
or (
|
||||
title_score > doc.secondary_slot.primary_slot + 5
|
||||
and title_score > page.primary_slot.primary_slot + 1
|
||||
and abs(height - next_item.avg_font_size()) < 0.1
|
||||
and title_block.bottom_edge() - next_item.top_edge() < 0.5 * height
|
||||
)
|
||||
):
|
||||
tolerance = 0.1 * height
|
||||
align_value = alignment_code(title_block)
|
||||
next_alignment = alignment_code(next_item)
|
||||
if (
|
||||
(left_aligned(title_block, next_item, tolerance)
|
||||
and align_value in (1, 2) and next_alignment in (1, 2))
|
||||
or (right_aligned(title_block, next_item, tolerance)
|
||||
and align_value in (1, 4) and next_alignment in (1, 4))
|
||||
or (center_aligned(title_block, next_item, tolerance)
|
||||
and title_block.alignment_slot and next_item.alignment_slot)
|
||||
):
|
||||
title_group.append(next_item)
|
||||
|
||||
group = title_group
|
||||
for measure_item in group:
|
||||
zp_state.primary_slot.add(id(measure_item))
|
||||
|
||||
doc_state = zp_state.tertiary_slot
|
||||
previous_block = blocks[index - 1] if index - 1 >= 0 else None
|
||||
|
||||
# Accumulate statistics over the title group
|
||||
max_heading_score = 0
|
||||
max_width = 0.0
|
||||
consecutive = 0
|
||||
max_consecutive = 0
|
||||
bracket_count = 0
|
||||
total_tokens = 0
|
||||
right_pen = 1.0
|
||||
email_count = 0
|
||||
|
||||
for result_value in group:
|
||||
max_heading_score = max(max_heading_score, heading_score(result_value))
|
||||
max_width = max(max_width, result_value.bbox_width())
|
||||
title_tokens_view = tokenize_block(result_value)
|
||||
for entry in enumerate_tokens(title_tokens_view):
|
||||
sample_item = entry["token"]
|
||||
total_tokens += 1
|
||||
if is_word_token(sample_item):
|
||||
consecutive += 1
|
||||
max_consecutive = max(max_consecutive, consecutive)
|
||||
if sample_item.boundary_slot:
|
||||
bracket_count += 1
|
||||
# email detection: "@" followed by word "." word (4 tokens)
|
||||
if sample_item.str == "@" and entry["index"] + 3 < title_tokens_view.length:
|
||||
next_token = title_tokens_view.token_at(entry["index"] + 1)
|
||||
dot = title_tokens_view.token_at(entry["index"] + 2)
|
||||
after = title_tokens_view.token_at(entry["index"] + 3)
|
||||
if (
|
||||
next_token is not None and dot is not None and after is not None
|
||||
and next_token.type == 2 and dot.str == "." and after.type == 2
|
||||
):
|
||||
email_count += 1
|
||||
else:
|
||||
consecutive = 0
|
||||
if alignment_code(result_value) == 4:
|
||||
# The line count is structurally positive here. Keep the fallback so
|
||||
# a degenerate line cannot raise during title scoring.
|
||||
|
||||
right_pen /= result_value.line_count() or 1
|
||||
|
||||
if total_tokens <= 0:
|
||||
return
|
||||
|
||||
# Multiplicative factors
|
||||
len_value = clamp_value(total_tokens * total_tokens / 16.0, 0.5, 1.0)
|
||||
# Page width should be positive. Keep IEEE-style Infinity/NaN behavior for
|
||||
# degenerate pages instead of raising during scoring.
|
||||
width_ratio_sq = (max_width / page.bounds.bbox_width()) if page.bounds.bbox_width() else (math.inf if max_width > 0 else math.nan)
|
||||
width_ratio_sq *= width_ratio_sq
|
||||
bracket = bracket_count / total_tokens
|
||||
bracket_factor = max(0.1, 1 - 9 * bracket * bracket) / max(1, max_consecutive - 2)
|
||||
page_pos = max(0.1, 1 - 2 * (page.page_index - 1) / max(1, len(doc_state.primary_slot)))
|
||||
# Doc-wide height is positive in normal inputs. The epsilon prevents a
|
||||
# degenerate input from raising and still yields the minimum density factor.
|
||||
page_density_ratio = page.primary_slot.secondary_slot / max(1e-6, doc_state.secondary_slot.secondary_slot)
|
||||
density_factor = max(0.5, 1 - page_density_ratio * page_density_ratio) * (1 + clamp_value((0.25 - page_density_ratio) / 0.15, 0, 1))
|
||||
# Page top/height is positive in normal inputs. Degenerate pages take the
|
||||
# minimum top-position factor instead of raising.
|
||||
|
||||
top = max(0.1, group[0].top_edge() / page.bounds.top_edge()) if page.bounds.top_edge() else 0.1
|
||||
|
||||
# Abbreviation penalty: count adjacent single-char + delimiter pairs
|
||||
abbrev = 0
|
||||
for block in group:
|
||||
tokens = tokenize_block(block)
|
||||
previous = None
|
||||
for token in tokens:
|
||||
if previous is not None and len(token.str) <= 1 and is_superscript_adjacent(previous, token):
|
||||
abbrev += 1
|
||||
previous = token
|
||||
factor = clamp_value(1.0 / max(1, abbrev), 0.3, 1.0)
|
||||
|
||||
# Recurrence penalty: first block's normalized text appears how often?
|
||||
# The histogram uses the same normalized text hash as the document-wide
|
||||
# ghost-text map.
|
||||
norm_text = jenkins_hash(deaccented_text(group[0]))
|
||||
recurrence_count = doc_state.tertiary_slot.get(norm_text, 0) if hasattr(doc_state, "tertiary_slot") and isinstance(doc_state.tertiary_slot, dict) else 0
|
||||
ratio = recurrence_count / max(1, len(doc_state.primary_slot))
|
||||
adj = total_tokens - 3
|
||||
recurrence_factor = 1 - 0.5 * clamp_value(ratio / 0.3, 0, 1) * (1 / max(1, adj * adj))
|
||||
|
||||
# Institution-word penalty (non-first-page)
|
||||
institution = 1.0
|
||||
if is_cover_like_page(doc_state, page):
|
||||
inst_hits = 0
|
||||
for block in group:
|
||||
for token in tokenize_block(block):
|
||||
# single-token
|
||||
# Match using the same lowercase + diacritic-stripped form as
|
||||
# the institution-word set.
|
||||
if _de_norm(token.str, True) in INSTITUTION_WORDS:
|
||||
inst_hits += 1
|
||||
institution = 1.0 / (1 + inst_hits)
|
||||
|
||||
# "Title:" label bonus from previous block
|
||||
label = 1.0
|
||||
if previous_block is not None:
|
||||
prev_tokens = tokenize_block(previous_block)
|
||||
if prev_tokens.length <= 3 and trie_prefix_match(TITLE_LABEL_TRIE, prev_tokens) is not None:
|
||||
label = 3.0
|
||||
|
||||
# Email penalty
|
||||
email = 1.0 / ((1 + email_count) ** 2)
|
||||
|
||||
# Script-family match: build a script histogram over the candidate group's text and
|
||||
# compare the candidate script family against the document script family.
|
||||
script_acc = ScriptHistogram()
|
||||
for result_value in group:
|
||||
tally_scripts(script_acc, block_text(result_value))
|
||||
script = 1.0 if dominant_script_family(script_acc) == doc_state.secondary_slot.tertiary_slot else 0.5
|
||||
|
||||
score = (
|
||||
max_heading_score * len_value * width_ratio_sq * right_pen * bracket_factor * page_pos
|
||||
* density_factor * top * factor * recurrence_factor * institution * label
|
||||
* email * script
|
||||
)
|
||||
|
||||
if zp_state.secondary_slot is None or score > zp_state.secondary_slot.score:
|
||||
zp_state.secondary_slot = TitleCandidate(page, group, score)
|
||||
@@ -0,0 +1,105 @@
|
||||
"""Tokenizer subsystem. The tokenizer is character-driven: it walks each character of each line,
|
||||
uses category-transition tolerances to decide when the current token can
|
||||
extend, and closes tokens when script, punctuation, or spacing transitions
|
||||
require a boundary. Tokens keep back-references to the contributing line and span offsets so the
|
||||
visible text can be reconstructed and cross-line tokens, such as a word broken
|
||||
by a hyphen across two lines, can be stitched.
|
||||
"""
|
||||
|
||||
import unicodedata
|
||||
from typing import Any, Iterable, Iterator, Optional
|
||||
|
||||
from ..model import (
|
||||
_strip_diacritics,
|
||||
avg_char_width2,
|
||||
intervals_overlap,
|
||||
to_number,
|
||||
rect_union,
|
||||
EMPTY_RECT,
|
||||
avg_char_width,
|
||||
Line,
|
||||
char_category,
|
||||
is_word_category,
|
||||
is_punct_category,
|
||||
letter_count,
|
||||
punct_count,
|
||||
info_weight,
|
||||
Block,
|
||||
)
|
||||
|
||||
from .token_types import (
|
||||
SCRIPT_FAMILY_MAP,
|
||||
_build_gap_tolerance_grid,
|
||||
GAP_TOLERANCE_GRID,
|
||||
can_extend_token,
|
||||
TokenAnchor,
|
||||
last_token_anchor,
|
||||
first_anchor_span,
|
||||
is_char_token,
|
||||
is_word_token,
|
||||
is_trimmable_token,
|
||||
token_numeric_value,
|
||||
Token,
|
||||
TokenView,
|
||||
wrap_tokens,
|
||||
enumerate_tokens,
|
||||
first_token,
|
||||
last_token,
|
||||
)
|
||||
from .tokenizer import (
|
||||
LineTokenizer,
|
||||
tokenize_block,
|
||||
clamp_value,
|
||||
is_superscript_adjacent,
|
||||
)
|
||||
from .tries import (
|
||||
_de_norm,
|
||||
TrieConfig,
|
||||
BuiltTrie,
|
||||
set_reverse,
|
||||
set_case_fold,
|
||||
TrieNode,
|
||||
trie_insert_step,
|
||||
trie_walk_step,
|
||||
aho_corasick_match,
|
||||
aho_corasick_tokens,
|
||||
TrieBuilder,
|
||||
_trie_insert_entry,
|
||||
trie_bulk_insert,
|
||||
_trie_finalize,
|
||||
build_trie,
|
||||
trie_prefix_match,
|
||||
_trie_full_match,
|
||||
trie_full_match,
|
||||
strip_trie_match,
|
||||
strip_leading_if_in,
|
||||
COMMA_CHARS,
|
||||
strip_trailing_comma,
|
||||
is_comma_token,
|
||||
trim_trailing_punct,
|
||||
)
|
||||
from .hashing import (
|
||||
_FH_MASK,
|
||||
_to_uint32,
|
||||
_to_int32,
|
||||
_int32_xor,
|
||||
_int32_left_shift,
|
||||
_uint32_right_shift,
|
||||
_little_endian_signed_word,
|
||||
_utf8_bytes_from_utf16_units,
|
||||
_jenkins_mix,
|
||||
jenkins_hash,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
# state machine
|
||||
"GAP_TOLERANCE_GRID", "can_extend_token", "TokenAnchor", "last_token_anchor", "first_anchor_span",
|
||||
"is_char_token", "is_word_token", "is_trimmable_token", "token_numeric_value", "Token",
|
||||
"TokenView", "wrap_tokens", "enumerate_tokens", "first_token", "last_token",
|
||||
"LineTokenizer", "tokenize_block",
|
||||
"clamp_value", "is_superscript_adjacent", "jenkins_hash",
|
||||
# trie
|
||||
"TrieConfig", "BuiltTrie", "set_reverse", "set_case_fold", "TrieNode", "trie_insert_step", "trie_walk_step", "build_trie", "trie_prefix_match", "trie_full_match",
|
||||
"strip_trie_match", "strip_leading_if_in", "strip_trailing_comma", "trim_trailing_punct", "COMMA_CHARS",
|
||||
"SCRIPT_FAMILY_MAP", "GAP_TOLERANCE_GRID",
|
||||
]
|
||||
@@ -0,0 +1,139 @@
|
||||
"""32-bit integer helpers and the Jenkins string hash."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Jenkins lookup2 string hash (UTF-8 bytes -> signed 32-bit int).
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
_FH_MASK = 0xFFFFFFFF
|
||||
|
||||
|
||||
def _to_uint32(number: int) -> int:
|
||||
return number & _FH_MASK
|
||||
|
||||
|
||||
def _to_int32(number: int) -> int:
|
||||
number &= _FH_MASK
|
||||
return number - 0x100000000 if number >= 0x80000000 else number
|
||||
|
||||
|
||||
def _int32_xor(number: int, other_number: int) -> int:
|
||||
"""ToInt32 of the 32-bit xor of the operands' uint32 forms."""
|
||||
return _to_int32(_to_uint32(number) ^ _to_uint32(other_number))
|
||||
|
||||
|
||||
def _int32_left_shift(number: int, other_number: int) -> int:
|
||||
"""(signed 32-bit result)."""
|
||||
return _to_int32((_to_uint32(number) << (other_number & 31)) & _FH_MASK)
|
||||
|
||||
|
||||
def _uint32_right_shift(number: int, other_number: int) -> int:
|
||||
"""(unsigned right shift)."""
|
||||
return _to_uint32(number) >> (other_number & 31)
|
||||
|
||||
|
||||
def _little_endian_signed_word(byte_values: list, off: int) -> int:
|
||||
"""Return a little-endian four-byte word with each byte sign-extended."""
|
||||
def _sign_extend_byte(number: int) -> int:
|
||||
return number - 256 if number > 127 else number
|
||||
return (_sign_extend_byte(byte_values[off]) + (_sign_extend_byte(byte_values[off + 1]) << 8)
|
||||
+ (_sign_extend_byte(byte_values[off + 2]) << 16) + (_sign_extend_byte(byte_values[off + 3]) << 24))
|
||||
|
||||
|
||||
def _utf8_bytes_from_utf16_units(text: str) -> list[int]:
|
||||
"""Encode by walking UTF-16 code units, preserving lone surrogates."""
|
||||
raw = text.encode("utf-16-le", "surrogatepass")
|
||||
units = [raw[index] | (raw[index + 1] << 8) for index in range(0, len(raw), 2)]
|
||||
output_bytes: list[int] = []
|
||||
index = 0
|
||||
while index < len(units):
|
||||
value = units[index]
|
||||
if value < 128:
|
||||
output_bytes.append(value)
|
||||
elif value < 2048:
|
||||
output_bytes.append((value >> 6) | 192)
|
||||
output_bytes.append((value & 63) | 128)
|
||||
else:
|
||||
if (
|
||||
(value & 0xFC00) == 0xD800
|
||||
and index + 1 < len(units)
|
||||
and (units[index + 1] & 0xFC00) == 0xDC00
|
||||
):
|
||||
index += 1
|
||||
value = 0x10000 + ((value & 1023) << 10) + (units[index] & 1023)
|
||||
output_bytes.append((value >> 18) | 240)
|
||||
output_bytes.append(((value >> 12) & 63) | 128)
|
||||
else:
|
||||
output_bytes.append((value >> 12) | 224)
|
||||
output_bytes.append(((value >> 6) & 63) | 128)
|
||||
output_bytes.append((value & 63) | 128)
|
||||
index += 1
|
||||
return output_bytes
|
||||
|
||||
|
||||
def _jenkins_mix(mix_state: list) -> int:
|
||||
"""Jenkins lookup2 mix over the 3-word state ``[a, b, c]``."""
|
||||
secondary_item, candidate_item, reference_item = mix_state
|
||||
secondary_item = _int32_xor(secondary_item - candidate_item - reference_item, _uint32_right_shift(reference_item, 13))
|
||||
candidate_item = _int32_xor(candidate_item - reference_item - secondary_item, _int32_left_shift(secondary_item, 8))
|
||||
reference_item = reference_item - secondary_item
|
||||
reference_item = _int32_xor(reference_item - candidate_item, _uint32_right_shift(candidate_item, 13))
|
||||
secondary_item = secondary_item - candidate_item
|
||||
secondary_item = secondary_item - reference_item
|
||||
secondary_item = _int32_xor(secondary_item, _uint32_right_shift(reference_item, 12))
|
||||
candidate_item = _int32_xor(candidate_item - reference_item - secondary_item, _int32_left_shift(secondary_item, 16))
|
||||
reference_item = reference_item - secondary_item
|
||||
reference_item = _int32_xor(reference_item - candidate_item, _uint32_right_shift(candidate_item, 5))
|
||||
secondary_item = secondary_item - candidate_item
|
||||
secondary_item = secondary_item - reference_item
|
||||
secondary_item = _int32_xor(secondary_item, _uint32_right_shift(reference_item, 3))
|
||||
candidate_item = _int32_xor(candidate_item - reference_item - secondary_item, _int32_left_shift(secondary_item, 10))
|
||||
reference_item = reference_item - secondary_item
|
||||
reference_item = _int32_xor(reference_item - candidate_item, _uint32_right_shift(candidate_item, 15))
|
||||
mix_state[0], mix_state[1], mix_state[2] = secondary_item, candidate_item, reference_item
|
||||
return reference_item
|
||||
|
||||
|
||||
def jenkins_hash(text: str) -> int:
|
||||
"""Encode text through the package UTF-16/UTF-8 byte path, then run Jenkins lookup2."""
|
||||
byte_values = _utf8_bytes_from_utf16_units(text)
|
||||
count_item = len(byte_values)
|
||||
mix_state = [-1640531527, -1640531527, 314159265] # 0x9E3779B9, 0x9E3779B9, seed
|
||||
off = 0
|
||||
entry_item = count_item
|
||||
while entry_item >= 12:
|
||||
mix_state[0] = mix_state[0] + _little_endian_signed_word(byte_values, off)
|
||||
mix_state[1] = mix_state[1] + _little_endian_signed_word(byte_values, off + 4)
|
||||
mix_state[2] = mix_state[2] + _little_endian_signed_word(byte_values, off + 8)
|
||||
_jenkins_mix(mix_state)
|
||||
entry_item -= 12
|
||||
off += 12
|
||||
mix_state[2] = mix_state[2] + count_item
|
||||
# Tail-byte mixing follows Jenkins lookup2's fall-through layout.
|
||||
if entry_item >= 11:
|
||||
mix_state[2] = mix_state[2] + _int32_left_shift(byte_values[off + 10], 24)
|
||||
if entry_item >= 10:
|
||||
mix_state[2] = mix_state[2] + ((byte_values[off + 9] & 255) << 16)
|
||||
if entry_item >= 9:
|
||||
mix_state[2] = mix_state[2] + ((byte_values[off + 8] & 255) << 8)
|
||||
if entry_item >= 8:
|
||||
mix_state[1] = mix_state[1] + _little_endian_signed_word(byte_values, off + 4)
|
||||
mix_state[0] = mix_state[0] + _little_endian_signed_word(byte_values, off)
|
||||
elif entry_item >= 4:
|
||||
if entry_item >= 7:
|
||||
mix_state[1] = mix_state[1] + ((byte_values[off + 6] & 255) << 16)
|
||||
if entry_item >= 6:
|
||||
mix_state[1] = mix_state[1] + ((byte_values[off + 5] & 255) << 8)
|
||||
if entry_item >= 5:
|
||||
mix_state[1] = mix_state[1] + (byte_values[off + 4] & 255)
|
||||
mix_state[0] = mix_state[0] + _little_endian_signed_word(byte_values, off)
|
||||
else:
|
||||
if entry_item >= 3:
|
||||
mix_state[0] = mix_state[0] + ((byte_values[off + 2] & 255) << 16)
|
||||
if entry_item >= 2:
|
||||
mix_state[0] = mix_state[0] + ((byte_values[off + 1] & 255) << 8)
|
||||
if entry_item >= 1:
|
||||
mix_state[0] = mix_state[0] + (byte_values[off] & 255)
|
||||
return _jenkins_mix(mix_state)
|
||||
@@ -0,0 +1,243 @@
|
||||
"""Token types, anchors, and token-view utilities."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Iterable, Iterator, Optional
|
||||
|
||||
from ..model import (
|
||||
_strip_diacritics,
|
||||
avg_char_width2,
|
||||
intervals_overlap,
|
||||
to_number,
|
||||
rect_union,
|
||||
EMPTY_RECT,
|
||||
avg_char_width,
|
||||
Line,
|
||||
char_category,
|
||||
is_word_category,
|
||||
is_punct_category,
|
||||
letter_count,
|
||||
punct_count,
|
||||
info_weight,
|
||||
Block,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Character-category transition table #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
# Map 12 character categories down to script-family buckets used by statistics.
|
||||
SCRIPT_FAMILY_MAP = [0, 1, 2, 2, 2, 2, 3, 4, 5, 6, 7, 7, 8]
|
||||
|
||||
|
||||
def _build_gap_tolerance_grid() -> list[list[float]]:
|
||||
"""Return the fractional gap tolerance for adjacent character categories."""
|
||||
token_value = [[0.0] * 12 for _ in range(12)]
|
||||
# Small punctuation-to-mark transition weights.
|
||||
for candidate_item in (1, 2, 3, 4):
|
||||
token_value[candidate_item][5] = 0.16
|
||||
token_value[candidate_item][6] = 0.16
|
||||
token_value[3][2] = 0.1
|
||||
token_value[6][2] = 0.1
|
||||
token_value[6][3] = 0.1
|
||||
token_value[8][2] = 0.1
|
||||
token_value[8][3] = 0.1
|
||||
return token_value
|
||||
|
||||
|
||||
GAP_TOLERANCE_GRID = _build_gap_tolerance_grid()
|
||||
|
||||
|
||||
def can_extend_token(number: int, other_number: int, candidate_text: str) -> bool:
|
||||
"""Return whether the current token can extend with ``candidate_text``."""
|
||||
from ..stats import char_script_bucket
|
||||
if other_number == 4 and char_script_bucket(candidate_text) == 5:
|
||||
return False
|
||||
if other_number == number and not is_punct_category(other_number):
|
||||
return True
|
||||
if is_word_category(number) and is_word_category(other_number):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Token span anchors.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class TokenAnchor:
|
||||
"""Cross-line anchor range attached to a token."""
|
||||
|
||||
__slots__ = ("line", "anchor_span", "start_offset", "primary_slot")
|
||||
|
||||
def __init__(self, line: Line, anchor_span_value, start_offset_value: int, next_number: int):
|
||||
self.line = line
|
||||
self.anchor_span = anchor_span_value
|
||||
self.start_offset = start_offset_value
|
||||
self.primary_slot = next_number
|
||||
|
||||
|
||||
def last_token_anchor(token: "Token") -> TokenAnchor:
|
||||
"""Return the token's last cross-line anchor entry."""
|
||||
return token.anchor_ranges[-1]
|
||||
|
||||
|
||||
def first_anchor_span(token: "Token"):
|
||||
"""Return the anchor span from the token's first cross-line entry."""
|
||||
return token.anchor_ranges[0].anchor_span
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Token kind predicates #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def is_char_token(token: "Token") -> bool:
|
||||
"""Return True for digit or letter tokens."""
|
||||
return token.type == 1 or token.type == 2
|
||||
|
||||
|
||||
def is_word_token(token: "Token") -> bool:
|
||||
"""Return True for merged word-like tokens: word, number-word, or symbolic token kinds."""
|
||||
return token.type in (3, 4, 5)
|
||||
|
||||
|
||||
def is_trimmable_token(token: "Token") -> bool:
|
||||
"""Return True for word, number-word, or colon tokens that can be trimmed from phrase edges."""
|
||||
return token.type == 3 or token.type == 4 or token.str == ":"
|
||||
|
||||
|
||||
def token_numeric_value(token: "Token") -> float:
|
||||
"""numeric value of token, NaN if non-numeric."""
|
||||
return to_number(token.str)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Token #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class Token:
|
||||
"""One token. It stores the token kind, raw text, contributing line/span anchors, bracket attachment flag, and first/last character categories."""
|
||||
|
||||
__slots__ = ("type", "str", "anchor_ranges", "boundary_slot", "primary_slot", "secondary_slot")
|
||||
|
||||
def __init__(self, type_: int, candidate_text: str, anchor_ranges_value: list[TokenAnchor], boundary_flag: bool, previous_number: int, limit_number: int):
|
||||
self.type = type_
|
||||
self.str = candidate_text
|
||||
self.anchor_ranges = anchor_ranges_value
|
||||
self.boundary_slot = boundary_flag
|
||||
self.primary_slot = previous_number
|
||||
self.secondary_slot = limit_number
|
||||
|
||||
def line(self) -> Line:
|
||||
"""Line of the first origin-span back-reference."""
|
||||
return self.anchor_ranges[0].line
|
||||
|
||||
def __repr__(self) -> str: # diagnostic
|
||||
return f"<Token t={self.type} {self.str!r} ba={self.boundary_slot}>"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Directional token view #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class TokenView:
|
||||
"""Sliceable, directional view over a token array. Supports forward / reverse iteration via ``dir`` = +1 / -1. ``slice`` and ``reverse`` produce new views without copying. """
|
||||
|
||||
__slots__ = ("primary_slot", "start", "end", "dir", "length")
|
||||
|
||||
def __init__(self, other_tokens: list[Token], start: int, end: int, dir_: int):
|
||||
self.primary_slot = other_tokens
|
||||
self.start = start
|
||||
self.end = end
|
||||
self.dir = dir_
|
||||
self.length = (end - start) // dir_ if dir_ != 0 else 0
|
||||
|
||||
def __iter__(self) -> Iterator[Token]:
|
||||
secondary_item = self.start
|
||||
while secondary_item != self.end:
|
||||
yield self.primary_slot[secondary_item]
|
||||
secondary_item += self.dir
|
||||
|
||||
def token_at(self, other_number: int) -> Optional[Token]:
|
||||
if other_number < 0 or other_number >= self.length:
|
||||
return None
|
||||
return self.primary_slot[self.start + other_number * self.dir]
|
||||
|
||||
def __getitem__(self, other_number: int) -> Optional[Token]:
|
||||
return self.token_at(other_number)
|
||||
|
||||
def __len__(self) -> int:
|
||||
return self.length
|
||||
|
||||
def __bool__(self) -> bool:
|
||||
return self.length > 0
|
||||
|
||||
def __str__(self) -> str:
|
||||
parts: list[str] = []
|
||||
for token in self:
|
||||
parts.append(token.str)
|
||||
if token.boundary_slot:
|
||||
parts.append(" ")
|
||||
return "".join(parts)
|
||||
|
||||
def slice(self, other_number: int = 0, candidate_number: int = 0) -> "TokenView":
|
||||
"""Bounds-clamped directional slice. Args follow Unicode-compatible semantics: a > 0 -> from index a a < 0 -> from end-relative a = 0 -> from start b > 0 -> to index b b < 0 -> end-relative b = 0 -> to end """
|
||||
if other_number > 0:
|
||||
slice_start = self.start + other_number * self.dir
|
||||
elif other_number < 0:
|
||||
slice_start = self.end + other_number * self.dir
|
||||
else:
|
||||
slice_start = self.start
|
||||
if slice_start * self.dir < self.start * self.dir:
|
||||
slice_start = self.start
|
||||
if slice_start * self.dir > self.end * self.dir:
|
||||
slice_start = self.end
|
||||
if candidate_number > 0:
|
||||
slice_end = self.start + candidate_number * self.dir
|
||||
elif candidate_number < 0:
|
||||
slice_end = self.end + candidate_number * self.dir
|
||||
else:
|
||||
slice_end = self.end
|
||||
if slice_end * self.dir < slice_start * self.dir:
|
||||
slice_end = slice_start
|
||||
if slice_end * self.dir > self.end * self.dir:
|
||||
slice_end = self.end
|
||||
return TokenView(self.primary_slot, slice_start, slice_end, self.dir)
|
||||
|
||||
def reverse(self) -> "TokenView":
|
||||
return TokenView(self.primary_slot, self.end - self.dir, self.start - self.dir, -self.dir)
|
||||
|
||||
def to_string(self) -> str:
|
||||
return str(self)
|
||||
|
||||
|
||||
def wrap_tokens(tokens: list[Token]) -> TokenView:
|
||||
"""Wrap a list of tokens as a forward token view."""
|
||||
return TokenView(tokens, 0, len(tokens), 1)
|
||||
|
||||
|
||||
def enumerate_tokens(tokens: TokenView) -> Iterator[dict]:
|
||||
"""Enumerate a token view yielding indexed token records."""
|
||||
token = tokens.primary_slot
|
||||
start = tokens.start
|
||||
end = tokens.end
|
||||
step = tokens.dir
|
||||
cursor = start
|
||||
while cursor != end:
|
||||
yield {"index": (cursor - start) // step, "token": token[cursor]}
|
||||
cursor += step
|
||||
|
||||
|
||||
def first_token(tokens: TokenView) -> Optional[Token]:
|
||||
"""Return the first token, or None."""
|
||||
return tokens.primary_slot[tokens.start] if tokens.length > 0 else None
|
||||
|
||||
|
||||
def last_token(tokens: TokenView) -> Optional[Token]:
|
||||
"""Return the last token, or None."""
|
||||
return tokens.primary_slot[tokens.end - tokens.dir] if tokens.length > 0 else None
|
||||
@@ -0,0 +1,271 @@
|
||||
"""Line tokenization into word, char, and number tokens."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import unicodedata
|
||||
from typing import Any, Iterable, Iterator, Optional
|
||||
|
||||
from ..model import (
|
||||
_strip_diacritics,
|
||||
avg_char_width2,
|
||||
intervals_overlap,
|
||||
to_number,
|
||||
rect_union,
|
||||
EMPTY_RECT,
|
||||
avg_char_width,
|
||||
Line,
|
||||
char_category,
|
||||
is_word_category,
|
||||
is_punct_category,
|
||||
letter_count,
|
||||
punct_count,
|
||||
info_weight,
|
||||
Block,
|
||||
)
|
||||
|
||||
from .token_types import (
|
||||
SCRIPT_FAMILY_MAP,
|
||||
GAP_TOLERANCE_GRID,
|
||||
can_extend_token,
|
||||
TokenAnchor,
|
||||
last_token_anchor,
|
||||
first_anchor_span,
|
||||
Token,
|
||||
TokenView,
|
||||
wrap_tokens,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Line tokenizer state machine #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
class LineTokenizer:
|
||||
"""Line-tokenizer state machine with line/span anchors for reconstruction."""
|
||||
|
||||
__slots__ = ("tertiary_slot", "secondary_slot", "cache_slot", "auxiliary_slot", "option_slot", "marker_slot", "primary_slot", "previous_slot", "state_slot", "style_slot", "measure_slot")
|
||||
|
||||
def __init__(self):
|
||||
self.tertiary_slot: list[Token] = []
|
||||
self.secondary_slot = None # last anchor span
|
||||
self.cache_slot: Optional[Line] = None
|
||||
self.auxiliary_slot = -1
|
||||
self.option_slot = -1
|
||||
self.marker_slot: list[TokenAnchor] = []
|
||||
self.primary_slot = ""
|
||||
self.previous_slot = False
|
||||
self.state_slot = 0 # last-char category
|
||||
self.style_slot = 0 # first-char category
|
||||
self.measure_slot = 0 # type-hint accumulator
|
||||
|
||||
# --- inner state ops --------------------------------------------------
|
||||
|
||||
def _close_anchor_range(self) -> None:
|
||||
"""Close the current anchor range into the in-flight token and reset offsets."""
|
||||
self.marker_slot.append(TokenAnchor(self.cache_slot, self.secondary_slot, self.auxiliary_slot, self.option_slot))
|
||||
self.auxiliary_slot = self.option_slot = -1
|
||||
|
||||
def _close_token(self, boundary_flag: bool) -> None:
|
||||
"""Close the in-flight token into the token list."""
|
||||
if self.auxiliary_slot >= 0:
|
||||
self._close_anchor_range()
|
||||
self.tertiary_slot.append(Token(self.measure_slot, self.primary_slot, self.marker_slot, boundary_flag, self.style_slot, self.state_slot))
|
||||
self.marker_slot = []
|
||||
self.primary_slot = ""
|
||||
self.measure_slot = 0
|
||||
self.style_slot = 0
|
||||
self.state_slot = 0
|
||||
|
||||
def _accumulate_char(self, other_text: str, candidate_number: int) -> None:
|
||||
"""Append a character and update the in-flight token kind from the category map."""
|
||||
if len(self.primary_slot) == 1 and self.state_slot == 5:
|
||||
# If the in-flight token is a single mark, attach it before the new
|
||||
# character so combining marks bind to the following letter.
|
||||
self.primary_slot = other_text + self.primary_slot
|
||||
self.style_slot = candidate_number
|
||||
else:
|
||||
if not self.primary_slot:
|
||||
self.style_slot = candidate_number
|
||||
self.primary_slot += other_text
|
||||
self.state_slot = candidate_number
|
||||
cat = SCRIPT_FAMILY_MAP[candidate_number]
|
||||
if self.measure_slot == 0:
|
||||
self.measure_slot = cat
|
||||
elif self.measure_slot == 1 and cat != 1:
|
||||
self.measure_slot = 2
|
||||
self.previous_slot = False
|
||||
|
||||
def _advance_char(self, other_text: str, candidate_number: int) -> None:
|
||||
"""Advance the tokenizer with one character. Whitespace sets the pending-boundary flag; non-whitespace either extends or closes the current token."""
|
||||
reference_item = char_category(other_text)
|
||||
if reference_item == 10:
|
||||
# whitespace
|
||||
self.previous_slot = True
|
||||
return
|
||||
|
||||
if self.previous_slot and self.primary_slot:
|
||||
# If the last non-whitespace category and the current category cannot
|
||||
# belong to the same word-like token, close the current token.
|
||||
|
||||
if not (reference_item == 5 and is_word_category(self.state_slot)):
|
||||
self._close_token(True)
|
||||
|
||||
# Soft-hyphen rejoin across lines: if there is no in-flight token, the
|
||||
# current char is lowercase, and the previous tokens were a word plus
|
||||
# "-" ending on another line, undo the split and continue that word.
|
||||
if not self.primary_slot and len(self.tertiary_slot) >= 2 and reference_item == 3:
|
||||
entry_item = self.tertiary_slot[-1]
|
||||
token = self.tertiary_slot[-2]
|
||||
if (
|
||||
token.secondary_slot == 3
|
||||
and not token.boundary_slot
|
||||
and entry_item.str == "-"
|
||||
and last_token_anchor(entry_item).line is not self.cache_slot
|
||||
):
|
||||
self.tertiary_slot.pop() # drop "-"
|
||||
entry_item = self.tertiary_slot.pop() # pop word
|
||||
self.measure_slot = entry_item.type
|
||||
self.primary_slot = entry_item.str
|
||||
self.marker_slot = entry_item.anchor_ranges
|
||||
self.style_slot = entry_item.primary_slot
|
||||
self.state_slot = entry_item.secondary_slot
|
||||
self.previous_slot = False
|
||||
self._accumulate_char(other_text, reference_item)
|
||||
self.auxiliary_slot = self.option_slot = candidate_number
|
||||
return
|
||||
|
||||
if self.primary_slot:
|
||||
if can_extend_token(self.state_slot, reference_item, other_text):
|
||||
self._accumulate_char(other_text, reference_item)
|
||||
if self.auxiliary_slot < 0:
|
||||
self.auxiliary_slot = candidate_number
|
||||
self.option_slot = candidate_number
|
||||
else:
|
||||
self._close_token(False)
|
||||
self._accumulate_char(other_text, reference_item)
|
||||
self.auxiliary_slot = self.option_slot = candidate_number
|
||||
else:
|
||||
self._accumulate_char(other_text, reference_item)
|
||||
self.auxiliary_slot = self.option_slot = candidate_number
|
||||
|
||||
# --- public API -------------------------------------------------------
|
||||
|
||||
def add_line(self, other_line: Line) -> "LineTokenizer":
|
||||
"""Walk one line and append its token contribution."""
|
||||
line = self.tertiary_slot[-1] if self.tertiary_slot else None
|
||||
if self.primary_slot:
|
||||
# Close in-flight; a trailing hyphen can glue to the next line only
|
||||
# when the previous token was not already bracket-attached.
|
||||
self._close_token(self.primary_slot != "-" or line is None or line.boundary_slot)
|
||||
self.cache_slot = other_line
|
||||
|
||||
# Single-codepoint pending combining mark.
|
||||
pending = None # type: Optional[Any]
|
||||
for index in range(len(other_line.primary_slot)):
|
||||
span = other_line.primary_slot[index]
|
||||
if span.char_count() <= 0:
|
||||
continue
|
||||
# Drop solitary combining marks (last-character category is 5)
|
||||
|
||||
if pending is None and span.char_count() == 1 and span.char_stats.secondary_slot == 5:
|
||||
pending = span
|
||||
continue
|
||||
# Drop the bullet-then-content kerning glitch (layout branch:
|
||||
# single-character token, previous category is 11, and next span overlaps horizontally)
|
||||
|
||||
if (
|
||||
index + 1 < len(other_line.primary_slot)
|
||||
and span.char_count() == 1
|
||||
and span.char_stats.secondary_slot == 11
|
||||
and span.left_edge() >= other_line.primary_slot[index + 1].left_edge()
|
||||
and span.center_x() < other_line.primary_slot[index + 1].right_edge()
|
||||
):
|
||||
continue
|
||||
|
||||
if self.secondary_slot is not None and self.primary_slot:
|
||||
# Decide whether the new span continues the same token
|
||||
if (
|
||||
span.left_edge() <= self.secondary_slot.right_edge() + 0.1 * avg_char_width2(self.secondary_slot)
|
||||
and (
|
||||
abs(span.bottom_edge() - self.secondary_slot.bottom_edge()) < 0.1
|
||||
or abs(span.center_y() - self.secondary_slot.center_y()) < 0.1
|
||||
)
|
||||
and self.secondary_slot.primary_slot == span.primary_slot
|
||||
):
|
||||
# Continue: close the current cross-line anchor entry and switch anchor.
|
||||
self._close_anchor_range()
|
||||
self.secondary_slot = span
|
||||
self.previous_slot = False
|
||||
else:
|
||||
gap_tolerance = (GAP_TOLERANCE_GRID[self.secondary_slot.char_stats.tertiary_slot][span.char_stats.secondary_slot] or 0.12) * avg_char_width(self.cache_slot)
|
||||
close = (
|
||||
self.previous_slot
|
||||
or abs(self.secondary_slot.bottom_edge() - span.bottom_edge()) > 1
|
||||
or span.left_edge() < self.secondary_slot.right_edge() - 1
|
||||
or span.left_edge() > self.secondary_slot.right_edge() + gap_tolerance
|
||||
)
|
||||
self._close_token(close)
|
||||
self.secondary_slot = span
|
||||
else:
|
||||
self.secondary_slot = span
|
||||
|
||||
for char_index in range(len(span.text)):
|
||||
char_value = span.text[char_index]
|
||||
if (
|
||||
char_index == 0
|
||||
and pending is not None
|
||||
and intervals_overlap(pending.left_edge(), pending.right_edge(), span.left_edge(), span.right_edge())
|
||||
):
|
||||
# Compose with the pending combining mark
|
||||
combined = unicodedata.normalize("NFC", char_value + pending.state_slot[0])
|
||||
self._advance_char(combined[0], 0)
|
||||
else:
|
||||
self._advance_char(char_value, char_index)
|
||||
pending = None
|
||||
return self
|
||||
|
||||
def tokens(self) -> TokenView:
|
||||
"""Finalize and return a token view."""
|
||||
if self.primary_slot:
|
||||
self._close_token(True)
|
||||
return wrap_tokens(self.tertiary_slot)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# X(block) -- cached token list for a block #
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def tokenize_block(block: Block) -> TokenView:
|
||||
"""tokenize all lines of a block, cached on the block token cache."""
|
||||
if block.tokens_cache is not None:
|
||||
return block.tokens_cache # type: ignore[return-value]
|
||||
token = LineTokenizer()
|
||||
for line in block.primary_slot:
|
||||
token.add_line(line)
|
||||
block.tokens_cache = token.tokens() # type: ignore[assignment]
|
||||
return block.tokens_cache # type: ignore[return-value]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Utility helpers.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def clamp_value(value: float, lower_bound: float, upper_bound: float) -> float:
|
||||
"""Clamp a value between lower and upper bounds. The lower bound wins when the bounds are inverted, and NaN propagates."""
|
||||
measure_item = upper_bound if upper_bound < value else value
|
||||
return lower_bound if lower_bound > measure_item else measure_item
|
||||
|
||||
|
||||
def is_superscript_adjacent(token: Token, other_token: Token) -> bool:
|
||||
"""Return whether the next token is a raised, shorter marker on the same line."""
|
||||
candidate_item = last_token_anchor(token).anchor_span
|
||||
reference_item = first_anchor_span(other_token)
|
||||
return (
|
||||
reference_item is not candidate_item
|
||||
and last_token_anchor(token).line is other_token.line()
|
||||
and reference_item.bbox_height() < candidate_item.bbox_height()
|
||||
and reference_item.bottom_edge() > candidate_item.bottom_edge() + 0.1 * candidate_item.bbox_height()
|
||||
)
|
||||
@@ -0,0 +1,336 @@
|
||||
"""Trie construction, matching, and token trimming utilities."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Iterable, Iterator, Optional
|
||||
|
||||
from ..model import (
|
||||
_strip_diacritics,
|
||||
avg_char_width2,
|
||||
intervals_overlap,
|
||||
to_number,
|
||||
rect_union,
|
||||
EMPTY_RECT,
|
||||
avg_char_width,
|
||||
Line,
|
||||
char_category,
|
||||
is_word_category,
|
||||
is_punct_category,
|
||||
letter_count,
|
||||
punct_count,
|
||||
info_weight,
|
||||
Block,
|
||||
)
|
||||
|
||||
from .token_types import (
|
||||
can_extend_token,
|
||||
is_trimmable_token,
|
||||
TokenView,
|
||||
wrap_tokens,
|
||||
enumerate_tokens,
|
||||
first_token,
|
||||
last_token,
|
||||
)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Token trie matcher and builder.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _de_norm(text: str, case_fold: bool) -> str:
|
||||
"""Normalize trie keys by optional case folding, NFD decomposition, combining-mark stripping, and NFC recomposition. This strips diacritics without applying compatibility normalization."""
|
||||
return _strip_diacritics(text.lower() if case_fold else text)
|
||||
|
||||
|
||||
class TrieConfig:
|
||||
"""Trie configuration: reverse-match mode and case-fold mode."""
|
||||
|
||||
__slots__ = ("primary_slot", "secondary_slot")
|
||||
|
||||
def __init__(self):
|
||||
self.primary_slot: bool = False
|
||||
self.secondary_slot: bool = False
|
||||
|
||||
|
||||
class BuiltTrie:
|
||||
"""Built trie wrapper containing the root node and a reverse-match flag."""
|
||||
|
||||
__slots__ = ("secondary_slot", "primary_slot")
|
||||
|
||||
def __init__(self, primary_item: "TrieNode", candidate_flag: bool):
|
||||
self.secondary_slot = primary_item # root node
|
||||
self.primary_slot = candidate_flag # reverse-match flag
|
||||
|
||||
|
||||
def set_reverse(primary_item: TrieConfig) -> TrieConfig:
|
||||
"""set reverse flag."""
|
||||
primary_item.primary_slot = True
|
||||
return primary_item
|
||||
|
||||
|
||||
def set_case_fold(primary_item: TrieConfig, other_flag: bool) -> TrieConfig:
|
||||
"""set case-fold flag."""
|
||||
primary_item.secondary_slot = other_flag
|
||||
return primary_item
|
||||
|
||||
|
||||
class TrieNode:
|
||||
"""- trie node."""
|
||||
|
||||
__slots__ = ("str", "depth", "primary_slot", "children", "dict_suffix_link", "failure_link", "is_terminal", "payload")
|
||||
|
||||
def __init__(self, other_text: str, depth: int, case_fold: bool):
|
||||
self.str = other_text
|
||||
self.depth = depth
|
||||
self.primary_slot = case_fold
|
||||
self.children: dict[str, "TrieNode"] = {}
|
||||
self.dict_suffix_link = None
|
||||
self.failure_link: Optional["TrieNode"] = None
|
||||
self.is_terminal = False
|
||||
self.payload = None
|
||||
|
||||
def normalize(self, other_text: str) -> str:
|
||||
return _de_norm(other_text, self.primary_slot)
|
||||
|
||||
|
||||
def trie_insert_step(node: TrieNode, other_text: str) -> TrieNode:
|
||||
"""walk one child, creating if absent."""
|
||||
key = node.normalize(other_text)
|
||||
child = node.children.get(key)
|
||||
if child is None:
|
||||
child = TrieNode(key, node.depth + 1, node.primary_slot)
|
||||
node.children[key] = child
|
||||
return child
|
||||
|
||||
|
||||
def trie_walk_step(node: TrieNode, other_text: str) -> TrieNode:
|
||||
"""Walk one child; if absent, fall back through failure links."""
|
||||
key = node.normalize(other_text)
|
||||
child = node.children.get(key)
|
||||
if child is not None:
|
||||
return child
|
||||
if node.failure_link is not None:
|
||||
return trie_walk_step(node.failure_link, other_text)
|
||||
return node
|
||||
|
||||
|
||||
def aho_corasick_match(trie: BuiltTrie, tokens) -> Optional[dict]:
|
||||
"""Aho-Corasick walk over a trie. Returns the shortest earliest terminal match and its payload. Dictionary-suffix matches use the suffix depth for match length while retaining the current node payload, which is load-bearing for edge cases."""
|
||||
if isinstance(tokens, list):
|
||||
tokens = wrap_tokens(tokens)
|
||||
if trie.primary_slot:
|
||||
tokens = tokens.reverse()
|
||||
matched_tokens: Optional[TokenView] = None
|
||||
matched_reverse = None
|
||||
earliest_start = -1
|
||||
node: TrieNode = trie.secondary_slot # root node
|
||||
for entry in enumerate_tokens(tokens):
|
||||
index = entry["index"]
|
||||
token = entry["token"]
|
||||
node = trie_walk_step(node, token.str)
|
||||
depth = node.depth if node.is_terminal else 0
|
||||
if depth > 0 and (earliest_start < 0 or index - depth + 1 <= earliest_start):
|
||||
earliest_start = index - depth + 1
|
||||
matched_tokens = tokens.slice(earliest_start, index + 1)
|
||||
matched_reverse = node.payload
|
||||
if trie.primary_slot:
|
||||
matched_tokens = matched_tokens.reverse()
|
||||
kb_node = node.dict_suffix_link
|
||||
kb_depth = kb_node.depth if kb_node is not None else 0
|
||||
if kb_depth > 0 and (earliest_start < 0 or index - kb_depth + 1 <= earliest_start):
|
||||
earliest_start = index - kb_depth + 1
|
||||
matched_tokens = tokens.slice(earliest_start, index + 1)
|
||||
matched_reverse = node.payload
|
||||
|
||||
if trie.primary_slot:
|
||||
matched_tokens = matched_tokens.reverse()
|
||||
# Once a match exists and the current path start has moved past the
|
||||
# earliest match start, no later token can produce an earlier match.
|
||||
if earliest_start >= 0 and index - node.depth + 1 > earliest_start:
|
||||
break
|
||||
if matched_tokens is None:
|
||||
return None
|
||||
return {"tokens": matched_tokens, "payload": matched_reverse}
|
||||
|
||||
|
||||
def aho_corasick_tokens(trie: BuiltTrie, tokens) -> Optional[TokenView]:
|
||||
"""Return only the matched token view from an Aho-Corasick match."""
|
||||
token = aho_corasick_match(trie, tokens)
|
||||
return token["tokens"] if token is not None else None
|
||||
|
||||
|
||||
class TrieBuilder:
|
||||
"""Trie builder context holding the root node and configuration."""
|
||||
|
||||
__slots__ = ("primary_slot", "secondary_slot")
|
||||
|
||||
def __init__(self, query_value: TrieConfig):
|
||||
self.primary_slot = TrieNode("", 0, query_value.secondary_slot) # root node
|
||||
self.secondary_slot = query_value # the config
|
||||
|
||||
|
||||
def _trie_insert_entry(builder: TrieBuilder, entry: str, payload: Optional[Any] = None) -> None:
|
||||
"""Insert one phrase into the trie after character-by-character tokenization. This keeps punctuation-attached phrases such as ``vol.`` and ``etc.`` aligned with document tokenization. The optional payload is stored only on an empty terminal payload slot."""
|
||||
node = builder.primary_slot
|
||||
tokens: list[str] = []
|
||||
trie = ""
|
||||
previous_category = 0
|
||||
for char in entry:
|
||||
cat = char_category(char)
|
||||
if cat == 10 or (trie and not can_extend_token(previous_category, cat, char)):
|
||||
if trie:
|
||||
tokens.append(trie)
|
||||
trie = ""
|
||||
if cat != 10:
|
||||
trie += char
|
||||
previous_category = cat
|
||||
if trie:
|
||||
tokens.append(trie)
|
||||
if builder.secondary_slot.primary_slot:
|
||||
tokens.reverse()
|
||||
for tok in tokens:
|
||||
node = trie_insert_step(node, tok)
|
||||
node.is_terminal = True
|
||||
# Payload assignment uses truthiness: falsy payloads are skipped, and falsy
|
||||
# existing payloads are overwritten. In this package payloads are non-empty
|
||||
# dictionary-like objects, so the truthiness contract is stable.
|
||||
if payload and not node.payload:
|
||||
node.payload = payload
|
||||
|
||||
|
||||
def trie_bulk_insert(builder: TrieBuilder, entries, payload: Optional[Any] = None) -> None:
|
||||
"""Bulk-insert phrases into ``builder`` with a shared terminal payload."""
|
||||
for entry in entries:
|
||||
_trie_insert_entry(builder, entry, payload)
|
||||
|
||||
|
||||
def _trie_finalize(builder: TrieBuilder) -> BuiltTrie:
|
||||
"""Assign Aho-Corasick failure links and dictionary-suffix links with breadth-first traversal, then return a built trie wrapper."""
|
||||
from collections import deque
|
||||
|
||||
root = builder.primary_slot
|
||||
queue: deque = deque([root])
|
||||
while queue:
|
||||
node = queue.popleft()
|
||||
for child in node.children.values():
|
||||
queue.append(child)
|
||||
# failure link: longest proper suffix that is a prefix in the trie
|
||||
trie = node
|
||||
while trie.failure_link is not None:
|
||||
child.failure_link = trie.failure_link.children.get(trie.failure_link.normalize(child.str))
|
||||
if child.failure_link is not None:
|
||||
break
|
||||
trie = trie.failure_link
|
||||
if child.failure_link is None:
|
||||
child.failure_link = root
|
||||
# dictionary-suffix link: nearest failure ancestor that is terminal
|
||||
trie = child.failure_link
|
||||
while trie is not None:
|
||||
if trie.is_terminal:
|
||||
child.dict_suffix_link = trie
|
||||
break
|
||||
trie = trie.failure_link
|
||||
|
||||
return BuiltTrie(builder.primary_slot, builder.secondary_slot.primary_slot)
|
||||
|
||||
|
||||
def build_trie(strings: Iterable[str], other_trie: Optional[TrieConfig] = None) -> BuiltTrie:
|
||||
"""Build a trie from a list of phrase strings."""
|
||||
if other_trie is None:
|
||||
other_trie = TrieConfig()
|
||||
builder = TrieBuilder(other_trie)
|
||||
for trie in strings:
|
||||
_trie_insert_entry(builder, trie)
|
||||
return _trie_finalize(builder)
|
||||
|
||||
|
||||
def trie_prefix_match(trie: BuiltTrie, tokens) -> Optional[TokenView]:
|
||||
"""Return the longest prefix match against the token trie."""
|
||||
# ``tokens`` may be a TokenView or a list; coerce.
|
||||
if isinstance(tokens, list):
|
||||
tokens = wrap_tokens(tokens)
|
||||
if trie.primary_slot:
|
||||
tokens = tokens.reverse()
|
||||
|
||||
matched: Optional[TokenView] = None
|
||||
node: TrieNode = trie.secondary_slot # root node
|
||||
for entry in enumerate_tokens(tokens):
|
||||
if not node.children:
|
||||
break
|
||||
index = entry["index"]
|
||||
token = entry["token"]
|
||||
next_node = node.children.get(node.normalize(token.str))
|
||||
if next_node is None:
|
||||
break
|
||||
node = next_node
|
||||
if node.is_terminal:
|
||||
slice_view = tokens.slice(0, index + 1)
|
||||
if trie.primary_slot:
|
||||
slice_view = slice_view.reverse()
|
||||
matched = slice_view
|
||||
return matched
|
||||
|
||||
|
||||
def _trie_full_match(trie: BuiltTrie, tokens) -> bool:
|
||||
"""full-match check."""
|
||||
result = trie_prefix_match(trie, tokens)
|
||||
if isinstance(tokens, list):
|
||||
tokens_view = wrap_tokens(tokens)
|
||||
else:
|
||||
tokens_view = tokens
|
||||
return result is not None and result.length == tokens_view.length
|
||||
|
||||
|
||||
trie_full_match = _trie_full_match
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Token-list strip helpers.
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def strip_trie_match(tokens: TokenView, other_trie: BuiltTrie) -> TokenView:
|
||||
"""Strip a matching keyword sequence from a token view."""
|
||||
trie = trie_prefix_match(other_trie, tokens)
|
||||
if trie is None:
|
||||
return tokens
|
||||
if other_trie.primary_slot:
|
||||
return tokens.slice(0, tokens.length - trie.length)
|
||||
return tokens.slice(trie.length)
|
||||
|
||||
|
||||
def strip_leading_if_in(tokens: TokenView, other_items: set) -> TokenView:
|
||||
"""Strip the leading token if its text is in the provided set."""
|
||||
first = first_token(tokens)
|
||||
if tokens.length > 0 and first is not None and first.str in other_items:
|
||||
return tokens.slice(1)
|
||||
return tokens
|
||||
|
||||
|
||||
# Six comma variants only, not general punctuation.
|
||||
COMMA_CHARS: set[str] = {",", "﹐", ",", "、", "﹑", "、"}
|
||||
|
||||
|
||||
def strip_trailing_comma(tokens: TokenView) -> TokenView:
|
||||
"""Strip a trailing comma token."""
|
||||
last = last_token(tokens)
|
||||
if tokens.length > 0 and last is not None and last.str in COMMA_CHARS:
|
||||
return tokens.slice(0, tokens.length - 1)
|
||||
return tokens
|
||||
|
||||
|
||||
def is_comma_token(token) -> bool:
|
||||
"""Return True when the token string is one of the supported comma variants."""
|
||||
return token is not None and token.str in COMMA_CHARS
|
||||
|
||||
|
||||
def trim_trailing_punct(tokens: TokenView) -> TokenView:
|
||||
"""Trim trailing punctuation-like tokens."""
|
||||
end = tokens.length
|
||||
while end > 0:
|
||||
tok = tokens.token_at(end - 1)
|
||||
if tok is None or not is_trimmable_token(tok):
|
||||
break
|
||||
end -= 1
|
||||
return tokens.slice(0, end)
|
||||
Reference in New Issue
Block a user