Add PageIndex Flash

This commit is contained in:
Ray
2026-07-30 20:35:19 +08:00
parent 0e4b68c92d
commit fffc2c5512
79 changed files with 16438 additions and 0 deletions
+50
View File
@@ -0,0 +1,50 @@
# PageIndex Flash
Builds a PageIndex tree structure from a PDF using layout statistics alone.
No LLM, no API key, no OCR, no network. Runs in seconds, fully offline.
## Usage
```python
from pageindex.flash import page_index_flash
tree = page_index_flash("paper.pdf")
```
```bash
python3 run_pageindex.py --pdf_path document.pdf --flash
```
Accepts a path (`str` or `pathlib.Path`) or an `io.BytesIO` stream. Raises on a
missing, non-PDF, encrypted, empty, or unreadable file.
## Output
```python
{
"doc_name": str,
"doc_title": str,
"structure": [
{
"title": str,
"node_id": str, # 4-digit, zero-padded
"start_index": int,
"end_index": int,
"nodes": [...], # absent on leaf nodes
}
],
}
```
Page indexes are 1-based. `nodes` nests the same shape recursively.
## Limits
- Scanned PDFs without embedded text are not supported.
- Encrypted PDFs need preprocessing first.
- Headings drawn as vector paths, or very decorative layouts, can be missed.
- Titles are taken from the document text as-is.
## Dependencies
`pypdfium2`, `PyPDF2`, `regex`, `sortedcontainers`.
+5
View File
@@ -0,0 +1,5 @@
"""PageIndex Flash: LLM-free tree structure extraction from PDF layout statistics."""
from .api import page_index_flash
__all__ = ["page_index_flash"]
+104
View File
@@ -0,0 +1,104 @@
"""Public API for PageIndex Flash. The only supported entry point is :func:`page_index_flash`. Everything else in this package is internal pipeline machinery."""
from __future__ import annotations
from io import BytesIO
from pathlib import Path
from typing import BinaryIO
import pypdfium2 as pdfium
from .main import extract_toc
def _is_pdfium_password_error(exc: Exception) -> bool:
msg = str(exc).lower()
return "password" in msg or "security" in msg or "encrypted" in msg
def _validate_path(path: Path) -> str:
if not path.exists():
raise FileNotFoundError(f"PDF file not found: {path}")
if not path.is_file():
raise ValueError(f"PDF path is not a file: {path}")
if path.suffix.lower() != ".pdf":
raise ValueError(f"PDF file must have a .pdf extension: {path}")
with path.open("rb") as score_value:
if score_value.read(5) != b"%PDF-":
raise ValueError(f"File does not look like a PDF: {path}")
return str(path)
def _validate_stream(stream: BinaryIO) -> BinaryIO:
try:
pos = stream.tell()
head = stream.read(5)
stream.seek(pos)
except Exception as exc: # noqa: BLE001 - normalize stream capability errors
raise TypeError("PDF stream must be seekable and readable") from exc
if head != b"%PDF-":
raise ValueError("Input stream does not look like a PDF")
return stream
def _validate_pdf(pdf):
if isinstance(pdf, (str, Path)):
handle = _validate_path(Path(pdf))
restore = None
elif isinstance(pdf, BytesIO):
handle = _validate_stream(pdf)
restore = pdf.tell()
else:
raise TypeError("page_index_flash(pdf) expects a PDF path or io.BytesIO stream")
doc = None
try:
doc = pdfium.PdfDocument(handle)
if len(doc) == 0:
raise ValueError("PDF contains no pages")
except pdfium.PdfiumError as exc:
if _is_pdfium_password_error(exc):
raise ValueError("PDF is encrypted or password-protected") from exc
raise ValueError(f"Could not open PDF: {exc}") from exc
finally:
if doc is not None:
doc.close()
if restore is not None:
pdf.seek(restore)
return pdf
def _thin(structure):
from ..utils import page_level_thinning, write_node_id
page_level_thinning(structure)
write_node_id(structure)
async def _summarize(structure, page_list, model):
from ..utils import add_node_text, generate_summaries_for_structure, remove_structure_text
add_node_text(structure, page_list)
await generate_summaries_for_structure(structure, model=model)
remove_structure_text(structure)
def page_index_flash(pdf, summary=True, summary_model=None) -> dict:
"""Build a PageIndex tree structure from a PDF using layout statistics, without an LLM. Args: pdf: path to a PDF file (``str`` or ``pathlib.Path``) or an in-memory binary stream (``io.BytesIO``). summary: if True, generate LLM summaries for each node (requires ``summary_model``). summary_model: the LLM model identifier to use for summary generation. Returns: dict with keys ``doc_name``, ``doc_title``, ``structure`` (a list of nested ``{"title", "start_index", "end_index", "nodes"}`` dicts; page indexes are 1-based) and ``has_abstract_or_references_section`` (True when a top-level entry is an abstract or references heading). """
result = extract_toc(_validate_pdf(pdf))
structure = result.get("structure", [])
if structure:
_thin(structure)
if summary and structure:
import asyncio
from ..utils import ConfigLoader
if summary_model is None:
cfg = ConfigLoader().load()
summary_model = getattr(cfg, 'summary_model', None) or cfg.model
page_texts = result.pop("page_texts", [])
page_list = [(text, 0) for text in page_texts]
asyncio.run(_summarize(structure, page_list, summary_model))
else:
result.pop("page_texts", None)
return result
__all__ = ["page_index_flash"]
+56
View File
@@ -0,0 +1,56 @@
"""Block clustering. This module walks page lines in reading order, extends nearby compatible
blocks, starts a new block when no neighbor fits, and then splits simple
"heading + body" two-line blocks where the first line is a standalone section
heading. The clustering pass must return blocks, not raw lines. Reading-order assignment
then uses each block's first line to find the column index; doing that on raw
lines would read an unrelated first-span flag.
"""
from typing import Optional
from sortedcontainers import SortedKeyList
import json
from pathlib import Path
from ..model import (
style_key,
magnitude_ratio,
left_aligned,
right_aligned,
center_aligned,
x_centers_close,
Rect,
last_span,
avg_char_width,
EMPTY_RECT,
left_edge_key,
reading_order_key,
numbering_kind,
Line,
case_signal,
last_line_of,
first_span_of,
letter_count,
dominant_style_of,
is_upper_dominant,
Block,
_max_nan_propagating,
)
from ..stats import DocStats, PageStats
from ..tokens import set_case_fold, TrieConfig, build_trie, tokenize_block
from .join_rules import (
_DICT_PATH,
_DICTS,
SECTION_HEADING_TRIE,
BlockClusterContext,
should_join_line_to_block,
)
from .build import (
split_heading_body_blocks,
_set_add,
cluster_lines_into_blocks,
)
__all__ = ["BlockClusterContext", "should_join_line_to_block", "cluster_lines_into_blocks", "split_heading_body_blocks", "SECTION_HEADING_TRIE"]
+173
View File
@@ -0,0 +1,173 @@
"""Clusters lines into blocks and splits heading-body blocks."""
from __future__ import annotations
from sortedcontainers import SortedKeyList
from ..model import (
style_key,
magnitude_ratio,
left_aligned,
right_aligned,
center_aligned,
x_centers_close,
Rect,
last_span,
avg_char_width,
EMPTY_RECT,
left_edge_key,
reading_order_key,
numbering_kind,
Line,
case_signal,
last_line_of,
first_span_of,
letter_count,
dominant_style_of,
is_upper_dominant,
Block,
_max_nan_propagating,
)
from ..tokens import set_case_fold, TrieConfig, build_trie, tokenize_block
from .join_rules import (
SECTION_HEADING_TRIE,
BlockClusterContext,
should_join_line_to_block,
)
# --------------------------------------------------------------------------- #
# Two-line block split post-process #
# --------------------------------------------------------------------------- #
def split_heading_body_blocks(input_blocks: list[Block]) -> list[Block]:
"""Split blocks whose first line is a section heading followed by body text."""
from ..labels import trie_matches_all, advance_past_line
split_output_blocks: list[Block] = []
for input_block in input_blocks:
first_line = input_block.line() # first line
# Skip blocks that obviously aren't "heading + body":
# - 1-line blocks
# - small/short blocks
# - first-span style == last-span style AND wide first line
if (
input_block.line_count() <= 1
or (input_block.bbox_height() >= 0.6 * input_block.bbox_width() and input_block.char_count() < 20 * input_block.line_count())
or (style_key(first_span_of(input_block)) == style_key(last_span(last_line_of(input_block))) and first_line.bbox_width() > 0.5 * input_block.bbox_width())
):
split_output_blocks.append(input_block)
continue
block_tokens = tokenize_block(input_block)
first_line_tokens = block_tokens.slice(0, advance_past_line(block_tokens, first_line, 0))
split_token = block_tokens.token_at(first_line_tokens.length)
if split_token is None or split_token.primary_slot == 3:
split_output_blocks.append(input_block)
continue
if not trie_matches_all(SECTION_HEADING_TRIE, first_line_tokens):
split_output_blocks.append(input_block)
continue
# Split: first block holds the heading line; second holds the rest.
split_heading_block = Block()
split_heading_block.add_line(first_line)
split_body_block = Block()
for line_idx in range(1, input_block.line_count()):
split_body_block.add_line(input_block.primary_slot[line_idx])
split_output_blocks.append(split_heading_block)
split_output_blocks.append(split_body_block)
return split_output_blocks
# --------------------------------------------------------------------------- #
# Block-clustering driver #
# --------------------------------------------------------------------------- #
def _set_add(tree: SortedKeyList, block: Block) -> None:
"""Sorted-set insertion semantics: when another block has the same left-edge ordering key, the new block is ignored instead of kept as a multiset duplicate."""
idx = tree.bisect_left(block)
if idx < len(tree) and left_edge_key(tree[idx]) == left_edge_key(block): # type: ignore[arg-type]
return # key collision -> sorted set.add drops the element
tree.add(block)
def cluster_lines_into_blocks(ctx: BlockClusterContext) -> list[Block]:
"""Walk lines, extend existing blocks when compatible, otherwise open a block. Returns blocks sorted bottom, then top, then left, then right before reading-order assignment."""
# Tree of *blocks* sorted by (left, right, top desc, bottom desc)
tree: SortedKeyList = SortedKeyList(key=left_edge_key)
clustered_blocks: list[Block] = []
lines = ctx.secondary_slot
line_count = len(lines)
for line_index in range(line_count):
candidate_line = lines[line_index]
next_line = lines[line_index + 1] if line_index + 1 < line_count else None
# The new line wrapped as a block (used as the tree key for lookups).
seed_block = Block().add_line(candidate_line)
# Collect candidate blocks whose horizontal interval overlaps e_line.
# * predecessors: walk backwards from g_seed_block's left, gather
# blocks whose right edge >= e_line.left.
# * successors: walk forwards, gather blocks whose left edge <= e_line.right.
candidate_blocks: list[Block] = []
# Predecessors by decreasing block-order key.
# Predecessor walk starts at the largest key <= the seed key.
idx_pred = tree.bisect_right(seed_block)
block = idx_pred - 1
while block >= 0:
existing_block: Block = tree[block] # type: ignore[assignment]
if existing_block.right_edge() < candidate_line.left_edge():
break
candidate_blocks.append(existing_block)
block -= 1
# Successors by increasing block-order key.
# Successor walk starts at the smallest key >= the seed key. An exact
# key-equal node is intentionally visited by both walks.
idx_succ = tree.bisect_left(seed_block)
block = idx_succ
while block < len(tree):
existing_block = tree[block] # type: ignore[assignment]
if existing_block.left_edge() > candidate_line.right_edge():
break
candidate_blocks.append(existing_block)
block += 1
# Sort candidates by bottom, then top, left, and right.
candidate_blocks.sort(key=lambda block: (block.bottom_edge(), block.top_edge(), block.left_edge(), block.right_edge()))
did_join = False
# Capture the first candidate (closest) before mutating the list
first_candidate = candidate_blocks[0] if candidate_blocks else None
for existing_block in candidate_blocks:
if not did_join and first_candidate is not None and should_join_line_to_block(
ctx, existing_block, candidate_line, next_line, first_candidate
):
# Join: remove m from tree, extend with e_line, re-add.
try:
tree.remove(existing_block)
except ValueError:
pass
existing_block.add_line(candidate_line)
_set_add(tree, existing_block)
did_join = True
else:
# Doesn't take this line -- block is "closed", emit it.
clustered_blocks.append(existing_block)
try:
tree.remove(existing_block)
except ValueError:
pass
if not did_join:
_set_add(tree, seed_block)
# Drain remaining open blocks
for block in tree:
clustered_blocks.append(block)
# Post-process to split 2-line "heading+body" blocks when the first line
# matches section, abstract, or references keywords.
clustered_blocks = split_heading_body_blocks(clustered_blocks)
clustered_blocks.sort(key=reading_order_key)
return clustered_blocks
+326
View File
@@ -0,0 +1,326 @@
"""Line-to-block joining rules and the section-heading trie."""
from __future__ import annotations
from typing import Optional
import json
from pathlib import Path
from ..model import (
style_key,
magnitude_ratio,
left_aligned,
right_aligned,
center_aligned,
x_centers_close,
Rect,
last_span,
avg_char_width,
EMPTY_RECT,
left_edge_key,
reading_order_key,
numbering_kind,
Line,
case_signal,
last_line_of,
first_span_of,
letter_count,
dominant_style_of,
is_upper_dominant,
Block,
_max_nan_propagating,
)
from ..stats import DocStats, PageStats
from ..tokens import set_case_fold, TrieConfig, build_trie, tokenize_block
# Combined heading trie used to detect "first line is a section header" patterns
# when splitting two-line blocks.
_DICT_PATH = Path(__file__).parent.parent / "data" / "dictionaries.json"
_DICTS = json.loads(_DICT_PATH.read_text(encoding="utf-8"))
SECTION_HEADING_TRIE = build_trie(
list(_DICTS.get("section_keywords", []))
+ list(_DICTS.get("abstract_keywords", []))
+ list(_DICTS.get("references", [])),
set_case_fold(TrieConfig(), True),
)
# --------------------------------------------------------------------------- #
# Block-clustering context bundle #
# --------------------------------------------------------------------------- #
class BlockClusterContext:
"""Block-clustering context. Fields: j document statistics o page bbox g page statistics h lines to cluster v column rectangles """
__slots__ = ("tertiary_slot", "auxiliary_slot", "primary_slot", "secondary_slot", "state_slot")
def __init__(self, doc_stats: DocStats, page_bbox: Rect, page_stats: PageStats, lines: list, columns: list):
self.tertiary_slot = doc_stats
self.auxiliary_slot = page_bbox
self.primary_slot = page_stats
self.secondary_slot = lines
self.state_slot = columns
# --------------------------------------------------------------------------- #
# Should a line join an existing block? #
# --------------------------------------------------------------------------- #
def should_join_line_to_block(
block_cluster_ctx: BlockClusterContext,
other_block: Block,
candidate_line: Line,
previous_line: Optional[Line],
first_candidate_block: Block,
) -> bool:
"""Return True iff the candidate line should be appended to the current block."""
# -- Step 1: reject incompatible skew ----------
if abs(other_block.skew_frac() - candidate_line.skew_frac()) > 1:
return False
# -- Step 2: size + alignment gates ----------------------------------
font_size_delta = candidate_line.avg_font_size() - other_block.avg_font_size()
left_edges_aligned = left_aligned(other_block, candidate_line, 1)
both_edges_aligned = left_edges_aligned or (other_block.line_count() == 1 and left_aligned(other_block, candidate_line, 8 * avg_char_width(other_block.line())))
right_edges_aligned = right_aligned(other_block, candidate_line, 2)
both_edges_aligned = both_edges_aligned and right_edges_aligned
# m = min size-excess over page body; k = min size-excess over doc body
page_body_font_delta = min(candidate_line.avg_font_size() - block_cluster_ctx.primary_slot.primary_slot, other_block.avg_font_size() - block_cluster_ctx.primary_slot.primary_slot)
doc_body_font_delta = min(candidate_line.avg_font_size() - block_cluster_ctx.tertiary_slot.primary_slot, other_block.avg_font_size() - block_cluster_ctx.tertiary_slot.primary_slot)
block_last_span = last_span(last_line_of(other_block))
line_first_span = candidate_line.primary_slot[0]
if (
abs(font_size_delta) > page_body_font_delta
and abs(font_size_delta) > doc_body_font_delta - 2
and not (style_key(block_last_span) == style_key(line_first_span) and block_last_span.char_count() > 1 and line_first_span.char_count() > 1)
and (
font_size_delta > 2
or (font_size_delta > 1 and not both_edges_aligned)
or font_size_delta < -5
or (font_size_delta < -2 and candidate_line.char_count() >= 5)
or (font_size_delta < -1 and candidate_line.char_count() >= 20 and not both_edges_aligned)
)
):
return False
# -- Step 3: font / bold mismatch ------------------------------------
block_last_line = last_line_of(other_block)
width_ratio = magnitude_ratio(other_block.bbox_width(), candidate_line.bbox_width())
bold_mismatch = (block_last_span.primary_slot != line_first_span.primary_slot)
font_mismatch = (
block_last_span.font_name != line_first_span.font_name
and dominant_style_of(other_block) != style_key(line_first_span)
)
if font_mismatch or bold_mismatch:
if bold_mismatch and width_ratio > 2:
return False
if (block_last_line.char_stats.secondary_slot == 1 or block_last_line.char_stats.secondary_slot == 2) and (
candidate_line.char_stats.secondary_slot == 2 or width_ratio > 4
):
return False
if block_last_line.char_stats.tertiary_slot == 6 or other_block.bbox_width() > 1.5 * block_last_line.bbox_width():
return False
if other_block.bold_frac() > 0.9 and candidate_line.bold_frac() < 0.8 and width_ratio > 2:
return False
# -- Step 4: spatial gates -------------------------------------------
centers_aligned = center_aligned(other_block, candidate_line, 1)
if not centers_aligned:
vertical_gap = other_block.bottom_edge() - candidate_line.top_edge()
horizontal_offset = candidate_line.left_edge() - other_block.left_edge()
if (vertical_gap > -1 and horizontal_offset > 0.33 * other_block.bbox_width()) or horizontal_offset > 0.98 * other_block.bbox_width():
return False
if candidate_line.center_x() < other_block.left_edge():
return False
# -- Step 5: tolerance base ------------------------------------------
bottom_edge_gap = other_block.bottom_edge() - candidate_line.bottom_edge()
join_tolerance = (
_max_nan_propagating(1.3 * (other_block.top_edge() - other_block.bottom_edge()) / other_block.line_count(), block_cluster_ctx.primary_slot.tertiary_slot)
+ 1.3 * other_block.avg_font_size()
) / 2.0
# -- Step 6: case-flip "hanging indent" detector ---------------------
block_case_signal = case_signal(other_block.char_stats)
line_case_signal = case_signal(candidate_line.char_stats)
# Capture the old block-last span before comparing both sides of the case
# transition.
case_signal_flip = (
((block_case_signal == 1 and line_case_signal == -1) or (line_case_signal == 1 and block_case_signal == -1))
and letter_count(candidate_line.char_stats) >= 3
and (is_upper_dominant(other_block.char_stats) != is_upper_dominant(line_first_span.char_stats) or letter_count(line_first_span.char_stats) < 3)
and (is_upper_dominant(block_last_span.char_stats) != is_upper_dominant(candidate_line.char_stats) or letter_count(block_last_span.char_stats) < 3)
)
if (
not font_mismatch and not bold_mismatch and not case_signal_flip
and (width_ratio <= 1.2 or left_aligned(block_last_line, candidate_line, 0.1))
# Preserve the no-guard width-ratio edge case: a zero-width block still
# allows a positive-width last line to increase the join tolerance.
and (block_last_line.bbox_width() / other_block.bbox_width() > 0.9 if other_block.bbox_width() != 0 else block_last_line.bbox_width() > 0)
):
join_tolerance *= 1.3
if page_body_font_delta > 0.5 * block_cluster_ctx.primary_slot.primary_slot and not case_signal_flip:
join_tolerance *= 2
# -- Step 7: column alignment ----------------------------------------
column_rect = (block_cluster_ctx.state_slot[candidate_line.measure_slot] if (0 <= candidate_line.measure_slot < len(block_cluster_ctx.state_slot)) else None) or EMPTY_RECT
line_left_aligned_to_column = left_aligned(candidate_line, column_rect, 4.5)
line_right_aligned_to_column = right_aligned(candidate_line, column_rect, 4.5)
block_left_aligned_to_column = left_aligned(other_block, column_rect, 4.5)
block_right_aligned_to_column = right_aligned(other_block, column_rect, 4.5)
block_column_justified = (
block_left_aligned_to_column == block_right_aligned_to_column
and other_block.alignment_slot
and x_centers_close(block_cluster_ctx.auxiliary_slot, other_block)
)
line_column_centered = (
line_left_aligned_to_column == line_right_aligned_to_column
and (x_centers_close(block_cluster_ctx.auxiliary_slot, candidate_line) or (block_column_justified and centers_aligned))
)
# -- Step 8: alignment multipliers -----------------------------------
if (
block_column_justified and line_column_centered
and other_block.bbox_width() > 0.5 * candidate_line.bbox_width()
and (previous_line is None or candidate_line.bottom_edge() - previous_line.bottom_edge() >= bottom_edge_gap)
and not font_mismatch
):
join_tolerance *= 1.3
if previous_line is not None and (
(other_block.bold_frac() > previous_line.bold_frac() and candidate_line.bold_frac() > previous_line.bold_frac())
or (other_block.avg_font_size() > previous_line.bbox_height() + 1 and candidate_line.bbox_height() > previous_line.bbox_height() + 1)
):
join_tolerance = max(join_tolerance, candidate_line.bottom_edge() - previous_line.top_edge())
elif block_right_aligned_to_column and line_left_aligned_to_column:
join_tolerance *= 1.3 if other_block.line_count() <= 1 else 1.2
elif block_left_aligned_to_column and line_left_aligned_to_column:
join_tolerance *= 1.1
elif block_right_aligned_to_column:
if other_block.line_count() <= 1:
join_tolerance *= 1.1
if candidate_line.char_stats.secondary_slot == 3:
join_tolerance *= 1.1
if other_block.line_count() <= 1 and candidate_line.char_stats.secondary_slot == 3:
join_tolerance *= 1.1
if (
candidate_line.left_edge() > other_block.left_edge()
and candidate_line.left_edge() <= other_block.left_edge() + 0.1 * other_block.bbox_width()
and (other_block.line_count() <= 1 or left_aligned(candidate_line, block_last_line, 1))
):
join_tolerance *= 1.2
elif candidate_line.bbox_width() < 0.9 * block_last_line.bbox_width() and center_aligned(other_block, candidate_line, 1):
join_tolerance *= 1.1
if left_edges_aligned and candidate_line.bbox_width() < 0.5 * other_block.bbox_width() and other_block.char_stats.tertiary_slot != 6 and candidate_line.char_stats.tertiary_slot == 6:
join_tolerance *= 1.3
# -- Step 9: numbering pattern checks --------------------------------
block_numbering_kind = numbering_kind(other_block.line())
block_has_numbering = (
numbering_kind(other_block.line()) != 0
and first_span_of(other_block).bbox_height() >= 0.8 * other_block.avg_font_size()
)
block_starts_with_digit = block_has_numbering and block_numbering_kind == 1
line_numbering_kind = numbering_kind(candidate_line)
line_has_numbering = (
numbering_kind(candidate_line) != 0
and candidate_line.primary_slot[0].bbox_height() >= 0.8 * candidate_line.avg_font_size()
)
line_starts_with_digit = line_has_numbering and line_numbering_kind == 1
if block_starts_with_digit and not line_starts_with_digit and font_size_delta <= -0.5:
join_tolerance /= 2
elif (
(block_starts_with_digit and (bold_mismatch or font_size_delta <= -0.5))
or (line_starts_with_digit and (bold_mismatch or font_size_delta >= 0.5))
):
join_tolerance /= 1.5
elif block_starts_with_digit and candidate_line.left_edge() >= other_block.left_edge() and 0.9 * candidate_line.bbox_width() > other_block.bbox_width():
join_tolerance /= 1.5
elif block_has_numbering and candidate_line.left_edge() >= other_block.left_edge() and 0.9 * candidate_line.bbox_width() > other_block.bbox_width():
join_tolerance /= 1.3
elif (block_starts_with_digit and candidate_line.char_stats.secondary_slot != 3 or line_starts_with_digit) and font_mismatch:
join_tolerance /= 1.3
elif block_starts_with_digit and left_edges_aligned and candidate_line.char_stats.secondary_slot == 2:
join_tolerance /= 1.3
elif (
(block_has_numbering and (font_mismatch or bold_mismatch or font_size_delta <= -0.5 or (left_edges_aligned and candidate_line.char_stats.secondary_slot == 2)))
or (line_has_numbering and (font_mismatch or bold_mismatch or font_size_delta >= 0.5))
):
join_tolerance /= 1.1
if block_has_numbering and line_has_numbering:
join_tolerance /= 1.3
# -- Step 10: hanging-indent + neighbour patches ---------------------
block_first_letter = other_block.line().alignment_slot
if (
block_numbering_kind == 1
and line_numbering_kind != 1
and not left_edges_aligned
and block_first_letter is not None
and left_aligned(block_first_letter, candidate_line, 1)
):
join_tolerance *= 2
if case_signal_flip:
join_tolerance /= 1.1
if other_block.line_count() == 1 or not left_edges_aligned:
divisor = 3 if width_ratio > 3 else (1.5 if width_ratio > 1.5 else 1)
join_tolerance /= divisor
if (is_upper_dominant(other_block.char_stats) and block_has_numbering) or (is_upper_dominant(candidate_line.char_stats) and line_has_numbering):
join_tolerance /= 2
if font_mismatch or bold_mismatch:
join_tolerance /= 1.5
if other_block is not first_candidate_block and bottom_edge_gap > 1.1 * (first_candidate_block.bottom_edge() - candidate_line.bottom_edge()):
join_tolerance /= 2
return bottom_edge_gap <= join_tolerance
+131
View File
@@ -0,0 +1,131 @@
"""
Block classification for header/footer, watermark, boilerplate, TOC-page, and
reference-list marking. The module combines recurrence hashes, page-number
patterns, body-paragraph gates, cross-page geometry, and numeric-column
clustering. The dot-leader and page-number gates intentionally use Unicode
number properties so fullwidth and non-Latin digits are handled consistently.
"""
import json
import math
import regex as regex_module # Unicode \p{...} property classes
from pathlib import Path
from typing import Optional
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_round_half_up_to_int,
magnitude_ratio,
intervals_overlap,
y_overlaps,
center_aligned,
to_number,
last_span,
heading_score,
text_of_line,
Line,
last_line_of,
first_span_of,
is_word_category,
block_text,
deaccented_text,
letter_count,
dominant_style_of,
punct_count,
info_weight,
is_upper_dominant,
is_caps_heavy,
alignment_code,
Block,
)
from ..stats import style_key, DocStats, weighted_percentile, column_index_of, char_script_bucket
from ..tokens import (
is_trimmable_token,
token_numeric_value,
Token,
TokenView,
wrap_tokens,
enumerate_tokens,
jenkins_hash,
trie_prefix_match,
strip_trie_match,
strip_leading_if_in,
COMMA_CHARS,
strip_trailing_comma,
trim_trailing_punct,
set_case_fold,
TrieConfig,
build_trie,
LineTokenizer,
tokenize_block,
BuiltTrie,
trie_full_match,
is_char_token,
is_word_token,
)
from .keyword_tables import (
_DICT_PATH,
_DICTS,
_dict_trie,
COPYRIGHT_TRIE,
VOLUME_WORDS_TRIE,
TOC_TITLES_TRIE,
FIGURE_KEYWORDS_TRIE,
_TABLE_KEYWORDS_TRIE,
TABLE_KEYWORDS_TRIE,
_CHART_KEYWORDS_TRIE,
CHART_KEYWORDS_TRIE,
APPENDIX_SECTION_TRIE,
INTRODUCTION_SECTION_TRIE,
BOX_KEYWORD_TRIE,
KEYWORDS_SECTION_TRIE,
_BOILERPLATE_PHRASES_PATH,
BOILERPLATE_TRIE,
DOT_LEADER_ROW_RE,
PAGE_NUMBER_ONLY_RE,
_search_trie,
_normalize_text_key,
)
from .body_text import (
record_recurring_text,
is_body_paragraph,
span_style_text_key,
normalized_block_text,
_ROMAN_NUMERALS,
span_page_number,
longest_word_and_number,
)
from .header_footer import (
PageMarkState,
record_marked_block,
is_header_positioned,
has_adjacent_page_numbers,
mark_header_footer,
walk_from_page_edge,
find_cross_page_match,
HeaderFooterContext,
bounded_edit_distance,
detect_header_footer,
)
from .toc_boilerplate import (
mark_watermarks,
_institution_thesis_words,
INSTITUTION_THESIS_TRIE,
PROFESSOR_TITLES_TRIE,
is_boilerplate_block,
NumberColumnCluster,
extract_number_column,
pick_nearer_cluster,
detect_toc_range,
mark_toc_and_boilerplate,
)
__all__ = [
"is_body_paragraph", "record_recurring_text", "span_style_text_key", "normalized_block_text", "span_page_number", "longest_word_and_number", "record_marked_block", "PageMarkState", "is_header_positioned", "has_adjacent_page_numbers", "mark_header_footer", "walk_from_page_edge", "find_cross_page_match",
"HeaderFooterContext", "detect_header_footer", "mark_watermarks",
"is_boilerplate_block", "NumberColumnCluster", "extract_number_column", "pick_nearer_cluster", "detect_toc_range", "mark_toc_and_boilerplate",
"bounded_edit_distance",
"COPYRIGHT_TRIE", "VOLUME_WORDS_TRIE", "TOC_TITLES_TRIE", "FIGURE_KEYWORDS_TRIE", "TABLE_KEYWORDS_TRIE", "CHART_KEYWORDS_TRIE", "APPENDIX_SECTION_TRIE", "INTRODUCTION_SECTION_TRIE", "BOX_KEYWORD_TRIE", "KEYWORDS_SECTION_TRIE",
]
+209
View File
@@ -0,0 +1,209 @@
"""Body-paragraph classification and recurring-text recording."""
from __future__ import annotations
import math
from typing import Optional
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_round_half_up_to_int,
magnitude_ratio,
intervals_overlap,
y_overlaps,
center_aligned,
to_number,
last_span,
heading_score,
text_of_line,
Line,
last_line_of,
first_span_of,
is_word_category,
block_text,
deaccented_text,
letter_count,
dominant_style_of,
punct_count,
info_weight,
is_upper_dominant,
is_caps_heavy,
alignment_code,
Block,
)
from ..stats import style_key, DocStats, weighted_percentile, column_index_of, char_script_bucket
from ..tokens import (
is_trimmable_token,
token_numeric_value,
Token,
TokenView,
wrap_tokens,
enumerate_tokens,
jenkins_hash,
trie_prefix_match,
strip_trie_match,
strip_leading_if_in,
COMMA_CHARS,
strip_trailing_comma,
trim_trailing_punct,
set_case_fold,
TrieConfig,
build_trie,
LineTokenizer,
tokenize_block,
BuiltTrie,
trie_full_match,
is_char_token,
is_word_token,
)
from .keyword_tables import (
BOILERPLATE_TRIE,
PAGE_NUMBER_ONLY_RE,
_normalize_text_key,
)
# --------------------------------------------------------------------------- #
# Recurring-text histogram updater #
# --------------------------------------------------------------------------- #
def record_recurring_text(doc, other_text: str) -> None:
"""Increment the recurring-text histogram under the Jenkins lookup2 hash key. Empty normalized text is a valid key and must not be skipped."""
key = jenkins_hash(other_text)
doc.tertiary_slot[key] = doc.tertiary_slot.get(key, 0) + 1
# --------------------------------------------------------------------------- #
# Body-paragraph predicate #
# --------------------------------------------------------------------------- #
# The phrase gate is intentionally narrow. Broad Latin keyword matching
# over-rejects normal body paragraphs, for example sentences starting with
# "figure", and then lets figure captions be treated as headings.
def is_body_paragraph(doc_stats: DocStats, page, block: Block) -> bool:
"""Return whether ``block`` is a substantive body paragraph."""
if block.weighted_ratio_primary < 0.6:
return False
width = info_weight(block.char_stats)
lines = block.line_count()
sentence_punct = block.char_stats.primary_slot[6]
# The width-per-line ratio uses IEEE-style division. For a zero-line block,
# d/e is +inf
# (d>0) or NaN (d==0), so every ``d/e < k`` test is False and the block is
# NOT rejected here (it falls through to the char-count gate below, which
# rejects an empty block). This is not an early return.
dw_per_line = (
width / lines if lines != 0
else (math.inf if width > 0 else math.nan)
)
if (
dw_per_line < 15
or (lines >= 10 and dw_per_line < 20)
or (lines >= 10 and dw_per_line < 25 and sentence_punct < lines / 8)
or (lines >= 20 and dw_per_line < 40 and sentence_punct < lines / 20)
):
return False
block_width = block.bbox_width()
if lines >= 4:
short = 0
for state_item in block:
if state_item.bbox_width() < 0.75 * block_width and not state_item.primary_slot[0].state_slot.startswith("•"):
short += 1
if short >= lines / 2 and sentence_punct < lines / 8:
return False
if block_width < page.bounds.bbox_width() / 7:
return False
chars = block.char_count()
if chars < 40 or (lines >= 3 and alignment_code(block) == 3) or letter_count(block.char_stats) < 0.1 * chars:
return False
size = block.avg_font_size()
body_size = min(page.primary_slot.primary_slot, doc_stats.primary_slot)
min_value = min(doc_stats.primary_slot, max(page.bounds.bbox_height(), page.bounds.bbox_width()) / 60)
min_value = min(0.7 * min_value, min_value - 3)
# Boilerplate phrases are rejected as non-body even when they otherwise look
# paragraph-like. This keeps acknowledgement/copyright/proceedings language
# out of body-density calculations without broad keyword matching.
if size < body_size - 2 or size < min_value or trie_prefix_match(BOILERPLATE_TRIE, tokenize_block(block)):
return False
if chars >= 250 and lines >= 4:
return True
if block_width < page.bounds.bbox_width() / 5 or size < body_size - 0.5:
return False
if chars >= 100 and lines >= 2 and sentence_punct >= 2:
return True
if (chars >= 100 or block.char_stats.tertiary_slot == 6) and (
size >= page.primary_slot.primary_slot - 0.5 or size > doc_stats.primary_slot - 0.1
):
return first_span_of(block).font_name == page.primary_slot.state_slot or last_span(last_line_of(block)).font_name == page.primary_slot.state_slot
return False
# --------------------------------------------------------------------------- #
# Header/footer helper keys and predicates #
# --------------------------------------------------------------------------- #
def span_style_text_key(span) -> str:
"""Span style hash including text content: font name, rounded height, bold flag, lowercase text."""
# Use exact half-up integer rounding; Python f"{x:.0f}" uses half-even.
return f"{span.font_name} {_round_half_up_to_int(span.bbox_height())} {'B' if span.primary_slot else 'R'} {span.text.lower()}"
def normalized_block_text(block: Block) -> str:
"""block normalized-text hash."""
out = []
for token in tokenize_block(block):
out.append(_normalize_text_key(token.str.lower()))
return "".join(out)
# Roman numeral lookup used for page-number-like header/footer spans.
_ROMAN_NUMERALS = {
"I": 1, "II": 2, "III": 3, "IV": 4, "V": 5, "VI": 6, "VII": 7,
"VIII": 8, "IX": 9, "X": 10, "XI": 11, "XII": 12, "XIII": 13,
"XIV": 14, "XV": 15, "XVI": 16, "XVII": 17, "XVIII": 18, "XIX": 19, "XX": 20,
}
def span_page_number(span) -> Optional[int]:
"""Extract a page number from a span using a digit gate, then Roman numeral lookup."""
text = span.text
match = PAGE_NUMBER_ONLY_RE.match(text)
if match:
page_number = to_number(match.group(1))
if not math.isnan(page_number) and page_number > 0 and page_number < 1e6 and page_number == math.ceil(page_number):
return int(page_number)
return None
return _ROMAN_NUMERALS.get(text.upper())
def longest_word_and_number(block: Block) -> list[str]:
"""extract longest letter-word and longest digit-string. Returns a list of 0-2 strings: lowercased longest word (if >3 chars), then the longest digit-string (raw). """
longest_word: Optional[str] = None
longest_number: Optional[str] = None
for tok in tokenize_block(block):
if tok.type == 2:
if longest_word is None or len(tok.str) > len(longest_word):
longest_word = tok.str
elif tok.type == 1:
if longest_number is None or len(tok.str) > len(longest_number):
longest_number = tok.str
out: list[str] = []
if longest_word and len(longest_word) > 3:
out.append(_normalize_text_key(longest_word.lower()))
if longest_number:
out.append(longest_number)
return out
@@ -0,0 +1,481 @@
"""Header and footer detection via cross-page recurrence."""
from __future__ import annotations
import math
from typing import Optional
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_round_half_up_to_int,
magnitude_ratio,
intervals_overlap,
y_overlaps,
center_aligned,
to_number,
last_span,
heading_score,
text_of_line,
Line,
last_line_of,
first_span_of,
is_word_category,
block_text,
deaccented_text,
letter_count,
dominant_style_of,
punct_count,
info_weight,
is_upper_dominant,
is_caps_heavy,
alignment_code,
Block,
)
from ..stats import style_key, DocStats, weighted_percentile, column_index_of, char_script_bucket
from ..tokens import (
is_trimmable_token,
token_numeric_value,
Token,
TokenView,
wrap_tokens,
enumerate_tokens,
jenkins_hash,
trie_prefix_match,
strip_trie_match,
strip_leading_if_in,
COMMA_CHARS,
strip_trailing_comma,
trim_trailing_punct,
set_case_fold,
TrieConfig,
build_trie,
LineTokenizer,
tokenize_block,
BuiltTrie,
trie_full_match,
is_char_token,
is_word_token,
)
from .keyword_tables import (
COPYRIGHT_TRIE,
VOLUME_WORDS_TRIE,
FIGURE_KEYWORDS_TRIE,
TABLE_KEYWORDS_TRIE,
CHART_KEYWORDS_TRIE,
_search_trie,
)
from .body_text import (
record_recurring_text,
is_body_paragraph,
span_style_text_key,
normalized_block_text,
span_page_number,
longest_word_and_number,
)
class PageMarkState:
"""Per-page classification state: first classified index, max heading score, and classified character count."""
__slots__ = ("primary_slot", "secondary_slot", "tertiary_slot")
def __init__(self):
self.primary_slot = -1
self.secondary_slot = 0
self.tertiary_slot = 0
def record_marked_block(state: PageMarkState, idx: int, block: Block) -> None:
"""Update per-page state after classifying ``block``."""
state.primary_slot = idx
state.secondary_slot = max(state.secondary_slot, heading_score(block))
state.tertiary_slot += block.char_count()
def is_header_positioned(ctx, other_block: Block, candidate_block: Optional[Block]) -> bool:
"""Return whether a block is header-positioned relative to the reference block, with content-density gates."""
if candidate_block is None:
cond = True
elif ctx.primary_slot == 1:
cond = other_block.top_edge() > candidate_block.bottom_edge()
else:
cond = other_block.bottom_edge() < candidate_block.top_edge()
return cond and other_block.line_count() == 1 and info_weight(other_block.char_stats) >= 8 and letter_count(other_block.char_stats) >= 5 and other_block.char_stats.primary_slot[1] >= 1
def has_adjacent_page_numbers(ctx, page: int, candidate_number: int, reference_flag: bool) -> bool:
"""Return whether nearby pages show a strong ``n±1`` / ``n±2`` / ``n±4`` page-number pattern."""
page_index = page - 1
page_count = len(ctx.secondary_slot.primary_slot)
adjacent_one = (
(page_index - 1 >= 0 and (candidate_number - 1) in ctx.tertiary_slot[page_index - 1])
or (page_index + 1 < page_count and (candidate_number + 1) in ctx.tertiary_slot[page_index + 1])
)
adjacent_two = (
(page_index - 2 >= 0 and (candidate_number - 2) in ctx.tertiary_slot[page_index - 2])
or (page_index + 2 < page_count and (candidate_number + 2) in ctx.tertiary_slot[page_index + 2])
)
if not adjacent_one and not adjacent_two:
return False
if adjacent_one and adjacent_two:
return True
adjacent_four = (
(page_index - 4 >= 0 and (candidate_number - 4) in ctx.tertiary_slot[page_index - 4])
or (page_index + 4 < page_count and (candidate_number + 4) in ctx.tertiary_slot[page_index + 4])
)
if candidate_number > page / 2 - 30:
return adjacent_one or (not reference_flag and adjacent_two) or (adjacent_two and adjacent_four)
return bool(adjacent_two and adjacent_four)
def mark_header_footer(ctx, other_block: Block) -> None:
"""mark block as classified + bump ghost-text count."""
record_recurring_text(ctx.secondary_slot, deaccented_text(other_block))
other_block.type = ctx.primary_slot
def walk_from_page_edge(ctx, blocks: list[Block], callback) -> None:
"""direction-aware iteration. HEADER (g=1) walks blocks in normal order from top; FOOTER (g=2) walks in reverse from bottom. ``callback`` returns True to halt. """
if ctx.primary_slot == 1:
for page in blocks:
if callback(page):
break
else:
block_index = len(blocks) - 1
while block_index >= 0:
if callback(blocks[block_index]):
break
block_index -= 1
def find_cross_page_match(ctx, page, block: Block, text_key: str, ref: Block) -> Optional[Block]:
"""Find a matching block on a nearby page by exact normalized text, then by longest word/number pieces."""
entries = ctx.auxiliary_slot.get(text_key) or []
for entry in entries:
entry_page_index = entry["page_index"]
entry_block: Block = entry["block"]
if entry_page_index < page.page_index - 3:
continue
if entry_page_index == page.page_index:
continue
if entry_page_index > page.page_index + 3:
break
distance_sq = entry_block.left_edge() - block.left_edge()
left_delta = entry_block.top_edge() - block.top_edge()
right_delta = entry_block.right_edge() - block.right_edge()
bottom_delta = entry_block.bottom_edge() - block.bottom_edge()
distance_sq = distance_sq * distance_sq + left_delta * left_delta + right_delta * right_delta + bottom_delta * bottom_delta
size = page.primary_slot.primary_slot
if not (
distance_sq >= 100
or (distance_sq >= 1 and (
(page.page_index == 1 and heading_score(block) >= size + 0.5)
or (entry_page_index == 1 and heading_score(entry_block) >= size + 0.5)
))
):
return entry_block
if is_header_positioned(ctx, block, ref):
for key in longest_word_and_number(block):
map_value = ctx.measure_slot.get(key)
if map_value is None or len(map_value) < max(4, len(ctx.secondary_slot.primary_slot) / 4):
continue
target = heading_score(block)
for nearby_page_index in range(page.page_index - 2, page.page_index + 3):
if nearby_page_index == page.page_index:
continue
nearby_entry = map_value.get(nearby_page_index)
if nearby_entry is None:
continue
body_font_size = page.primary_slot.primary_slot
if (abs(target - heading_score(nearby_entry["block"])) > 1
or (page.page_index == 1 and target >= body_font_size + 0.5)
or (nearby_page_index == 1 and heading_score(nearby_entry["block"]) >= body_font_size + 0.5)):
continue
threshold = min(len(text_key), len(nearby_entry["text_key"])) / 5
if bounded_edit_distance(text_key, nearby_entry["text_key"], threshold) >= threshold:
continue
return nearby_entry["block"]
return None
# --------------------------------------------------------------------------- #
# Header/footer detection context #
# --------------------------------------------------------------------------- #
class HeaderFooterContext:
"""Per-pass header/footer state."""
__slots__ = ("secondary_slot", "primary_slot", "previous_slot", "option_slot", "tertiary_slot", "auxiliary_slot", "measure_slot", "state_slot")
def __init__(self, doc, candidate_number: int):
self.secondary_slot = doc
self.primary_slot = candidate_number
self.previous_slot = "HEADER" if candidate_number == 1 else "FOOTER"
self.option_slot: dict[str, int] = {} # span style/text key -> page count
self.tertiary_slot: list[set[int]] = [] # per-page page-number set
self.auxiliary_slot: dict[str, list[dict]] = {} # normalized text key -> location/block records
self.measure_slot: dict[str, dict[int, dict]] = {} # word/number key -> page -> text/block record
self.state_slot: list[list[Block]] = [] # per-page candidate blocks
# --------------------------------------------------------------------------- #
# Bounded edit distance for fuzzy block-key comparison. #
# --------------------------------------------------------------------------- #
def bounded_edit_distance(text: str, other_text: str, candidate_item: float) -> float:
"""Bounded banded edit distance. Returns the limit when the strings differ by more than that many edits; otherwise returns the exact Levenshtein distance."""
candidate_item = max(len(text), len(other_text)) if candidate_item <= 0 else math.ceil(candidate_item)
if len(text) <= 0:
return min(len(other_text), candidate_item)
if len(other_text) <= 0:
return min(len(text), candidate_item)
if len(text) < len(other_text):
text, other_text = other_text, text # a is the longer string (columns)
if len(text) - len(other_text) >= candidate_item:
return candidate_item
reference_item = 0 # leftmost band column
entry_item = 0 # rightmost band column
score_value = [0] * (len(text) + 1) # previous row
group_value = [0] * (len(text) + 1) # current row
for state_item in range(len(text) + 1): # seed row 0, but only out to column c
score_value[state_item] = state_item
if state_item > candidate_item:
break
entry_item = state_item
for state_item in range(1, len(other_text) + 1):
compare_char = other_text[state_item - 1]
key_value = len(text) # leftmost column kept < c this row
measure_item = 0 # rightmost column kept < c this row
for line_value in range(reference_item, min(entry_item + 1, len(text)) + 1):
if line_value == reference_item:
group_value[line_value] = 1 + score_value[line_value]
elif text[line_value - 1] == compare_char:
group_value[line_value] = score_value[line_value - 1]
else:
group_value[line_value] = 1 + min(group_value[line_value - 1], score_value[line_value - 1])
if line_value <= entry_item:
group_value[line_value] = min(group_value[line_value], 1 + score_value[line_value])
if group_value[line_value] < candidate_item:
key_value = min(key_value, line_value)
measure_item = line_value
if key_value > measure_item: # whole band reached c -> distance >= c
return candidate_item
score_value, group_value = group_value, score_value
reference_item = key_value
entry_item = measure_item
return min(score_value[entry_item] + len(text) - entry_item, candidate_item)
# --------------------------------------------------------------------------- #
# Header / footer detection #
# --------------------------------------------------------------------------- #
def detect_header_footer(ctx: HeaderFooterContext) -> None:
"""Run the three-pass header/footer detector."""
# ----- Pass 1: per-page candidate collection -----------------------
for page in ctx.secondary_slot.primary_slot:
seen_style_keys: set[str] = set()
page_numbers: set[int] = set()
ctx.tertiary_slot.append(page_numbers)
page_candidates: list[Block] = []
ctx.state_slot.append(page_candidates)
first_substantive_ref: list[Optional[Block]] = [None] # closure-friendly
def walk_cb(block: Block) -> bool:
if block.skew_frac() >= 1 or block.area() <= 0:
return False
# Page height should be positive. If a degenerate page appears, keep
# IEEE-style Infinity/NaN behavior so the comparisons below stay inert.
den = page.bounds.bbox_height()
num = block.top_edge() if ctx.primary_slot == 1 else block.bottom_edge()
relative = (num / den) if den else (math.copysign(math.inf, num) if num else math.nan)
if (ctx.primary_slot == 1 and relative < 0.8) or (ctx.primary_slot == 2 and relative > 0.2):
pass_value = False
else:
tokens = tokenize_block(block)
if _search_trie(COPYRIGHT_TRIE, tokens):
pass_value = True
elif (
block.line_count() >= 3
or info_weight(block.char_stats) * (1 + block.bold_frac()) >= 200
or trie_prefix_match(FIGURE_KEYWORDS_TRIE, tokens)
or trie_prefix_match(CHART_KEYWORDS_TRIE, tokens)
or trie_prefix_match(TABLE_KEYWORDS_TRIE, tokens)
):
pass_value = False
else:
pass_value = True
if not pass_value:
return True
if letter_count(block.char_stats) >= 5 and first_substantive_ref[0] is None:
first_substantive_ref[0] = block
ref = first_substantive_ref[0]
if block.type == 0 and block.char_count() > 0:
page_candidates.append(block)
if letter_count(block.char_stats) >= 5:
text_key = normalized_block_text(block)
item_list = ctx.auxiliary_slot.get(text_key)
if item_list is None:
item_list = []
ctx.auxiliary_slot[text_key] = item_list
item_list.append({"page_index": page.page_index, "block": block})
if is_header_positioned(ctx, block, ref):
for key in longest_word_and_number(block):
inner = ctx.measure_slot.get(key)
if inner is None:
inner = {}
ctx.measure_slot[key] = inner
if page.page_index not in inner:
inner[page.page_index] = {"text_key": text_key, "block": block}
for line in block:
for span in line:
if span.char_count() <= 0:
continue
ok = span_style_text_key(span)
if block.char_count() >= 4 and ok not in seen_style_keys:
ctx.option_slot[ok] = ctx.option_slot.get(ok, 0) + 1
seen_style_keys.add(ok)
detected_page_number = span_page_number(span)
if detected_page_number is not None:
page_numbers.add(detected_page_number)
return False
walk_from_page_edge(ctx, page.output_slot, walk_cb)
# ----- Pass 2: per-page rejection sweep ----------------------------
text_counts: dict[str, int] = {}
samples: list[tuple[float, float]] = []
for page in ctx.secondary_slot.primary_slot:
candidates = ctx.state_slot[page.page_index - 1]
state = PageMarkState()
seen_page_number = False
first_substantive: list[Optional[Block]] = [None]
for candidate_index in range(len(candidates)):
candidate_block = candidates[candidate_index]
if candidate_block.char_count() <= 0:
continue
if candidate_block.type == ctx.primary_slot:
record_marked_block(state, candidate_index, candidate_block)
continue
if candidate_block.type != 0:
continue
candidate_tokens = tokenize_block(candidate_block)
# Copyright terms must appear at the start of the block, not merely
# anywhere inside it.
if candidate_tokens.length < 10 and trie_prefix_match(COPYRIGHT_TRIE, candidate_tokens):
mark_header_footer(ctx, candidate_block)
record_marked_block(state, candidate_index, candidate_block)
continue
if letter_count(candidate_block.char_stats) >= 5:
if first_substantive[0] is None:
first_substantive[0] = candidate_block
pk_hash = normalized_block_text(candidate_block)
match = find_cross_page_match(ctx, page, candidate_block, pk_hash, first_substantive[0])
if match is not None:
mark_header_footer(ctx, candidate_block)
record_marked_block(state, candidate_index, candidate_block)
other_unmarked = match.type != ctx.primary_slot
if other_unmarked:
mark_header_footer(ctx, match)
if len(pk_hash) >= 5:
previous_count = text_counts.get(pk_hash, 0)
text_counts[pk_hash] = 2 if (previous_count or other_unmarked) else 1
continue
style_threshold = max(2.0, min(len(ctx.secondary_slot.primary_slot) / 3.0, 5.0))
chars = 0
for candidate_line in candidate_block:
for candidate_span in candidate_line:
if candidate_span.char_count() <= 0:
continue
style_hash = span_style_text_key(candidate_span)
if ctx.option_slot.get(style_hash, 0) >= style_threshold:
chars += candidate_span.char_count()
continue
page_number = span_page_number(candidate_span)
if page_number is not None and has_adjacent_page_numbers(ctx, page.page_index, page_number, seen_page_number):
seen_page_number = True
chars += candidate_span.char_count()
if chars >= candidate_block.char_count():
mark_header_footer(ctx, candidate_block)
record_marked_block(state, candidate_index, candidate_block)
pk_again = normalized_block_text(candidate_block)
if len(pk_again) >= 5:
text_counts[pk_again] = text_counts.get(pk_again, 0) + 1
if state.primary_slot < 0:
continue
first_block = candidates[state.primary_slot]
samples.append((first_block.center_y(), float(state.tertiary_slot)))
# Also classify earlier non-confirmed blocks
for index in range(state.primary_slot):
block = candidates[index]
if block.type == ctx.primary_slot:
continue
if ctx.primary_slot == 1 and block.bottom_edge() < first_block.bottom_edge():
continue
if block.bbox_width() >= page.bounds.bbox_width() / 2:
continue
if is_body_paragraph(ctx.secondary_slot.secondary_slot, page, block):
continue
if letter_count(block.char_stats) > 0 and heading_score(block) >= state.secondary_slot + 1:
continue
block.type = ctx.primary_slot
record_recurring_text(ctx.secondary_slot, deaccented_text(block))
# ----- Pass 3: cutoff line + top-3 text-hash sweep -----------------
if len(samples) < len(ctx.secondary_slot.primary_slot) / 20:
return
cutoff = weighted_percentile(samples, 20 if ctx.primary_slot == 1 else 80)
top: list[tuple[str, int]] = []
for text_hash, count in text_counts.items():
if count < len(ctx.secondary_slot.primary_slot) / 20:
continue
top.append((text_hash, count))
if not top:
return
top.sort(key=lambda item_pair: -item_pair[1])
if len(top) > 3:
top = top[:3]
for page in ctx.secondary_slot.primary_slot:
for recurring_block in ctx.state_slot[page.page_index - 1]:
if ctx.primary_slot == 1 and recurring_block.top_edge() < cutoff:
break
if ctx.primary_slot == 2 and recurring_block.bottom_edge() > cutoff:
break
if recurring_block.type != 0:
continue
recurring_tokens = tokenize_block(recurring_block)
stripped = _search_trie(VOLUME_WORDS_TRIE, recurring_tokens)
if stripped is not None:
# ``stripped.end`` is absolute in the forward token view, so this
# drops the matched volume phrase and keeps the tail.
tail = recurring_tokens.slice(stripped.end)
head_tok = tail.token_at(0) if tail.length > 0 else None
if head_tok is not None and head_tok.type == 1:
recurring_block.type = ctx.primary_slot
record_recurring_text(ctx.secondary_slot, deaccented_text(recurring_block))
# Deliberately fall through: the same block can also match the
# top recurring-text sweep below.
if letter_count(recurring_block.char_stats) < 5:
continue
if page.page_index <= 1 and heading_score(recurring_block) > ctx.secondary_slot.secondary_slot.primary_slot + 1:
continue
text_key = normalized_block_text(recurring_block)
for text_hash, _ in top:
threshold = min(len(text_key), len(text_hash)) / 2.0
if bounded_edit_distance(text_key, text_hash, threshold) >= threshold:
continue
# No break: every sufficiently similar recurring key contributes
# to the recurring-text histogram.
recurring_block.type = ctx.primary_slot
record_recurring_text(ctx.secondary_slot, deaccented_text(recurring_block))
@@ -0,0 +1,113 @@
"""Dictionary-backed keyword tries and shared regexes."""
from __future__ import annotations
import json
import regex as regex_module # Unicode \p{...} property classes
from pathlib import Path
from typing import Optional
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_round_half_up_to_int,
magnitude_ratio,
intervals_overlap,
y_overlaps,
center_aligned,
to_number,
last_span,
heading_score,
text_of_line,
Line,
last_line_of,
first_span_of,
is_word_category,
block_text,
deaccented_text,
letter_count,
dominant_style_of,
punct_count,
info_weight,
is_upper_dominant,
is_caps_heavy,
alignment_code,
Block,
)
from ..tokens import (
is_trimmable_token,
token_numeric_value,
Token,
TokenView,
wrap_tokens,
enumerate_tokens,
jenkins_hash,
trie_prefix_match,
strip_trie_match,
strip_leading_if_in,
COMMA_CHARS,
strip_trailing_comma,
trim_trailing_punct,
set_case_fold,
TrieConfig,
build_trie,
LineTokenizer,
tokenize_block,
BuiltTrie,
trie_full_match,
is_char_token,
is_word_token,
)
# --------------------------------------------------------------------------- #
# Load dictionaries (built into tries on first use) #
# --------------------------------------------------------------------------- #
_DICT_PATH = Path(__file__).parent.parent / "data" / "dictionaries.json"
_DICTS = json.loads(_DICT_PATH.read_text(encoding="utf-8"))
def _dict_trie(key: str) -> BuiltTrie:
"""Build a case-folded trie from a dictionary entry."""
return build_trie(_DICTS.get(key, []), set_case_fold(TrieConfig(), True))
COPYRIGHT_TRIE = build_trie(["Copyright", "©"], set_case_fold(TrieConfig(), True)) # inline list
VOLUME_WORDS_TRIE = _dict_trie("volume_words")
TOC_TITLES_TRIE = _dict_trie("toc_titles")
FIGURE_KEYWORDS_TRIE = _dict_trie("ai_section_keywords")
_TABLE_KEYWORDS_TRIE = _dict_trie("table_keywords")
TABLE_KEYWORDS_TRIE = _TABLE_KEYWORDS_TRIE
_CHART_KEYWORDS_TRIE = _dict_trie("chart_keywords")
CHART_KEYWORDS_TRIE = _CHART_KEYWORDS_TRIE
APPENDIX_SECTION_TRIE = _dict_trie("appendices_dict")
INTRODUCTION_SECTION_TRIE = _dict_trie("introduction_dict")
BOX_KEYWORD_TRIE = build_trie(["box"], set_case_fold(TrieConfig(), True)) # inline list
KEYWORDS_SECTION_TRIE = _dict_trie("keywords_dict")
# Multilingual boilerplate phrase trie: publisher and proceeding headers plus
# stock acknowledgement openers such as "First of all I would like to thank".
# Used by the body-paragraph gate to reject boilerplate as non-body.
# Phrase list stored as a data asset.
_BOILERPLATE_PHRASES_PATH = Path(__file__).parent.parent / "data" / "boilerplate_phrases.json"
BOILERPLATE_TRIE = build_trie(json.loads(_BOILERPLATE_PHRASES_PATH.read_text(encoding="utf-8")), set_case_fold(TrieConfig(), True))
# Regular expressions for the dot-leader and page-number gates (Unicode \p{Number} -> ``regex`` module).
# Leading class is ASCII 1-9 + fullwidth 1-9 (U+FF11-FF19); it must NOT admit
# fullwidth zero U+FF10, so it is [1-91-9], not [1-90-9].
DOT_LEADER_ROW_RE = regex_module.compile(r"([.][" + _UNICODE_WHITESPACE_CLASS + r"]*){5,}[" + _UNICODE_WHITESPACE_CLASS + r"]*[1-91-9]\p{Number}*\Z")
PAGE_NUMBER_ONLY_RE = regex_module.compile(r"^[ |]*([1-91-9]\p{Number}*)[ |]*\Z")
def _search_trie(trie: BuiltTrie, tokens) -> Optional[TokenView]:
"""Return the shortest earliest Aho-Corasick trie match for ``tokens``."""
from ..tokens import aho_corasick_tokens as _real_bh
return _real_bh(trie, tokens)
def _normalize_text_key(text: str) -> str:
"""Strip diacritics only; callers lowercase first when a case-folded key is needed."""
return _strip_diacritics(text)
@@ -0,0 +1,375 @@
"""Watermark, boilerplate, and TOC-range detection."""
from __future__ import annotations
import math
from typing import Optional
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_round_half_up_to_int,
magnitude_ratio,
intervals_overlap,
y_overlaps,
center_aligned,
to_number,
last_span,
heading_score,
text_of_line,
Line,
last_line_of,
first_span_of,
is_word_category,
block_text,
deaccented_text,
letter_count,
dominant_style_of,
punct_count,
info_weight,
is_upper_dominant,
is_caps_heavy,
alignment_code,
Block,
)
from ..tokens import (
is_trimmable_token,
token_numeric_value,
Token,
TokenView,
wrap_tokens,
enumerate_tokens,
jenkins_hash,
trie_prefix_match,
strip_trie_match,
strip_leading_if_in,
COMMA_CHARS,
strip_trailing_comma,
trim_trailing_punct,
set_case_fold,
TrieConfig,
build_trie,
LineTokenizer,
tokenize_block,
BuiltTrie,
trie_full_match,
is_char_token,
is_word_token,
)
from .keyword_tables import (
_DICTS,
_dict_trie,
TOC_TITLES_TRIE,
DOT_LEADER_ROW_RE,
_search_trie,
)
from .body_text import (
is_body_paragraph,
normalized_block_text,
)
# --------------------------------------------------------------------------- #
# Side-rail watermark detector #
# --------------------------------------------------------------------------- #
def mark_watermarks(doc) -> None:
"""Bucket skewed side-rail blocks by normalized text; recurring groups are marked as watermarks."""
buckets: dict[str, list[Block]] = {}
for page in doc.primary_slot:
for block in page.output_slot:
if block.skew_frac() < 1:
continue
if letter_count(block.char_stats) < 5:
continue
# Skip blocks in the central 80% of the page width.
horizontal_offset = block.center_x()
page_width = page.bounds.bbox_width()
if 0.1 * page_width < horizontal_offset < 0.9 * page_width:
continue
# No empty-string guard: empty normalized-text keys bucket together.
key = normalized_block_text(block)
buckets.setdefault(key, []).append(block)
for group in buckets.values():
if len(group) < 3:
continue
for block in group:
block.type = 12
# --------------------------------------------------------------------------- #
# Boilerplate block predicate #
# --------------------------------------------------------------------------- #
# Institution/thesis trie combines institution words with thesis-specific terms.
_institution_thesis_words = list(_DICTS.get("institution_words", [])) + list(_DICTS.get("nk_thesis_words", []))
INSTITUTION_THESIS_TRIE = build_trie(_institution_thesis_words, set_case_fold(TrieConfig(), True))
PROFESSOR_TITLES_TRIE = _dict_trie("professor_titles")
def is_boilerplate_block(block: Block) -> bool:
"""line/block looks like boilerplate (committee members, author affiliations, journal volume info etc.)."""
tokens = tokenize_block(block)
if info_weight(block.char_stats) >= 200 or tokens.length >= 100:
return False
if _search_trie(INSTITUTION_THESIS_TRIE, tokens):
return True
# Strip leading lines that match professor/title boilerplate.
while tokens.length > 0:
# Count tokens belonging to the first token's line and strip that line.
first_line = tokens.token_at(0).line() if tokens.token_at(0) else None
if first_line is None:
break
line_end = 0
while line_end < tokens.length:
tok = tokens.token_at(line_end)
if tok is None or tok.line() is not first_line:
break
line_end += 1
if not trie_prefix_match(PROFESSOR_TITLES_TRIE, tokens.slice(0, line_end)):
return False
tokens = tokens.slice(line_end)
return True
# --------------------------------------------------------------------------- #
# TOC-page detection chain #
# --------------------------------------------------------------------------- #
class NumberColumnCluster:
"""numeric-leading-token cluster."""
__slots__ = ("anchor_x", "width", "secondary_slot", "primary_slot", "length", "tertiary_slot")
def __init__(self, anchor_x_value: float, width: float, reference_number: int, next_number: int, length: int, limit_flag: bool):
self.anchor_x = anchor_x_value # anchor x-position
self.width = width # cluster typical width
self.secondary_slot = reference_number # first value seen
self.primary_slot = next_number # last value seen
self.length = length
self.tertiary_slot = limit_flag # is increasing
def extract_number_column(block) -> Optional[NumberColumnCluster]:
"""Extract a numeric-leading cluster if block lines form an increasing page-number sequence."""
column = 0
last_number = 0
sequence_length = 0
for line in block:
line_number = to_number(text_of_line(line))
if math.isnan(line_number):
return None
if not (line_number > 0 and line_number < 1e6 and line_number == math.ceil(line_number)) or line_number >= 1e4 or last_number > line_number:
return NumberColumnCluster(block.center_x(), block.bbox_width(), column, last_number, sequence_length, False)
if column <= 0:
column = int(line_number)
last_number = int(line_number)
sequence_length += 1
return NumberColumnCluster(block.center_x(), block.bbox_width(), column, last_number, sequence_length, True)
def pick_nearer_cluster(cluster: NumberColumnCluster, other_cluster: Optional[NumberColumnCluster], other: Optional[NumberColumnCluster]) -> Optional[NumberColumnCluster]:
"""Pick the closer neighbor cluster within the current cluster width."""
distance = (cluster.anchor_x - other_cluster.anchor_x) if other_cluster is not None else math.inf
candidate_distance = (other.anchor_x - cluster.anchor_x) if other is not None else math.inf
if distance > cluster.width and candidate_distance > cluster.width:
return None
return other_cluster if distance < candidate_distance else other
def detect_toc_range(doc, page, index) -> Optional[dict]:
"""Detect a TOC-like block range within ``page``. The detector combines dot-leader rows, blocks ending in dot-leader page numbers, contents-like titles, and same-x-range numeric clusters. Returns a ``{start_index, end_index}`` range or ``None``."""
blocks = page.output_slot
lines = 0
dot_leader_blocks = 0
weight = 0.0
last_multiline = -1
contents = -1
pre_contents = -1
last_toc = -1
seen_body = False
body_stop_y = page.bounds.top_edge()
is_last_page = (index == page.page_index - 1) if isinstance(index, int) and index >= 0 else False
clusters: list[NumberColumnCluster] = [] # sorted by anchor_x
def _add_cluster(cluster: NumberColumnCluster) -> None:
# Sorted-set semantics: an equal anchor_x is a no-op.
import bisect
keys = [existing_cluster.anchor_x for existing_cluster in clusters]
insert_index = bisect.bisect_left(keys, cluster.anchor_x)
if insert_index < len(clusters) and clusters[insert_index].anchor_x == cluster.anchor_x:
return
clusters.insert(insert_index, cluster)
def _next_number_column_cluster(cluster: NumberColumnCluster) -> Optional[NumberColumnCluster]:
# Non-strict successor: an equal anchor_x entry is returned.
import bisect
keys = [column.anchor_x for column in clusters]
cluster_index = bisect.bisect_left(keys, cluster.anchor_x)
return clusters[cluster_index] if cluster_index < len(clusters) else None
def _prev_number_column_cluster(cluster: NumberColumnCluster) -> Optional[NumberColumnCluster]:
# Non-strict predecessor: an equal anchor_x entry is returned.
import bisect
keys = [column.anchor_x for column in clusters]
cluster_index = bisect.bisect_right(keys, cluster.anchor_x)
return clusters[cluster_index - 1] if cluster_index > 0 else None
def _remove(cluster: NumberColumnCluster) -> None:
try:
clusters.remove(cluster)
except ValueError:
pass
for codepoint, block in enumerate(blocks):
is_toc = False
# Count dot-leader rows across all lines in the block.
for line in block:
if DOT_LEADER_ROW_RE.search(text_of_line(line)):
is_toc = True
if last_toc >= 0:
last_toc = codepoint
else:
lines += 1
if lines >= 5 or (lines >= 3 and is_last_page):
last_toc = codepoint
# A dot-leader on the last line also starts or extends the TOC range.
last_line = block.primary_slot[-1] if block.primary_slot else None
if last_line is not None and DOT_LEADER_ROW_RE.search(text_of_line(last_line)):
is_toc = True
if last_toc >= 0:
last_toc = codepoint
continue
else:
dot_leader_blocks += 1
weight += info_weight(block.char_stats)
if is_last_page and dot_leader_blocks >= 2 and weight >= 0.8 * page.primary_slot.secondary_slot:
last_toc = codepoint
# Body block tracking
if not is_toc and is_body_paragraph(doc.secondary_slot, page, block):
seen_body = True
body_stop_y = min(body_stop_y, block.bottom_edge())
if block.line_count() > 1 and not is_toc:
last_multiline = codepoint
# "Contents"-like title must consume the whole block, not just a prefix.
if contents < 0 and block.line_count() <= 1 and trie_full_match(TOC_TITLES_TRIE, tokenize_block(block)):
contents = codepoint
pre_contents = last_multiline
if not seen_body and block.top_edge() > 3 * page.bounds.bbox_height() / 4:
return {"start_index": pre_contents + 1, "end_index": len(blocks) - 1}
# Numeric-column clustering on unclassified blocks below the body line.
if block.right_edge() < page.bounds.center_x():
continue
if block.top_edge() > body_stop_y:
continue
# Extract even from an empty-looking block; the extractor decides whether
# a usable numeric sequence exists.
cluster = extract_number_column(block)
if cluster is not None:
successor = _next_number_column_cluster(cluster)
predecessor = _prev_number_column_cluster(cluster)
picked = pick_nearer_cluster(cluster, predecessor, successor)
if picked is not None:
_remove(picked)
new_value = NumberColumnCluster(
picked.anchor_x,
picked.width,
picked.secondary_slot,
cluster.primary_slot,
picked.length + cluster.length,
picked.tertiary_slot and cluster.tertiary_slot and picked.primary_slot <= cluster.secondary_slot,
)
else:
new_value = cluster
_add_cluster(new_value)
if (new_value.tertiary_slot
and (new_value.length >= 10 or (new_value.length >= 5 and is_last_page))
and new_value.primary_slot - new_value.secondary_slot > 0.01 * new_value.primary_slot):
last_toc = codepoint
if last_toc < 0:
return None
return {"start_index": pre_contents + 1 if pre_contents >= 0 else 0, "end_index": last_toc}
# --------------------------------------------------------------------------- #
# TOC pages, references lists, and figure/table captions #
# --------------------------------------------------------------------------- #
def mark_toc_and_boilerplate(doc) -> None:
"""Mark TOC blocks as type=9 and captions/boilerplate as type=12."""
previous_toc_page = -math.inf
for page in doc.primary_slot:
if (
page.page_index - 1 >= len(doc.primary_slot) / 2
and page.primary_slot.secondary_slot >= 0.9 * doc.secondary_slot.secondary_slot
):
continue
# "Most-boilerplate" check for front-matter pages.
if (
page.page_index > 1
and page.page_index < 50
and page.primary_slot.secondary_slot < max(200, min(0.75 * doc.secondary_slot.secondary_slot, 1000))
):
total = 0.0
body_paragraph = 0.0
body_paragraph_lines = 0
for line_or_block in page.output_slot:
if line_or_block.type != 0 or line_or_block.skew_frac() >= 1:
continue
width_value = info_weight(line_or_block.char_stats) * heading_score(line_or_block)
total += width_value
if is_boilerplate_block(line_or_block):
body_paragraph += width_value
body_paragraph_lines += 1
if body_paragraph >= 0.8 * total and body_paragraph_lines >= 3:
for block in page.output_slot:
block.type = 12
continue
toc = detect_toc_range(doc, page, previous_toc_page)
if toc is None:
continue
previous_toc_page = page.page_index
start = toc["start_index"]
end = toc["end_index"]
blocks = page.output_slot
# Track centered-block count, weighted font sum, total weight, and the
# running bottom edge used by walk-forward break conditions.
centered_flag = 0
width_flag = 0.0
width = 0.0
walk_break_y = page.bounds.top_edge()
for idx in range(start, len(blocks)):
block = blocks[idx]
score = heading_score(block)
if idx <= end:
if block.isolated_centered:
centered_flag += 1
heading_weight = info_weight(block.char_stats)
width_flag += block.avg_font_size() * heading_weight
width += heading_weight
walk_break_y = min(walk_break_y, block.bottom_edge())
block.type = 9
continue
# Walk forward with four break conditions: prominent heading, centered
# block, dense body text, or a large vertical gap to a prominent block.
if block.skew_frac() < 1 and score > doc.secondary_slot.primary_slot + 4 and score > page.primary_slot.primary_slot + 4:
break
if centered_flag <= 1 and block.isolated_centered:
break
if info_weight(block.char_stats) > 300 and block.char_stats.primary_slot[6] > 2 and block.weighted_ratio_secondary > 0.5:
break
if width > 0:
average = width_flag / width
if walk_break_y - block.top_edge() > average and block.skew_frac() < 1 and score > average + 1.5:
break
walk_break_y = min(walk_break_y, block.bottom_edge())
block.type = 9
+68
View File
@@ -0,0 +1,68 @@
"""Line clustering pipeline.
The initial pass walks spans in document order and groups them into lines using
an in-line continuation test, while also collapsing overstrike duplicates
(artificial-bold rendering where the same glyph is painted twice). The merge
pass inserts lines into a sorted structure keyed by top-desc reading order,
looks up predecessor/successor neighbors, and either merges the new line into a
neighbor or keeps it separate. Neighbor lookup is inclusive of an exact
reading-order key match, so the successor uses ``bisect_left`` and the
predecessor uses ``bisect_right - 1``.
"""
import re
from dataclasses import dataclass, field
from typing import Optional
from sortedcontainers import SortedKeyList
from ..model import (
_UNICODE_WHITESPACE_CLASS,
avg_char_width2,
Span,
magnitude_ratio,
same_x_extent,
same_y_extent,
append_span,
last_span,
avg_char_width,
raw_text_of_line,
text_of_line,
reading_order_key,
left_edge_key,
numbering_kind,
Line,
letter_count,
is_upper_dominant,
)
from .merge_rules import (
TRAILING_DOT_LEADER_RE,
span_continues_line,
vertical_distance_in_line_heights,
pick_closer_neighbor,
should_merge_lines,
)
from .build import (
_skip_mark_only,
build_initial_lines,
_is_label_stack,
LinesContainer,
_set_add,
cluster_lines,
)
# --------------------------------------------------------------------------- #
# Combined helper #
# --------------------------------------------------------------------------- #
__all__ = [
"span_continues_line",
"vertical_distance_in_line_heights",
"pick_closer_neighbor",
"should_merge_lines",
"build_initial_lines",
"cluster_lines",
"TRAILING_DOT_LEADER_RE",
]
+211
View File
@@ -0,0 +1,211 @@
"""Builds initial lines and clusters them into merged lines."""
from __future__ import annotations
from dataclasses import dataclass, field
from typing import Optional
from sortedcontainers import SortedKeyList
from ..model import (
_UNICODE_WHITESPACE_CLASS,
avg_char_width2,
Span,
magnitude_ratio,
same_x_extent,
same_y_extent,
append_span,
last_span,
avg_char_width,
raw_text_of_line,
text_of_line,
reading_order_key,
left_edge_key,
numbering_kind,
Line,
letter_count,
is_upper_dominant,
)
from .merge_rules import (
span_continues_line,
pick_closer_neighbor,
should_merge_lines,
)
# --------------------------------------------------------------------------- #
# Initial line builder.
# --------------------------------------------------------------------------- #
def _skip_mark_only(span: Span, page_area: float) -> bool:
"""Return True for mark-heavy tiny glyphs whose area is below one part per million of the page area."""
return (span.char_count() - span.char_stats.primary_slot[5]) > 1 and span.area() < page_area * 1e-6
def build_initial_lines(spans: list[Span], page_bbox) -> list[Line]:
"""Build initial lines from flat spans. Returns the list of initial lines. """
line: list[Line] = []
pending_line = Line()
pending_span: Optional[Span] = None
page_area = page_bbox.area()
for span in spans:
if span.text == "" or _skip_mark_only(span, page_area):
continue
if pending_span is not None:
# Overstrike duplicate detection: same trimmed text, both edges +
# both top/bottom within 10% of f's geometry -> f gets the bold
# bit and h is discarded.
if (
pending_span.char_count() > 0
and pending_span.state_slot == span.state_slot
and same_x_extent(pending_span, span, 0.1 * pending_span.bbox_width())
and same_y_extent(pending_span, span, 0.1 * pending_span.bbox_height())
):
pending_span.primary_slot = True
continue
# End current line if e is non-empty AND tn says NOT to continue
if not (len(pending_line.primary_slot) <= 0 or span_continues_line(pending_line, pending_span)):
line.append(pending_line)
pending_line = Line()
append_span(pending_line, pending_span)
pending_span = span
else:
pending_span = span
if pending_span is not None:
if not (len(pending_line.primary_slot) <= 0 or span_continues_line(pending_line, pending_span)):
line.append(pending_line)
pending_line = Line()
append_span(pending_line, pending_span)
line.append(pending_line)
# If no current span exists, the pending line is intentionally dropped.
# This path is currently unreachable from the loop logic.
return line
# --------------------------------------------------------------------------- #
# xn -- line clustering driver #
# --------------------------------------------------------------------------- #
def _is_label_stack(line: Line, other_line: Line, body_ma: float) -> bool:
"""Detect a display-sized label stacked directly above the text it labels. The geometry must overlap horizontally while sitting on a different baseline; the upper piece must be display-sized relative to body text and larger than the lower text. This captures chapter numbers and drop caps that should be read before the title below them."""
if body_ma <= 0:
return False
overlap = min(line.right_edge(), other_line.right_edge()) - max(line.left_edge(), other_line.left_edge())
frac = overlap / max(1e-6, min(line.bbox_width(), other_line.bbox_width()))
vertical_overlap = min(line.top_edge(), other_line.top_edge()) - max(line.bottom_edge(), other_line.bottom_edge())
vertical_overlap_fraction = vertical_overlap / max(1e-6, min(line.bbox_height(), other_line.bbox_height()))
if not (frac > 0.5 and vertical_overlap_fraction < 0.5):
return False
upper, lower = (line, other_line) if line.center_y() > other_line.center_y() else (other_line, line)
# display-type (>= 2x body) AND larger than the text it sits above
# (>= 1.5x lower): a leading label over smaller text. The second clause
# drops same-size display stacks (e.g. chart axis numbers over each other).
return upper.avg_font_size() >= 2.0 * body_ma and upper.avg_font_size() >= 1.5 * lower.avg_font_size()
@dataclass
class LinesContainer:
"""Mutable line container used by the clustering pass."""
primary_slot: list[Line] = field(default_factory=list)
def _set_add(tree: SortedKeyList, line: Line) -> None:
"""Set-style insertion into the sorted line index. Lines with identical top, bottom, left, and right ordering keys are dropped instead of duplicated."""
idx = tree.bisect_left(line)
if idx < len(tree) and reading_order_key(tree[idx]) == reading_order_key(line): # type: ignore[arg-type]
return # reading-order key collision -> sorted set insertion drops the element
tree.add(line)
def cluster_lines(lines_container: LinesContainer, other_item: float, candidate_items: list) -> list[Line]:
"""Mutate the contained line list by merging nearby compatible lines."""
# Sort input lines by reading order.
lines_container.primary_slot.sort(key=left_edge_key)
# Body-text reference for the display-size test in _is_label_stack: the
# median glyph font size across the page (dominated by body text).
merged_accent_spans = sorted(
span_value.font_size for line in lines_container.primary_slot for span_value in line.primary_slot
if getattr(span_value, "font_size", 0) > 0
)
body_ma = merged_accent_spans[len(merged_accent_spans) // 2] if merged_accent_spans else 0.0
# Tree of lines, ordered by reading position (top desc, bottom desc, left, right).
tree: SortedKeyList = SortedKeyList(key=reading_order_key)
merged_lines: list[Line] = [] # output (lines that won't merge further)
for candidate_line in lines_container.primary_slot:
# Rotated / skewed lines: don't try to cluster, just emit
if last_span(candidate_line).previous_slot > 1:
merged_lines.append(candidate_line)
continue
# successor (just below f vertically) and predecessor (just above).
# predecessor/successor search are INCLUSIVE floor/ceiling, so a reading-order-key-equal line already
# in the tree is the zero-distance neighbour: successor = bisect_left
# (first key >= f), predecessor = bisect_right - 1 (last key <= f).
idx_succ = tree.bisect_left(candidate_line)
successor_line = tree[idx_succ] if idx_succ < len(tree) else None
idx_pred = tree.bisect_right(candidate_line)
line_item = tree[idx_pred - 1] if idx_pred > 0 else None
neighbor_line = pick_closer_neighbor(line_item, successor_line, candidate_line, other_item)
if neighbor_line is None:
_set_add(tree, candidate_line)
continue
tree.remove(neighbor_line)
neighbor_last_span = last_span(neighbor_line) # last span of k
# Subscript / overstrike case (single-span f duplicating k's last span)
if (
len(candidate_line.primary_slot) == 1
and len(neighbor_line.primary_slot) <= 5
and neighbor_last_span.char_count() > 0
and neighbor_last_span.state_slot == candidate_line.primary_slot[0].state_slot
and same_x_extent(neighbor_last_span, candidate_line, 0.1 * neighbor_last_span.bbox_width())
and same_y_extent(neighbor_last_span, candidate_line, 0.1 * neighbor_last_span.bbox_height())
):
if abs(neighbor_last_span.left_edge() - candidate_line.left_edge()) < 0.01 and abs(neighbor_last_span.top_edge() - candidate_line.top_edge()) < 0.01:
# exact duplicate -> keep the original line unchanged
_set_add(tree, neighbor_line)
continue
# Otherwise create a new line carrying k's spans with m marked bold
new_line = Line()
neighbor_last_span.primary_slot = True
for source_span in neighbor_line:
append_span(new_line, source_span)
_set_add(tree, new_line)
elif should_merge_lines(neighbor_line, candidate_line, candidate_items):
# Continuation merge. Normally append f after k (left-to-right).
# If the candidate is a display-sized label stacked above the text,
# reading order is top-to-bottom, so the label leads. Reorder spans
# only; the merge and block/line structure stay unchanged.
if _is_label_stack(neighbor_line, candidate_line, body_ma) and candidate_line.center_y() > neighbor_line.center_y():
merged = Line()
for span in candidate_line:
append_span(merged, span)
for span in neighbor_line:
append_span(merged, span)
_set_add(tree, merged)
else:
for span in candidate_line:
append_span(neighbor_line, span)
_set_add(tree, neighbor_line)
else:
# Cannot merge: emit k as a finalized line, start fresh with f
merged_lines.append(neighbor_line)
_set_add(tree, candidate_line)
# Drain remaining
merged_lines.extend(tree)
# Final sort by reading order.
merged_lines.sort(key=reading_order_key)
lines_container.primary_slot = merged_lines
return merged_lines
+190
View File
@@ -0,0 +1,190 @@
"""Span continuation and line-merge predicates."""
from __future__ import annotations
import re
from typing import Optional
from ..model import (
_UNICODE_WHITESPACE_CLASS,
avg_char_width2,
Span,
magnitude_ratio,
same_x_extent,
same_y_extent,
append_span,
last_span,
avg_char_width,
raw_text_of_line,
text_of_line,
reading_order_key,
left_edge_key,
numbering_kind,
Line,
letter_count,
is_upper_dominant,
)
# Matches "...." dot-leader trails used in TOC entries: "Chapter 1 ........"
TRAILING_DOT_LEADER_RE = re.compile(r"([.][" + _UNICODE_WHITESPACE_CLASS + r"]*){4,}\Z")
# --------------------------------------------------------------------------- #
# In-line continuation predicate.
# --------------------------------------------------------------------------- #
def span_continues_line(line: Line, other_span: Span) -> bool:
"""Return whether ``span`` continues the current line. The test requires matching skew, overlapping vertical intervals, and a horizontal gap within a per-character tolerance that widens after sentence-ending punctuation."""
if last_span(line).previous_slot != other_span.previous_slot:
return False
line_center_y = line.center_y() # a's y-center
span_center_y = other_span.center_y() # b's y-center
# Vertical disjointness check: if both centers fall outside the other box,
# the spans are not on the same line.
if (line_center_y > other_span.top_edge() or line_center_y < other_span.bottom_edge()) and (span_center_y > line.top_edge() or span_center_y < line.bottom_edge()):
return False
# tolerance from per-char height
tolerance = min(5.0, max(0.1, avg_char_width(line), avg_char_width2(other_span)))
wide_tolerance = 2.0 * tolerance
# When a's last char is sentence-end punctuation, widen the tolerance
if line.char_stats.tertiary_slot == 5:
wide_tolerance *= 2.0
return other_span.left_edge() > line.right_edge() - wide_tolerance and other_span.left_edge() < line.right_edge() + tolerance
# --------------------------------------------------------------------------- #
# Neighbor distance and picker.
# --------------------------------------------------------------------------- #
def vertical_distance_in_line_heights(line: Line, other_line: Line) -> float:
"""normalized vertical-center distance between two lines. ``|a.center_y - b.center_y| / max(a.bbox_height, b.bbox_height)``: how many line-heights apart the centres are. Returns 0 when centres coincide. """
line_center_y = line.center_y()
other_center_y = other_line.center_y()
if line_center_y == other_center_y:
return 0.0
denom = max(line.bbox_height(), other_line.bbox_height())
if denom == 0.0:
# Empty lines carry an inverted-sentinel bbox. Preserve IEEE division
# edge cases so the later distance comparison simply does not merge.
diff = line_center_y - other_center_y
return float("nan") if diff != diff else float("inf")
return abs(line_center_y - other_center_y) / denom
def pick_closer_neighbor(
line: Optional[Line],
other_line: Optional[Line],
candidate_line: Line,
reference_item: float,
) -> Optional[Line]:
"""Pick the closer neighboring line to the current line when it falls within the merge tolerance. Returns the closer candidate when the distance is below the threshold, else ``None``. Either or both candidates may be ``None`` (e.g. c is at the top of the tree -> no predecessor). """
if line is None and other_line is None:
return None
entry_item = vertical_distance_in_line_heights(line, candidate_line) if line is not None else float("inf")
second_candidate = vertical_distance_in_line_heights(other_line, candidate_line) if other_line is not None else float("inf")
if entry_item >= reference_item and second_candidate >= reference_item:
return None
return line if entry_item < second_candidate else other_line
# --------------------------------------------------------------------------- #
# Line merge predicate.
# --------------------------------------------------------------------------- #
def should_merge_lines(line: Line, other_line: Line, candidate_items: list) -> bool:
"""Return whether ``other_line`` should merge into ``line``. The decision compares the horizontal gap against a tolerance based on harmonic mean character width, then adjusts for style mismatch, script category, dot leaders, column membership, short continuations, bracketed starts, sentence endings, and uppercase dominance."""
if line.char_count() > 0 and other_line.char_count() > 0:
# Different skew/rotation -> never merge
if magnitude_ratio(line.previous_slot, other_line.previous_slot) > 2 and abs(line.previous_slot - other_line.previous_slot) > 10:
return False
# Harmonic mean of character heights with no clamp. A zero char-height
# contributes an infinite inverse, driving the merge tolerance to zero.
line_projection = 1.0 / avg_char_width(line) if avg_char_width(line) != 0 else float("inf")
other_projection = 1.0 / avg_char_width(other_line) if avg_char_width(other_line) != 0 else float("inf")
harmonic_char_width = 2.0 / (line_projection + other_projection)
horizontal_gap = other_line.left_edge() - line.right_edge() # horizontal gap
gap_factor = 2.0
# italic mismatch
italic = line.bold_frac() > 0
other_italic = other_line.bold_frac() > 0
if italic != other_italic:
gap_factor /= 1.5
# last-char category 4 = other-letter (Lo, CJK/syllabics)
# OR more than half of a's chars are category 4
if line.char_stats.tertiary_slot == 4 or line.char_stats.primary_slot[4] > line.char_count() / 2:
gap_factor /= 2.0
# sentence-end + all-digits + dot leader pattern -> TOC row, don't merge
sent_end = line.char_stats.tertiary_slot == 6
if sent_end:
# candidate numeric-token test: the candidate has digits and all characters are digits
all_digits = other_line.char_stats.auxiliary_slot > 0 and other_line.char_stats.auxiliary_slot == other_line.char_stats.primary_slot[1]
if all_digits and TRAILING_DOT_LEADER_RE.search(raw_text_of_line(line)):
gap_factor *= 3.0
else:
all_digits = False
# Column-based bonuses ----------------------------------------------------
if candidate_items and 0 <= line.measure_slot < len(candidate_items):
line_column = candidate_items[line.measure_slot]
col_left = line_column.get("left", float("inf"))
col_right = line_column.get("right", float("-inf"))
else:
col_left = float("inf")
col_right = float("-inf")
inside_col = (
line.left_edge() >= col_left
and line.right_edge() <= col_right
and other_line.left_edge() >= col_left
and other_line.right_edge() <= col_right
)
if (line.char_count() < 40 or inside_col) and (
same_y_extent(line, other_line, 0.1) or same_y_extent(last_span(line), other_line, 0.1)
):
gap_factor *= 1.5
if line.char_count() < 40 and inside_col:
gap_factor *= 2.0
# At-column-edge demotion
if 0 <= other_line.measure_slot < len(candidate_items):
other_column = candidate_items[other_line.measure_slot]
else:
other_column = None
if (
len(candidate_items) <= 0
or (
abs(line.right_edge() - col_right) < 5
and (line.measure_slot >= len(candidate_items) - 1 or not other_column or abs(other_line.left_edge() - other_column.get("left", float("inf"))) < 5)
)
):
gap_factor /= 2.0
# Very short leading line with continuation evidence: short, low aspect,
# numbering-like, and followed by text with letters. The inside-column flag
# controls whether this gets the stronger multiplier.
if line.char_count() <= 8 and line.bbox_width() <= 10 * line.avg_font_size() and numbering_kind(line) != 0 and letter_count(other_line.char_stats) > 0:
gap_factor *= 3.0 if inside_col else 2.0
# Bracketed short line or uppercase sentence-period inside a column.
if line.char_count() <= 10:
text = text_of_line(line)
if text.startswith("[") and text.endswith("]"):
gap_factor *= 2.0
elif inside_col and line.char_stats.secondary_slot == 2 and text.endswith("."):
gap_factor *= 2.0
# Both lines are uppercase-dominant inside the same column.
if inside_col and is_upper_dominant(line.char_stats) and is_upper_dominant(other_line.char_stats):
gap_factor *= 1.5
return horizontal_gap <= gap_factor * harmonic_char_width
+49
View File
@@ -0,0 +1,49 @@
"""
Column detection via sweep-line gutter scoring and recursive page splitting.
The detector builds horizontal and vertical sweep events, scores candidate
gutters, assigns column indexes to lines, and returns column rectangles used by
the second line-clustering pass. Direction code 0 scans vertical positions to
find row breaks; direction code 1 scans horizontal positions to find column
breaks.
"""
import math
from typing import Optional
from ..model import (
Rect, rect_union, EMPTY_RECT, Line, info_weight, text_of_line, numbering_kind, numbering_value, _UNICODE_WHITESPACE_CLASS, _max_nan_propagating, _min_nan_propagating,
)
# Detect TOC dot leaders ("... 5", "....3"). Gutter scoring rejects a split
# candidate when too many dot-leader lines straddle the gap, because a TOC page
# should remain in one reading region.
# The regular expression is end-anchored only; use re.search rather than re.match.
import re as re_module
from .gutters import (
SweepEvent,
SplitCandidate,
ColumnDetectionContext,
DOT_LEADER_RE,
collect_gutter_candidates,
_score_gutter_gap,
)
from .splitting import (
assign_column_index,
recursive_split,
detect_columns,
columns_to_x_bounds,
)
__all__ = [
"SweepEvent",
"SplitCandidate",
"ColumnDetectionContext",
"collect_gutter_candidates",
"assign_column_index",
"recursive_split",
"detect_columns",
"columns_to_x_bounds",
]
+345
View File
@@ -0,0 +1,345 @@
"""Gutter-gap candidates and scoring for column detection."""
from __future__ import annotations
import math
from typing import Optional
from ..model import (
Rect, rect_union, EMPTY_RECT, Line, info_weight, text_of_line, numbering_kind, numbering_value, _UNICODE_WHITESPACE_CLASS, _max_nan_propagating, _min_nan_propagating,
)
# Detect TOC dot leaders ("... 5", "....3"). Gutter scoring rejects a split
# candidate when too many dot-leader lines straddle the gap, because a TOC page
# should remain in one reading region.
# The regular expression is end-anchored only; use re.search rather than re.match.
import re as re_module
# --------------------------------------------------------------------------- #
# Sweep event. ``is_start=True`` means "line enters" at a start edge; False means
# "line leaves" at an end edge.
# --------------------------------------------------------------------------- #
class SweepEvent:
__slots__ = ("line", "position", "is_start")
def __init__(self, line: Line, position: float, is_start_flag: bool):
self.line = line
self.position = position
self.is_start = is_start_flag
# --------------------------------------------------------------------------- #
# Column-split candidate. Direction 0 is a vertical sweep; direction 1 is a
# horizontal sweep. Higher score is better.
# --------------------------------------------------------------------------- #
class SplitCandidate:
__slots__ = ("start", "end", "direction", "score")
def __init__(self, start: float, end: float, direction_value: int, score: float):
self.start = start
self.end = end
self.direction = direction_value
self.score = score
# --------------------------------------------------------------------------- #
# Detection context. Thresholds derived from page geometry and page statistics.
# --------------------------------------------------------------------------- #
class ColumnDetectionContext:
"""Page-level thresholds used while recursively scoring gutter candidates."""
__slots__ = ("secondary_slot", "primary_slot", "tertiary_slot", "state_slot", "auxiliary_slot", "option_slot", "measure_slot")
def __init__(self, primary_item, secondary_item, candidate_item):
self.secondary_slot = secondary_item
self.primary_slot = candidate_item
# ``log2(0)`` would be -inf -- guard against empty input.
self.tertiary_slot = math.floor(2 * math.log2(len(candidate_item))) if candidate_item else 0
self.state_slot = primary_item.bbox_width() / 6.0
self.auxiliary_slot = _max_nan_propagating(0.5 * secondary_item.primary_slot, _min_nan_propagating(1.1 * (secondary_item.tertiary_slot - secondary_item.primary_slot), 3.0 * secondary_item.primary_slot))
self.option_slot = secondary_item.measure_slot
self.measure_slot = 1.5 * secondary_item.primary_slot
DOT_LEADER_RE = re_module.compile(r"([.][" + _UNICODE_WHITESPACE_CLASS + r"]*){5,}\Z")
# --------------------------------------------------------------------------- #
# Score split candidates in a sweep.
# --------------------------------------------------------------------------- #
def collect_gutter_candidates(
context: ColumnDetectionContext,
other_rect: Rect, # root rect (page bbox)
candidate_rect: Rect, # current sub-rect
events: list[SweepEvent], # sorted events
direction: int, # direction: 0 vert sweep / 1 horiz sweep
extent: float, # extent (height or width)
min_gap: float, # minimum gutter size
out_candidates: list[SplitCandidate], # output: candidates to append to
) -> None:
"""Walk adjacent event pairs looking for column gutters. Each candidate gap receives a multiplicative score from line weight, font-size balance, indent/outdent structure, citation markers, and edge proximity; the best viable score wins."""
active_count = 0
for size_value in range(len(events) - 1):
if events[size_value].is_start:
active_count += 1
else:
active_count -= 1
if active_count > 0:
continue
# The gap between adjacent event positions is a candidate gutter.
score = _score_gutter_gap(context, other_rect, candidate_rect, events, direction, extent, min_gap, size_value)
if score is not None:
key_value = events[size_value].position
score_value = events[size_value + 1].position
out_candidates.append(SplitCandidate(key_value, score_value, direction, score))
def _score_gutter_gap(
primary_item: ColumnDetectionContext,
other_rect: Rect, # root rect
candidate_rect: Rect, # current sub-rect (variable name p follows the extraction rule)
reference_items: list[SweepEvent], # events
next_number: int, # direction
extent: float, # extent
min_gap: float, # min gutter
limit_number: int, # current event index
) -> Optional[float]:
"""Score one candidate gap at adjacent sweep events, or return None when it is not viable."""
key_value = reference_items[limit_number].position
score_value = reference_items[limit_number + 1].position
item_value = score_value - key_value
if item_value < min_gap:
return None
# --- backward pass: lines that close before this gap -----------------
measure_item = preceding_max_width = 0
secondary_item = reference_item = distance_accumulator = width_value = wide_accumulator = 0.0
min_edge = math.inf
preceding_max_trailing_edge = -math.inf
min_edge_position = math.inf
max_edge = -math.inf
event_count = 0
lower_accumulator = math.inf
candidate_item = group_value = max_char_count = state_item = 0
sample_item = limit_number
while sample_item >= 0:
sweep_event = reference_items[sample_item]
event_position = sweep_event.position
event_is_start = sweep_event.is_start
sweep_line = sweep_event.line
if event_position < key_value - item_value:
break
if event_is_start:
sample_item -= 1
continue
measure_item += 1
preceding_max_width = max(preceding_max_width, sweep_line.bbox_width())
if next_number == 1:
leading_edge, trailing_edge = sweep_line.bottom_edge(), sweep_line.top_edge()
else:
leading_edge, trailing_edge = sweep_line.left_edge(), sweep_line.right_edge()
min_edge = min(min_edge, leading_edge)
preceding_max_trailing_edge = max(preceding_max_trailing_edge, trailing_edge)
entry_item = info_weight(sweep_line.char_stats)
secondary_item += entry_item
if sweep_line.avg_font_size() > reference_item:
reference_item = sweep_line.avg_font_size()
distance_accumulator = entry_item
elif sweep_line.avg_font_size() == reference_item:
distance_accumulator += entry_item
if key_value - event_position < 1:
width_value += 1
wide_accumulator = max(wide_accumulator, sweep_line.bbox_width())
min_edge_position = min(min_edge_position, leading_edge)
max_edge = max(max_edge, trailing_edge)
if numbering_kind(sweep_line) == 1:
event_count += 1
if numbering_value(sweep_line) == 1:
lower_accumulator = min(lower_accumulator, sweep_line.left_edge())
if next_number == 1:
if sweep_line.char_count() <= 5 and numbering_kind(sweep_line) != 0:
candidate_item += 1
if sweep_line.char_count() <= 10:
text = text_of_line(sweep_line)
if (text.startswith("[") and text.endswith("]")) or (
sweep_line.char_stats.secondary_slot == 2 and text.endswith(".")
):
group_value += 1
if DOT_LEADER_RE.search(text_of_line(sweep_line)):
state_item += 1
max_char_count = max(max_char_count, sweep_line.char_count())
sample_item -= 1
# Sanity gates
if (
candidate_item >= measure_item
or candidate_item >= max(2, measure_item / 2)
or group_value >= measure_item
or state_item >= max(2, measure_item / 2)
or (next_number == 1 and max_char_count <= 1)
):
return None
# --- forward pass: lines that open after this gap --------------------
next_gap = 0.0
following_line_count = 0
other_gap = math.inf
following_max_trailing_edge = -math.inf
page_gap = following_max_font_size = after = 0
quantity = 0
numbering_score = following_numbering_count = following_edge_max_width = 0
right_gap = 0.0
count_item = limit_number + 1
while count_item < len(reference_items):
sweep_event = reference_items[count_item]
event_position = sweep_event.position
event_is_start = sweep_event.is_start
sweep_line = sweep_event.line
if event_position > score_value + item_value:
break
if not event_is_start:
count_item += 1
continue
following_line_count += 1
next_gap = max(next_gap, sweep_line.bbox_width())
if next_number == 1:
leading_edge, trailing_edge = sweep_line.bottom_edge(), sweep_line.top_edge()
else:
leading_edge, trailing_edge = sweep_line.left_edge(), sweep_line.right_edge()
other_gap = min(other_gap, leading_edge)
following_max_trailing_edge = max(following_max_trailing_edge, trailing_edge)
width = info_weight(sweep_line.char_stats)
after += width
if sweep_line.avg_font_size() > following_max_font_size:
following_max_font_size = sweep_line.avg_font_size()
page_gap = width
elif sweep_line.avg_font_size() == following_max_font_size:
page_gap += width
if event_position - score_value < 1:
quantity += 1
following_edge_max_width = max(following_edge_max_width, sweep_line.bbox_width())
if numbering_kind(sweep_line) == 1:
following_numbering_count += 1
numbering_score = _max_nan_propagating(numbering_score, numbering_value(sweep_line))
right_gap = _max_nan_propagating(right_gap, sweep_line.top_edge())
count_item += 1
if measure_item <= 0 or following_line_count <= 0:
return None
# Combined scoring mixes the root page rectangle for page-level thresholds
# with the current recursive sub-rectangle for split geometry. Root and sub
# coincide before the first split, but diverge on genuinely multi-column
# pages; keeping both frames is part of the column decision model.
root_width = other_rect.bbox_width()
height = other_rect.bbox_height()
root_center_x = other_rect.center_x()
if next_number == 1:
# vertical sweep: special pre-gate for single-line columns
# The top-edge gate compares the sub-rectangle to the root page height.
if width_value <= 1 and quantity <= 1 and not (
candidate_rect.top_edge() < other_rect.bottom_edge() + 0.3 * height
and lower_accumulator < math.inf
and numbering_score <= 4
):
return None
if (secondary_item <= 100 and after <= 100) and (
key_value < other_rect.left + 0.2 * root_width or score_value > other_rect.left + 0.8 * root_width
):
return None
mid = (reference_item + following_max_font_size) / 2
if next_number == 0 and distance_accumulator >= 0.8 * secondary_item and page_gap >= 0.8 * after and (
(
abs(reference_item - following_max_font_size) < 0.1
and reference_item >= primary_item.secondary_slot.primary_slot + 0.5
and following_max_font_size >= primary_item.secondary_slot.primary_slot + 0.5
and item_value < max(1.3 * mid, min_gap * 2)
)
or (
reference_item >= primary_item.secondary_slot.primary_slot + 2
and following_max_font_size >= primary_item.secondary_slot.primary_slot + 2
and item_value < max(1.5 * mid, min_gap * 3)
)
):
return None
line_value = extent * extent * item_value
if next_number == 1:
line_value *= min(width_value, quantity)
if secondary_item <= 50 and measure_item <= 1:
line_value /= 100
candidate_height = candidate_rect.bbox_height()
threshold = candidate_rect.top_edge() - 0.2 * candidate_height
if following_numbering_count >= 3 and numbering_score >= 6 and right_gap < threshold:
# Preserve IEEE division here: a zero denominator yields +inf and a
# negative denominator is clamped below. The branch selection depends
# on those numeric edge cases.
denom = numbering_score - following_numbering_count
inv = (1 / denom) if denom != 0 else math.inf
factor = max(0.3, min(1.0, inv))
factor *= factor
line_value *= factor
elif candidate_height > height / 2:
factor = candidate_height / height
factor *= factor
line_value *= 1 + factor
line_value *= max(1, 2 - abs(root_center_x - (key_value + score_value) / 2) / root_width * 10)
if next_number == 0:
candidate_width = candidate_rect.bbox_width()
line_value *= max(wide_accumulator, following_edge_max_width) / candidate_width * (max(preceding_max_width, next_gap) / candidate_width)
min_value = min(min_edge, other_gap)
max_value = max(preceding_max_trailing_edge, following_max_trailing_edge)
if min_value < root_center_x and max_value > root_center_x:
left_center_distance = root_center_x - min_value
value = max_value - root_center_x
line_value *= 1 + min(left_center_distance, value) / max(left_center_distance, value)
# Edge bands are measured from the root page rectangle.
edge_top = other_rect.primary_slot + 0.2 * height
edge_bot = other_rect.primary_slot + 0.8 * height
if key_value < edge_top or score_value > edge_bot:
line_value *= 4
# The lower-edge boost uses forward-pass numbering and width counts,
# because it is testing the material below the candidate gap.
if (key_value < edge_top and event_count >= 1 and wide_accumulator < root_width / 4) or (
score_value > edge_bot and following_numbering_count >= 1 and following_edge_max_width < root_width / 4
):
line_value *= 9
if 2 * width_value >= limit_number and reference_item > following_max_font_size + 0.5:
line_value *= 100
if lower_accumulator < math.inf:
if lower_accumulator < root_center_x:
line_value *= 100
elif event_count >= 2 or following_numbering_count >= 2:
line_value /= 4
size = max(reference_item, following_max_font_size)
# This test uses the full sweep extent, not the minimum gutter size.
if (
extent >= 0.99 * root_width
and min_edge_position < root_center_x
and max_edge > root_center_x
and reference_item >= following_max_font_size + 0.5
and item_value > size
):
line_value *= item_value / size
if secondary_item < 1 or after < 1:
line_value *= 10
return _max_nan_propagating(0.0, line_value)
+223
View File
@@ -0,0 +1,223 @@
"""Recursive column splitting and column index assignment."""
from __future__ import annotations
import math
from typing import Optional
from ..model import (
Rect, rect_union, EMPTY_RECT, Line, info_weight, text_of_line, numbering_kind, numbering_value, _UNICODE_WHITESPACE_CLASS, _max_nan_propagating, _min_nan_propagating,
)
from .gutters import (
SweepEvent,
SplitCandidate,
ColumnDetectionContext,
collect_gutter_candidates,
)
# --------------------------------------------------------------------------- #
# Assign column indexes to lines whose event is a start edge.
# --------------------------------------------------------------------------- #
def assign_column_index(items: list[SweepEvent], other_number: int) -> None:
"""Assign ``column_index`` to each gutter event that starts a column-owned line."""
for column in items:
if column.is_start:
column.line.measure_slot = other_number
# --------------------------------------------------------------------------- #
# Recursive split driver.
# --------------------------------------------------------------------------- #
def recursive_split(
context: ColumnDetectionContext,
horizontal_events: list[SweepEvent], # horizontal events (sorted by F/L)
vertical_events: list[SweepEvent], # vertical events (sorted by C/D)
reference_rect: Rect, # root rect
current_rect: Rect, # current sub-rect
depth: int, # depth
column_offset: int, # column-index offset
) -> list[Rect]:
if depth >= context.tertiary_slot:
assign_column_index(horizontal_events, column_offset)
return [current_rect]
split_candidates: list[SplitCandidate] = []
if current_rect.bbox_height() >= context.measure_slot:
collect_gutter_candidates(context, reference_rect, current_rect, horizontal_events, 1, current_rect.bbox_height(), context.option_slot, split_candidates)
if current_rect.bbox_width() >= context.state_slot:
collect_gutter_candidates(context, reference_rect, current_rect, vertical_events, 0, current_rect.bbox_width(), context.auxiliary_slot, split_candidates)
if len(split_candidates) <= 0:
if current_rect.bbox_width() < 0.8 * reference_rect.bbox_width():
assign_column_index(horizontal_events, column_offset)
return [current_rect]
# Examine gaps in b for vertical gutters (fallback)
key_value = active_overlap_count = 0
measure_item = local = 0
gap_indices: list[int] = []
gap_scan_index = 0
while gap_scan_index < len(horizontal_events) - 1:
width_value = horizontal_events[gap_scan_index].position
event_is_start = horizontal_events[gap_scan_index].is_start
candidate_item = horizontal_events[gap_scan_index].line
if width_value > reference_rect.left + reference_rect.bbox_width() * 5 / 6:
break
if event_is_start:
active_overlap_count += 1
key_value += info_weight(candidate_item.char_stats)
else:
active_overlap_count -= 1
key_value -= info_weight(candidate_item.char_stats)
local = max(local, active_overlap_count)
measure_item = max(measure_item, key_value)
if event_is_start or active_overlap_count > 2 or width_value < reference_rect.left + reference_rect.bbox_width() / 6:
gap_scan_index += 1
continue
previous_gap_index = gap_indices[-1] if gap_indices else None
if previous_gap_index is not None and width_value < horizontal_events[previous_gap_index].position + reference_rect.bbox_width() / 10:
gap_indices[-1] = gap_scan_index
elif local >= 8 and measure_item >= 100:
gap_indices.append(gap_scan_index)
local = measure_item = 0
gap_scan_index += 1
if len(gap_indices) <= 0 or len(gap_indices) > 2:
assign_column_index(horizontal_events, column_offset)
return [current_rect]
if local < 4 or measure_item < 50:
assign_column_index(horizontal_events, column_offset)
return [current_rect]
# Assign columns based on the discovered gaps
out: list[Rect] = []
cursor = 0
for index in range(len(gap_indices) + 1):
pos = gap_indices[index] if index < len(gap_indices) else len(horizontal_events)
for event_index in range(cursor, pos):
sweep_event = horizontal_events[event_index]
if sweep_event.is_start:
sweep_event.line.measure_slot = column_offset + len(out)
end = horizontal_events[pos].position if pos < len(horizontal_events) else current_rect.right_edge()
out.append(Rect(horizontal_events[cursor].position, end, current_rect.top, current_rect.bottom_edge()))
cursor = pos + 1
return out
# Pick best candidate split
best: Optional[SplitCandidate] = None
for count_item in split_candidates:
if best is None or best.score < count_item.score:
best = count_item
assert best is not None # h is non-empty here
if best.direction == 0:
# Horizontal split (vertical gutter): divide events into top / bottom halves
upper_left, split_max = math.inf, -math.inf
value, lower_right = math.inf, -math.inf
upper_horizontal_events: list[SweepEvent] = []
lower_horizontal_events: list[SweepEvent] = []
for horizontal_event in horizontal_events:
line = horizontal_event.line
if line.top_edge() > best.start:
upper_horizontal_events.append(horizontal_event)
upper_left = min(upper_left, line.left_edge())
split_max = max(split_max, line.right_edge())
elif line.bottom_edge() < best.end:
lower_horizontal_events.append(horizontal_event)
value = min(value, line.left_edge())
lower_right = max(lower_right, line.right_edge())
upper_vertical_events: list[SweepEvent] = []
split: list[SweepEvent] = []
for vertical_event in vertical_events:
if vertical_event.position > best.start:
upper_vertical_events.append(vertical_event)
elif vertical_event.position < best.end:
split.append(vertical_event)
upper = recursive_split(context, upper_horizontal_events, upper_vertical_events, reference_rect, Rect(upper_left, split_max, current_rect.top, best.end), depth + 1, column_offset)
lower = recursive_split(
context,
lower_horizontal_events,
split,
reference_rect,
Rect(value, lower_right, best.start, current_rect.bottom_edge()),
depth + 1,
column_offset + len(upper),
)
return upper + lower
# Vertical split (horizontal gutter): divide events into left / right halves
split_max, left_bottom = -math.inf, math.inf
right_top, right_bottom = -math.inf, math.inf
left_events: list[SweepEvent] = []
right_events: list[SweepEvent] = []
left_vert: list[SweepEvent] = []
right_vert: list[SweepEvent] = []
for split_event in horizontal_events:
if split_event.position < best.end:
left_events.append(split_event)
elif split_event.position > best.start:
right_events.append(split_event)
for event in vertical_events:
line = event.line
if line.left_edge() < best.end:
left_vert.append(event)
split_max = max(split_max, line.top_edge())
left_bottom = min(left_bottom, line.bottom_edge())
elif line.right_edge() > best.start:
right_vert.append(event)
right_top = max(right_top, line.top_edge())
right_bottom = min(right_bottom, line.bottom_edge())
left = recursive_split(
context, left_events, left_vert, reference_rect, Rect(current_rect.left, best.start, split_max, left_bottom), depth + 1, column_offset
)
right = recursive_split(
context,
right_events,
right_vert,
reference_rect,
Rect(best.end, current_rect.right_edge(), right_top, right_bottom),
depth + 1,
column_offset + len(left),
)
return left + right
# --------------------------------------------------------------------------- #
# Build events, sort them, then start recursive splitting.
# --------------------------------------------------------------------------- #
def detect_columns(column: ColumnDetectionContext) -> list[Rect]:
"""Detect column rectangles and populate each line's column index."""
horizontal_events: list[SweepEvent] = []
vertical_events: list[SweepEvent] = []
bbox = EMPTY_RECT
for line in column.primary_slot:
if line.bbox_width() <= 0 or line.bbox_height() <= 0:
continue
bbox = rect_union(bbox, line.secondary_slot)
horizontal_events.append(SweepEvent(line, line.left_edge(), True))
horizontal_events.append(SweepEvent(line, line.right_edge(), False))
vertical_events.append(SweepEvent(line, line.bottom_edge(), True))
vertical_events.append(SweepEvent(line, line.top_edge(), False))
# Sort by position, start events before end events, then line width.
horizontal_events.sort(key=lambda split_event: (split_event.position, 0 if split_event.is_start else 1, split_event.line.bbox_width()))
# Sort by position, start events before end events, then line height.
vertical_events.sort(key=lambda split_event: (split_event.position, 0 if split_event.is_start else 1, split_event.line.bbox_height()))
return recursive_split(column, horizontal_events, vertical_events, bbox, bbox, 0, 0)
# --------------------------------------------------------------------------- #
# Public helper: produce the {left, right} dict list used by line merging #
# --------------------------------------------------------------------------- #
def columns_to_x_bounds(column_rects: list[Rect]) -> list[dict]:
"""Convert column rectangles to a ``[{left, right}, ...]`` table."""
return [{"left": column.left, "right": column.right} for column in column_rects]
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,120 @@
"""Per-page heading-candidate detection. This module builds and filters heading candidates from page blocks. It combines
numbering recognition, chapter/appendix keywords, local neighbor geometry,
font/style signals, cross-page rejection, and page-level candidate filtering
before handing candidates to outline assembly.
"""
import json
import math
import re
import regex as regex_module # Unicode \p{...} property classes.
from pathlib import Path
from typing import Any, Optional
from ..outline_assembly import HeadingCandidate, OutlineNode
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_trim_unicode_ws,
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
)
from ..tokens import (
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
)
from .keyword_tables import (
_DICT_PATH,
_DICTS,
SECTION_KEYWORDS_TRIE,
ABSTRACT_KEYWORDS_TRIE,
REFERENCES_TRIE,
APPENDIX_SECTION_TRIE,
INTRODUCTION_SECTION_TRIE,
BOX_KEYWORD_TRIE,
KEYWORDS_SECTION_TRIE,
CHAPTER_WORDS_TRIE,
APPENDIX_KEYWORDS_TRIE,
_normalize_text_key,
ABSTRACT_KEYWORDS_SET,
REFERENCES_SET,
NUMBERED_PREFIX_RE,
DEAD_DIGIT_RE,
EQUATION_KEYWORDS_TRIE,
ENGLISH_WORD_TO_NUMBER,
ROMAN_NUMERAL_MAP,
FORMULA_CHAR_WEIGHTS,
)
from .text_checks import (
token_text_of_block,
similar_style,
is_heading_continuation,
matches_abstract,
matches_references,
vertically_close,
is_equation_adjacent_line,
has_substantive_content,
is_cover_page,
clamp,
token_to_number,
letter_to_ordinal,
)
from .neighbors import (
BlockNeighborCache,
compute_bucket_span,
neighbor_above,
body_neighbor_above,
neighbor_right,
neighbor_right_peer,
closest_body_neighbor_above,
PageNeighborMap,
)
from .candidates import (
PageScanState,
push_candidate,
make_heading_candidate,
make_plain_candidate,
make_body_heading_candidate,
make_numbered_candidate,
_di_count,
_number_at_token_index,
)
from .detectors import (
detect_numbered_heading,
detect_labeled_heading,
detect_chapter_appendix,
detect_box_heading,
classify_heading,
is_acceptable_heading,
safe_column_index,
try_classify_heading,
is_too_wide_for_heading,
passes_neighbor_check,
has_competing_labeled_heading,
is_year_string,
is_bibliography_entry,
)
from .style_detectors import (
detect_font_heading,
detect_heading_with_body,
)
from .page_scan import (
scan_page_headings,
DocCandidateCollector,
filter_page_candidates,
build_doc_heading_candidates,
find_section_openers,
)
__all__ = [
"SECTION_KEYWORDS_TRIE", "ABSTRACT_KEYWORDS_TRIE", "ABSTRACT_KEYWORDS_SET", "REFERENCES_TRIE", "REFERENCES_SET", "APPENDIX_SECTION_TRIE", "INTRODUCTION_SECTION_TRIE", "BOX_KEYWORD_TRIE", "KEYWORDS_SECTION_TRIE", "CHAPTER_WORDS_TRIE", "APPENDIX_KEYWORDS_TRIE",
"ROMAN_NUMERAL_MAP", "ENGLISH_WORD_TO_NUMBER", "FORMULA_CHAR_WEIGHTS", "NUMBERED_PREFIX_RE", "DEAD_DIGIT_RE",
"is_heading_continuation", "similar_style", "matches_abstract", "matches_references", "vertically_close", "is_equation_adjacent_line", "has_substantive_content", "is_cover_page", "token_to_number", "letter_to_ordinal",
"BlockNeighborCache", "compute_bucket_span", "neighbor_above", "body_neighbor_above", "neighbor_right", "closest_body_neighbor_above",
"PageNeighborMap", "PageScanState", "DocCandidateCollector", "filter_page_candidates",
"classify_heading", "is_acceptable_heading", "try_classify_heading", "push_candidate", "detect_numbered_heading", "make_heading_candidate", "detect_labeled_heading", "make_body_heading_candidate", "detect_heading_with_body", "detect_chapter_appendix",
"is_too_wide_for_heading", "passes_neighbor_check", "make_plain_candidate", "make_numbered_candidate", "has_competing_labeled_heading", "detect_font_heading", "scan_page_headings", "detect_box_heading",
"build_doc_heading_candidates",
]
@@ -0,0 +1,214 @@
"""Page scan state and heading-candidate constructors."""
from __future__ import annotations
import math
from typing import Any, Optional
from ..outline_assembly import HeadingCandidate, OutlineNode
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_trim_unicode_ws,
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
)
from ..tokens import (
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
)
from .text_checks import matches_references
from .neighbors import (
neighbor_right,
closest_body_neighbor_above,
PageNeighborMap,
)
# --------------------------------------------------------------------------- #
# Main per-page heading state #
# --------------------------------------------------------------------------- #
class PageScanState:
"""Per-page heading scan state."""
__slots__ = ("secondary_slot", "primary_slot", "state_slot", "auxiliary_slot", "tertiary_slot", "option_slot", "measure_slot")
def __init__(self, doc, page):
self.secondary_slot = doc # document state
self.primary_slot = page
self.state_slot = doc.primary_slot[page.page_index - 2] if page.page_index >= 2 else None # prev page
self.auxiliary_slot = page.output_slot # blocks in original order
self.tertiary_slot = PageNeighborMap(page) # neighbor map
self.option_slot: list[HeadingCandidate] = [] # output candidates
self.measure_slot: set = set() # set of block ids already pushed
# --------------------------------------------------------------------------- #
# Push heading candidate into page state #
# --------------------------------------------------------------------------- #
def push_candidate(page_scan: PageScanState, candidate: HeadingCandidate) -> None:
"""Push a candidate into the page scan state."""
page_scan.option_slot.append(candidate)
page_scan.measure_slot.add(candidate.group_slot)
# --------------------------------------------------------------------------- #
# Heading-candidate builder.
# --------------------------------------------------------------------------- #
def make_heading_candidate(page_scan: PageScanState, type_: int, block: Block, item_list: list[int],
tokens: Optional[TokenView], title_tokens: Optional[TokenView], has_numbering_flag: bool = False) -> HeadingCandidate:
"""Build a heading candidate and apply the spatial promotion rule."""
neighbor = page_scan.tertiary_slot
right_neighbor = neighbor_right(neighbor, block)
# Spatial promotion to structural numbering: if a right-side neighbour exists, the block
# has high skew (real horizontal text), its title ends in a colon-like
# symbol, and its last line is nearly as wide as and right-aligned to
# the neighbour -> promote the flag to true.
if (
not has_numbering_flag and right_neighbor is not None
and block.previous_slot > 0.9 and title_tokens is not None
):
last_title_token = last_token(title_tokens)
if last_title_token is not None and is_trimmable_token(last_title_token):
mh_block = last_line_of(block)
if (
mh_block.bbox_width() > 0.7 * right_neighbor.bbox_width()
and abs(mh_block.right_edge() - right_neighbor.right_edge()) < 2 * avg_char_width(mh_block)
):
has_numbering_flag = True
prominent_flag = (
type_ == 7
or (len(item_list) > 0 and title_tokens is not None and matches_references(title_tokens))
)
return HeadingCandidate(
type_=type_,
page=page_scan.primary_slot,
group_value=block,
anchor=closest_body_neighbor_above(neighbor, block),
numbering_value=item_list,
tokens=tokens,
title_tokens=title_tokens,
has_numbering_flag=has_numbering_flag,
prominent_flag=prominent_flag,
)
# --------------------------------------------------------------------------- #
# Shorthand heading-candidate builders #
# --------------------------------------------------------------------------- #
def make_plain_candidate(page_scan: PageScanState, type_: int, block: Block) -> HeadingCandidate:
"""Build a type-only candidate using the full block text."""
return make_heading_candidate(page_scan, type_, block, [], None, tokenize_block(block), False)
def make_body_heading_candidate(page_scan: PageScanState, type_: int, block: Block, tokens: TokenView) -> HeadingCandidate:
"""Build a candidate from body-heading tokens."""
return make_heading_candidate(page_scan, type_, block, [], None, trim_trailing_punct(tokens), True)
# --------------------------------------------------------------------------- #
# Composed-number heading-candidate builder #
# --------------------------------------------------------------------------- #
def make_numbered_candidate(page_scan: PageScanState, block: Block, item_list: list[int], tokens: TokenView, title_tokens: TokenView) -> Optional[HeadingCandidate]:
"""Build a numbered-heading candidate after the full reject-guard chain. The guard rejects empty numbering, weak single-token numbering, unsupported top-of-page continuations, alignment failures, and trailing-number continuation conflicts."""
from ..labels import extract_structural_number # numbering-prefix detector
# Basic reject branch for empty, weak, or top-of-page continuation markers.
if title_tokens.length <= 0:
return None
first_title_token = first_token(title_tokens)
if (title_tokens.length == 1 and first_title_token is not None
and first_title_token.primary_slot != 2 and first_title_token.primary_slot != 4 and first_title_token.secondary_slot != 2
and not block.isolated_centered):
return None
if len(item_list) == 1 and item_list[0] == 1 and block.top_edge() < 0.3 * page_scan.primary_slot.bounds.bbox_height():
from ..heading_detection import neighbor_right
if neighbor_right(page_scan.tertiary_slot, block) is None:
last_title_token = last_token(title_tokens)
if last_title_token is not None and last_title_token.anchor_ranges and last_title_token.anchor_ranges[-1].line is last_line_of(block):
return None
# If basic guards didn't trigger, examine multi-line patterns.
reject = False
if block.line_count() > 1:
second_line = block.primary_slot[1]
first_number_token = first_token(tokens)
first_title_token = first_token(title_tokens)
if first_number_token is not None and first_title_token is not None:
left = first_anchor_span(first_number_token).left_edge()
title_left = first_anchor_span(first_title_token).left_edge()
if not (left < title_left and second_line.left_edge() > (left + title_left) / 2):
# Check trailing tokens for c+1 continuation
trailing_tokens = tokenize_block(block)
trailing_tokens = trailing_tokens.slice(_di_count(trailing_tokens, block.line()))
trailing_tokens = extract_structural_number(trailing_tokens)
if trailing_tokens is None or trailing_tokens.length <= 0:
reject = False
elif block.measure_slot:
reject = True
else:
if len(item_list) == 1 and trailing_tokens.length <= 2:
trailing_first_token = trailing_tokens.token_at(0)
if trailing_first_token is not None:
value = token_numeric_value(trailing_first_token)
# Strict equality on the raw Number, no truncation
# (a fractional value never
# equals the integer c[0]+1).
reject = (not math.isnan(value) and value == item_list[0] + 1)
else:
reject = False
else:
reject = False
if reject:
return None
return make_heading_candidate(page_scan, 1, block, item_list, tokens, title_tokens, False)
def _di_count(tokens: TokenView, line) -> int:
"""count tokens belonging to ``line`` starting from index 0."""
count_item = 0
for index_value in range(tokens.length):
token_value = tokens.token_at(index_value)
if token_value is None or token_value.line() is not line:
break
count_item += 1
return count_item
# --------------------------------------------------------------------------- #
# Numbered heading detector #
# --------------------------------------------------------------------------- #
def _number_at_token_index(tokens: TokenView, index: int) -> int:
"""Try to extract a numbering value at index ``b_idx`` of a token view. Returns 0 if not a number-followed-by-separator, else the number. """
if tokens.length < index + 2:
return 0
token = tokens.token_at(index)
if token is None or token.type != 1:
return 0
next_tok = tokens.token_at(index + 1)
if next_tok is None:
return 0
from ..labels import PERIOD_CHARS as period_chars
if not (
next_tok.str in period_chars
or next_tok.str in (")", "]", ".", "。", "。", ")", "]", "】")
):
return 0
val = token_numeric_value(token)
if math.isnan(val) or val <= 0 or val >= 1000:
return 0
return int(val)
@@ -0,0 +1,492 @@
"""Numbered, labeled, chapter/appendix, and box heading detectors plus acceptability checks."""
from __future__ import annotations
import math
from typing import Any, Optional
from ..outline_assembly import HeadingCandidate, OutlineNode
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_trim_unicode_ws,
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
)
from ..tokens import (
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
)
from .keyword_tables import (
APPENDIX_SECTION_TRIE,
BOX_KEYWORD_TRIE,
CHAPTER_WORDS_TRIE,
APPENDIX_KEYWORDS_TRIE,
ROMAN_NUMERAL_MAP,
)
from .text_checks import (
similar_style,
matches_abstract,
matches_references,
has_substantive_content,
clamp,
token_to_number,
letter_to_ordinal,
)
from .neighbors import (
neighbor_above,
body_neighbor_above,
neighbor_right,
closest_body_neighbor_above,
)
from .candidates import (
PageScanState,
make_heading_candidate,
make_plain_candidate,
make_numbered_candidate,
)
def detect_numbered_heading(page_scan: PageScanState, block: Block, tokens: TokenView) -> Optional[HeadingCandidate]:
"""Identify "1.2.3" / "[1]" style numbered heading prefixes."""
item_list: list[int] = []
at_value = None
for entry in enumerate_tokens(tokens):
index = entry["index"]
token = entry["token"]
if token.type == 1:
if len(item_list) >= 4:
break
val = token_numeric_value(token)
if math.isnan(val) or val <= 0 or val >= 20:
break
if len(token.str) >= 3:
break
item_list.append(int(val))
if not token.boundary_slot:
continue
at_value = tokens.token_at(index + 1)
if (
index + 2 < tokens.length
and at_value is not None
and (is_word_token(at_value) or at_value.type == 6)
and tokens.token_at(index + 2) is not None
and tokens.token_at(index + 2).type == 1
):
break
# Strong numbering, prominent style, or a viable separator token is
# enough to build a numbered-heading candidate.
if (
len(item_list) > 1
or heading_score(block) > page_scan.primary_slot.primary_slot.primary_slot + 1
or (not block.measure_slot and at_value is not None and (
at_value.primary_slot in (2, 4) or at_value.secondary_slot == 2 or at_value.type == 4
or at_value.str == "." or at_value.str == "|"
))
):
return make_numbered_candidate(
page_scan, block, item_list,
tokens.slice(0, index + 1),
tokens.slice(index + 1),
)
return None
# Accept period-like punctuation or a symbol token as a numbering
# separator.
if token.str in (".", ".", "。", "。") or token.type == 4:
prev = tokens.token_at(index - 1) # token_at(-1) returns None
if prev is None or prev.type != 1:
break
if not token.boundary_slot:
continue
return make_numbered_candidate(
page_scan, block, item_list,
tokens.slice(0, index + 1), tokens.slice(index + 1),
)
if len(item_list) <= 0 or token.type != 2:
break
if token.primary_slot not in (2, 4):
break
if (
len(item_list) > 1
or len(token.str) >= 3
or tokens.length - index >= 3
):
return make_numbered_candidate(
page_scan, block, item_list,
tokens.slice(0, index), tokens.slice(index),
)
return None
return None
# --------------------------------------------------------------------------- #
# Complex numbering format detector.
# --------------------------------------------------------------------------- #
def detect_labeled_heading(page_scan: PageScanState, block: Block, tokens: TokenView) -> Optional[HeadingCandidate]:
"""Detect Roman, letter, CJK, and mixed-numbering headings."""
if tokens.length <= 1:
return None
first = tokens.token_at(0)
second = tokens.token_at(1)
if first is None or second is None:
return None
# Roman numeral path
roman = ROMAN_NUMERAL_MAP.get(first.str)
if roman is not None and is_word_token(second) and second.str in "..。。:)":
prefix = tokens.slice(0, 2)
return make_heading_candidate(page_scan, 2, block, [roman], prefix, tokens.slice(prefix.length))
# CJK number path
cjk_pos = "一二三四五六七八九十".find(first.str)
if cjk_pos >= 0 and is_word_token(second):
prefix = tokens.slice(0, 2)
rest = tokens.slice(prefix.length)
if rest.length <= 0:
return None
return make_heading_candidate(page_scan, 3, block, [cjk_pos + 1], prefix, rest)
# Letter path
if tokens.length <= 1 or (block.char_stats.secondary_slot == 3 and (block.line_count() > 1 or is_punct_category(block.char_stats.tertiary_slot))):
return None
letter_val = letter_to_ordinal(first.str)
if letter_val is None:
return None
if second.str == "." or second.str == ")":
value = letter_val
else:
first_anchor = first_anchor_span(first)
second_anchor = first_anchor_span(second)
if (
not first.boundary_slot
or first_anchor is second_anchor
or second_anchor.left_edge() < first_anchor.right_edge() + first_anchor.bbox_width()
or heading_score(block) < page_scan.primary_slot.primary_slot.primary_slot + 1
or letter_count(block.char_stats) / tokens.length < 2
):
return None
value = letter_val
item_list: list[int] = [value]
prefix = tokens.slice(0, 2 if is_word_token(second) else 1)
rest = tokens.slice(prefix.length)
if second.str == "." and not second.boundary_slot and rest.length >= 2:
first_rest = first_token(rest)
if first_rest is not None and first_rest.type == 1:
heading = token_numeric_value(first_rest)
if math.isnan(heading) or heading <= 0 or heading >= 20:
return None
item_list.append(int(heading))
rest = rest.slice(1)
first_rest = first_token(rest)
if rest.length > 0 and first_rest is not None and is_word_token(first_rest):
rest = rest.slice(1)
if rest.length <= 0:
return None
prefix = tokens.slice(0, tokens.length - rest.length)
return make_heading_candidate(page_scan, 4, block, item_list, prefix, rest)
# --------------------------------------------------------------------------- #
# Chapter, appendix, and box-style dispatch.
# --------------------------------------------------------------------------- #
def detect_chapter_appendix(page_scan: PageScanState, other_block: Block) -> Optional[HeadingCandidate]:
""". Match "Chapter X" / "Appendix X" / box-N / etc."""
candidate_item = heading_score(other_block)
if candidate_item <= page_scan.primary_slot.primary_slot.primary_slot + 0.1:
return None
flag = (
other_block.isolated_centered or candidate_item > page_scan.secondary_slot.secondary_slot.primary_slot + 0.1
and (other_block.bold_frac() > 0.9 or is_upper_dominant(other_block.char_stats) or candidate_item > 1.5 * page_scan.secondary_slot.secondary_slot.primary_slot)
)
tokens = tokenize_block(other_block)
match = None
if flag:
match = trie_prefix_match(CHAPTER_WORDS_TRIE, tokens)
if flag and match is not None:
value = token_to_number(tokens.token_at(match.length))
if value is None:
return None
prefix = tokens.slice(0, skip_bracketed_word(tokens, match.length + 1))
return make_heading_candidate(page_scan, 8, other_block, [value], prefix, tokens.slice(prefix.length))
if flag:
match = trie_prefix_match(APPENDIX_SECTION_TRIE, tokens)
if match is not None:
prefix = tokens.slice(0, skip_bracketed_word(tokens, match.length))
return make_heading_candidate(page_scan, 9, other_block, [], prefix, tokens.slice(prefix.length))
match = trie_prefix_match(APPENDIX_KEYWORDS_TRIE, tokens)
if match is not None:
next_item = tokens.token_at(match.length)
val = token_to_number(next_item) or (letter_to_ordinal(next_item.str) if next_item is not None else None)
if not flag and val is None:
return None
item_list = [val] if val is not None else []
prefix = tokens.slice(0, skip_bracketed_word(tokens, match.length + (1 if val is not None else 0)))
return make_heading_candidate(page_scan, 10, other_block, item_list, prefix, tokens.slice(prefix.length))
return None
# --------------------------------------------------------------------------- #
# Box-format heading.
# --------------------------------------------------------------------------- #
def detect_box_heading(page_scan: PageScanState, other_block: Block) -> Optional[HeadingCandidate]:
"""match "Box N" pattern."""
tokens = tokenize_block(other_block)
match = trie_prefix_match(BOX_KEYWORD_TRIE, tokens)
if match is None:
return None
rest = tokens.slice(match.length)
if rest.length <= 0 or rest.token_at(0).type != 1:
return None
val = token_numeric_value(rest.token_at(0))
if math.isnan(val) or val <= 0:
return None
prefix = tokens.slice(0, skip_bracketed_word(tokens, match.length + 1))
return make_heading_candidate(page_scan, 12, other_block, [int(val)], prefix, tokens.slice(prefix.length))
# --------------------------------------------------------------------------- #
# Heading-type dispatcher.
# --------------------------------------------------------------------------- #
def classify_heading(page_scan: PageScanState, other_block: Block) -> HeadingCandidate:
""". Sequential dispatch through type detectors; fallback to the font-position classifier."""
tokens = tokenize_block(other_block)
heading = detect_chapter_appendix(page_scan, other_block)
if heading is None:
heading = detect_box_heading(page_scan, other_block)
if heading is None:
heading = detect_numbered_heading(page_scan, other_block, tokens)
if heading is None:
heading = detect_labeled_heading(page_scan, other_block, tokens)
if heading is not None:
return heading
type_code = 7 if matches_references(tokens) else (5 if matches_abstract(tokens) else 0)
return make_plain_candidate(page_scan, type_code, other_block)
# --------------------------------------------------------------------------- #
# Heading acceptance gate.
# --------------------------------------------------------------------------- #
def is_acceptable_heading(page_scan: PageScanState, other_heading_candidate: HeadingCandidate) -> bool:
""". The big "is this an acceptable heading?" gate."""
heading = other_heading_candidate.group_slot
if heading.bbox_height() >= 2 * heading.bbox_width() or info_weight(heading.char_stats) <= 3 or heading.line_count() > 5 or heading.char_count() >= 300:
return False
page_height = page_scan.primary_slot.bounds.bbox_height()
if heading.bottom_edge() > 0.95 * page_height:
return False
score = heading_score(heading)
doc_group = page_scan.secondary_slot.secondary_slot.primary_slot
if score <= page_scan.primary_slot.primary_slot.primary_slot + 0.5 and score <= doc_group + 0.5 and not other_heading_candidate.is_prominent:
return False
width = page_scan.primary_slot.bounds.bbox_width()
if (
(heading.left_edge() > 0.55 * width and score <= doc_group + 5)
or heading.left_edge() > 0.75 * width
or (
heading.left_edge() > 0.4 * width
and heading.center_x() > 0.6 * width
and page_scan.primary_slot.primary_slot.secondary_slot > min(1000, page_scan.secondary_slot.secondary_slot.secondary_slot)
)
):
return False
col_bottom = page_scan.primary_slot.tertiary_slot[safe_column_index(heading)] if 0 <= safe_column_index(heading) < len(page_scan.primary_slot.tertiary_slot) else None
if (
heading.bbox_width() < 0.2 * width and col_bottom is not None
and col_bottom.bbox_width() < 0.2 * width and col_bottom.bbox_height() > 1.5 * col_bottom.bbox_width()
):
return False
neighbor = neighbor_right(page_scan.tertiary_slot, heading)
gap = heading.bottom_edge() - neighbor.top_edge() if neighbor is not None else math.inf
line_gap = page_scan.primary_slot.primary_slot.tertiary_slot - page_scan.primary_slot.primary_slot.primary_slot
if gap < 0.9 * line_gap:
return False
above = neighbor_above(page_scan.tertiary_slot, heading)
above_gap = above.bottom_edge() - heading.top_edge() if above is not None else math.inf
if above_gap < 0.9 * line_gap:
return False
if (
page_scan.state_slot is not None
and not page_scan.state_slot.measure_slot
and (
page_scan.state_slot.primary_slot.secondary_slot < clamp(page_scan.secondary_slot.secondary_slot.secondary_slot, 200, 500)
or not page_scan.state_slot.state_slot
)
and score > doc_group + 0.5
):
return True
doc_right_neighbor = page_scan.secondary_slot.secondary_slot.measure_slot
previous_page_left_neighbor = page_scan.state_slot.primary_slot.option_slot if page_scan.state_slot is not None else math.nan
previous_page_height = page_scan.state_slot.bounds.bbox_height() if page_scan.state_slot is not None else math.nan
centered_flag = heading.isolated_centered
if (
previous_page_left_neighbor <= doc_right_neighbor
and (score <= doc_group + 1.5 or (score <= doc_group + 5 and not centered_flag))
or heading.weighted_ratio_primary < 0.5 * page_scan.secondary_slot.secondary_slot.auxiliary_slot
):
return False
body_neighbor = closest_body_neighbor_above(page_scan.tertiary_slot, heading)
# Reject candidates that are separated from a classified above-neighbor, or
# whose own content is more formula-like than heading-like.
if body_neighbor is not None and heading.bottom_edge() - body_neighbor.top_edge() > 2 * heading.bbox_height() and body_neighbor.marker_slot != 0:
return False
previous_block = page_scan.auxiliary_slot[heading.orig_index - 1] if 0 <= heading.orig_index - 1 < len(page_scan.auxiliary_slot) else None
next_block = page_scan.auxiliary_slot[heading.orig_index + 1] if 0 <= heading.orig_index + 1 < len(page_scan.auxiliary_slot) else None
if has_substantive_content(heading, previous_block, next_block):
return False
return (
(previous_page_left_neighbor > doc_right_neighbor + 0.1 * previous_page_height
and (score > doc_group + 2
or (previous_page_left_neighbor > doc_right_neighbor + 0.2 * previous_page_height
and body_neighbor is not None and neighbor is not None
and gap > neighbor.avg_font_size())))
or (centered_flag and (neighbor is None or neighbor.marker_slot == 0))
or score > 1.5 * doc_group
or (len(other_heading_candidate.numbering) == 1 and other_heading_candidate.numbering[0] == 1 and above is None)
)
def safe_column_index(block) -> int:
"""Safe wrapper for hh that handles missing H field."""
from ..stats import column_index_of
# Empty containers must return -1; returning 0 would index a real column.
return column_index_of(block)
# --------------------------------------------------------------------------- #
# Heading classification + acceptability gate #
# --------------------------------------------------------------------------- #
def try_classify_heading(page_scan: PageScanState, other_block: Block) -> Optional[HeadingCandidate]:
"""Try to build a candidate for a block, then apply rejection gates."""
candidate = classify_heading(page_scan, other_block)
if len(candidate.numbering) > 1:
return None
if candidate.type in (8, 9, 10):
return candidate
if candidate.type == 12:
return None
return candidate if is_acceptable_heading(page_scan, candidate) else None
# --------------------------------------------------------------------------- #
# Additional heading detectors and neighbor gates.
# --------------------------------------------------------------------------- #
def is_too_wide_for_heading(page_scan: PageScanState, other_block: Block) -> bool:
""". Block is too wide / central to be a heading."""
width = other_block.bbox_width()
if width > 0.7 * page_scan.primary_slot.bounds.bbox_width() / 2 or width > 0.7 * page_scan.secondary_slot.secondary_slot.option_slot:
return True
count = 0
for heading in range(page_scan.primary_slot.page_index - 1, page_scan.primary_slot.page_index + 2):
if 0 < heading <= len(page_scan.secondary_slot.primary_slot):
page = page_scan.secondary_slot.primary_slot[heading - 1]
if width > 0.7 * page.primary_slot.previous_slot:
count += 1
return count >= 2
def passes_neighbor_check(page_scan: PageScanState, other_block: Block) -> bool:
"""Block-level neighbor-aware acceptance gate. Returns True when the caller should reject the block."""
if is_too_wide_for_heading(page_scan, other_block):
return False
blocks = page_scan.auxiliary_slot
prev_idx = other_block.orig_index - 1
candidate_item = blocks[prev_idx] if 0 <= prev_idx < len(blocks) else None
overlap = candidate_item is not None and y_overlaps(other_block, candidate_item)
if overlap and is_too_wide_for_heading(page_scan, candidate_item):
return False
next_idx = other_block.orig_index + 1
candidate_item = blocks[next_idx] if 0 <= next_idx < len(blocks) else None
next_overlap = candidate_item is not None and y_overlaps(other_block, candidate_item)
if next_overlap and is_too_wide_for_heading(page_scan, candidate_item):
return False
if not overlap and not next_overlap:
return False
candidate_item = neighbor_right(page_scan.tertiary_slot, other_block)
if candidate_item is not None and is_too_wide_for_heading(page_scan, candidate_item):
return False
if candidate_item is not None and not candidate_item.is_body_paragraph and candidate_item.line_count() > 3 and candidate_item.bbox_height() > 0.8 * candidate_item.bbox_width():
return True
# Compare against the closest above-neighbor with a width threshold derived
# from this block's first line.
above_or_overlap = closest_body_neighbor_above(page_scan.tertiary_slot, other_block)
threshold = 4 * avg_char_width(other_block.line())
if (above_or_overlap is not None and candidate_item is not above_or_overlap
and x_aligned(other_block, above_or_overlap, threshold)
and above_or_overlap.state_slot == 0
and is_too_wide_for_heading(page_scan, above_or_overlap)):
return False
keyword_match = body_neighbor_above(page_scan.tertiary_slot, other_block)
if (keyword_match is not None
and x_aligned(other_block, keyword_match, threshold)
and keyword_match.state_slot == 0
and is_too_wide_for_heading(page_scan, keyword_match)):
return False
return True
def has_competing_labeled_heading(page_scan: PageScanState, other_heading_candidate: HeadingCandidate, candidate_block: Block) -> bool:
"""Cross-page reject check for competing labeled-heading siblings."""
if candidate_block.type != 0 or candidate_block.char_count() >= 500:
return False
block = other_heading_candidate.group_slot
if not similar_style(block, candidate_block) or abs(block.top_edge() - candidate_block.top_edge()) >= 5 * block.bbox_height():
return False
other_candidate = detect_labeled_heading(page_scan, candidate_block, tokenize_block(candidate_block))
if other_candidate is None or other_heading_candidate.type != other_candidate.type:
return False
return abs(other_candidate.numbering[0] - other_heading_candidate.numbering[0]) >= 1
def is_year_string(text: str) -> bool:
"""True iff the text parses to a plausible year (1700..2100)."""
value = to_number(text)
return not math.isnan(value) and 1700 < value < 2100
def is_bibliography_entry(block: Block, other_number: int = -1) -> bool:
"""True iff ``block`` looks like a bibliography entry."""
if other_number < 0:
other_number = 0
for line in block:
reference_item = numbering_value(line)
if not math.isnan(reference_item) and 0 < reference_item <= 9999:
other_number += 1
if other_number < 2 and block.char_count() / max(1, other_number) > 300:
return False
year = 0
digit = 0
word = 0
period_after_word = 0
word_state = 0
tokens = tokenize_block(block)
for entry in enumerate_tokens(tokens):
state_item = entry["token"]
if is_word_token(state_item):
if state_item.type == 3 and word_state == 1:
period_after_word += 1
word_state = 0
elif state_item.type == 1:
key_value = token_numeric_value(state_item)
if 0 < key_value < 1000:
digit += 1
elif is_year_string(state_item.str):
year += 1
elif state_item.type == 2:
word += 1
word_state += 1
if word < 0.1 * tokens.length:
return False
return digit >= 1.5 * other_number or year >= 0.5 * other_number or period_after_word >= 0.5 * other_number
@@ -0,0 +1,96 @@
"""Dictionary-backed keyword tries, keyword sets, and numbering tables."""
from __future__ import annotations
import json
import re
import regex as regex_module # Unicode \p{...} property classes.
from pathlib import Path
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_trim_unicode_ws,
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
)
from ..tokens import (
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
)
# --------------------------------------------------------------------------- #
# Dictionary tries (case-folded) #
# --------------------------------------------------------------------------- #
_DICT_PATH = Path(__file__).parent.parent / "data" / "dictionaries.json"
_DICTS = json.loads(_DICT_PATH.read_text(encoding="utf-8"))
SECTION_KEYWORDS_TRIE = build_trie(_DICTS.get("section_keywords", []), set_case_fold(TrieConfig(), True)) # general sections
ABSTRACT_KEYWORDS_TRIE = build_trie(_DICTS.get("abstract_keywords", []), set_case_fold(TrieConfig(), True)) # abstract
REFERENCES_TRIE = build_trie(_DICTS.get("references", []), set_case_fold(TrieConfig(), True)) # references
APPENDIX_SECTION_TRIE = build_trie(_DICTS.get("appendices_dict", []), set_case_fold(TrieConfig(), True)) # appendix
INTRODUCTION_SECTION_TRIE = build_trie(_DICTS.get("introduction_dict", []), set_case_fold(TrieConfig(), True)) # introduction
BOX_KEYWORD_TRIE = build_trie(["box"], set_case_fold(TrieConfig(), True))
KEYWORDS_SECTION_TRIE = build_trie(_DICTS.get("keywords_dict", []), set_case_fold(TrieConfig(), True)) # keywords
CHAPTER_WORDS_TRIE = build_trie(_DICTS.get("chapter_words", []), set_case_fold(TrieConfig(), True)) # chapter
APPENDIX_KEYWORDS_TRIE = build_trie(_DICTS.get("appendix_keywords", []), set_case_fold(TrieConfig(), True)) # appendix (hi)
# Whole-text lookup sets use normalized lowercase strings. The normalization is
# NFD -> strip combining marks (U+0300-U+036F) -> NFC; it is diacritic stripping,
# not compatibility folding.
def _normalize_text_key(text: str) -> str:
return _strip_diacritics(text)
# Whole-text lookup sets for abstract and references headings.
# Abstract headings are matched diacritic-insensitively; references are not.
ABSTRACT_KEYWORDS_SET = frozenset(_strip_diacritics(text_value.lower()) for text_value in _DICTS.get("abstract_keywords", []))
REFERENCES_SET = frozenset(text_value.lower() for text_value in _DICTS.get("references", []))
# Numbered heading prefix: leading ASCII/fullwidth 1-9, followed by Unicode
# numeric code points, punctuation, and whitespace or uppercase lookahead. The
# leading class deliberately excludes fullwidth zero (U+FF10).
NUMBERED_PREFIX_RE = regex_module.compile(r"^([1-91-9]\p{Number}*)[ .-](?:[" + _UNICODE_WHITESPACE_CLASS + r"]|\p{Lu})")
# Equation separator fallback. This intentionally matches only the literal
# string pattern around ``p{Number}``, so the branch remains inert for ordinary
# numeric text.
DEAD_DIGIT_RE = re.compile(r"^.p\{Number\}+.$")
# Trie of equation-like keywords ("equation", "eqn", "eq", plus multilingual
# variants).
EQUATION_KEYWORDS_TRIE = build_trie(
[
"equation", "equation.", "eqn", "eqn.", "eq", "eq.",
"ecuación", "equação", "gleichung", "equazione", "ekvation",
"yhtälö", "ligning", "persamaan", "denklem", "ecuația",
"equació", "rovnica", "rovnice", "równanie", "vergelijking",
"jednadžba", "jöfnu", "võrrand", "vienādojums", "lygtis",
"enačba", "egyenlet", "phương trình", "εξίσωση",
"方程", "방정식", "уравнение", "рівняння", "раўнанне", "једначина",
],
set_case_fold(TrieConfig(), True),
)
# Roman and English number words used by heading numbering detectors.
ENGLISH_WORD_TO_NUMBER = {
"one": 1, "two": 2, "three": 3, "four": 4, "five": 5, "six": 6,
"seven": 7, "eight": 8, "nine": 9, "ten": 10, "eleven": 11,
"twelve": 12, "thirteen": 13, "fourteen": 14, "fifteen": 15,
"sixteen": 16, "seventeen": 17, "eighteen": 18, "nineteen": 19, "twenty": 20,
}
ROMAN_NUMERAL_MAP = {
"I": 1, "II": 2, "III": 3, "IV": 4, "V": 5, "VI": 6, "VII": 7,
"VIII": 8, "IX": 9, "X": 10, "XI": 11, "XII": 12, "XIII": 13,
"XIV": 14, "XV": 15, "XVI": 16, "XVII": 17, "XVIII": 18, "XIX": 19, "XX": 20,
}
# Special-character weights used by equation-content scoring.
FORMULA_CHAR_WEIGHTS = {
"=": 10, "{": 5, "}": 5, "+": 5, "/": 3, "*": 3,
"-": 1, "~": 1, "[": 1, "]": 1, "(": 1, ")": 1,
}
@@ -0,0 +1,155 @@
"""Per-page block neighborhood maps and neighbor lookups."""
from __future__ import annotations
import math
from typing import Any, Optional
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_trim_unicode_ws,
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
)
from .text_checks import clamp
# --------------------------------------------------------------------------- #
# Per-page neighbor map.
# --------------------------------------------------------------------------- #
class BlockNeighborCache:
"""Per-block neighbor cache populated by the page neighbor map."""
__slots__ = ("state_slot", "tertiary_slot", "measure_slot", "auxiliary_slot", "primary_slot", "secondary_slot", "option_slot")
def __init__(self):
self.state_slot = False # initialized flag
self.tertiary_slot = None # closest body block below
self.measure_slot = None # block 1-column-left
self.auxiliary_slot = None # next block to the right
self.primary_slot = None # earlier body block above
self.secondary_slot = None # nearest body block above
self.option_slot = None # block 1-column-right peer
def compute_bucket_span(neighbor_map, block) -> dict:
"""Compute the inclusive horizontal bucket span for a block."""
start_bucket = int(clamp(math.floor(block.left_edge() / neighbor_map.tertiary_slot), 0, neighbor_map.secondary_slot - 1))
end_bucket = int(clamp(math.ceil(block.right_edge() / neighbor_map.tertiary_slot), 0, neighbor_map.secondary_slot - 1))
return {"start_bucket": start_bucket, "end_bucket": end_bucket}
def neighbor_above(neighbor_map, other_block: Block) -> Optional[Block]:
"""closest 'j' neighbor (block above)."""
width_value = neighbor_map.primary_slot[other_block.orig_index] if other_block.orig_index < len(neighbor_map.primary_slot) else None
return width_value.tertiary_slot if width_value is not None else None
def body_neighbor_above(neighbor_map, other_block: Block) -> Optional[Block]:
"""closest 'g' neighbor."""
width_value = neighbor_map.primary_slot[other_block.orig_index] if other_block.orig_index < len(neighbor_map.primary_slot) else None
return width_value.primary_slot if width_value is not None else None
def neighbor_right(neighbor_map, other_block: Block) -> Optional[Block]:
"""Closest right-side peer neighbor."""
width_value = neighbor_map.primary_slot[other_block.orig_index] if other_block.orig_index < len(neighbor_map.primary_slot) else None
return width_value.auxiliary_slot if width_value is not None else None
def neighbor_right_peer(neighbor_map, secondary_item):
return neighbor_right(neighbor_map, secondary_item)
def closest_body_neighbor_above(neighbor_map, other_block: Block) -> Optional[Block]:
"""Closest stored neighbor above."""
width_value = neighbor_map.primary_slot[other_block.orig_index] if other_block.orig_index < len(neighbor_map.primary_slot) else None
return width_value.secondary_slot if width_value is not None else None
class PageNeighborMap:
"""Per-page horizontal-bucket neighbor map for constant-time nearby-block queries."""
__slots__ = ("tertiary_slot", "secondary_slot", "primary_slot")
def __init__(self, page):
blocks = page.output_slot
self.tertiary_slot = max(5, page.bounds.bbox_width() / 300) # bucket width
self.secondary_slot = int(math.floor(page.bounds.bbox_width() / self.tertiary_slot)) # bucket count
self.primary_slot: list[Optional[BlockNeighborCache]] = [None] * (max(len(blocks), 1) + 1)
# mark buckets crossed by body-marked blocks
marked = [False] * self.secondary_slot
for candidate_item in blocks:
if not candidate_item.is_body_paragraph:
continue
spans = compute_bucket_span(self, candidate_item)
for index in range(spans["start_bucket"], spans["end_bucket"]):
if 0 <= index < self.secondary_slot:
marked[index] = True
recent_height: list[int] = [-1] * self.secondary_slot # most recent body block height at bucket
# Reads past the end of this list must behave like an unset slot: -1 is
# falsy at the ``>= 0`` tests below just as a missing entry is, and a
# write to it extends the list. On a degenerate page with zero buckets
# every clamped index is 0, so one slot reproduces that growth.
recent_block_index: list[int] = [-1] * max(self.secondary_slot, 1) # most-recent block V (j-direction)
recent: list[Optional[Block]] = [None] * self.secondary_slot # most-recent block at bucket
pending: list[list[int]] = [[] for _ in range(self.secondary_slot)] # pending V's per bucket
for block_index, current_block in enumerate(blocks):
if (
current_block.char_count() <= 0
or current_block.skew_frac() > 1
or current_block.type in (1, 2, 12)
):
continue
width = BlockNeighborCache()
self.primary_slot[current_block.orig_index] = width
spans = compute_bucket_span(self, current_block)
left = spans["start_bucket"]
right = spans["end_bucket"]
# H field: block 1-column-left or right
value = recent_block_index[left]
adjacent_bucket_index = recent_block_index[
left - 1 if (left > 0 and current_block.left_edge() < (left + 0.5) * self.tertiary_slot)
else (left + 1 if left < self.secondary_slot - 1 else left)
]
if value >= 0 or adjacent_bucket_index >= 0:
same_bucket_block = blocks[value] if 0 <= value < len(blocks) else None
adjacent_bucket_block = blocks[adjacent_bucket_index] if 0 <= adjacent_bucket_index < len(blocks) else None
if same_bucket_block is not None and (adjacent_bucket_block is None or same_bucket_block.bottom_edge() < adjacent_bucket_block.bottom_edge()):
picked_index = value
else:
picked_index = adjacent_bucket_index
if 0 <= picked_index < len(blocks):
width.measure_slot = blocks[picked_index]
if self.primary_slot[picked_index] is not None:
self.primary_slot[picked_index].option_slot = current_block
recent_block_index[left] = current_block.orig_index
for col in range(left, right):
if 0 <= col < self.secondary_slot:
width.state_slot = width.state_slot or marked[col]
recent_block = recent[col]
if recent_block is not None and (width.primary_slot is None or recent_block.bottom_edge() < width.primary_slot.bottom_edge()):
width.primary_slot = recent_block
previous_body_index = recent_height[col]
recent_height[col] = current_block.orig_index
if previous_body_index >= 0 and previous_body_index < len(blocks):
same_bucket_block = blocks[previous_body_index]
if width.tertiary_slot is None or same_bucket_block.bottom_edge() < width.tertiary_slot.bottom_edge():
width.tertiary_slot = same_bucket_block
previous_cache = self.primary_slot[previous_body_index]
if previous_cache is not None and (previous_cache.auxiliary_slot is None or current_block.top_edge() > previous_cache.auxiliary_slot.top_edge()):
previous_cache.auxiliary_slot = current_block
if current_block.is_body_paragraph:
for pending_index in pending[col]:
if 0 <= pending_index < len(self.primary_slot):
pending_neighbor = self.primary_slot[pending_index]
if pending_neighbor is not None and (pending_neighbor.secondary_slot is None or current_block.top_edge() > pending_neighbor.secondary_slot.top_edge()):
pending_neighbor.secondary_slot = current_block
recent[col] = current_block
pending[col].clear()
pending[col].append(current_block.orig_index)
@@ -0,0 +1,474 @@
"""Whole-page heading scan and document-level candidate collection/filtering."""
from __future__ import annotations
import math
from typing import Any, Optional
from ..outline_assembly import HeadingCandidate, OutlineNode
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_trim_unicode_ws,
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
)
from ..tokens import (
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
)
from .keyword_tables import (
SECTION_KEYWORDS_TRIE,
INTRODUCTION_SECTION_TRIE,
)
from .text_checks import (
is_heading_continuation,
matches_abstract,
matches_references,
has_substantive_content,
is_cover_page,
)
from .neighbors import (
neighbor_above,
neighbor_right,
closest_body_neighbor_above,
)
from .candidates import (
PageScanState,
push_candidate,
make_plain_candidate,
make_body_heading_candidate,
)
from .detectors import (
detect_numbered_heading,
detect_labeled_heading,
detect_chapter_appendix,
try_classify_heading,
passes_neighbor_check,
has_competing_labeled_heading,
)
from .style_detectors import (
detect_font_heading,
detect_heading_with_body,
)
# --------------------------------------------------------------------------- #
# Main per-page heading scan #
# --------------------------------------------------------------------------- #
def scan_page_headings(page_scan: PageScanState) -> list[HeadingCandidate]:
"""Return heading candidates found on this page."""
if is_cover_page(page_scan.secondary_slot, page_scan.primary_slot):
return []
page_scan.option_slot.clear()
page_scan.measure_slot.clear()
blocks = page_scan.auxiliary_slot
for block in blocks:
if block.char_count() <= 0 or block.skew_frac() > 1 or block.type != 0:
continue
if block.state_slot != 0: # already classified
continue
above = neighbor_above(page_scan.tertiary_slot, block)
if above is not None and above.secondary_slot.contains(block.secondary_slot):
continue
if block.char_count() <= 1 and block.char_stats.secondary_slot != 4:
continue
wn_entry = page_scan.tertiary_slot.primary_slot[block.orig_index] if 0 <= block.orig_index < len(page_scan.tertiary_slot.primary_slot) else None
has_da_above = wn_entry is not None and wn_entry.state_slot
rows = block.line_count()
# Try lo for 2-line heading-body patterns
if has_da_above and rows > 1 and (
(0 < block.bold_frac() < 1)
or style_key(first_span_of(block)) != style_key(last_span(last_line_of(block)))
):
lo_result = detect_heading_with_body(page_scan, block)
if lo_result is not None:
push_candidate(page_scan, lo_result)
continue
tokens = tokenize_block(block)
first_line = block.line()
first_line_tokens = tokens.slice(0, advance_past_line(tokens, first_line, 0))
if has_da_above and not block.measure_slot and rows >= 3\
and first_line.bbox_width() <= 0.2 * min(block.primary_slot[1].bbox_width(), block.primary_slot[2].bbox_width())\
and matches_abstract(first_line_tokens):
push_candidate(page_scan, make_body_heading_candidate(page_scan, 5, block, first_line_tokens))
continue
if block.char_count() >= 200:
continue
if block.char_count() >= 100 and rows > 1 and block.char_stats.primary_slot[6] - first_line.char_stats.primary_slot[6] > 1:
continue
size = block.avg_font_size()
page_width = page_scan.primary_slot.bounds.bbox_width()
lots_caps = block.char_stats.primary_slot[2] >= max(3, letter_count(block.char_stats) / 2)
if rows > 4 or (rows >= 3 and not (size >= 1.5 * page_scan.primary_slot.primary_slot.primary_slot or lots_caps)):
continue
if block.weighted_ratio_primary < 0.5 * page_scan.secondary_slot.secondary_slot.auxiliary_slot:
continue
if letter_count(block.char_stats) <= 0:
continue
if block.bold_frac() < 0.1 and not lots_caps and size < page_scan.secondary_slot.secondary_slot.primary_slot - 2:
continue
layout_gate = detect_chapter_appendix(page_scan, block)
if layout_gate is not None:
push_candidate(page_scan, layout_gate)
continue
if passes_neighbor_check(page_scan, block):
continue
top_gap = above.bottom_edge() - block.top_edge() if above is not None else math.inf
isolated = (
not block.measure_slot and rows <= 2
and (above is None or top_gap > 1.5 * block.avg_font_size()
or (above.state_slot != 5 and above.state_slot != 11))
)
if isolated:
if matches_references(tokens):
push_candidate(page_scan, make_plain_candidate(page_scan, 7, block))
continue
if has_da_above and trie_matches_all(INTRODUCTION_SECTION_TRIE, tokens):
push_candidate(page_scan, make_plain_candidate(page_scan, 11, block))
continue
above_index = block.orig_index - 1
below_index = block.orig_index + 1
previous_block = blocks[above_index] if 0 <= above_index < len(blocks) else None
below = blocks[below_index] if 0 <= below_index < len(blocks) else None
if not has_da_above and (
not (heading_score(block) >= page_scan.primary_slot.primary_slot.primary_slot + 1.5)
or (above is not None and above.type != 1)
or (previous_block is not None and previous_block.type != 1)
or (below is not None and not (below.top_edge() < block.bottom_edge() - size))
):
continue
if block.bold_frac() < 0.1 and not lots_caps\
and first_span_of(block).font_name == page_scan.primary_slot.primary_slot.state_slot\
and size < page_scan.secondary_slot.secondary_slot.primary_slot - 0.5:
continue
predecessor = neighbor_right(page_scan.tertiary_slot, block)
if predecessor is not None and predecessor.skew_frac() > 1:
continue
if top_gap < 0:
continue
predecessor_gap = block.bottom_edge() - predecessor.top_edge() if predecessor is not None else math.inf
if predecessor_gap < -0.9 * block.bbox_height():
continue
line_gap = page_scan.primary_slot.primary_slot.tertiary_slot - page_scan.primary_slot.primary_slot.primary_slot
if top_gap < line_gap and size < page_scan.primary_slot.primary_slot.primary_slot - 1:
continue
# Sibling/peer block pointers from the neighbor cache.
right_neighbor_sib = wn_entry.measure_slot if wn_entry is not None else None
left_neighbor_sib = wn_entry.option_slot if wn_entry is not None else None
heading_kind = detect_numbered_heading(page_scan, block, tokens)
if heading_kind is not None:
from ..stats import column_index_of as _column_index
col_idx = _column_index(block) if block.primary_slot else 0
column_rect = page_scan.primary_slot.tertiary_slot[col_idx] if 0 <= col_idx < len(page_scan.primary_slot.tertiary_slot) else None
# Narrow-column heading-vs-prev-numbering check
if (size < page_scan.secondary_slot.secondary_slot.primary_slot - 1
and block.bbox_width() < 0.2 * page_width
and column_rect is not None
and column_rect.bbox_width() < 0.2 * page_width
and column_rect.bbox_height() > 1.5 * column_rect.bbox_width()):
first_number = heading_kind.numbering[0] if heading_kind.numbering else 0
if above is not None:
first = first_token(tokenize_block(above))
if first is not None and first.type == 1 and token_numeric_value(first) != first_number:
continue
if predecessor is not None:
predecessor_first_token = first_token(tokenize_block(predecessor))
if predecessor_first_token is not None and predecessor_first_token.type == 1 and token_numeric_value(predecessor_first_token) != first_number:
continue
# Prev-block continuation check via Kn (numbered-sequence test)
first_token_value = tokens.token_at(0)
second_token_value = tokens.token_at(1) if len(tokens) > 1 else None
if len(heading_kind.numbering) <= 1 and (
(first_token_value is not None and first_token_value.boundary_slot)
or (second_token_value is not None and is_word_token(second_token_value))):
first_number = heading_kind.numbering[0] if heading_kind.numbering else 0
if above is not None and is_heading_continuation(above, block, first_number):
continue
if (right_neighbor_sib is not None and above is not right_neighbor_sib
and not center_aligned(block, right_neighbor_sib, 1)
and (above.line_count() < 10 or above.char_count() < 300)
and is_heading_continuation(right_neighbor_sib, block, first_number)):
continue
if predecessor is not None and is_heading_continuation(predecessor, block, first_number):
continue
if (left_neighbor_sib is not None and predecessor is not left_neighbor_sib
and not center_aligned(block, left_neighbor_sib, 1)
and (predecessor.line_count() < 10 or predecessor.char_count() < 300)
and is_heading_continuation(left_neighbor_sib, block, first_number)):
continue
# Top-of-page small-font footnote-marker rejection
second = tokens.token_at(1) if len(tokens) > 1 else None
if (block.top_edge() < page_scan.primary_slot.bounds.bbox_height() / 4
and size <= page_scan.primary_slot.primary_slot.primary_slot
and first_span_of(block).char_stats.secondary_slot == 1
and first_span_of(block).bbox_height() < size - 0.5
and len(heading_kind.numbering) <= 1
and second is not None and is_char_token(second)):
continue
push_candidate(page_scan, heading_kind)
continue
if block.measure_slot:
continue
if block.char_count() >= 120:
continue
if top_gap <= line_gap - 0.1:
continue
heading_signature = detect_labeled_heading(page_scan, block, tokens)
if heading_signature is not None:
if above is not None and has_competing_labeled_heading(page_scan, heading_signature, above):
continue
if (right_neighbor_sib is not None and above is not right_neighbor_sib
and has_competing_labeled_heading(page_scan, heading_signature, right_neighbor_sib)):
continue
if predecessor is not None and has_competing_labeled_heading(page_scan, heading_signature, predecessor):
continue
if (left_neighbor_sib is not None and predecessor is not left_neighbor_sib
and has_competing_labeled_heading(page_scan, heading_signature, left_neighbor_sib)):
continue
push_candidate(page_scan, heading_signature)
continue
if isolated:
caps_heavy = size + 2 * block.bold_frac() >= page_scan.secondary_slot.secondary_slot.primary_slot + 4 or is_upper_dominant(block.char_stats)
above_or_overlap = closest_body_neighbor_above(page_scan.tertiary_slot, block)
page_width = page_scan.primary_slot.bounds.bbox_width()
# Abstract-heading acceptance uses the closest above-overlap block
# as the guard; when it exists, the predecessor exists too.
type5_cond = caps_heavy or (
above_or_overlap is not None
and (block.bottom_edge() - above_or_overlap.top_edge() < 3 * (block.bottom_edge() - predecessor.top_edge())
or info_weight(predecessor.char_stats) >= 30)
)
if type5_cond and matches_abstract(tokens):
push_candidate(page_scan, make_plain_candidate(page_scan, 5, block))
continue
type6_cond = (
caps_heavy
or (above is not None and x_aligned(block, above, 1)
and (above.bbox_width() >= page_width / 6
or above.bold_frac() > 0.9
or is_upper_dominant(above.char_stats)))
or (predecessor is not None and x_aligned(block, predecessor, 1)
and (predecessor.bbox_width() >= page_width / 6
or predecessor.bold_frac() > 0.9
or is_upper_dominant(predecessor.char_stats)
or (block.previous_slot < 0.1 and predecessor.previous_slot > 0.9)))
)
if type6_cond and trie_full_match(SECTION_KEYWORDS_TRIE, tokens):
push_candidate(page_scan, make_plain_candidate(page_scan, 6, block))
continue
if info_weight(block.char_stats) <= 3:
continue
if has_substantive_content(block, previous_block, below):
continue
outline_context = detect_font_heading(page_scan, block)
if outline_context is not None:
push_candidate(page_scan, outline_context)
return page_scan.option_slot
# --------------------------------------------------------------------------- #
# Document-wide outline-candidate collector #
# --------------------------------------------------------------------------- #
class DocCandidateCollector:
"""Document-level state aggregating per-page heading candidates."""
__slots__ = ("previous_slot", "measure_slot", "option_slot", "auxiliary_slot", "primary_slot", "tertiary_slot", "secondary_slot", "state_slot")
def __init__(self, doc, labeled):
self.previous_slot = doc
self.measure_slot = labeled
self.option_slot = 0
self.auxiliary_slot = False
self.primary_slot = 0
self.secondary_slot = False
self.tertiary_slot = False
self.state_slot: list[HeadingCandidate] = []
def filter_page_candidates(doc_collector: DocCandidateCollector, page, page_candidates: list[HeadingCandidate]) -> None:
"""Per-page candidate filter for noisy pages, title overlap, page headers, and numbering continuity."""
from ..outline_assembly import is_script_compatible
from ..model import intervals_overlap, is_caps_heavy
from ..stats import column_index_of
# Advance document-level numbering state through outline entries up to this page.
while doc_collector.option_slot < len(doc_collector.measure_slot):
outline_entry = doc_collector.measure_slot[doc_collector.option_slot]
current_candidate = outline_entry.heading
if current_candidate.page.page_index > page.page_index:
break
if current_candidate.type == 2:
if not doc_collector.tertiary_slot:
doc_collector.tertiary_slot = (len(current_candidate.numbering) > 0 and current_candidate.numbering[0] == 1)
elif current_candidate.type == 4:
if not doc_collector.secondary_slot:
doc_collector.secondary_slot = (len(current_candidate.numbering) > 0 and current_candidate.numbering[0] == 1)
elif current_candidate.type == 1 and len(current_candidate.numbering) > 0:
doc_collector.primary_slot = max(doc_collector.primary_slot, current_candidate.numbering[0])
doc_collector.option_slot += 1
count = len(page_candidates)
if count >= 20:
return
# Sort candidates by column, vertical position, then horizontal position.
def _ih_key(heading_candidate: HeadingCandidate):
first_line = heading_candidate.group_slot.primary_slot[0] if heading_candidate.group_slot.primary_slot else None
column_index = first_line.measure_slot if first_line is not None else -1
return (column_index, -heading_candidate.group_slot.top_edge(), -heading_candidate.group_slot.bottom_edge(), heading_candidate.group_slot.left_edge(), heading_candidate.group_slot.right_edge())
page_candidates.sort(key=_ih_key)
accepted: list[HeadingCandidate] = []
title: Optional[Block] = None
if not doc_collector.auxiliary_slot and getattr(page, "auxiliary_slot", False):
for block in page.output_slot:
if block.type == 3:
title = block
break
min_first_number = math.inf
total_bottom = math.inf
single_numbering_count = 0
for page_candidate in page_candidates:
if total_bottom == math.inf and page_candidate.type == 5:
total_bottom = page_candidate.group_slot.top_edge() + page_candidate.group_slot.avg_font_size()
if page_candidate.type == 1 and len(page_candidate.numbering) > 0:
min_first_number = min(min_first_number, page_candidate.numbering[0])
if len(page_candidate.numbering) == 1:
single_numbering_count += 1
max_first_number = 0
for index in range(count):
active_candidate = page_candidates[index]
next_item = page_candidates[index + 1] if index + 1 < count else None
if is_script_compatible(doc_collector.previous_slot.secondary_slot.tertiary_slot, active_candidate):
continue
if not doc_collector.auxiliary_slot and active_candidate.type != 11:
bottom = active_candidate.group_slot.top_edge()
if (title is not None and bottom > title.top_edge()
and intervals_overlap(active_candidate.group_slot.left_edge(), active_candidate.group_slot.right_edge(), title.left_edge(), title.right_edge())):
continue
if bottom > total_bottom:
continue
if (active_candidate.type == 0 and next_item is not None and next_item.type == 5
and active_candidate.group_slot.left_edge() <= next_item.group_slot.right_edge() and active_candidate.group_slot.right_edge() >= next_item.group_slot.left_edge()
and active_candidate.group_slot.bottom_edge() - next_item.group_slot.top_edge() < 2 * active_candidate.group_slot.bbox_height()
and len(tokenize_block(active_candidate.group_slot)) > 1):
continue
if active_candidate.type == 2:
if len(active_candidate.numbering) > 0 and active_candidate.numbering[0] == 1:
doc_collector.tertiary_slot = True
elif not doc_collector.tertiary_slot:
continue
accepted.append(active_candidate)
continue
if active_candidate.type == 4:
if len(active_candidate.numbering) > 0 and active_candidate.numbering[0] == 1:
doc_collector.secondary_slot = True
elif not doc_collector.secondary_slot:
continue
accepted.append(active_candidate)
continue
if active_candidate.type != 1:
accepted.append(active_candidate)
continue
# type == 1
if single_numbering_count >= 5:
continue
first_number = active_candidate.numbering[0] if len(active_candidate.numbering) > 0 else 0
if (active_candidate.group_slot.bold_frac() < 0.9 and not is_caps_heavy(active_candidate.group_slot)
and active_candidate.group_slot.avg_font_size() < page.primary_slot.primary_slot + 1):
if doc_collector.primary_slot <= 0 and min_first_number > 1 and len(active_candidate.numbering) <= 1:
continue
if first_number > 3 * page.page_index:
continue
max_first_number = max(max_first_number, first_number)
accepted.append(active_candidate)
doc_collector.state_slot.extend(accepted)
doc_collector.auxiliary_slot = True
doc_collector.primary_slot = max(doc_collector.primary_slot, max_first_number)
# --------------------------------------------------------------------------- #
# Public entry point: build heading candidates for the whole document #
# --------------------------------------------------------------------------- #
def build_doc_heading_candidates(doc, labeled: Optional[list] = None) -> list[HeadingCandidate]:
"""Run per-page heading detection across the document."""
doc_collector = DocCandidateCollector(doc, labeled if labeled is not None else [])
saw_body = False
for page in doc.primary_slot:
# Skip initial cover-like pages until the first body-like page is reached.
from ..title import is_cover_like_page
if not saw_body and is_cover_like_page(doc, page):
continue
saw_body = True
page_vo = PageScanState(doc, page)
page_candidates = scan_page_headings(page_vo)
filter_page_candidates(doc_collector, page, page_candidates)
return doc_collector.state_slot
def find_section_openers(doc, start_page_idx: int) -> list:
"""Find the first valid heading on each page, then clique-filter the result."""
from ..outline_assembly import is_script_compatible, has_conflict_in_context, OutlineContext, OutlineNode
item_list: list[HeadingCandidate] = []
index = start_page_idx
while index < len(doc.primary_slot):
page = doc.primary_slot[index]
current_candidate: Optional[HeadingCandidate] = None
page_scan_state = PageScanState(doc, page)
if not page_scan_state.primary_slot.auxiliary_slot:
for block in page_scan_state.auxiliary_slot:
if block.char_count() <= 0 or block.skew_frac() > 1 or block.type != 0:
continue
if block.is_body_paragraph or block.top_edge() < 0.5 * page_scan_state.primary_slot.bounds.bbox_height():
break
if block.marker_slot != 0:
break
detected_candidate = try_classify_heading(page_scan_state, block)
if detected_candidate is not None:
current_candidate = detected_candidate
break
if block.line_count() > 2:
break
if current_candidate is not None and not is_script_compatible(doc.secondary_slot.tertiary_slot, current_candidate):
item_list.append(current_candidate)
index += 1
if len(item_list) <= 1:
return []
# Compare candidates against the full context and against the accepted subset.
bundle = OutlineContext(item_list)
accepted_context = OutlineContext([])
out = []
for current_candidate in item_list:
if accepted_context.has_nearby_duplicate(current_candidate):
current_candidate.group_slot.type = 12
continue
if has_conflict_in_context(bundle, current_candidate):
continue
accepted_context.add(current_candidate)
out.append(OutlineNode(current_candidate))
current_candidate.group_slot.type = 7
current_candidate.group_slot.used_as_heading = True
return out
@@ -0,0 +1,383 @@
"""Font-change and body-embedded heading detectors."""
from __future__ import annotations
import math
from typing import Any, Optional
from ..outline_assembly import HeadingCandidate, OutlineNode
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_trim_unicode_ws,
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
)
from ..tokens import (
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
)
from .keyword_tables import (
SECTION_KEYWORDS_TRIE,
INTRODUCTION_SECTION_TRIE,
KEYWORDS_SECTION_TRIE,
)
from .text_checks import (
matches_abstract,
vertically_close,
)
from .neighbors import (
neighbor_above,
body_neighbor_above,
neighbor_right,
closest_body_neighbor_above,
)
from .candidates import (
PageScanState,
make_heading_candidate,
make_plain_candidate,
make_body_heading_candidate,
)
from .detectors import (
detect_numbered_heading,
detect_labeled_heading,
is_bibliography_entry,
)
def detect_font_heading(page_scan: PageScanState, other_block: Block) -> Optional[HeadingCandidate]:
"""Detailed font/position-based fallback heading classifier."""
from ..model import x_aligned, last_span, last_line_of, first_span_of, letter_count, punct_count, dominant_style_of, is_upper_dominant, is_caps_heavy, is_sentence_like, alignment_code
from ..tokens import last_token, is_comma_token
above = neighbor_above(page_scan.tertiary_slot, other_block)
top_gap = above.bottom_edge() - other_block.top_edge() if above is not None else math.inf
predecessor = neighbor_right(page_scan.tertiary_slot, other_block)
predecessor_gap = other_block.bottom_edge() - predecessor.top_edge() if predecessor is not None else math.inf
keyword_match = body_neighbor_above(page_scan.tertiary_slot, other_block)
above_or_overlap = closest_body_neighbor_above(page_scan.tertiary_slot, other_block)
# Initial gate: one of On OR bold/centered tall block.
if not (
vertically_close(keyword_match, other_block) or vertically_close(above_or_overlap, other_block)
or (other_block.bottom_edge() >= 0.8 * page_scan.primary_slot.bounds.bbox_height()
and other_block.avg_font_size() >= page_scan.secondary_slot.secondary_slot.primary_slot + 1
and other_block.bold_frac() > 0.9
and (above is None or above.type == 1))
):
return None
page = page_scan.primary_slot.primary_slot # page statistics
far = 10 * min(other_block.avg_font_size(), page.tertiary_slot)
if top_gap < math.inf and top_gap > far and above.state_slot == 0:
return None
if predecessor is not None and predecessor.state_slot != 0:
return None
# Compound rejection for candidates sitting above non-body predecessors.
# Keep the explicit short-circuit structure: each inner predicate requires
# the predecessor to exist.
inner_reject = False
if above is not None and above.state_slot != 0:
inner_reject = (
predecessor_gap > 5 * page.tertiary_slot
or (predecessor is not None and not predecessor.is_body_paragraph)
or (predecessor is not None and predecessor.bbox_width() < page_scan.primary_slot.bounds.bbox_width() / 5)
or (predecessor is not None and predecessor.char_count() < 0.5 * other_block.char_count())
or (predecessor is not None and predecessor.weighted_ratio_secondary < 0.33)
or (predecessor is not None and predecessor.char_count() < 500
and predecessor.weighted_ratio_secondary < 0.5 and alignment_code(predecessor) != 1)
or (predecessor is not None and predecessor.char_count() < 250 and predecessor.weighted_ratio_secondary < 0.5)
)
if (inner_reject
or (predecessor is not None and (
predecessor.weighted_ratio_primary < 0.67 * page_scan.secondary_slot.secondary_slot.auxiliary_slot
or (other_block.char_count() < 30 and predecessor.char_count() < 300
and predecessor.weighted_ratio_primary < 0.8 * page_scan.secondary_slot.secondary_slot.auxiliary_slot)))):
return None
last_tok = last_token(tokenize_block(other_block))
if last_tok is not None and is_comma_token(last_tok):
return None
# Branch 1: tall first-line + big-font heading
if (predecessor_gap < math.inf and predecessor_gap > 0
and other_block.style_slot >= page_scan.secondary_slot.secondary_slot.primary_slot + 2
and other_block.avg_font_size() >= page_scan.secondary_slot.secondary_slot.primary_slot + 1.5
and other_block.avg_font_size() >= page.primary_slot + 0.5
and (above is None or (other_block.style_slot >= above.style_slot and other_block.avg_font_size() >= above.avg_font_size()))
and predecessor is not None
and other_block.style_slot >= predecessor.style_slot and other_block.avg_font_size() >= predecessor.avg_font_size()):
return make_plain_candidate(page_scan, 0, other_block)
caps_heavy = is_caps_heavy(other_block)
# Branch 2 reject: matches body-style and not all-caps, OR clearly
# smaller font than predecessor near it.
if ((dominant_style_of(other_block) in page_scan.primary_slot.style_slot and not caps_heavy
and (page.auxiliary_slot == dominant_style_of(other_block)
or (other_block.bold_frac() < 0.9 and other_block.previous_slot < 0.9
and alignment_code(other_block) != 3 and not is_sentence_like(other_block))))
or (above is not None and predecessor is not None
and other_block.avg_font_size() <= predecessor.avg_font_size()
and top_gap < predecessor_gap / 4)):
return None
# Branch 3: medium-confidence font-size heading
if (predecessor_gap < math.inf and predecessor_gap > 0
and other_block.avg_font_size() >= page_scan.secondary_slot.secondary_slot.primary_slot + 0.5
and predecessor is not None
and other_block.style_slot >= predecessor.style_slot and other_block.avg_font_size() >= predecessor.avg_font_size()
and predecessor.avg_font_size() >= page.primary_slot - 0.5
and other_block.bbox_width() < 0.95 * predecessor.bbox_width()
and predecessor.bbox_width() >= 0.25 * page_scan.primary_slot.bounds.bbox_width()):
return make_plain_candidate(page_scan, 0, other_block)
line_height = page.tertiary_slot - page.primary_slot
# Branch 4: moderate-gap large-font heading
if (predecessor_gap > line_height and predecessor_gap < 5 * line_height
and (above is None or other_block.avg_font_size() >= above.avg_font_size() + 0.5)
and predecessor is not None
and other_block.avg_font_size() >= predecessor.avg_font_size() + 0.5
and other_block.avg_font_size() >= page_scan.secondary_slot.secondary_slot.primary_slot - 0.5
and other_block.bbox_width() < 0.95 * predecessor.bbox_width()
and predecessor.is_body_paragraph
and predecessor.avg_font_size() >= page.primary_slot - 0.5
and predecessor.char_stats.secondary_slot != 1):
return make_plain_candidate(page_scan, 0, other_block)
# Reject: many letters with low density signals body para
letters = other_block.char_stats.primary_slot[6]
if other_block.char_stats.primary_slot[10] != 0:
ratio = (letters + other_block.char_stats.primary_slot[8]) / other_block.char_stats.primary_slot[10]
else:
# IEEE division edge case: positive numerator over zero behaves as +inf,
# which keeps the low-density rejection active.
ratio = math.inf if (letters + other_block.char_stats.primary_slot[8]) > 0 else math.nan
if letters > 1 and ratio > 0.3:
return None
symbol_count = punct_count(other_block.char_stats)
letter_total = letter_count(other_block.char_stats)
# Same IEEE division edge case as the letter-density ratio above.
symbol_ratio = symbol_count / letter_total if letter_total != 0 else (math.inf if symbol_count > 0 else math.nan)
if (symbol_count >= 5 and symbol_ratio > 0.2
or top_gap < 0.2 * other_block.avg_font_size()
or top_gap < min(other_block.avg_font_size(), 0.7 * predecessor_gap)):
return None
centered = other_block.char_stats.secondary_slot == 2
neg = -0.2 * last_span(last_line_of(other_block)).bbox_height() if caps_heavy else 0
# Branch A: tight criteria with neighbor analysis
neighbor_heading_cue = (
predecessor_gap < math.inf and predecessor_gap > neg
and other_block.avg_font_size() >= page.primary_slot - 0.1
and predecessor is not None and other_block.avg_font_size() >= predecessor.avg_font_size() - 0.1
and other_block.avg_font_size() >= page_scan.secondary_slot.secondary_slot.primary_slot - 0.5
and ((predecessor.is_body_paragraph and other_block.bbox_width() < 0.95 * predecessor.bbox_width()
and x_aligned(other_block, predecessor, max(1, other_block.bbox_width() / 10))
and predecessor_gap < 6 * other_block.bbox_height())
or (top_gap < math.inf and above is not None and above.is_body_paragraph
and other_block.bbox_width() < 0.95 * above.bbox_width()
and x_aligned(other_block, above, max(1, other_block.bbox_width() / 10))
and top_gap < 6 * other_block.bbox_height()))
and (centered or caps_heavy)
and ((other_block.bold_frac() > predecessor.bold_frac() and other_block.bold_frac() > 0.5
and (not first_span_of(predecessor).primary_slot
or (above is not None and other_block.bold_frac() > above.bold_frac())))
or caps_heavy)
)
# Nearby body text with the dominant style changed is a strong heading cue.
difference_style = bool(
predecessor is not None and predecessor.is_body_paragraph
and predecessor.avg_font_size() > page.primary_slot - 0.5
and predecessor_gap > 0 and predecessor_gap < 3 * other_block.bbox_height()
and dominant_style_of(other_block) != dominant_style_of(predecessor)
)
style_change_cue = (
difference_style
and above is not None and above.is_body_paragraph
and top_gap > 0 and top_gap < 3 * other_block.bbox_height()
and centered and dominant_style_of(above) == dominant_style_of(predecessor) if predecessor is not None else False
)
if neighbor_heading_cue or style_change_cue:
return make_plain_candidate(page_scan, 0, other_block)
# Top-like context: there is no above block, or the above block is already a
# title/heading marker.
topnum = above is None or above.type == 1
branch_C1 = (
topnum and centered and difference_style
and predecessor_gap < other_block.bbox_height()
and predecessor is not None and dominant_style_of(predecessor) == page.auxiliary_slot
)
branch_C2 = (
topnum and centered
and predecessor is not None and above_or_overlap is not None
and predecessor is not above_or_overlap
and predecessor.bottom_edge() - above_or_overlap.top_edge() < predecessor.avg_font_size()
and dominant_style_of(above_or_overlap) == page.auxiliary_slot and dominant_style_of(other_block) != page.auxiliary_slot
and (other_block.avg_font_size() >= predecessor.avg_font_size() + 0.5
or (caps_heavy and not is_upper_dominant(predecessor.char_stats)))
)
branch_C3 = (
above is not None
and (above.used_as_heading or above in page_scan.measure_slot)
and (above.avg_font_size() >= other_block.avg_font_size() + 0.5
or (is_upper_dominant(above.char_stats) and not caps_heavy))
and centered and difference_style
and predecessor is not None and dominant_style_of(predecessor) == page.auxiliary_slot
)
if branch_C1 or branch_C2 or branch_C3:
return make_plain_candidate(page_scan, 0, other_block)
return None
def detect_heading_with_body(page_scan: PageScanState, other_block: Block) -> Optional[HeadingCandidate]:
""". Detect heading-with-body 2-line patterns."""
if other_block.line_count() < 2:
return None
tokens = tokenize_block(other_block)
first_line = other_block.line()
second_line = other_block.primary_slot[1]
split = 0
letter_count = 0
font = first_line.primary_slot[0].font_name if first_line.primary_slot else ""
if first_line.bold_frac() > 0 and first_line.bold_frac() < 1:
for entry in enumerate_tokens(tokens):
index = entry["index"]
anchor_token = entry["token"]
if anchor_token.line() is not first_line or not first_anchor_span(anchor_token).primary_slot:
break
if anchor_token.type == 2 and len(anchor_token.str) > 1:
letter_count += 1
split = index + 1
elif last_span(last_line_of(other_block)).font_name != font:
other_count = 0
for candidate_line in other_block:
if candidate_line is not first_line and candidate_line.primary_slot[0].font_name == font:
other_count += 1
if other_count > other_block.line_count() / 4:
return None
for entry in enumerate_tokens(tokens):
index = entry["index"]
anchor_token = entry["token"]
line = anchor_token.line()
if first_anchor_span(anchor_token).font_name != font or (line is not first_line and line is not second_line):
break
if anchor_token.type == 2 and (len(anchor_token.str) > 1 or anchor_token.primary_slot == 4):
letter_count += 1
split = index + 1
if split <= 0 or split >= tokens.length:
return None
# Allow up to two punctuation-like tokens to stay with the prefix when they
# remain on the same line and bracket attachment permits it.
token = tokens.token_at(split - 1)
next_token = tokens.token_at(split)
for _ in range(2):
if token is None or next_token is None:
return None
last_anchor = last_token_anchor(token)
if not (is_word_token(next_token)
and getattr(last_anchor, "line", None) is next_token.line()
and (not token.boundary_slot or next_token.boundary_slot)):
break
split += 1
token = next_token
next_token = tokens.token_at(split)
if token is None or next_token is None:
return None
if letter_count <= 0:
return None
prefix = tokens.slice(0, split)
# First-token style check for prefix/body split confidence.
first = prefix.token_at(0)
first_anchor = first_anchor_span(first) if first is not None else None
if first_anchor is not None:
if not first_anchor.primary_slot and not first_anchor.measure_slot and other_block.previous_slot > 0.5:
return None
if (not first_anchor.primary_slot and first_anchor.font_size < other_block.avg_font_size() + 1):
rest = tokens.slice(split)
if rest.length <= 0 or (rest.token_at(0) is not None and rest.token_at(0).primary_slot == 3):
return None
# Reject prefixes that are only section keywords and contain no extra text.
hn_match = trie_prefix_match(KEYWORDS_SECTION_TRIE, prefix)
if hn_match is not None and len(hn_match) >= letter_count:
return None
# Reuse numbered-heading detection on the prefix.
heading_kind = detect_numbered_heading(page_scan, other_block, prefix)
if heading_kind is not None and len(heading_kind.numbering) > 1:
return make_heading_candidate(page_scan, heading_kind.type, heading_kind.group_slot, heading_kind.numbering, heading_kind.secondary_slot, heading_kind.primary_slot, True)
if heading_kind is not None and (trie_matches_all(INTRODUCTION_SECTION_TRIE, heading_kind.primary_slot) or (is_uppercase_dominant(heading_kind.primary_slot) and not is_bibliography_entry(other_block))):
return make_heading_candidate(page_scan, heading_kind.type, heading_kind.group_slot, heading_kind.numbering, heading_kind.secondary_slot, heading_kind.primary_slot, True)
# Reuse the labeled-heading detector on the prefix with body-heading status.
if prefix.length > 3:
heading_signature = detect_labeled_heading(page_scan, other_block, prefix)
if heading_signature is not None:
return make_heading_candidate(page_scan, heading_signature.type, heading_signature.group_slot, heading_signature.numbering, heading_signature.secondary_slot, trim_trailing_punct(heading_signature.primary_slot), True)
if matches_abstract(prefix):
return make_body_heading_candidate(page_scan, 5, other_block, prefix)
if trie_matches_all(INTRODUCTION_SECTION_TRIE, prefix):
return make_body_heading_candidate(page_scan, 11, other_block, prefix)
# Final font-size and trailing-token reject gates.
if first_line.avg_font_size() < page_scan.secondary_slot.secondary_slot.primary_slot - 2:
return None
if token is not None and is_word_token(token) and not is_trimmable_token(token):
return None
# Body paragraphs can still contain an all-caps heading prefix.
if (not is_upper_dominant(other_block.char_stats) and other_block.is_body_paragraph and info_weight(other_block.char_stats) >= 100):
all_caps_vf = CharStats(prefix.to_string())
if is_upper_dominant(all_caps_vf) and all_caps_vf.primary_slot[2] <= other_block.char_stats.primary_slot[3]:
if heading_kind is not None:
return make_heading_candidate(page_scan, heading_kind.type, heading_kind.group_slot, heading_kind.numbering, heading_kind.secondary_slot, heading_kind.primary_slot, True)
if trie_matches_all(SECTION_KEYWORDS_TRIE, prefix):
return make_body_heading_candidate(page_scan, 6, other_block, prefix)
return make_body_heading_candidate(page_scan, 0, other_block, prefix)
# Centered two-line heading branch.
above_neighbor = neighbor_above(page_scan.tertiary_slot, other_block)
gap = (above_neighbor.bottom_edge() - first_line.top_edge()) if above_neighbor is not None else math.inf
intersection = first_line.bottom_edge() - second_line.top_edge()
per_char = avg_char_width(first_line)
centered_flag = False
# When there is no above block, the infinite gap is sufficient for this
# branch and later above-block checks must remain guarded.
if first_anchor is not None:
cond_outer = (
first_anchor.measure_slot
and not first_anchor_span(next_token).measure_slot if next_token is not None else False
)
# First-line anchor, second-line anchor, gap, neighbor, and punctuation
# checks together identify a centered heading prefix.
if (first_anchor.measure_slot
and next_token is not None and not first_anchor_span(next_token).measure_slot
and (gap > 1.1 * intersection
or (last_token(tokenize_block(above_neighbor)) is not None and is_word_token(last_token(tokenize_block(above_neighbor))))
or last_line_of(above_neighbor).right_edge() < first_line.right_edge() - 8 * per_char)
and (first_line.right_edge() > second_line.right_edge() - 4 * per_char
or first_line.char_stats.tertiary_slot != 6
or second_line.char_stats.secondary_slot == 3)):
for prefix_token in prefix:
if prefix_token.primary_slot == 2:
centered_flag = True
break
if prefix_token.type == 2 or prefix_token.boundary_slot:
break
if (centered_flag
and token is not None and is_trimmable_token(token)
and next_token is not None and next_token.primary_slot == 2):
if trie_matches_all(SECTION_KEYWORDS_TRIE, prefix):
return make_body_heading_candidate(page_scan, 6, other_block, prefix)
if letter_count > 1:
return make_body_heading_candidate(page_scan, 0, other_block, prefix)
return None
@@ -0,0 +1,221 @@
"""Block-text predicates: keyword matches, continuation, content, and number parsing."""
from __future__ import annotations
import math
from typing import Any, Optional
from ..labels import is_uppercase_dominant, trie_matches_all, advance_past_line, skip_bracketed_word, token_case_signal, format_caption_label, CaptionEntry, extract_structural_number
from ..model import (
_UNICODE_WHITESPACE_CLASS,
_strip_diacritics,
_trim_unicode_ws,
style_key, magnitude_ratio, same_x_extent, same_y_extent, y_overlaps, left_aligned, right_aligned, center_aligned, x_aligned, x_centers_close, to_number,
last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind, Line, last_line_of, first_span_of, is_word_category, block_text, is_punct_category, deaccented_text, letter_count, punct_count, dominant_style_of,
info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, CharStats, alignment_code, Block,
)
from ..tokens import (
is_trimmable_token, token_numeric_value, Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_trie_match, strip_leading_if_in, COMMA_CHARS, strip_trailing_comma, first_token, trim_trailing_punct, set_case_fold, TrieConfig, build_trie, tokenize_block,
trie_full_match, last_token_anchor, first_anchor_span, is_char_token, is_word_token,
)
from .keyword_tables import (
ABSTRACT_KEYWORDS_TRIE,
REFERENCES_TRIE,
_normalize_text_key,
ABSTRACT_KEYWORDS_SET,
REFERENCES_SET,
NUMBERED_PREFIX_RE,
DEAD_DIGIT_RE,
EQUATION_KEYWORDS_TRIE,
ENGLISH_WORD_TO_NUMBER,
ROMAN_NUMERAL_MAP,
FORMULA_CHAR_WEIGHTS,
)
def token_text_of_block(block: Block) -> str:
"""Tokenize ``block``, join tokens using their stored spacing flags, trim the result, and memoize it on the block."""
if block.token_text_cache is not None:
return block.token_text_cache
block.token_text_cache = _trim_unicode_ws(tokenize_block(block).to_string())
return block.token_text_cache
# --------------------------------------------------------------------------- #
# Simple heading and equation predicates.
# --------------------------------------------------------------------------- #
def similar_style(block: Block, other_block: Block) -> bool:
"""Return whether two blocks have very similar bold ratio and font size."""
return abs(block.bold_frac() - other_block.bold_frac()) < 0.5 and abs(block.avg_font_size() - other_block.avg_font_size()) < 1
def is_heading_continuation(block: Block, other_block: Block, candidate_number: int) -> bool:
"""Return whether ``block`` is the next numbered heading continuation of ``other_block``."""
if block.type != 0 or block.char_count() >= 500 or not similar_style(other_block, block):
return False
text = token_text_of_block(block)
if block.left_edge() >= other_block.left_edge() and text.startswith("•"):
return True
if is_upper_dominant(other_block.char_stats) and is_upper_dominant(block.char_stats) and not left_aligned(other_block, block, 1) and not right_aligned(other_block, block, 1) and center_aligned(other_block, block, 1):
return False
heading = NUMBERED_PREFIX_RE.match(text)
if heading and len(heading.groups()) >= 1:
matched_number = to_number(heading.group(1))
return abs(candidate_number - matched_number) == 1
return False
def matches_abstract(tokens: TokenView) -> bool:
"""Token sequence matches abstract keywords or their normalized text set."""
if trie_matches_all(ABSTRACT_KEYWORDS_TRIE, tokens):
return True
if tokens.length > 10:
return False
normalized = ""
for candidate_item in tokens:
if is_word_token(candidate_item):
continue
if candidate_item.type != 2 or len(normalized) + len(candidate_item.str) > 20:
return False
normalized += _normalize_text_key(candidate_item.str.lower())
return normalized in ABSTRACT_KEYWORDS_SET
def matches_references(tokens: TokenView) -> bool:
"""Token sequence matches references keywords or their whole-text set."""
secondary_item = trie_prefix_match(REFERENCES_TRIE, tokens)
if secondary_item is None:
if tokens.length <= 15:
normalized = ""
for candidate_item in tokens:
if is_word_token(candidate_item):
continue
if candidate_item.type != 2 or len(normalized) + len(candidate_item.str) > 20:
return False
normalized += candidate_item.str.lower()
return normalized in REFERENCES_SET
return False
if secondary_item.length == tokens.length:
return True
rest = tokens.slice(secondary_item.length)
if rest.length == 1:
first = rest.token_at(0)
if first is not None and is_word_token(first):
return True
return trie_matches_all(REFERENCES_TRIE, rest)
def vertically_close(block: Optional[Block], other_block: Block) -> bool:
"""a is vertically very close to b."""
if block is None:
return False
candidate_item = block.bottom_edge() - other_block.top_edge() if block.top_edge() > other_block.top_edge() else other_block.bottom_edge() - block.top_edge()
return candidate_item < 2 * other_block.avg_font_size() or (x_aligned(block, other_block, 1) and candidate_item < 5 * other_block.avg_font_size())
def is_equation_adjacent_line(line: Optional[Line], block: Block) -> bool:
"""Return whether a line is adjacent to an equation block: it overlaps and follows the block, matches the equation-separator pattern, or consists entirely of equation-keyword tokens after trimming wrapper punctuation."""
from ..labels import extract_structural_number
if line is None or line.line_count() != 1:
return False
if line.left_edge() < block.right_edge() or not y_overlaps(block, line):
return False
if DEAD_DIGIT_RE.match(block_text(line)):
return True # Equation separator match is enough to accept.
tokens = tokenize_block(line)
# Drop single non-digit chars at both edges when token-count is >= 3.
if (tokens.length >= 3
and (first := first_token(tokens)) is not None and len(first.str) <= 1
and first.type != 1
and (last := last_token(tokens)) is not None and len(last.str) <= 1
and last.type != 1):
tokens = tokens.slice(1, tokens.length - 1)
tokens = strip_trie_match(tokens, EQUATION_KEYWORDS_TRIE)
yi_match = extract_structural_number(tokens)
return yi_match is not None and yi_match.length == tokens.length
def has_substantive_content(block: Block, other_block: Optional[Block], candidate_block: Optional[Block]) -> bool:
"""heuristic "this block has substantive content?" score >= 5."""
entry_item = 0
for token in tokenize_block(block):
anchor = first_anchor_span(token)
line = token.line()
size = line.previous_slot
flag = anchor.top_edge() < line.bottom_edge() + 0.8 * size or anchor.bottom_edge() > line.top_edge() - 0.8 * size
if token.type == 1:
entry_item += 2 if flag else 1
continue
weight = FORMULA_CHAR_WEIGHTS.get(token.str)
if weight is not None:
entry_item += (3 if flag else 1) * weight
continue
if token.type == 6:
entry_item += (3 if flag else 1) * 5
continue
if len(token.str) <= 3 and token.primary_slot != 4:
if flag:
entry_item += 5 if is_word_token(token) else 1
continue
if flag:
continue
len_value = (2 if anchor.primary_slot else 1) * len(token.str)
if token.primary_slot == 4:
entry_item -= 2 * len_value
elif token.primary_slot == 2:
entry_item -= len_value
elif token.primary_slot == 3:
entry_item -= 0.5 * len_value
if entry_item < 0:
return False
if entry_item >= 5:
return True
return is_equation_adjacent_line(other_block, block) or is_equation_adjacent_line(candidate_block, block)
def is_cover_page(doc, page) -> bool:
"""Return whether ``page`` behaves like a cover page: it is title-marked, appears early, and has light content or no body text."""
return (
page.auxiliary_slot
and page.page_index < max(2, len(doc.primary_slot) / 2)
and (
page.primary_slot.secondary_slot < clamp(0.5 * doc.secondary_slot.secondary_slot, 200, 1000)
or not page.state_slot
)
)
def clamp(value: float, lower_bound: float, upper_bound: float) -> float:
"""``max(lo, min(hi, v))``. NaN propagates."""
measure_item = upper_bound if upper_bound < value else value
return lower_bound if lower_bound > measure_item else measure_item
def token_to_number(tok: Optional[Token]) -> Optional[int | float]:
"""extract numeric value from a token (digit, Roman, or English)."""
if tok is None:
return None
if tok.type == 1:
token = token_numeric_value(tok)
if not math.isnan(token) and token > 0:
return int(token) if token.is_integer() else token
return None
return ROMAN_NUMERAL_MAP.get(tok.str) or ENGLISH_WORD_TO_NUMBER.get(tok.str.lower())
def letter_to_ordinal(tok_str: str) -> Optional[int]:
"""'a'/'A' -> 1, 'b' -> 2, ..., 'h' -> 8. None otherwise."""
if len(tok_str) != 1:
return None
# Only the FIRST UTF-16 code unit of the lowercased character counts: a
# case mapping that expands to several units (U+0130) contributes just its
# first, and an astral lowercase contributes its high surrogate.
low = tok_str[0].lower()
code_unit = ord(low[0])
if code_unit > 0xFFFF:
code_unit = 0xD800 + ((code_unit - 0x10000) >> 10)
value = code_unit - 96
return value if 1 <= value <= 8 else None
+48
View File
@@ -0,0 +1,48 @@
"""Keyword-labeled section and caption-region detection. This module finds blocks that look like figure/table/chart labels or named
sections, then extends each label forward or backward to claim the associated
body blocks. The resulting regions are used by classification and outline
assembly to avoid treating captions or labeled content as ordinary headings.
"""
import regex as regex_module # Unicode \p{...} property classes.
from typing import Optional
from ..classification import FIGURE_KEYWORDS_TRIE, TABLE_KEYWORDS_TRIE, CHART_KEYWORDS_TRIE
from ..model import (
Rect, rect_union, extend_top_to, extend_bottom_to, EMPTY_RECT, Bounded,
_trim_unicode_ws,
center_aligned, last_span, heading_score, reading_order_key, numbering_text, Line, last_line_of, first_span_of, dominant_style_of, info_weight, Block,
)
from ..stats import column_index_of
from ..tokens import Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_leading_if_in, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, BuiltTrie, is_word_token
from .caption_text import (
PERIOD_CHARS,
STRUCTURAL_NUMBER_RE,
is_number_separator,
extract_structural_number,
format_caption_label,
REFERENCE_PHRASE_TRIE,
is_uppercase_dominant,
trie_matches_all,
advance_past_line,
skip_bracketed_word,
token_case_signal,
caption_outranks,
)
from .caption_regions import (
CaptionedRegion,
dedupe_caption_entries,
extend_caption_region,
build_caption_regions,
CaptionEntry,
CaptionContext,
iter_page_blocks,
detect_captions,
)
__all__ = [
"PERIOD_CHARS", "STRUCTURAL_NUMBER_RE", "is_number_separator", "extract_structural_number", "format_caption_label", "REFERENCE_PHRASE_TRIE",
"is_uppercase_dominant", "trie_matches_all", "advance_past_line", "skip_bracketed_word", "token_case_signal", "caption_outranks",
"CaptionEntry", "CaptionedRegion", "CaptionContext", "iter_page_blocks", "detect_captions", "dedupe_caption_entries", "extend_caption_region", "build_caption_regions",
]
+366
View File
@@ -0,0 +1,366 @@
"""Caption region growth, deduplication, and detection."""
from __future__ import annotations
from typing import Optional
from ..classification import FIGURE_KEYWORDS_TRIE, TABLE_KEYWORDS_TRIE, CHART_KEYWORDS_TRIE
from ..model import (
Rect, rect_union, extend_top_to, extend_bottom_to, EMPTY_RECT, Bounded,
_trim_unicode_ws,
center_aligned, last_span, heading_score, reading_order_key, numbering_text, Line, last_line_of, first_span_of, dominant_style_of, info_weight, Block,
)
from ..stats import column_index_of
from ..tokens import Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_leading_if_in, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, BuiltTrie, is_word_token
from .caption_text import (
PERIOD_CHARS,
extract_structural_number,
format_caption_label,
REFERENCE_PHRASE_TRIE,
caption_outranks,
)
# --------------------------------------------------------------------------- #
# Captioned/labeled region wrapper #
# --------------------------------------------------------------------------- #
class CaptionedRegion(Bounded):
"""Captioned or labeled region plus its body blocks. The region stores the document context, page, heading block, body blocks, neighboring block reference, label flag, label type, and an area-weighted score used to choose forward vs backward extension."""
__slots__ = ("weighted_ratio_primary", "page", "primary_slot", "output_slot", "state_slot", "alignment_slot", "type", "score")
def __init__(self, primary_item, secondary_item, candidate_item, bbox: Rect, blocks, next_item, flag):
super().__init__(bbox)
self.weighted_ratio_primary = primary_item
self.page = secondary_item
self.primary_slot = candidate_item # the original heading block
self.output_slot = blocks # list of body blocks
self.state_slot = next_item
self.alignment_slot = flag
# Caption label type is carried by the heading block marker.
# ``Block.type`` is a later classification label and is still zero here.
self.type = candidate_item.marker_slot
# Region score formula.
area_pct = 100.0 * self.area() / self.page.bounds.area() if self.page.bounds.area() > 0 else 0.0
if area_pct <= 0:
score = 0.0
else:
if (self.state_slot is not None
and self.state_slot.top_edge() < self.top_edge()
and self.state_slot.right_edge() > self.left_edge()
and self.alignment_slot):
area_pct /= 5.0
if self.type == 4:
inner = 0.0
for block in self.output_slot:
if block.skew_frac() > 1:
continue
inner += block.area()
score = area_pct * max(0.1, 1 - inner / self.area()) if self.area() > 0 else 0.0
else:
# Span text is a string, so every span contributes its character
# count to the caption-region score.
count = 1.0
for block in self.output_slot:
for line in block:
for span in line:
count += span.char_count()
score = count * area_pct
self.score = score
# --------------------------------------------------------------------------- #
# Deduplicate caption entries and keep the best entry for each label.
# --------------------------------------------------------------------------- #
def dedupe_caption_entries(caption_context: "CaptionContext") -> list["CaptionEntry"]:
"""Deduplicate structural-number entries by label while preserving page order."""
if not caption_context.state_slot:
return caption_context.auxiliary_slot
captions_by_label: dict[str, CaptionEntry] = {}
for caption in caption_context.auxiliary_slot:
if len(caption.primary_slot) <= 1:
continue
existing = captions_by_label.get(caption.primary_slot)
if existing is None or caption_outranks(caption, existing):
captions_by_label[caption.primary_slot] = caption
out = list(captions_by_label.values())
out.sort(key=lambda caption_sort_key: (caption_sort_key.page_index, caption_sort_key.group_slot.reading_order_index))
return out
# --------------------------------------------------------------------------- #
# Extend a labeled section forward or backward.
# --------------------------------------------------------------------------- #
def extend_caption_region(
caption_context: "CaptionContext",
entry: "CaptionEntry",
prior_regions: list,
page_set: Optional[set],
direction: int,
) -> Optional[CaptionedRegion]:
"""Walk page blocks forward or backward from a labeled entry, accumulating a region until an already-classified block, claimed block, deep body block, fresh top-level heading, or size/gap boundary is reached."""
page = caption_context.primary_slot.primary_slot[entry.page_index - 1]
origin = entry.group_slot
anchor = origin.bottom_edge() if direction > 0 else origin.top_edge()
bbox = Rect(origin.left_edge(), origin.right_edge(), anchor, anchor)
blocks: list[Block] = []
sorted_value = page.secondary_slot
index = entry.group_slot.reading_order_index + direction
previous: Block = origin
while 0 <= index < len(sorted_value):
caption = sorted_value[index]
caption_column = column_index_of(caption)
if caption_column < 0:
break
# layout branch: when crossing the column band, walk page.j (column
# rects) to the nearest column that horizontally overlaps the
# bbox and extend the bbox vertically to that column's edge.
if direction < 0 and caption_column < column_index_of(entry.group_slot) and caption.bottom_edge() < anchor:
col_idx = caption_column - 1
column_rect = page.tertiary_slot[col_idx] if 0 <= col_idx < len(page.tertiary_slot) else None
while column_rect is not None and (
column_rect.bottom_edge() < bbox.top_edge()
or column_rect.right_edge() < bbox.left_edge()
or column_rect.left_edge() > bbox.right_edge()
):
col_idx -= 1
column_rect = page.tertiary_slot[col_idx] if 0 <= col_idx < len(page.tertiary_slot) else None
if column_rect is not None:
bbox = extend_top_to(bbox, column_rect.bottom_edge())
else:
bbox = extend_top_to(bbox, page.bounds.top_edge())
break
if direction > 0 and caption_column > column_index_of(entry.group_slot) and caption.top_edge() > anchor:
col_idx = caption_column + 1
column_rect = page.tertiary_slot[col_idx] if 0 <= col_idx < len(page.tertiary_slot) else None
while column_rect is not None and (
column_rect.top_edge() > bbox.bottom_edge()
or column_rect.right_edge() < bbox.left_edge()
or column_rect.left_edge() > bbox.right_edge()
):
col_idx += 1
column_rect = page.tertiary_slot[col_idx] if 0 <= col_idx < len(page.tertiary_slot) else None
if column_rect is not None:
bbox = extend_bottom_to(bbox, column_rect.top_edge())
else:
bbox = extend_bottom_to(bbox, page.bounds.bottom_edge())
break
# Grow the bbox to include n
if direction < 0:
bbox = extend_top_to(bbox, caption.bottom_edge())
else:
bbox = extend_bottom_to(bbox, caption.top_edge())
# Stop conditions
if caption.type != 0 or caption.reading_order_index in caption_context.secondary_slot:
break
if page_set is not None and index in page_set:
break
size = min(caption_context.primary_slot.secondary_slot.primary_slot, entry.group_slot.avg_font_size())
if caption.is_body_paragraph and caption.avg_font_size() > min(0.9 * size, size - 1.5):
break
next_block = sorted_value[index + 1] if index + 1 < len(sorted_value) else None
gap = previous.bottom_edge() - caption.top_edge() if direction > 0 else 0
line_gap = page.primary_slot.tertiary_slot - page.primary_slot.primary_slot
if (
direction > 0 and next_block is not None and caption.line_count() <= 4 and caption.char_stats.secondary_slot != 3
and gap > line_gap
and (previous is entry.group_slot or gap > min(3 * line_gap, caption.bottom_edge() - next_block.top_edge()))
):
next_item = sorted_value[index + 2] if index + 2 < len(sorted_value) else None
if heading_score(caption) >= heading_score(previous) + 0.5 and (next_block.is_body_paragraph or (next_item is not None and next_item.is_body_paragraph)):
break
# A numbering-like line with enough trailing text can stop this
# backward body-paragraph scan.
line_text = numbering_text(caption.line())
if (line_text
and caption.char_stats.secondary_slot == 2
and heading_score(caption) >= size
and gap > 2 * caption.avg_font_size()
and caption.char_count() - len(line_text) > 2):
break
blocks.append(caption)
bbox = rect_union(bbox, caption.secondary_slot)
index += direction
previous = caption
if direction < 0 and index < 0:
bbox = extend_top_to(bbox, page.bounds.top_edge())
elif direction > 0 and index >= len(sorted_value):
bbox = extend_bottom_to(bbox, page.bounds.bottom_edge())
# When a backward extension expands the region, also consume forward
# neighbours whose geometric center sits inside the grown bbox.
if direction < 0:
fwd_idx = entry.group_slot.reading_order_index + 1
while fwd_idx < len(sorted_value):
block = sorted_value[fwd_idx]
center_x = block.center_x()
center_y = block.center_y()
if (center_x < bbox.left_edge() or center_x > bbox.right_edge()
or center_y < bbox.bottom_edge() or center_y > bbox.top_edge()):
break
blocks.append(block)
bbox = rect_union(bbox, block.secondary_slot)
fwd_idx += 1
area = bbox.area()
if area <= 0:
return None
# Check overlap with prior regions; if heavy overlap, reject.
for prior in prior_regions:
overlap_area = max(
0.0,
min(bbox.right, prior.secondary_slot.right) - max(bbox.left, prior.secondary_slot.left),
) * max(
0.0,
min(bbox.top, prior.secondary_slot.top) - max(bbox.primary_slot, prior.secondary_slot.primary_slot),
)
if overlap_area >= 0.25 * min(area, prior.area()):
return None
next_block = sorted_value[index] if 0 <= index < len(sorted_value) else None
on_page_set = page_set is not None and index in page_set
return CaptionedRegion(
primary_item=caption_context.primary_slot, secondary_item=page, candidate_item=entry.group_slot,
bbox=bbox, blocks=blocks, next_item=next_block, flag=on_page_set,
)
# --------------------------------------------------------------------------- #
# Extend all deduplicated labeled-section entries.
# --------------------------------------------------------------------------- #
def build_caption_regions(caption_context: "CaptionContext") -> list[CaptionedRegion]:
"""Build caption regions by extending each labeled entry in both directions."""
caption_context.tertiary_slot.clear()
caption_context.secondary_slot.clear()
entries = dedupe_caption_entries(caption_context)
for caption in entries:
set_value = caption_context.tertiary_slot.get(caption.page_index)
if set_value is None:
set_value = set()
caption_context.tertiary_slot[caption.page_index] = set_value
set_value.add(caption.group_slot.reading_order_index)
out: list[CaptionedRegion] = []
page = 0
prior_regions: list[CaptionedRegion] = []
for entry in entries:
if entry.page_index != page:
prior_regions = []
caption_context.secondary_slot.clear()
page = entry.page_index
if len(prior_regions) >= 8:
continue
page_set = caption_context.tertiary_slot.get(entry.page_index)
back = extend_caption_region(caption_context, entry, prior_regions, page_set, -1)
forward = extend_caption_region(caption_context, entry, prior_regions, page_set, 1)
winner = (
back if (back is not None and (forward is None or back.score > forward.score))
else forward
)
if winner is not None:
for body_block in winner.output_slot:
caption_context.secondary_slot.add(body_block.reading_order_index)
prior_regions.append(winner)
out.append(winner)
return out
# --------------------------------------------------------------------------- #
# Labeled-section entry.
# --------------------------------------------------------------------------- #
class CaptionEntry:
"""One labeled-section entry with label, type, page, block, and remainder tokens."""
__slots__ = ("primary_slot", "type", "page_index", "group_slot", "secondary_slot")
def __init__(self, label: str, type_: int, page: int, block: Block, remainder: TokenView):
self.primary_slot = label
self.type = type_
self.page_index = page
self.group_slot = block
self.secondary_slot = remainder
# --------------------------------------------------------------------------- #
# Labeled-section context.
# --------------------------------------------------------------------------- #
class CaptionContext:
"""Document-level state for labeled-section detection."""
__slots__ = ("primary_slot", "auxiliary_slot", "state_slot", "tertiary_slot", "secondary_slot")
def __init__(self, doc):
self.primary_slot = doc
self.auxiliary_slot: list[CaptionEntry] = []
self.state_slot: bool = False
self.tertiary_slot: dict = {} # page -> set of heading-block ga
self.secondary_slot: set = set() # set of heading-block ga across doc
# --------------------------------------------------------------------------- #
# Document-wide (page, block) iterator.
# --------------------------------------------------------------------------- #
def iter_page_blocks(doc):
"""Yield ``{'page': page, 'G': block}`` records in reading order."""
for page in doc.primary_slot:
for block in (page.secondary_slot or []):
yield {"page": page, "block": block}
# --------------------------------------------------------------------------- #
# Labeled-section detection driver.
# --------------------------------------------------------------------------- #
def detect_captions(caption_context: CaptionContext) -> None:
"""Find figure, table, and chart labels and record their structural prefixes."""
for entry in iter_page_blocks(caption_context.primary_slot):
page = entry["page"]
block = entry["block"]
if block.type != 0:
continue
tokens = tokenize_block(block)
type_value: Optional[int] = None
prefix = trie_prefix_match(FIGURE_KEYWORDS_TRIE, tokens)
if prefix is not None:
type_value = 4
else:
prefix = trie_prefix_match(TABLE_KEYWORDS_TRIE, tokens)
if prefix is not None:
type_value = 5
else:
prefix = trie_prefix_match(CHART_KEYWORDS_TRIE, tokens)
if prefix is not None:
type_value = 11
if type_value is None:
continue
remainder = strip_leading_if_in(tokens.slice(prefix.length), PERIOD_CHARS)
number = extract_structural_number(remainder)
label = format_caption_label(type_value, number)
if number is not None:
caption_context.state_slot = True
remainder = remainder.slice(number.length)
if trie_prefix_match(REFERENCE_PHRASE_TRIE, remainder) is not None:
continue
page.measure_slot = True
caption_context.auxiliary_slot.append(CaptionEntry(label, type_value, page.page_index, block, remainder))
# Mark the block's Y category (used by outline.py heading filter)
block.marker_slot = type_value
+171
View File
@@ -0,0 +1,171 @@
"""Caption label text helpers and structural-number parsing."""
from __future__ import annotations
import regex as regex_module # Unicode \p{...} property classes.
from typing import Optional
from ..model import (
Rect, rect_union, extend_top_to, extend_bottom_to, EMPTY_RECT, Bounded,
_trim_unicode_ws,
center_aligned, last_span, heading_score, reading_order_key, numbering_text, Line, last_line_of, first_span_of, dominant_style_of, info_weight, Block,
)
from ..tokens import Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, strip_leading_if_in, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, BuiltTrie, is_word_token
# --------------------------------------------------------------------------- #
# Helpers #
# --------------------------------------------------------------------------- #
PERIOD_CHARS = {".", ".", "。", "。"} # period-character set
# Structural-number pattern: Unicode numeric code points, optional letter
# affixes, or Roman numerals. ``\Z`` anchors at the absolute end of string, not
# before a trailing newline.
STRUCTURAL_NUMBER_RE = regex_module.compile(
r"^(?:[A-M]*\p{Number}+[A-Ma-m]?|[A-Ma-m]\p{Number}*|[IVX]+)\Z"
)
def is_number_separator(token: Optional[Token], other_flag: bool = True) -> bool:
"""Return whether the token is a structural-number separator candidate."""
if token is None:
return False
if token.boundary_slot:
return False
if token.type == 3:
return True
if other_flag and token.type == 4:
return True
return False
def extract_structural_number(tokens: TokenView, other_flag: bool = True) -> Optional[TokenView]:
"""extract a leading structural-number prefix from tokens. Returns the matched prefix as a token-view slice, or None. """
if tokens.length < 1:
return None
candidate_item = tokens
first = tokens.token_at(0)
if first is None:
return None
reference_item = first.str
if len(reference_item) == 1 and "A" <= reference_item[0] <= "H":
if not is_number_separator(tokens.token_at(1), other_flag):
return None
candidate_item = tokens.slice(2)
if candidate_item.length < 1:
return None
head = first_token(candidate_item)
if head is None or not STRUCTURAL_NUMBER_RE.match(head.str):
return None
candidate_item = candidate_item.slice(1)
while candidate_item.length >= 2 and is_number_separator(candidate_item.token_at(0), other_flag) and STRUCTURAL_NUMBER_RE.match(candidate_item.token_at(1).str): # type: ignore[union-attr]
candidate_item = candidate_item.slice(2)
return tokens.slice(0, tokens.length - candidate_item.length)
# - format code label
def format_caption_label(type_: int, num: Optional[TokenView]) -> str:
"""format the section-type letter prefix + number. type_ 4 -> "F", 5 -> "T", 11 -> "Q". Append the number string if any. """
if type_ == 4:
letter = "F"
elif type_ == 5:
letter = "T"
elif type_ == 11:
letter = "Q"
else:
return ""
if num is not None:
letter += _trim_unicode_ws(str(num))
return letter
# - case-sensitive trie of phrases that indicate "this is a
# reference TO a figure/table, not a label OF one".
REFERENCE_PHRASE_TRIE = build_trie(["lists the", "presents", "show the", "showed the", "shows"], set_case_fold(TrieConfig(), False))
# --------------------------------------------------------------------------- #
# Token helpers for caption-entry ranking.
# --------------------------------------------------------------------------- #
def is_uppercase_dominant(tokens: TokenView) -> bool:
"""Return True when the token sequence is dominated by uppercase words. Multi-character lowercase-start words whose second character is not uppercase reject the sequence as body-like text."""
from ..tokens import char_category
secondary_item = candidate_item = 0
for reference_item in tokens:
if reference_item.type != 2:
continue
if reference_item.primary_slot == 2:
secondary_item += 1
elif reference_item.primary_slot == 3:
if len(reference_item.str) > 4 and len(reference_item.str) >= 2 and char_category(reference_item.str[1]) != 2:
return False
candidate_item += 1
return secondary_item > max(2, candidate_item)
def trie_matches_all(trie: BuiltTrie, tokens: TokenView) -> bool:
"""tokens fully match ``trie`` (or all but a final word-y token)."""
match = trie_prefix_match(trie, tokens)
if match is None:
return False
if match.length == tokens.length:
return True
if match.length == tokens.length - 1:
last = last_token(tokens)
return last is not None and is_word_token(last)
return False
def advance_past_line(tokens: TokenView, line: Line, index: int) -> int:
"""Advance while the token at the current index belongs to ``line``."""
while index < tokens.length:
tok = tokens.token_at(index)
if tok is None:
break
if tok.line() is not line:
break
index += 1
return index
def skip_bracketed_word(tokens: TokenView, index: int) -> int:
"""advance over bracket-attached word token."""
tok = tokens.token_at(index)
if tok is not None and tok.boundary_slot and is_word_token(tok):
return index + 1
return index
def token_case_signal(token: Optional[Token]) -> int:
"""per-token "direction signal". Returns 2 if g==7/6 (sentence end), 1 if g==2 (uppercase), -1 if g==3 (lowercase), 0 otherwise. """
if token is None:
return 0
token_kind = token.primary_slot
if token_kind == 7 or token_kind == 6:
return 2
if token_kind == 2:
return 1
if token_kind == 3:
return -1
return 0
def caption_outranks(caption_entry: "CaptionEntry", other_caption_entry: "CaptionEntry") -> bool:
"""Return True when the first caption entry ranks better than the second."""
caption = is_uppercase_dominant(tokenize_block(caption_entry.group_slot))
other_is_uppercase = is_uppercase_dominant(tokenize_block(other_caption_entry.group_slot))
if caption != other_is_uppercase:
return caption
caption_first_token = first_token(caption_entry.secondary_slot) if caption_entry.secondary_slot.length > 0 else None
other_first_token = first_token(other_caption_entry.secondary_slot) if other_caption_entry.secondary_slot.length > 0 else None
group = token_case_signal(caption_first_token)
other_case_signal = token_case_signal(other_first_token)
if group != other_case_signal:
return group > other_case_signal
if caption_entry.page_index != other_caption_entry.page_index:
return caption_entry.page_index < other_caption_entry.page_index
return caption_entry.group_slot.reading_order_index < other_caption_entry.group_slot.reading_order_index
+314
View File
@@ -0,0 +1,314 @@
"""
End-to-end orchestrator for the TOC extraction pipeline. Pipeline order: 1. parse character-level spans and page viewport metadata 2. cluster spans into lines 3. compute page statistics 4. detect columns and recluster lines with column awareness 5. remove line-number artifacts and recompute statistics 6. compute document-level statistics 7. cluster lines into blocks and assign reading order 8. classify headers, footers, watermarks, TOC-like pages, captions, references, and body paragraphs 9. detect the document title 10. collect heading candidates and assemble the final outline The ordering is load-bearing: title selection, labeled-section detection,
heading candidate collection, and outline assembly each consume annotations
from the previous stages. ``to_pageindex_tree`` serializes the final outline
into the JSON shape that ``run_pageindex.py`` writes.
"""
from __future__ import annotations
import json
import re
import unicodedata
from io import BytesIO
from pathlib import Path
from typing import Optional, Union
# (re is used by the title-reject regex below)
from .blocks import cluster_lines_into_blocks, BlockClusterContext
from .classification import is_body_paragraph, detect_header_footer, HeaderFooterContext, mark_watermarks, mark_toc_and_boilerplate
from .labels import detect_captions, build_caption_regions, CaptionContext
from .model import Rect, numbering_kind, block_text, deaccented_text, Block
from .outline_assembly import (
build_heading_from_block, is_landscape_or_empty, is_outline_valid, is_chapter_outline_valid, mark_outline_block_types, assemble_outline, compute_max_heading_gap, has_table_or_prominent, OutlineNode, outline_to_dict_tree,
)
from .parser_pdfium_parallel import parse_charlevel_meta_parallel
from .phases import assign_reading_order, PageView, process_page
PageView = PageView # re-export for type hints
from .stats import compute_doc_stats
from .title import detect_title
# --------------------------------------------------------------------------- #
# References-section dictionary (load once) #
# --------------------------------------------------------------------------- #
_DICT_PATH = Path(__file__).parent / "data" / "dictionaries.json"
def _normalize_text_key(text: str) -> str:
return " ".join(unicodedata.normalize("NFKC", text).strip().split()).lower()
_REFS_DICT_RAW = json.loads(_DICT_PATH.read_text(encoding="utf-8"))
REFERENCES_KEYWORDS = frozenset(_normalize_text_key(text_value) for text_value in _REFS_DICT_RAW.get("references", []) if text_value)
# --------------------------------------------------------------------------- #
# Document container #
# --------------------------------------------------------------------------- #
class DocumentState:
"""Document-level extraction state: pages, document statistics, and recurring-text frequency map. """
__slots__ = ("primary_slot", "secondary_slot", "tertiary_slot")
def __init__(self, pages: list[PageView]):
self.primary_slot = pages
self.secondary_slot = None # set after document statistics are computed
self.tertiary_slot: dict = {}
# --------------------------------------------------------------------------- #
# References-section detection #
# --------------------------------------------------------------------------- #
def find_references(doc: DocumentState) -> Optional[tuple[int, Block]]:
"""Return ``(page_num, block)`` for the first references heading in reading order."""
for page in doc.primary_slot:
for block in (page.secondary_slot or []):
if block.type != 0:
continue
normalized = deaccented_text(block)
if not normalized or len(normalized) > 80:
continue
if normalized in REFERENCES_KEYWORDS:
return page.page_index, block
# Allow short numbered prefix: "12. References"
parts = normalized.split()
if 1 <= len(parts) <= 4 and parts[-1] in REFERENCES_KEYWORDS:
return page.page_index, block
return None
def mark_references(doc: DocumentState, ref: Optional[tuple[int, Block]]) -> None:
"""Tag the references heading itself + everything after as type=3."""
if ref is None:
return
ref_page, ref_block = ref
seen = False
for page in doc.primary_slot:
if page.page_index < ref_page:
continue
for block in (page.secondary_slot or []):
if not seen and block is ref_block:
seen = True
block.type = 3
continue
if seen:
block.type = 3
# --------------------------------------------------------------------------- #
# Repeated-text accumulator #
# --------------------------------------------------------------------------- #
def page_by_block_lookup(pages, block) -> Optional[PageView]:
"""Find which page owns ``block``. Used for wrapping labeled blocks."""
for page in pages:
if block in (page.secondary_slot or []):
return page
return None
# --------------------------------------------------------------------------- #
# End-to-end entry point #
# --------------------------------------------------------------------------- #
def extract_toc(
doc_handle: Union[str, Path, BytesIO],
workers: Optional[int] = None,
) -> dict:
"""Run the full pipeline. Returns a dict shaped like:: { "doc_name": "...", "doc_title": "...", "structure": [ {"title": "...", "start_index": 1, "end_index": 3, "nodes": [...]}, ... ], "has_abstract_or_references_section": False } ``has_abstract_or_references_section`` is True when any TOP-LEVEL outline entry is an abstract-keyword heading or carries the prominent-heading flag (a references-keyword heading, plain or numbered). The near-empty bail and the valid-outline branch both report False. ``workers`` sets the process count for the per-page parallel parser: None = auto (CPU count - 1), 1 forces the sequential path; output is identical either way. """
# ----- 1) Parse PDF -> flat spans per page --------------------------
# per-page (view box, /Rotate) comes from the same engine (PDFium) that
# produced the block coordinates, so the geometry frame is consistent.
parsed, page_meta = parse_charlevel_meta_parallel(doc_handle, workers=workers)
# ----- 2) Per-page layout classification ----------------------------------
# Heading coordinate projection uses the page viewport.
pages: list[PageView] = []
for index_value, spans in enumerate(parsed):
viewport_box_value, rot = page_meta[index_value]
viewport_x0, viewport_y0, viewport_x1, viewport_y1 = viewport_box_value
# page bbox uses DISPLAYED (post-/Rotate) dims.
viewport_width, viewport_height = abs(viewport_x1 - viewport_x0), abs(viewport_y1 - viewport_y0)
page_width, page_height = (viewport_height, viewport_width) if rot % 180 == 90 else (viewport_width, viewport_height)
page_bbox = Rect(0, page_width, page_height, 0)
page = process_page(spans, page_num=index_value + 1, page_bbox=page_bbox)
if viewport_box_value is not None:
page.viewport_box, page.rot = viewport_box_value, rot
pages.append(page)
# ----- 3) Document-level stats --------------------------------------
doc = DocumentState(pages)
doc.secondary_slot = compute_doc_stats(pages)
# ----- 4) Block clustering per page, then reading order -------------
for page in pages:
ctx = BlockClusterContext(doc.secondary_slot, page.bounds, page.primary_slot, page.lines, page.tertiary_slot)
page.blocks = cluster_lines_into_blocks(ctx)
assign_reading_order(page, page.blocks)
# ----- Early empty-outline gate ------------------------------------
# Short, near-empty, unsupported-script, or mostly-landscape documents
# emit an empty outline rather than a fabricated structure.
if (doc.secondary_slot.state_slot <= 300 or doc.secondary_slot.previous_slot <= 200
or doc.secondary_slot.tertiary_slot in (0, 2, 10) or is_landscape_or_empty(doc)):
if isinstance(doc_handle, (str, Path)):
doc_name = Path(str(doc_handle)).name
else:
doc_name = "document.pdf"
return {
"doc_name": doc_name,
"doc_title": None,
"structure": [],
"has_abstract_or_references_section": False,
}
# ----- 5) Classification: header / footer / watermark / TOC pages ---
detect_header_footer(HeaderFooterContext(doc, 1)) # HEADER
detect_header_footer(HeaderFooterContext(doc, 2)) # FOOTER
mark_watermarks(doc)
mark_toc_and_boilerplate(doc)
# ----- 6) Body-paragraph flagging (post-classification) -------------
# Populates body-paragraph flags, page substantive-body flags,
# and per-page body-style hashes.
from .model import dominant_style_of as span_style_hash
for page in pages:
for block in (page.output_slot or []):
if block.type == 0:
block.is_body_paragraph = is_body_paragraph(doc.secondary_slot, page, block)
if block.is_body_paragraph:
page.state_slot = True
# The empty style hash is significant for later page-level
# membership checks, so it must be retained.
page.style_slot.add(span_style_hash(block))
# ----- 7) Title selection ------------------------------------------
# Title selection and title-echo marking must run before labeled-section
# detection and heading collection so title blocks are excluded from both.
from .classification import bounded_edit_distance, _normalize_text_key
doc_title: Optional[str] = None
title_winner = detect_title(doc)
if title_winner is not None:
# Emit the full joined title string, preserving inter-block spaces.
doc_title = title_winner.to_string()
title_winner.page.auxiliary_slot = True
for block in title_winner.output_slot:
block.type = 3
title_norm = _normalize_text_key(title_winner.to_string()).lower()
# The body-paragraph break exits only the inner block loop; later
# pages are still scanned for title-echo headers.
for candidate_page in doc.primary_slot:
for candidate_block in (candidate_page.output_slot or []):
if candidate_block.type != 0:
continue
normalized = deaccented_text(candidate_block).lower()
if (len(normalized) > 20 and len(title_norm) > 20 and (
normalized.startswith(title_norm)
or title_norm.startswith(normalized)
or title_norm.endswith(normalized))):
candidate_page.auxiliary_slot = True
candidate_block.type = 3
continue
threshold = 0.2 * min(len(normalized), len(title_norm))
if bounded_edit_distance(normalized, title_norm, threshold) < threshold:
candidate_page.auxiliary_slot = True
candidate_block.type = 3
elif candidate_block.is_body_paragraph:
break
# ----- 8) Keyword-labeled section detection -------------------------
# Labeled section regions are built here but extended after heading collection.
caption_context = CaptionContext(doc)
detect_captions(caption_context)
# ----- 9) General heading collection --------------------------------
# Heading collection runs before labeled regions claim their body blocks.
# The start page skips the title page when a title was found.
page_lookup: dict[int, int] = {}
for page in pages:
for block in (page.secondary_slot or []):
page_lookup[id(block)] = page.page_index
from .heading_detection import find_section_openers as _find_section_openers
title_page_idx = title_winner.page.page_index if title_winner is not None else 0
section_openers = _find_section_openers(doc, title_page_idx)
# ----- 10) Extend labeled sections and claim body blocks ------------
# Each labeled heading keeps its label type; body blocks are marked with
# the used-as-heading flag so heading collection skips claimed caption/section bodies.
# The head block type is preserved; claimed body blocks are not retyped.
caption_regions = build_caption_regions(caption_context)
for caption_region in caption_regions:
head_block = caption_region.primary_slot
head_block.state_slot = head_block.marker_slot
for body_block in caption_region.output_slot:
body_block.measure_slot = True
# NOTE: References-section detection -- intentionally absent ---------
# Bulk-marking everything after a references heading would hide later
# appendix headings in some documents, so references detection remains off.
# ref = find_references(doc)
# mark_references(doc, ref)
# NOTE: Ghost-text histogram -- intentionally absent -----------------
# Recurring text is counted during header/footer/watermark marking. A
# second doc-wide pass would double-count headers and pollute title scoring.
# ----- 11) Outline assembly and validation gate ---------------------
outline_nodes = assemble_outline(doc, section_openers)
# Validate the assembled outline. Structured outlines must cover enough
# chapters; unstructured outlines are filtered by script and density gap.
# The abstract/references signal rides along with this gate: it is False on
# the valid-outline branch, and on the other branch it is read off the
# possibly-emptied list once the density filter has run.
if is_outline_valid(doc, outline_nodes):
if not is_chapter_outline_valid(doc, outline_nodes):
outline_nodes = []
has_abstract_or_references = False
else:
mark_outline_block_types(outline_nodes)
page_count = len(doc.primary_slot)
if doc.secondary_slot.tertiary_slot == 7 or (
page_count >= 3
and compute_max_heading_gap(outline_nodes, 1)["max_gap"] > (0.65 if doc.secondary_slot.tertiary_slot == 4 else 0.85) * page_count
):
outline_nodes = []
has_abstract_or_references = has_table_or_prominent(outline_nodes)
if outline_nodes:
structure = outline_to_dict_tree(outline_nodes, total_pages=len(pages))
else:
structure = []
# ----- 12) Output ---------------------------------------------------
if isinstance(doc_handle, (str, Path)):
doc_name = Path(str(doc_handle)).name
else:
doc_name = "document.pdf"
page_texts = []
for page in pages:
parts = []
for block in (page.secondary_slot or []):
parts.append(block_text(block))
page_texts.append("\n".join(parts))
return {
"doc_name": doc_name,
"doc_title": doc_title,
"structure": structure,
"has_abstract_or_references_section": has_abstract_or_references,
"page_texts": page_texts,
}
__all__ = ["extract_toc", "DocumentState", "find_references", "mark_references"]
+121
View File
@@ -0,0 +1,121 @@
"""
Data model for rectangles, spans, lines, blocks, character categories, and
alignment predicates. Coordinate convention follows PDF (origin bottom-left, y increases upward).
``Rect`` is constructed as ``Rect(left, right, top, bottom)``. A few internal
storage fields are implementation details; public callers should use the semantic accessors.
"""
import math
import re
import unicodedata
from decimal import Decimal, ROUND_HALF_UP
from typing import Any, Iterator, Optional, Protocol
import regex as regex_module # supports Unicode \p{...} property classes
from .char_stats import (
_SENTENCE_END_CHARS,
_MINUS_SIGN_CHARS,
_max_nan_propagating,
_min_nan_propagating,
char_category,
is_word_category,
is_punct_category,
_UNICODE_WHITESPACE_CHARS,
_trim_unicode_ws,
_UNICODE_WHITESPACE_CLASS,
_round_half_up_to_int,
CharStats,
merge_char_stats,
letter_count,
punct_count,
info_weight,
is_upper_dominant,
)
from .rects import (
RectLike,
Rect,
EMPTY_RECT,
Bounded,
rect_union,
rect_intersection,
extend_top_to,
extend_bottom_to,
cmp_left_edge,
left_edge_key,
cmp_reading_order,
reading_order_key,
cmp_bottom_edge,
magnitude_ratio,
same_x_extent,
same_y_extent,
intervals_overlap,
y_overlaps,
left_aligned,
right_aligned,
center_aligned,
x_aligned,
x_centers_close,
)
from .span_line import (
_bold_font_re,
_italic_font_re,
_font_name_aliases,
_subset_prefix_re,
Span,
Line,
append_span,
last_span,
_HasCharCount,
avg_char_width,
raw_text_of_line,
text_of_line,
avg_char_width2,
_ONE_DECIMAL_QUANTUM,
_format_half_up_one_decimal,
style_key,
)
from .block import (
Block,
iter_sorted_children,
argmax_key,
last_line_of,
first_span_of,
dominant_style_of,
dominant_font_size,
is_caps_heavy,
is_sentence_like,
heading_score,
case_signal,
alignment_code,
block_text,
deaccented_text,
_COMBINING_MARKS,
_strip_diacritics,
)
from .numbering import (
_NUMBERING_PREFIX_RE,
_BRACKETED_NUM_RE,
_TO_NUMBER_DEC,
_TO_NUMBER_INF,
_TO_NUMBER_HEX,
_TO_NUMBER_OCT,
_TO_NUMBER_BIN,
to_number,
_detect_numbering,
numbering_text,
numbering_value,
numbering_kind,
)
__all__ = [
"RectLike", "Rect", "Bounded", "EMPTY_RECT", "rect_union", "rect_intersection", "extend_top_to", "extend_bottom_to",
"CharStats", "merge_char_stats", "letter_count", "punct_count", "info_weight", "is_upper_dominant",
"char_category", "is_word_category", "is_punct_category",
"Span", "Line", "Block",
"append_span", "last_span", "avg_char_width", "raw_text_of_line", "text_of_line", "avg_char_width2", "style_key", "iter_sorted_children",
"magnitude_ratio", "same_x_extent", "same_y_extent", "intervals_overlap", "y_overlaps", "left_aligned", "right_aligned", "center_aligned", "x_aligned", "x_centers_close",
"to_number", "numbering_text", "numbering_value", "numbering_kind",
"argmax_key", "last_line_of", "first_span_of", "dominant_style_of", "dominant_font_size", "is_caps_heavy", "heading_score", "case_signal", "alignment_code", "block_text", "deaccented_text",
"cmp_left_edge", "left_edge_key", "cmp_reading_order", "reading_order_key", "cmp_bottom_edge",
]
+288
View File
@@ -0,0 +1,288 @@
"""Block type with text, style, and alignment helpers."""
from __future__ import annotations
import math
import re
import unicodedata
from typing import Any, Iterator, Optional, Protocol
from .char_stats import (
is_punct_category,
CharStats,
merge_char_stats,
letter_count,
punct_count,
info_weight,
is_upper_dominant,
)
from .rects import (
EMPTY_RECT,
Bounded,
rect_union,
left_aligned,
right_aligned,
center_aligned,
)
from .span_line import (
Span,
Line,
text_of_line,
style_key,
)
class Block(Bounded):
"""A vertically contiguous group of lines that share layout, such as a paragraph or heading run. Adding lines maintains weighted style, size, text, bbox, reading-order, classification, and cache fields."""
__slots__ = (
"primary_slot", "char_stats", "alignment_slot", "weighted_ratio_tertiary", "previous_slot", "weighted_skew", "weighted_font_size", "weighted_ratio_primary", "weighted_ratio_secondary", "style_slot",
"style_char_counts", "size_char_counts", "reading_order_index", "orig_index", "type", "isolated_centered", "is_body_paragraph", "measure_slot", "used_as_heading",
"state_slot", "marker_slot", "metric_slot",
"dominant_style_cache", "dominant_size_cache", "token_text_cache", "deaccented_text_cache", "cache_slot", "tokens_cache",
)
def __init__(self):
super().__init__(EMPTY_RECT)
self.primary_slot: list = []
self.char_stats: CharStats = CharStats("")
self.alignment_slot: bool = True
self.weighted_ratio_tertiary: float = 0.0
self.previous_slot: float = 0.0
self.weighted_skew: float = 0.0
self.weighted_font_size: float = 0.0
self.weighted_ratio_primary: float = 0.0
self.weighted_ratio_secondary: float = 0.0
self.style_slot: float = 0.0
self.style_char_counts: dict = {}
self.size_char_counts: dict = {}
self.reading_order_index: int = 0
self.orig_index: int = 0
self.type: int = 0
self.isolated_centered: bool = False
self.is_body_paragraph: bool = False
self.measure_slot: bool = False
self.used_as_heading: bool = False
self.state_slot: int = 0
self.marker_slot: int = 0
self.metric_slot: float = 0.0
# caches, invalidated on every add_line
self.dominant_style_cache: Optional[str] = None
self.dominant_size_cache: Optional[float] = None
self.token_text_cache: Optional[str] = None
self.deaccented_text_cache: Optional[str] = None
self.cache_slot: Optional[str] = None
self.tokens_cache: Optional[Any] = None
def __iter__(self):
return iter(self.primary_slot)
def line_count(self) -> int:
"""Line count -- ."""
return len(self.primary_slot)
def line(self):
"""First line -- ."""
return self.primary_slot[0]
def char_count(self) -> int: # type: ignore[override]
"""Total char count across all child lines."""
return self.char_stats.auxiliary_slot
def avg_font_size(self) -> float:
"""Weighted average font size -- ."""
return self.weighted_font_size
def bold_frac(self) -> float:
"""Weighted bold fraction -- ."""
return self.weighted_ratio_tertiary
def skew_frac(self) -> float:
"""Weighted skew fraction -- ."""
return self.weighted_skew
def add_line(self, other_line) -> "Block":
"""Add a line while maintaining weighted style, size, character, bbox, and per-style histograms."""
self.alignment_slot = self.alignment_slot and (len(self.primary_slot) <= 0 or center_aligned(self, other_line, 1))
self.primary_slot.append(other_line)
line = info_weight(self.char_stats)
added_weight = info_weight(other_line.char_stats)
total_weight = line + added_weight
if total_weight > 0:
self.weighted_ratio_tertiary = (self.weighted_ratio_tertiary * line + other_line.bold_frac() * added_weight) / total_weight
self.previous_slot = (self.previous_slot * line + other_line.weighted_ratio_secondary * added_weight) / total_weight
self.weighted_skew = (self.weighted_skew * line + other_line.skew_frac() * added_weight) / total_weight
self.weighted_font_size = (self.weighted_font_size * line + other_line.avg_font_size() * added_weight) / total_weight
self.weighted_ratio_primary = (self.weighted_ratio_primary * line + other_line.cache_slot * added_weight) / total_weight
merge_char_stats(self.char_stats, other_line.char_stats)
if other_line.char_count() <= 0:
return self
line = self.area() # area before union
self.style_slot = max(self.style_slot, other_line.previous_slot)
self.secondary_slot = rect_union(self.secondary_slot, other_line.secondary_slot)
added_weight = self.area() # area after union
if added_weight > 0:
self.weighted_ratio_secondary = (self.weighted_ratio_secondary * line + other_line.cache_slot * other_line.area()) / added_weight
for span in other_line:
sty = style_key(span)
self.style_char_counts[sty] = self.style_char_counts.get(sty, 0) + span.char_count()
# Font-size buckets use half-up rounding to one decimal place.
# Python round is half-to-even, so use floor(x + 0.5) on the
# scaled non-negative font size.
size_key = math.floor(span.font_size * 10 + 0.5) / 10
self.size_char_counts[size_key] = self.size_char_counts.get(size_key, 0) + span.char_count()
# invalidate caches
self.dominant_style_cache = self.dominant_size_cache = self.token_text_cache = self.deaccented_text_cache = self.cache_slot = self.tokens_cache = None
self.metric_slot = 0.0
return self
# Sorted child iterator.
def iter_sorted_children(primary_item):
"""Iterate a page-like object's sorted children as indexed item records."""
for idx, item in enumerate(primary_item.secondary_slot):
yield {"index": idx, "block": item}
# --------------------------------------------------------------------------- #
# Block-level accessors and derived text/style helpers #
# --------------------------------------------------------------------------- #
def argmax_key(items) -> Optional[str]:
"""return the key with max value. ``None`` if empty. ``items`` may be a ``dict`` (in which case we iterate ``.items``) or any iterable of ``(key, value)`` pairs. """
pairs = items.items() if isinstance(items, dict) else items
best: Optional[str] = None
candidate_item = float("-inf")
for reference_item, entry_item in pairs:
if entry_item <= candidate_item:
continue
best = reference_item
candidate_item = entry_item
return best
def last_line_of(block: Block) -> Line:
"""last child line of a block."""
return block.primary_slot[-1]
def first_span_of(block: Block) -> Span:
"""first span of a block's first line."""
return block.line().primary_slot[0]
def dominant_style_of(block: Block) -> str:
"""Cached dominant style hash from the block's style histogram."""
if block.dominant_style_cache is None:
block.dominant_style_cache = argmax_key(block.style_char_counts) or ""
return block.dominant_style_cache
def dominant_font_size(block: Block) -> float:
"""Return cached dominant font size from rounded-size character counts."""
if block.dominant_size_cache is None:
block.dominant_size_cache = float(argmax_key(block.size_char_counts) or 0)
return block.dominant_size_cache
def is_caps_heavy(primary_item) -> bool:
"""Return True if a line or block is uppercase-dominant."""
return is_upper_dominant(primary_item.char_stats) or primary_item.char_stats.primary_slot[2] >= max(2, primary_item.char_stats.auxiliary_slot)
def is_sentence_like(primary_item) -> bool:
"""Return whether a block looks like mixed-case body text rather than a heading. The test requires enough tokens, enough uppercase letters, and rejects long lowercase words."""
from ..tokens import tokenize_block
tokens = tokenize_block(primary_item)
if tokens.length < 3 or is_caps_heavy(primary_item):
return False
upper_count = primary_item.char_stats.primary_slot[2]
if upper_count <= 2 or upper_count < tokens.length / 10:
return False
match = 0
for token in tokens:
# Skip non-word tokens, short tokens, or g==4 (special)
if token.type != 2 or len(token.str) <= 2 or token.primary_slot == 4:
continue
if token.primary_slot == 2:
match += 1
elif len(token.str) >= 5:
return False
return match >= 3
def heading_score(heading) -> float:
"""Line/block heading score: dominant font size plus caps-heavy and bold bonuses."""
return dominant_font_size(heading) + (2 if is_caps_heavy(heading) else 0) + (1 if heading.weighted_ratio_tertiary > 0.5 else 0)
def case_signal(char_stats: CharStats) -> int:
"""Return an uppercase, lowercase, or neutral case signal from character statistics."""
if is_upper_dominant(char_stats) and not is_punct_category(char_stats.secondary_slot) and letter_count(char_stats) > 3 * char_stats.auxiliary_slot / 4 and punct_count(char_stats) < 5:
return 1
if char_stats.primary_slot[3] > 0:
return -1
return 0
def alignment_code(primary_item) -> int:
"""cached block-level alignment code. Returns: 1 fully-justified (every line aligned with the block on left or right) 2 left-aligned (every line shares the block's left) 3 flag-set justified (the block center-alignment flag is set) 4 right-aligned 5 mixed / other """
if primary_item.metric_slot != 0 or len(primary_item.primary_slot) <= 0:
return primary_item.metric_slot
left = True
right = True
any_value = True
for score_value in primary_item.primary_slot:
tolerance = max(1.0, score_value.bbox_width() / 20.0)
line_left_aligned = left_aligned(primary_item, score_value, tolerance)
line_right_aligned = right_aligned(primary_item, score_value, tolerance)
if not line_left_aligned:
left = False
if not line_right_aligned:
right = False
if not (line_left_aligned or line_right_aligned):
any_value = False
if left and not right:
primary_item.metric_slot = 2
elif right and not left:
primary_item.metric_slot = 4
elif any_value:
primary_item.metric_slot = 1
elif primary_item.alignment_slot:
primary_item.metric_slot = 3
else:
primary_item.metric_slot = 5
return primary_item.metric_slot
def block_text(block: Block) -> str:
"""cached joined trimmed text of a block (space-separated)."""
if block.cache_slot is not None:
return block.cache_slot
parts = []
for line_index, line_value in enumerate(block.primary_slot):
parts.append(text_of_line(line_value))
if line_index < len(block.primary_slot) - 1:
parts.append(" ")
block.cache_slot = "".join(parts)
return block.cache_slot
def deaccented_text(block: Block) -> str:
"""Cached diacritic-stripped block text; case and spacing are preserved."""
if block.deaccented_text_cache is not None:
return block.deaccented_text_cache
block.deaccented_text_cache = _strip_diacritics(block_text(block))
return block.deaccented_text_cache
_COMBINING_MARKS = re.compile("[̀-ͯ]")
def _strip_diacritics(text: str) -> str:
"""Strip combining diacritics only while preserving case and internal spacing."""
return unicodedata.normalize(
"NFC", _COMBINING_MARKS.sub("", unicodedata.normalize("NFD", text))
)
+184
View File
@@ -0,0 +1,184 @@
"""Character categories and per-run character statistics."""
from __future__ import annotations
import math
import unicodedata
# --------------------------------------------------------------------------- #
# Character classifier #
# --------------------------------------------------------------------------- #
# Character categories used by tokenization:
# 0 empty
# 1 number (digit / numeral)
# 2 uppercase letter (Lu, Lt)
# 3 lowercase letter (Ll)
# 4 other letter (Lo) -- CJK ideographs, syllabics, etc.
# 5 mark (Mc, Me, Mn)
# 6 sentence-end punct -- . ? ! 。 。 ? ! .
# 7 connector / dash -- _ - — − ⁻ ₋ etc.
# 8 other punctuation
# 9 math symbol (Sm)
# 10 whitespace
# 11 other (symbols, format, control, unassigned)
_SENTENCE_END_CHARS = frozenset(".?!。。?!.")
_MINUS_SIGN_CHARS = frozenset("−⁻₋") # minus, superscript/subscript minus
def _max_nan_propagating(value: float, other_item: float) -> float:
"""propagates NaN (Python ``max`` swallows it)."""
if math.isnan(value) or math.isnan(other_item):
return math.nan
return value if value >= other_item else other_item
def _min_nan_propagating(value: float, other_item: float) -> float:
"""propagates NaN (Python ``min`` swallows it)."""
if math.isnan(value) or math.isnan(other_item):
return math.nan
return value if value <= other_item else other_item
def char_category(char_value: str) -> int:
"""Return the tokenizer character category code from Unicode General_Category."""
if not char_value:
return 0
cat = unicodedata.category(char_value)
# Letters ------------------------------------------------------------------
if cat == "Ll":
return 3
if cat == "Lu" or cat == "Lt":
return 2
if cat == "Lo":
return 4
# Whitespace ---------------------------------------------------------------
# The whitespace set is the Unicode WhiteSpace + LineTerminator set:
# the C0 set \t\n\v\f\r, the BOM , and
# Unicode Space/Line/Paragraph separators (Zs/Zl/Zp). NOT Python's
# str.isspace, which also matches the C0 separators U+001C-U+001F and NEL
# U+0085, which this tokenizer intentionally excludes, and misses .
if char_value in "\t\n\x0b\x0c\r" or char_value == "\ufeff" or cat in ("Zs", "Zl", "Zp"):
return 10
# Sentence-end punctuation -------------------------------------------------
if char_value in _SENTENCE_END_CHARS:
return 6
# Dash / connector punctuation ---------------------------------------------
if cat in ("Pc", "Pd") or char_value in _MINUS_SIGN_CHARS:
return 7
# General punctuation ------------------------------------------------------
if cat.startswith("P"):
return 8
# Number -------------------------------------------------------------------
if cat.startswith("N"):
return 1
# Mark ---------------------------------------------------------------------
if cat.startswith("M"):
return 5
# Math symbol --------------------------------------------------------------
if cat == "Sm":
return 9
return 11
def is_word_category(number: int) -> bool:
"""is c a 'word-y' category (letter / digit / mark)?"""
return number == 3 or number == 2 or number == 1 or number == 5
def is_punct_category(number: int) -> bool:
"""is c a punctuation-y category (dash / punct / sentence)?"""
return number == 7 or number == 8 or number == 6
# Unicode trim strips the package whitespace set used by text parsing.
# Python str.strip uses a DIFFERENT set: it ALSO strips U+001C-001F and U+0085
# Trim keeps U+001C..U+001F and strips U+FEFF to match the intended whitespace set.
# (Same set as parser_pdfium_charlevel._UNICODE_WHITESPACE; defined here to avoid a
# circular import -- parser imports from model, not vice-versa.)
_UNICODE_WHITESPACE_CHARS = (
"\t\n\x0b\x0c\r \xa0 "
"           "
"

   "
)
def _trim_unicode_ws(text: str) -> str:
"""Strip the package whitespace set, not Python's broader ``str.strip`` set."""
return text.strip(_UNICODE_WHITESPACE_CHARS)
# Unicode-compatible ``\s`` = WhiteSpace + LineTerminator = the same 25-cp set as
# _UNICODE_WHITESPACE_CHARS. Bare Python ``\s`` differs: stdlib ``re`` ``\s`` ALSO matches
# U+001C-U+001F and U+0085, the ``regex`` module ``\s`` matches U+0085, and
# NEITHER matches U+FEFF (which does). Splice this char-class BODY into
# regex definitions ("[" + _UNICODE_WHITESPACE_CLASS + "]") instead of a bare ``\s``.
_UNICODE_WHITESPACE_CLASS = r"\t\n\x0b\x0c\r\x20\xa0  - 

   "
def _round_half_up_to_int(value: float) -> int:
"""Round a non-negative finite number to an integer using exact half-up semantics. The ``floor(x + 0.5)`` idiom is not equivalent at the single double ``0.49999999999999994``: adding 0.5 rounds up to ``1.0`` so floor gives 1. Compute the fractional part directly (exact for x >= 0 by Sterbenz) and compare to 0.5."""
score_value = math.floor(value)
frac = value - score_value
if frac < 0.5:
return score_value
return score_value + 1 # frac > 0.5, or an exact 0.5 tie
# --------------------------------------------------------------------------- #
# Per-string character-category accumulator #
# --------------------------------------------------------------------------- #
class CharStats:
"""Collect first/last character category, per-category counts, and total character count."""
__slots__ = ("secondary_slot", "tertiary_slot", "primary_slot", "auxiliary_slot")
def __init__(self, other_text: str):
self.secondary_slot = 0
self.tertiary_slot = 0
self.primary_slot = [0] * 12
self.auxiliary_slot = 0
for secondary_item in other_text:
cat = char_category(secondary_item)
if self.secondary_slot == 0:
self.secondary_slot = cat
self.tertiary_slot = cat
self.primary_slot[cat] += 1
self.auxiliary_slot += 1
def merge_char_stats(char_stats: CharStats, other_char_stats: CharStats) -> None:
"""merge b into a in place."""
if char_stats.secondary_slot == 0:
char_stats.secondary_slot = other_char_stats.secondary_slot
if other_char_stats.tertiary_slot != 0:
char_stats.tertiary_slot = other_char_stats.tertiary_slot
for candidate_item in range(12):
char_stats.primary_slot[candidate_item] += other_char_stats.primary_slot[candidate_item]
char_stats.auxiliary_slot += other_char_stats.auxiliary_slot
def letter_count(char_stats: CharStats) -> int:
"""count of letter-like chars (uppercase + lowercase + other-letter)."""
return char_stats.primary_slot[3] + char_stats.primary_slot[2] + char_stats.primary_slot[4]
def punct_count(char_stats: CharStats) -> int:
"""count of sentence-punctuation chars (6 + 7 + 8)."""
return char_stats.primary_slot[6] + char_stats.primary_slot[7] + char_stats.primary_slot[8]
def info_weight(char_stats: CharStats) -> float:
"""'informational' weight. ``letters + 2*other_letter + 0.5*(non-letter)`` -- biases towards alphabetic content; non-letter chars contribute half. """
return char_stats.primary_slot[3] + char_stats.primary_slot[2] + 2 * char_stats.primary_slot[4] + 0.5 * (char_stats.auxiliary_slot - letter_count(char_stats))
def is_upper_dominant(char_stats: CharStats) -> bool:
"""uppercase-dominant string detector. True iff (uppercase chars) > max(letters*3/4, letters-4) and (uppercase chars) > max(3, total/3). """
secondary_item = char_stats.primary_slot[2]
candidate_item = letter_count(char_stats)
return secondary_item > max(candidate_item * 3 / 4, candidate_item - 4) and secondary_item > max(3, char_stats.auxiliary_slot / 3)
+131
View File
@@ -0,0 +1,131 @@
"""Numbering-prefix detection and numeric parsing."""
from __future__ import annotations
import math
import re
import unicodedata
import regex as regex_module # supports Unicode \p{...} property classes
from .char_stats import (
_trim_unicode_ws,
_UNICODE_WHITESPACE_CLASS,
)
from .span_line import (
Line,
raw_text_of_line,
)
# --------------------------------------------------------------------------- #
# Numbering detection #
# --------------------------------------------------------------------------- #
# Uses Unicode property classes (\p{Number} / \P{Number}), compiled with the
# ``regex`` module (stdlib ``re`` can't express them). Matches:
# - leading roman or digit (group 1)
# - dotted lowercase a-h (group 2)
# - dotted lowercase ivx (group 3)
_NUMBERING_PREFIX_RE = regex_module.compile(
r"^(?:"
r"([IVX]+|[1-91-9]\p{Number}?)(?:[..。。):]|-\P{Number}|-$|[" + _UNICODE_WHITESPACE_CLASS + r"]|$)"
r"|(?:([A-Ha-h])|([ivx]))[..。。)]"
r")"
)
# Bracketed numeric labels such as "[1]" or "(1)".
_BRACKETED_NUM_RE = re.compile(r"^[\[\(] *([1-9][0-9]?) *[\)\]]")
# string grammar (ToNumber). ASCII digits ONLY: Python's
# ``\d`` and ``float`` both accept Unicode decimal digits (e.g. Arabic-Indic
# ٢) and ``float`` also accepts ``1_000`` / ``inf`` / ``nan``, none of which
# ``Number`` accepts -- hence the explicit ``[0-9]`` classes.
_TO_NUMBER_DEC = re.compile(r"^[+-]?(?:[0-9]+\.?[0-9]*|\.[0-9]+)(?:[eE][+-]?[0-9]+)?$")
_TO_NUMBER_INF = re.compile(r"^[+-]?Infinity$")
_TO_NUMBER_HEX = re.compile(r"^0[xX][0-9a-fA-F]+$")
_TO_NUMBER_OCT = re.compile(r"^0[oO][0-7]+$")
_TO_NUMBER_BIN = re.compile(r"^0[bB][01]+$")
def to_number(text: str) -> float:
"""NFKC-normalized numeric conversion with decimal, exponent, hex, octal, binary, and Infinity forms."""
if text is None:
return math.nan
token_value = _trim_unicode_ws(unicodedata.normalize("NFKC", text))
if token_value == "":
return 0.0
if _TO_NUMBER_INF.match(token_value):
return -math.inf if token_value[0] == "-" else math.inf
if _TO_NUMBER_HEX.match(token_value):
return float(int(token_value[2:], 16))
if _TO_NUMBER_OCT.match(token_value):
return float(int(token_value[2:], 8))
if _TO_NUMBER_BIN.match(token_value):
return float(int(token_value[2:], 2))
if _TO_NUMBER_DEC.match(token_value):
return float(token_value)
return math.nan
def _detect_numbering(line: Line) -> None:
"""Detect leading section numbering and cache the numbering kind and text on the line."""
if line.state_slot != -1:
return # already computed
line.state_slot = 0
if line.char_count() <= 0:
return
# Drop-capital / large-first-char detection (layout branch).
# If first span is smaller, sits above the next non-empty span, and is
# numeric -> use that span's text as the numbering.
if len(line.primary_slot) > 1:
secondary_item = line.primary_slot[0]
candidate_item = line.primary_slot[2] if (line.primary_slot[1].char_count() <= 0 and len(line.primary_slot) > 2) else line.primary_slot[1]
if (
secondary_item.bbox_height() < candidate_item.bbox_height()
and secondary_item.bottom_edge() > candidate_item.bottom_edge() + 0.05 * candidate_item.bbox_height()
and not math.isnan(to_number(secondary_item.text))
):
line.state_slot = 1
line.style_slot = secondary_item.text
return
text = raw_text_of_line(line)
measure_item = _NUMBERING_PREFIX_RE.match(text)
if measure_item and measure_item.group(1) and "1" <= measure_item.group(1)[0] <= "9":
line.state_slot = 1
line.style_slot = measure_item.group(1)
return
if measure_item and (measure_item.group(1) or measure_item.group(3)):
# Roman uppercase (group 1) or other -- both uppercase-ish
line.state_slot = 2
line.style_slot = measure_item.group(1) or measure_item.group(3)
return
if measure_item and measure_item.group(2):
line.state_slot = 3
line.style_slot = measure_item.group(2)
return
second_matrix = _BRACKETED_NUM_RE.match(text)
if second_matrix:
line.state_slot = 1
line.style_slot = second_matrix.group(1)
return
def numbering_text(line: Line) -> str:
"""get the cached numbering string."""
_detect_numbering(line)
return line.style_slot
def numbering_value(line: Line) -> float:
"""get numbering as a number, NaN if non-digit numbering."""
text = numbering_text(line)
return to_number(text) if line.state_slot == 1 else math.nan
def numbering_kind(line: Line) -> int:
"""get numbering type (0 none, 1 digit, 2 upper, 3 lower)."""
_detect_numbering(line)
return line.state_slot
+229
View File
@@ -0,0 +1,229 @@
"""Rectangle types, geometry predicates, and ordering comparators."""
from __future__ import annotations
import math
from .char_stats import (
_max_nan_propagating,
_min_nan_propagating,
)
# --------------------------------------------------------------------------- #
# Rectangle model #
# --------------------------------------------------------------------------- #
class RectLike:
"""Empty base for objects that expose bbox accessors."""
pass
class Rect(RectLike):
"""Axis-aligned bbox. PDF coordinates: top > bottom (y increases upward). """
__slots__ = ("left", "right", "top", "primary_slot")
def __init__(self, other_item: float, candidate_item: float, reference_item: float, next_item: float):
self.left = other_item
self.right = candidate_item
self.top = reference_item
self.primary_slot = next_item # bottom
# --- geometry accessors ----------------------------------
def left_edge(self) -> float: return self.left
def right_edge(self) -> float: return self.right
def top_edge(self) -> float: return self.top
def bottom_edge(self) -> float: return self.primary_slot # bottom
def bbox_width(self) -> float: return _max_nan_propagating(0.0, self.right - self.left) # width
def bbox_height(self) -> float: return _max_nan_propagating(0.0, self.top - self.primary_slot) # height
def area(self) -> float: return self.bbox_width() * self.bbox_height() # area
def center_x(self) -> float: return (self.left + self.right) / 2 # x-center
def center_y(self) -> float: return (self.top + self.primary_slot) / 2 # y-center
def contains(self, other_rect: "Rect") -> bool:
return (
self.left <= other_rect.left
and self.right >= other_rect.right
and self.top >= other_rect.top
and self.primary_slot <= other_rect.primary_slot
)
# Shared empty / inverted rectangle used to initialize accumulators.
EMPTY_RECT = Rect(math.inf, -math.inf, -math.inf, math.inf)
class Bounded(RectLike):
"""Mixin-style wrapper around an owned ``Rect``."""
__slots__ = ("secondary_slot",)
def __init__(self, other_rect: Rect):
self.secondary_slot = other_rect
def left_edge(self) -> float: return self.secondary_slot.left
def right_edge(self) -> float: return self.secondary_slot.right
def top_edge(self) -> float: return self.secondary_slot.top
def bottom_edge(self) -> float: return self.secondary_slot.primary_slot
def bbox_width(self) -> float: return self.secondary_slot.bbox_width()
def bbox_height(self) -> float: return self.secondary_slot.bbox_height()
def area(self) -> float: return self.secondary_slot.area()
def center_x(self) -> float: return self.secondary_slot.center_x()
def center_y(self) -> float: return self.secondary_slot.center_y()
def rect_union(rect: Rect, other_rect: Rect) -> Rect:
"""bbox union."""
return Rect(
_min_nan_propagating(rect.left, other_rect.left),
_max_nan_propagating(rect.right, other_rect.right),
_max_nan_propagating(rect.top, other_rect.top),
_min_nan_propagating(rect.primary_slot, other_rect.primary_slot),
)
def rect_intersection(rect: Rect, other_rect: Rect) -> Rect:
"""bbox intersection; disjoint boxes may have inverted horizontal or vertical edges."""
return Rect(
_max_nan_propagating(rect.left, other_rect.left),
_min_nan_propagating(rect.right, other_rect.right),
_min_nan_propagating(rect.top, other_rect.top),
_max_nan_propagating(rect.primary_slot, other_rect.primary_slot),
)
def extend_top_to(rect: Rect, other_item: float) -> Rect:
"""Clip the rectangle top to be at least ``other_value``."""
return Rect(rect.left, rect.right, _max_nan_propagating(rect.top, other_item), rect.primary_slot)
def extend_bottom_to(rect: Rect, other_item: float) -> Rect:
"""Clip the rectangle bottom to be at most ``other_value``."""
return Rect(rect.left, rect.right, rect.top, _min_nan_propagating(rect.primary_slot, other_item))
# --------------------------------------------------------------------------- #
# Sort comparators #
# --------------------------------------------------------------------------- #
def cmp_left_edge(left_value: Bounded, right_value: Bounded) -> float:
"""Order by (left asc, right asc, top desc, bottom desc). Returns the raw delta, not a normalised -1/0/1, because callers only consume the sign."""
if left_value.left_edge() != right_value.left_edge():
return left_value.left_edge() - right_value.left_edge()
if left_value.right_edge() != right_value.right_edge():
return left_value.right_edge() - right_value.right_edge()
if left_value.top_edge() != right_value.top_edge():
return right_value.top_edge() - left_value.top_edge()
return right_value.bottom_edge() - left_value.bottom_edge()
# Python's ``sorted`` accepts a key, not a cmp. Provide key functions too.
def left_edge_key(primary_item: Bounded) -> tuple:
return (primary_item.left_edge(), primary_item.right_edge(), -primary_item.top_edge(), -primary_item.bottom_edge())
def cmp_reading_order(left_value: Bounded, right_value: Bounded) -> float:
"""Order by (top desc, bottom desc, left asc, right asc). Top-of-page rows come first; within a row, leftmost first. Returns the raw delta because callers only consume the sign."""
if left_value.top_edge() != right_value.top_edge():
return right_value.top_edge() - left_value.top_edge()
if left_value.bottom_edge() != right_value.bottom_edge():
return right_value.bottom_edge() - left_value.bottom_edge()
if left_value.left_edge() != right_value.left_edge():
return left_value.left_edge() - right_value.left_edge()
return left_value.right_edge() - right_value.right_edge()
def reading_order_key(primary_item: Bounded) -> tuple:
return (-primary_item.top_edge(), -primary_item.bottom_edge(), primary_item.left_edge(), primary_item.right_edge())
def cmp_bottom_edge(left_value: Bounded, right_value: Bounded) -> float:
"""Order by (bottom asc, top asc, left asc, right asc). Returns the raw delta because callers only consume the sign."""
if left_value.bottom_edge() != right_value.bottom_edge():
return left_value.bottom_edge() - right_value.bottom_edge()
if left_value.top_edge() != right_value.top_edge():
return left_value.top_edge() - right_value.top_edge()
if left_value.left_edge() != right_value.left_edge():
return left_value.left_edge() - right_value.left_edge()
return left_value.right_edge() - right_value.right_edge()
# --------------------------------------------------------------------------- #
# Alignment / overlap predicates #
# --------------------------------------------------------------------------- #
def magnitude_ratio(value: float, other_item: float) -> float:
"""Return the larger-magnitude-over-smaller-magnitude ratio with IEEE-754 division semantics. Division by zero yields +/-Infinity for a nonzero non-NaN numerator and NaN for +/-0 over +/-0 and NaN over +/-0. Downstream threshold tests rely on signed infinity, so divide-by-zero must not be collapsed to NaN. """
# NaN comparisons take the false arm, which selects ``other_value / value``.
if abs(value) > abs(other_item):
num, den = value, other_item
else:
num, den = other_item, value
# raw `num/den`. Python raises ZeroDivisionError on den == +/-0, so the
# IEEE cases are spelled out: x/±0 = ±Infinity with sign(x) XOR sign(±0)
# 5/-0 = -Infinity, ±0/±0 = NaN, NaN/±0 = NaN. A NaN denominator passes
# `den != 0` and divides through to NaN.
if den != 0:
return num / den
if num == 0 or math.isnan(num):
return math.nan
return math.copysign(math.inf, num) * math.copysign(1.0, den)
def same_x_extent(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
"""Return whether both horizontal edges are within the tolerance."""
return abs(primary_item.left_edge() - secondary_item.left_edge()) <= candidate_item and abs(primary_item.right_edge() - secondary_item.right_edge()) <= candidate_item
def same_y_extent(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
"""Return whether both vertical edges are within the tolerance."""
return abs(primary_item.top_edge() - secondary_item.top_edge()) <= candidate_item and abs(primary_item.bottom_edge() - secondary_item.bottom_edge()) <= candidate_item
def intervals_overlap(value: float, other_item: float, candidate_item: float, reference_item: float) -> bool:
"""Return whether the two closed ranges overlap by either endpoint."""
return (value <= candidate_item and candidate_item <= other_item) or (candidate_item <= value and value <= reference_item)
def y_overlaps(primary_item: Bounded, secondary_item: Bounded) -> bool:
"""Return whether the vertical intervals of two boxes overlap."""
return intervals_overlap(primary_item.bottom_edge(), primary_item.top_edge(), secondary_item.bottom_edge(), secondary_item.top_edge())
def left_aligned(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
"""Return whether left edges match within the tolerance."""
return abs(primary_item.left_edge() - secondary_item.left_edge()) <= candidate_item
def right_aligned(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
"""Return whether right edges match within the tolerance."""
return abs(primary_item.right_edge() - secondary_item.right_edge()) <= candidate_item
def center_aligned(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
"""Return whether two boxes are center-aligned within the tolerance. Their left and right edge offsets must have opposite signs, then pass the center-distance tolerance."""
reference_item = primary_item.left_edge() - secondary_item.left_edge()
entry_item = primary_item.right_edge() - secondary_item.right_edge()
def sign(signed_delta):
if signed_delta > 0: return 1
if signed_delta < 0: return -1
return 0
if sign(reference_item) != -sign(entry_item):
return False
return abs(primary_item.center_x() - secondary_item.center_x()) <= max(candidate_item, min(abs(reference_item), abs(entry_item)) / 2)
def x_aligned(primary_item: Bounded, secondary_item: Bounded, candidate_item: float) -> bool:
"""any of left / right / center aligned."""
return left_aligned(primary_item, secondary_item, candidate_item) or right_aligned(primary_item, secondary_item, candidate_item) or center_aligned(primary_item, secondary_item, candidate_item)
def x_centers_close(primary_item: Bounded, secondary_item: Bounded) -> bool:
"""Return whether x-centers match within the secondary box width tolerance."""
return abs(secondary_item.center_x() - primary_item.center_x()) <= max(1, secondary_item.bbox_width() / 10)
+228
View File
@@ -0,0 +1,228 @@
"""Span and Line types with text and style helpers."""
from __future__ import annotations
import re
from decimal import Decimal, ROUND_HALF_UP
from typing import Any, Iterator, Optional, Protocol
from .char_stats import (
_trim_unicode_ws,
CharStats,
merge_char_stats,
letter_count,
info_weight,
)
from .rects import (
Rect,
EMPTY_RECT,
Bounded,
rect_union,
)
# --------------------------------------------------------------------------- #
# Text span #
# --------------------------------------------------------------------------- #
# Font-style detectors. Neither pattern is multiline or Unicode-aware: the end
# anchor binds at end of INPUT (Python's `$` would also match before a trailing
# newline, hence `\Z`), and case folding stays ASCII-only, so U+017F, U+0130
# and U+0131 do not fold onto "s"/"i". The digit classes are spelled out, so
# the ASCII flag touches nothing else here.
_bold_font_re = re.compile(r"(bold|timesb)", re.IGNORECASE | re.ASCII)
_italic_font_re = re.compile(r"(ital|it\Z|i[1-9][0-9]*\Z|obliq)", re.IGNORECASE | re.ASCII)
# Font-name canonicalization map.
_font_name_aliases = {
"timesnewroman": "Times",
"times-new-roman": "Times",
"timesroman": "Times",
"times-roman": "Times",
"timesnew": "Times",
"times-new": "Times",
}
# subset prefix regex: 6 uppercase letters + plus sign
_subset_prefix_re = re.compile(r"^[A-Z]{6}\+")
class Span(Bounded):
"""Span emitted by one text-showing item. Stores raw and trimmed text, character statistics, skew, font family/name, font size, bold/italic flags, and bbox helpers."""
__slots__ = (
"text", "state_slot", "char_stats", "previous_slot", "font_family", "font_name", "font_size", "primary_slot", "measure_slot",
)
def __init__(
self,
bbox: Rect,
text: str,
font_name_raw: str,
font_size: float,
bold: bool,
italic: bool,
skew: float = 0.0,
font_family: str = "",
):
"""Create a span from parser-normalized text, font, style, skew, and bounding-box fields."""
super().__init__(bbox)
self.text = text
self.state_slot = _trim_unicode_ws(text)
self.char_stats = CharStats(self.state_slot)
self.previous_slot = skew
self.font_family = font_family
# ---- font name normalisation -----------
reference_item = font_name_raw
if _subset_prefix_re.match(reference_item):
reference_item = reference_item[7:]
reference_item = _font_name_aliases.get(reference_item.lower(), reference_item)
self.font_name = reference_item
self.font_size = font_size
# Bold comes from the adapter flag or from the normalized font name.
self.primary_slot = bool(bold) or bool(_bold_font_re.search(self.font_name))
# Italic is name-derived only. The ``italic`` parameter is accepted
# for adapter compatibility but is not consulted.
del italic # noqa: F841 -- explicitly drop the arg
self.measure_slot = bool(_italic_font_re.search(self.font_name))
def char_count(self) -> int: # type: ignore[override]
"""Span char count."""
return self.char_stats.auxiliary_slot
def font_style(self) -> str:
""""<fontName> B" or "<fontName> R"."""
return f"{self.font_name} {'B' if self.primary_slot else 'R'}"
# --------------------------------------------------------------------------- #
# Text line #
# --------------------------------------------------------------------------- #
class Line(Bounded):
"""A list of spans on roughly the same baseline, with line-wide character statistics, first letter-bearing span, weighted bold/italic/skew/font-size aggregates, cached text, numbering state, column index, and span list."""
__slots__ = (
"primary_slot", "char_stats", "alignment_slot", "weighted_ratio_primary", "weighted_ratio_secondary", "weighted_ratio_tertiary", "metric_slot", "previous_slot", "measure_slot", "marker_slot", "state_slot", "style_slot", "cache_slot",
)
def __init__(self):
super().__init__(EMPTY_RECT)
self.primary_slot: list[Span] = []
self.char_stats: CharStats = CharStats("")
self.alignment_slot: Optional[Span] = None
self.weighted_ratio_primary: float = 0.0
self.weighted_ratio_secondary: float = 0.0
self.weighted_ratio_tertiary: float = 0.0
self.metric_slot: float = 0.0
self.previous_slot: float = 0.0
# Column index assigned by the column pass; -1 means unassigned.
self.measure_slot: int = -1
self.marker_slot: Optional[str] = None
self.state_slot: int = -1
self.style_slot: str = ""
self.cache_slot: float = 0.0
def __iter__(self) -> Iterator[Span]:
return iter(self.primary_slot)
def char_count(self) -> int: # type: ignore[override]
return self.char_stats.auxiliary_slot
def avg_font_size(self) -> float:
return self.metric_slot
def bold_frac(self) -> float:
return self.weighted_ratio_primary
def skew_frac(self) -> float:
return self.weighted_ratio_tertiary
def append_span(line: Line, other_span: Span) -> Line:
"""append span b into line a, updating weighted fields. Every aggregate field is updated in one pass so downstream line scoring sees the same weighted style, size, and geometry summaries. """
line.primary_slot.append(other_span)
span = info_weight(line.char_stats)
added_weight = info_weight(other_span.char_stats)
total_weight = span + added_weight
if total_weight > 0:
line.weighted_ratio_primary = (line.weighted_ratio_primary * span + (1 if other_span.primary_slot else 0) * added_weight) / total_weight
line.weighted_ratio_secondary = (line.weighted_ratio_secondary * span + (1 if other_span.measure_slot else 0) * added_weight) / total_weight
line.weighted_ratio_tertiary = (line.weighted_ratio_tertiary * span + other_span.previous_slot * added_weight) / total_weight
line.metric_slot = (line.metric_slot * span + other_span.font_size * added_weight) / total_weight
merge_char_stats(line.char_stats, other_span.char_stats)
if line.alignment_slot is None and letter_count(other_span.char_stats) > 0:
line.alignment_slot = other_span
if other_span.char_count() <= 0:
return line
span = line.area()
line.previous_slot = max(line.previous_slot, other_span.bbox_height())
line.secondary_slot = rect_union(line.secondary_slot, other_span.secondary_slot)
line.cache_slot = min(1.0, (line.cache_slot * span + other_span.area()) / max(1.0, line.area()))
line.marker_slot = None
line.state_slot = -1
line.style_slot = ""
return line
def last_span(line: Line) -> Span:
"""last span of line."""
return line.primary_slot[-1]
class _HasCharCount(Protocol):
"""Anything with K (char count), A (width), N (height)."""
def char_count(self) -> int: ...
def bbox_width(self) -> float: ...
def bbox_height(self) -> float: ...
def avg_char_width(primary_item: _HasCharCount) -> float:
"""Width per character. Returns 0 if there are no characters."""
return 0.0 if primary_item.char_count() <= 0 else primary_item.bbox_width() / primary_item.char_count()
def raw_text_of_line(line: Line) -> str:
"""concatenate raw text of all spans (no trimming)."""
parts = []
for span in line.primary_slot:
parts.append(span.text)
return "".join(parts)
def text_of_line(line: Line) -> str:
"""cached trimmed line text."""
if line.marker_slot is not None:
return line.marker_slot
line.marker_slot = _trim_unicode_ws(raw_text_of_line(line))
return line.marker_slot
def avg_char_width2(primary_item: _HasCharCount) -> float:
"""Width per character. Returns 0 for empty text."""
return 0.0 if primary_item.char_count() <= 0 else primary_item.bbox_width() / primary_item.char_count()
# --------------------------------------------------------------------------- #
# Text block #
# --------------------------------------------------------------------------- #
_ONE_DECIMAL_QUANTUM = Decimal("0.1")
def _format_half_up_one_decimal(value: float) -> str:
"""Round the exact double half-away-from-zero; the stats module uses the same helper."""
return str(Decimal(value).quantize(_ONE_DECIMAL_QUANTUM, rounding=ROUND_HALF_UP))
def style_key(span: "Span") -> str:
"""Style hash ``"<fontName> <B|R> <size rounded to 0.1>"`` using shared half-up rounding."""
return f"{span.font_style()} {_format_half_up_one_decimal(span.font_size)}"
+54
View File
@@ -0,0 +1,54 @@
"""Heading predicates and section-keyword helpers. The full outline tree is assembled in ``outline_assembly``. This module keeps
the lower-level heading checks that decide whether a block is a plausible
outline heading based on numbering, style, geometry, and section-keyword tries.
"""
import re
from collections import defaultdict
from typing import Optional
from ..labels import extract_structural_number
from ..model import numbering_text, numbering_kind, block_text, is_caps_heavy, Block
from ..stats import column_index_of
from ..tokens import set_case_fold, TrieConfig, build_trie, tokenize_block, trie_full_match
from .filtering import (
SECTION_KEYWORD_TRIE,
heading_order_key,
_NON_HEADING_TYPES,
_DOT_LEADER_RE,
_CAPTION_LABEL_RE,
_EQUATION_LABEL_RE,
_PAREN_FRAGMENT_RE,
_looks_like_pseudo_code,
is_heading_candidate,
_style_key,
_numbering_depth,
collect_headings,
_heading_signature,
_matches_section_keywords,
_PSEUDO_CODE_PATTERNS,
_AUTHOR_PATTERNS,
_BULLET_LIST_RE,
filter_by_clique,
)
from .tree import (
extract_top_level_headings,
assign_levels,
_heading_title,
_heading_page_num,
build_tree,
validate,
)
__all__ = [
"is_heading_candidate",
"collect_headings",
"filter_by_clique",
"assign_levels",
"build_tree",
"validate",
"extract_top_level_headings",
"SECTION_KEYWORD_TRIE",
"heading_order_key",
]
+241
View File
@@ -0,0 +1,241 @@
"""Heading candidate collection, filtering, and keyword screening."""
from __future__ import annotations
import re
from collections import defaultdict
from typing import Optional
from ..labels import extract_structural_number
from ..model import numbering_text, numbering_kind, block_text, is_caps_heavy, Block
from ..stats import column_index_of
from ..tokens import set_case_fold, TrieConfig, build_trie, tokenize_block, trie_full_match
# English section keywords loaded into a case-folded trie matching tokenized
# block text exactly.
SECTION_KEYWORD_TRIE = build_trie(
[
"acknowledgements", "acknowledgments",
"background",
"conclusion", "conclusions",
"discussion",
"introduction",
"materials and methods",
"method", "methods",
"results",
],
set_case_fold(TrieConfig(), True),
)
def heading_order_key(block: Block, page_lookup: dict[int, int]) -> tuple:
"""Sort by page, then column index and reading position."""
return (
page_lookup.get(id(block), 1),
column_index_of(block),
-block.top_edge(),
-block.bottom_edge(),
block.left_edge(),
block.right_edge(),
)
# --------------------------------------------------------------------------- #
# Heading candidate gates #
# --------------------------------------------------------------------------- #
# Block types excluded from heading candidacy:
# 1 header, 2 footer, 3 references body, 9 TOC page content,
# 12 watermark/caption, 13 claimed labeled-section body, 99 title.
_NON_HEADING_TYPES = frozenset({1, 2, 3, 9, 12, 13, 99})
_DOT_LEADER_RE = re.compile(r"\.{4,}\s*\d+\s*$")
_CAPTION_LABEL_RE = re.compile(
r"^\s*(?:figure|fig\.?|table|tab\.?|algorithm|alg\.?|equation|eq\.?|listing)\s+\d",
re.IGNORECASE,
)
# Equation labels like "(1)", "(2.3)", "(a)", "(i)", "(*)" -- parenthesised
# short labels that the numbering detector mistakes for "1." section starts.
_EQUATION_LABEL_RE = re.compile(
r"^\s*[\[\(]\s*(?:[0-9]+(?:\.\d+)?[a-z]?|[a-z]|[ivx]+)\s*[\)\]]\s*$",
re.IGNORECASE,
)
# Parenthesised body fragments like "(current cost)", "(estimated cost)".
_PAREN_FRAGMENT_RE = re.compile(r"^\s*[\[\(][^\]\)]{1,40}[\]\)]\s*$")
def _looks_like_pseudo_code(text: str) -> bool:
"""Reject pseudo-code and math fragments that can resemble numbered headings."""
if any(pat.search(text) for pat in _PSEUDO_CODE_PATTERNS):
return True
# No alphabetic word of >= 3 letters? Reject.
if not re.search(r"[A-Za-zÀ-ÿ一-鿿가-힯]{3,}", text):
return True
# Bullet-list item: "1. long flowing prose..."
if _BULLET_LIST_RE.match(text) and len(text) > 80:
return True
# Parenthesised fragment: "(current cost)", "(maximum flow)"
if _PAREN_FRAGMENT_RE.match(text):
return True
# Author-block heuristics
if any(pat.search(text) for pat in _AUTHOR_PATTERNS):
return True
return False
def is_heading_candidate(block: Block, body_size: float, body_bold: bool) -> bool:
"""Return whether a block has the visual and textual shape of a heading."""
if block.type in _NON_HEADING_TYPES:
return False
if block.line_count() > 6:
return False
text = block_text(block).strip()
if len(text) < 2 or len(text) > 200:
return False
if _DOT_LEADER_RE.search(text):
return False
if _CAPTION_LABEL_RE.match(text):
return False
if _EQUATION_LABEL_RE.match(text):
return False
if _looks_like_pseudo_code(text):
return False
marker_type = getattr(block, "marker_slot", 0)
if marker_type == 4:
return True
block_size = block.avg_font_size() or body_size
size_gain = block_size / max(body_size, 1e-3)
if size_gain >= 1.08:
return True
if block.bold_frac() > 0.5 and not body_bold and size_gain >= 0.95:
return True
if block.line_count() >= 1 and numbering_kind(block.line()) != 0 and len(text) <= 120 and (
block.bold_frac() > 0.3 or size_gain >= 1.0
):
return True
if is_caps_heavy(block) and len(text) <= 80 and size_gain >= 1.0:
return True
return False
# --------------------------------------------------------------------------- #
# Style buckets + level assignment #
# --------------------------------------------------------------------------- #
def _style_key(block: Block) -> tuple[str, float, bool]:
"""Hashable signature for grouping headings into hierarchy levels."""
first_span = block.line().primary_slot[0] if block.line().primary_slot else None
font = first_span.font_name if first_span else ""
return (font, round(block.avg_font_size(), 1), block.bold_frac() > 0.5)
def _numbering_depth(block: Block) -> Optional[int]:
"""Return the section-numbering depth, e.g. ``1.2.3-> 3. None if the block doesn't start with a digit-style number (only digit chains use ``.``-separated depth; Roman / letter labels return 1). """
if numbering_kind(block.line()) != 1:
return None
text = numbering_text(block.line())
if not text:
return None
if re.match(r"^\d+(?:\.\d+)*$", text):
return text.count(".") + 1
return 1
def collect_headings(doc) -> list[Block]:
"""Walk all pages, gather heading-candidate blocks in reading order."""
body_size = doc.secondary_slot.primary_slot
body_bold = doc.secondary_slot.tertiary_slot == 0
out: list[Block] = []
for page in doc.primary_slot:
for block in (page.secondary_slot or []):
if is_heading_candidate(block, body_size, body_bold):
out.append(block)
return out
# --------------------------------------------------------------------------- #
# Clique selection #
# --------------------------------------------------------------------------- #
def _heading_signature(block: Block) -> str:
"""Return the first visible span's font style for a heading block."""
if not block.primary_slot or not block.primary_slot[0].primary_slot:
return ""
return block.primary_slot[0].primary_slot[0].font_style()
def _matches_section_keywords(block: Block) -> bool:
"""Full-match canonical English section names after stripping a leading structural number. For example, "1 Introduction" tokenizes as ["1", "Introduction"], and the numeric prefix must be removed before keyword matching."""
tokens = tokenize_block(block)
prefix = extract_structural_number(tokens)
if prefix is not None:
tokens = tokens.slice(prefix.length)
return trie_full_match(SECTION_KEYWORD_TRIE, tokens)
_PSEUDO_CODE_PATTERNS = (
re.compile(r"^\s*\d+\s*:"), # "2:" / "10:" lead -> pseudo-code step
re.compile(r"[∀-⋿←-⇿≤≥≠∈∉∂∇∑∏√]"), # math operators
re.compile(r"^\s*\d+[a-z]"), # "9else", "15return" (no space)
re.compile(r"^[\d.\s/]+$"), # pure numbers / decimals
re.compile(r"^\s*\d+\s*[+\-*/=]\s*\d"), # arithmetic
re.compile(r"^\s*[a-z]+\s*[+\-*/=]\s*"), # variable assignments
re.compile(r"\bwhile\b|\bif\b|\belse\b|\bfor\b|\breturn\b|\bdo\b", re.IGNORECASE), # code keywords
# Subfigure captions like "(a) RETINA", "(b) IRMA", "(i) plot"
re.compile(r"^\s*[\(\[]\s*[a-zivx]+\s*[\)\]]\s+\w"),
)
# Author block / affiliation patterns:
# * "Yu Tang†, Leong Hou U‡, ..." -- multiple comma-separated names with
# affiliation markers
# * "Kimi Team" -- short "X Team" / "X Lab" / "X Group" naming
# * "†The University of ..." -- starts with affiliation marker
# * "{user, another}@domain" -- email block
_AUTHOR_PATTERNS = (
re.compile(r"[†‡§¶∗*]"), # affiliation markers
re.compile(r"@\S+\."), # contains email
re.compile(r"^\s*\S+\s+(?:Team|Group|Lab|Labs|Inc\.|Corp\.|Co\.)\s*$"),
)
# Bullet-list items: "1. " followed by long flowing text (>80 chars total)
_BULLET_LIST_RE = re.compile(r"^\s*\d+\.\s+\w")
def filter_by_clique(headings: list[Block]) -> list[Block]:
"""Keep headings that match the dominant section-keyword style group."""
if not headings:
return headings
anchor_groups: dict[str, list[Block]] = defaultdict(list)
for state_item in headings:
if _matches_section_keywords(state_item):
anchor_groups[_heading_signature(state_item)].append(state_item)
if not anchor_groups:
return headings
winner_sig, winner_group = max(anchor_groups.items(), key=lambda item_pair: len(item_pair[1]))
if len(winner_group) <= 1:
return headings
out: list[Block] = []
for state_item in headings:
if _heading_signature(state_item) == winner_sig:
out.append(state_item)
continue
# Different font from heading clique. Only keep if it's a clearly
# numbered heading that doesn't smell of pseudo-code / math.
if numbering_kind(state_item.line()) != 1:
continue
numbering = numbering_text(state_item.line())
if not re.match(r"^\d+(?:\.\d+){0,2}$", numbering):
continue
text = block_text(state_item)
if any(pat.search(text) for pat in _PSEUDO_CODE_PATTERNS):
continue
# Also require: at least one alphabetic word AFTER the number
# ("2 Introduction" yes, "9else" no, "1.804 1.737 1.692" no)
after_num = re.sub(r"^\s*\d+(?:\.\d+){0,2}\s*[.:)]?\s*", "", text)
if not re.search(r"[A-Za-zÀ-ÿ一-鿿가-힯]{3,}", after_num):
continue
out.append(state_item)
return out
+132
View File
@@ -0,0 +1,132 @@
"""Level assignment and outline tree construction."""
from __future__ import annotations
import re
from collections import defaultdict
from ..model import numbering_text, numbering_kind, block_text, is_caps_heavy, Block
from .filtering import (
_style_key,
_numbering_depth,
)
def extract_top_level_headings(headings: list[Block], levels: dict[int, int]) -> list[Block]:
"""Flatten the heading tree, returning only top-level headings."""
return [heading for heading in headings if levels.get(id(heading), 6) <= 1]
def assign_levels(headings: list[Block]) -> dict[int, int]:
"""Return ``{id(block) -> level}``. 1. Bucket by style key. 2. Rank styles by (size DESC, bold DESC) and assign level 1..6 in that order (anything below the 6th distinct style is clamped to 6). 3. If a heading has digit-numbering, its level is overridden to min(numbering_depth, style_level) -- numbering wins for deeper grouping but never promotes a heading above its style rank. """
buckets: dict[tuple[str, float, bool], list[Block]] = defaultdict(list)
for state_item in headings:
buckets[_style_key(state_item)].append(state_item)
ranked = sorted(buckets.keys(), key=lambda key_value: (-key_value[1], not key_value[2]))
style_level = {key_value: min(index_value + 1, 6) for index_value, key_value in enumerate(ranked)}
out: dict[int, int] = {}
for state_item in headings:
lvl = style_level.get(_style_key(state_item), 6)
depth = _numbering_depth(state_item)
if depth is not None:
lvl = max(1, min(lvl, depth))
out[id(state_item)] = lvl
return out
# --------------------------------------------------------------------------- #
# Tree assembly #
# --------------------------------------------------------------------------- #
def _heading_title(block: Block) -> str:
"""Cleaned title text for output (no dot leaders, single-line)."""
text = block_text(block).strip()
text = re.sub(r"\s+", " ", text)
return text
def _heading_page_num(block: Block, page_lookup) -> int:
"""Find the 1-based page number that owns this block. ``page_lookup`` is a dict ``{id(block) -> page.u}`` precomputed by the caller for O(1) lookup. """
return page_lookup.get(id(block), 1)
def build_tree(headings: list[Block], levels: dict[int, int], page_lookup, total_pages: int) -> list[dict]:
"""Assemble nested ``{title, start_index, end_index, nodes}`` tree."""
if not headings:
return []
root: list[dict] = []
stack: list[tuple[int, dict]] = []
for state_item in headings:
title = _heading_title(state_item)
if not title:
continue
node = {
"title": title,
"start_index": _heading_page_num(state_item, page_lookup),
"end_index": _heading_page_num(state_item, page_lookup),
"nodes": [],
}
lvl = levels.get(id(state_item), 6)
while stack and stack[-1][0] >= lvl:
stack.pop()
if not stack:
root.append(node)
else:
stack[-1][1]["nodes"].append(node)
stack.append((lvl, node))
# Fill end_index in DFS order.
flat: list[dict] = []
def _walk_nodes(nodes: list[dict]) -> None:
for count_item in nodes:
flat.append(count_item)
_walk_nodes(count_item["nodes"])
_walk_nodes(root)
for index_value, count_item in enumerate(flat):
next_start = flat[index_value + 1]["start_index"] if index_value + 1 < len(flat) else total_pages
count_item["end_index"] = max(count_item["start_index"], next_start - 1 if next_start > count_item["start_index"] else count_item["start_index"])
if flat:
flat[-1]["end_index"] = max(flat[-1]["start_index"], total_pages)
# Drop empty children so the JSON matches the shape the rest of PageIndex emits.
def _drop_empty_children(nodes: list[dict]) -> list[dict]:
for count_item in nodes:
if count_item["nodes"]:
_drop_empty_children(count_item["nodes"])
else:
del count_item["nodes"]
return nodes
return _drop_empty_children(root)
# --------------------------------------------------------------------------- #
# Outline validation #
# --------------------------------------------------------------------------- #
def validate(headings: list[Block], levels: dict[int, int], doc) -> bool:
"""Return whether the outline has enough top-level headings spanning a meaningful fraction of the document."""
top = [state_item for state_item in headings if levels.get(id(state_item), 6) <= 2]
if len(top) < 3:
return False
if len(top) >= 5:
return True
last_page = 1
for state_item in top:
# Direct page lookup would need a page back-reference; we use the document
# order proxy (top is already in reading order).
# Find by scanning document pages for the page containing the block.
page_num = 1
for page in doc.primary_slot:
if state_item in (page.secondary_slot or []):
page_num = page.page_index
break
if page_num - last_page > 0.5 * len(doc.primary_slot):
return False
last_page = page_num
return True
@@ -0,0 +1,110 @@
"""Outline assembly chain. This module turns heading candidates and labeled section regions into the final
nested outline tree. It groups candidates by numbering depth, style signature,
script compatibility, document order, and local clusters, then serializes the
tree into the public PageIndex JSON shape.
"""
import math
from typing import Any, Callable, Optional
from sortedcontainers import SortedKeyList
from ..model import (
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
)
from ..stats import style_key as style_key_fn, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
from ..tokens import (
Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, avg_char_width as avg_char_width_fn, trie_full_match, first_anchor_span, is_char_token, is_word_token,
)
# --------------------------------------------------------------------------- #
# Numbering-pattern clique selection.
# --------------------------------------------------------------------------- #
# Section-keyword trie shared with outline filtering.
from ..outline import SECTION_KEYWORD_TRIE
from .candidates import (
_viewport_y_fraction,
HeadingCandidate,
OutlineNode,
compare_heading_order,
_compare_block_order,
heading_order_key,
is_script_compatible,
heading_signature,
parent_signature,
cached_signature,
is_in_oo_range,
has_style_neighbor,
)
from .style_context import (
StyleCluster,
pick_style_bucket,
has_conflict_in_context,
is_compatible_with_context,
OutlineContext,
NumberingTrie,
insert_numbering,
count_sibling_numberings,
OutlineState,
_apply_heading_to_state,
compare_heading_depth,
)
from .cliques import (
find_keyword_clique,
CliqueTreeNode,
find_ancestor_next_sibling,
descend_to_deepest_last,
append_tree_child,
CliqueTreeBuilder,
block_style_signature,
is_member_of_tree,
can_share_heading_style,
compare_block_order,
heading_precedes_line,
CliqueFilterContext,
detect_body_headings,
partition_candidates,
interleave_clusters,
)
from .selection import (
min_font_distance,
should_reject_heading,
push_heading_to_state,
HierarchyStack,
find_parent_heading,
is_appendix_nesting_ok,
extract_sub_headings,
extract_top_level_headings,
is_outline_valid,
is_chapter_outline_valid,
)
from .assembly import (
mark_outline_block_types,
compute_max_heading_gap,
has_table_or_prominent,
is_landscape_or_empty,
build_heading_from_block,
assemble_outline,
_flatten_outline_nodes,
_heading_appears_at_page_top,
outline_to_dict_tree,
)
__all__ = [
"HeadingCandidate", "OutlineNode",
"compare_heading_order", "heading_order_key", "compare_heading_depth",
"is_script_compatible", "heading_signature", "parent_signature", "cached_signature", "is_in_oo_range", "has_style_neighbor", "pick_style_bucket", "has_conflict_in_context", "is_compatible_with_context",
"StyleCluster", "OutlineContext", "NumberingTrie", "insert_numbering", "count_sibling_numberings",
"OutlineState",
"find_keyword_clique", "detect_body_headings", "CliqueFilterContext",
"partition_candidates", "interleave_clusters", "push_heading_to_state", "should_reject_heading", "find_parent_heading", "HierarchyStack", "extract_sub_headings", "min_font_distance",
"extract_top_level_headings", "is_outline_valid", "is_chapter_outline_valid", "mark_outline_block_types", "compute_max_heading_gap", "has_table_or_prominent",
"build_heading_from_block",
"assemble_outline",
"outline_to_dict_tree",
]
@@ -0,0 +1,345 @@
"""Final outline assembly and conversion to the output dict tree."""
from __future__ import annotations
from typing import Any, Callable, Optional
from ..model import (
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
)
from ..tokens import (
Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, avg_char_width as avg_char_width_fn, trie_full_match, first_anchor_span, is_char_token, is_word_token,
)
from .candidates import (
HeadingCandidate,
OutlineNode,
heading_order_key,
)
from .style_context import (
OutlineState,
compare_heading_depth,
)
from .cliques import (
find_keyword_clique,
CliqueFilterContext,
detect_body_headings,
partition_candidates,
interleave_clusters,
)
from .selection import (
should_reject_heading,
push_heading_to_state,
HierarchyStack,
find_parent_heading,
extract_sub_headings,
)
def mark_outline_block_types(item_list: list[OutlineNode]) -> None:
"""Mark outline blocks as numbered or unnumbered headings."""
for block in item_list:
block.heading.group_slot.type = 8 if block.heading.has_numbering else 7
mark_outline_block_types(block.child_nodes)
def compute_max_heading_gap(outline_nodes: list[OutlineNode], other_number: int) -> dict:
"""Compute the maximum page-position gap between outline nodes."""
if not outline_nodes:
return {"max_gap": 0, "last_page_position": other_number}
heading = 0
for stack_outline_node in outline_nodes:
page_pos = stack_outline_node.heading.page.page_index + stack_outline_node.heading.auxiliary_slot
heading = max(heading, page_pos - other_number)
other_number = page_pos
rec = compute_max_heading_gap(stack_outline_node.child_nodes, other_number)
heading = max(heading, rec["max_gap"])
other_number = rec["last_page_position"]
return {"max_gap": heading, "last_page_position": other_number}
def has_table_or_prominent(outline_nodes: list[OutlineNode]) -> bool:
"""Return True if any heading is a table-like or prominent entry."""
return any(secondary_item.heading.type == 5 or secondary_item.heading.is_prominent for secondary_item in outline_nodes)
def is_landscape_or_empty(doc) -> bool:
"""Return True for mostly-landscape or near-empty documents with little outline text."""
if doc.secondary_slot.secondary_slot >= 1e3:
return False
secondary_item = 0
candidate_item = 0.0
for page in doc.primary_slot:
if page.bounds.bbox_width() > page.bounds.bbox_height() and page.primary_slot.secondary_slot < 1e3:
secondary_item += 1
candidate_item += page.primary_slot.secondary_slot
count_item = len(doc.primary_slot)
return secondary_item >= 0.9 * count_item or (secondary_item >= 0.7 * count_item and candidate_item >= 0.5 * doc.secondary_slot.state_slot)
# --------------------------------------------------------------------------- #
# Build a heading candidate from a block #
# --------------------------------------------------------------------------- #
def build_heading_from_block(block: Block, page, anchor: Optional[Block] = None) -> HeadingCandidate:
"""Build a heading candidate wrapper for a heading block."""
tokens = tokenize_block(block)
# Extract structural numbering from the leading line.
item_list: list[int] = []
has_numbering = False
prefix: Optional[TokenView] = None
title: TokenView = tokens
if numbering_kind(block.line()) == 1:
num_str = numbering_text(block.line())
if num_str:
try:
parts = [int(number_part) for number_part in num_str.replace(".", ".").split(".") if number_part.strip()]
if all(0 <= number_part < 1000 for number_part in parts):
item_list = parts
has_numbering = True
# Strip the leading number tokens from g
skip = 0
while skip < tokens.length:
tok = tokens.token_at(skip)
if tok is None:
break
if tok.type == 1 or tok.str in "..":
skip += 1
else:
break
title = tokens.slice(skip)
except (ValueError, AttributeError):
pass
# Type from labeled-section classification or from numbering.
marker_type = getattr(block, "marker_slot", 0) or 0
if marker_type == 4:
type_ = 4
elif marker_type == 5:
type_ = 5
elif marker_type == 11:
type_ = 11
elif item_list:
type_ = 1
elif is_caps_heavy(block) and block.line_count() == 1:
type_ = 2 # uppercase short heading
else:
type_ = 0
# Prominence flag: big font / bold-and-prominent.
body_size_threshold = page.primary_slot.primary_slot + 0.5 if page.primary_slot else 0
ja_flag = (
block.avg_font_size() > body_size_threshold + 1.5
or (block.bold_frac() > 0.5 and block.avg_font_size() >= body_size_threshold)
)
return HeadingCandidate(
type_=type_,
page=page,
group_value=block,
anchor=anchor,
numbering_value=item_list,
tokens=prefix,
title_tokens=title,
has_numbering_flag=has_numbering,
prominent_flag=ja_flag,
)
# --------------------------------------------------------------------------- #
# Main outline assembler #
# --------------------------------------------------------------------------- #
def assemble_outline(doc, labeled: list[OutlineNode]) -> list[OutlineNode]:
"""Produce the outline tree as a list of outline nodes. Arguments: ``doc`` is the document state; ``labeled`` is the list of outline nodes wrapping labeled headings. Output is a list of root outline nodes. Each node contains child nodes recursively. """
# ----- Stage 1: collect general headings.
from ..heading_detection import build_doc_heading_candidates
# Labeled headings prime the type gates used by general heading filtering.
general: list[HeadingCandidate] = build_doc_heading_candidates(doc, labeled)
# ----- Stage 2: merge with labeled
if len(labeled) + len(general) > 0:
combined = list(general)
for labeled_region_node in labeled:
combined.append(labeled_region_node.heading)
combined.sort(key=heading_order_key)
# Build the keyword clique before body-heading filtering so the filter
# can test whether a block is already represented in the candidate tree.
clique = find_keyword_clique(combined)
filtered = detect_body_headings(CliqueFilterContext(doc, combined, lambda line, other_line: compare_heading_depth(line, other_line, clique)))
general.extend(filtered)
# No dedup here: duplicate candidates that wrap the same block are
# collapsed downstream by partitioning and already-placed-block checks.
general = sorted(general, key=heading_order_key)
# ----- Stage 3: partition + cluster
if labeled:
# No pre-filter: partitioning re-separates labeled vs general, so any
# labeled block backfilled into the general list is handled there.
result = partition_candidates(general, labeled)
general = result["remaining"]
labeled = result["labeled"]
clusters = interleave_clusters(general, labeled)
else:
clusters = [{"labeled_anchor": None, "cluster_candidates": general}]
# ----- Stage 4: assemble tree
state = OutlineState(clusters)
if not state.measure_slot and state.option_slot <= state.previous_slot:
return []
out: list[OutlineNode] = []
for cluster in clusters:
cluster_anchor = cluster.get("labeled_anchor")
cluster_candidates = cluster.get("cluster_candidates", [])
if cluster_anchor is not None:
push_heading_to_state(state, cluster_anchor.heading)
out.append(cluster_anchor)
sub = extract_sub_headings(doc, state, cluster_anchor, cluster_candidates)
target = cluster_anchor.child_nodes if cluster_anchor is not None else out
target.extend(sub)
sub_clique = find_keyword_clique(cluster_candidates) if cluster_candidates else None
stack = HierarchyStack(sub_clique)
for insertion_candidate in cluster_candidates:
if should_reject_heading(state, insertion_candidate):
continue
push_heading_to_state(state, insertion_candidate)
insertion_candidate.group_slot.used_as_heading = True
stack_outline_node = OutlineNode(insertion_candidate)
parent = find_parent_heading(stack, insertion_candidate)
if parent is not None:
parent.child_nodes.append(stack_outline_node)
elif cluster_anchor is not None:
cluster_anchor.child_nodes.append(stack_outline_node)
else:
out.append(stack_outline_node)
stack.push(stack_outline_node)
return out
# --------------------------------------------------------------------------- #
# Outline tree -> PageIndex dict tree #
# --------------------------------------------------------------------------- #
def _flatten_outline_nodes(outline_node_list: list[OutlineNode]) -> list[OutlineNode]:
"""Walk an outline tree DFS to a flat list, preserving order."""
out: list[OutlineNode] = []
def _walk_nodes(items: list[OutlineNode]) -> None:
for item in items:
out.append(item)
if item.child_nodes:
_walk_nodes(item.child_nodes)
_walk_nodes(outline_node_list)
return out
def _heading_appears_at_page_top(heading: HeadingCandidate) -> bool:
"""Return whether a heading begins its page with no flowing content above it."""
top_heading = heading.group_slot
page = heading.page
if top_heading is None or page is None:
return True
group_index = getattr(top_heading, "reading_order_index", 0)
for block in (page.secondary_slot or []):
if block is top_heading or getattr(block, "reading_order_index", 0) >= group_index:
continue # only blocks before the heading
if block.char_count() <= 0:
continue # no text
if block.type in (1, 2, 12): # header / footer / watermark
continue
return False # real content precedes the heading
return True
def outline_to_dict_tree(outline_node_list: list[OutlineNode], total_pages: int) -> list[dict]:
"""Convert the outline tree directly to PageIndex JSON shape. Preserves the natural outline nesting without font-overlay rewriting. """
flat_nodes: list[dict] = []
def _walk_nodes(items: list[OutlineNode]) -> list[dict]:
result: list[dict] = []
for item in items:
# Title text is the numbering prefix plus the heading tokens, but
# the two are carried as separate fields and trimmed one by one,
# then rejoined with a single space and only for a non-empty
# prefix. A prefix's string form ends in a space after every
# space-flagged token, so trimming the parts separately is what
# keeps that space out of the join.
# Trim with the Unicode WhiteSpace+LineTerminator set, not Python's
# str.strip set: they differ on U+FEFF, U+0085, and U+001C-1F.
prefix_tokens = item.heading.secondary_slot
token = item.heading.primary_slot
child = _trim_unicode_ws(str(prefix_tokens)) if prefix_tokens is not None else ""
node = _trim_unicode_ws(str(token)) if token is not None else ""
title = (child + " " if child else "") + node
if not title:
if item.child_nodes:
result.extend(_walk_nodes(item.child_nodes))
continue
node = {
"title": title,
"node_id": "",
"start_index": item.heading.page.page_index,
"end_index": item.heading.page.page_index,
"nodes": _walk_nodes(item.child_nodes) if item.child_nodes else [],
"_appear_start": _heading_appears_at_page_top(item.heading),
}
flat_nodes.append(node)
result.append(node)
return result
root = _walk_nodes(outline_node_list)
# Fill end_index via DFS-order next-start - 1; last node extends to doc end.
flat: list[dict] = []
def _collect(nodes: list[dict]) -> None:
for count_item in nodes:
flat.append(count_item)
_collect(count_item["nodes"])
_collect(root)
for line, outline_entry in enumerate(flat):
if line + 1 < len(flat):
nxt = flat[line + 1]
# page_index post_processing (utils.post_processing): if the next
# heading starts at the top of its page, this section ends the page
# before it; otherwise the next heading sits below this section's
# tail, so the two share that boundary page and the end extends onto
# it.
boundary = (
nxt["start_index"] - 1
if nxt["_appear_start"]
else nxt["start_index"]
)
else:
boundary = total_pages
outline_entry["end_index"] = max(
outline_entry["start_index"],
boundary if boundary > outline_entry["start_index"] else outline_entry["start_index"],
)
if flat:
flat[-1]["end_index"] = max(flat[-1]["start_index"], total_pages)
# Stable DFS pre-order node ids, zero-padded to 4 (PageIndex convention;
# uses zero-padded depth-first ids). Drop the
# transient appear_start marker now that end_index is settled.
for line, outline_entry in enumerate(flat):
outline_entry["node_id"] = str(line).zfill(4)
del outline_entry["_appear_start"]
def _drop_empty_children(nodes: list[dict]) -> list[dict]:
for count_item in nodes:
if count_item["nodes"]:
_drop_empty_children(count_item["nodes"])
else:
del count_item["nodes"]
return nodes
return _drop_empty_children(root)
@@ -0,0 +1,255 @@
"""Heading candidate and outline node types plus ordering and signature helpers."""
from __future__ import annotations
from ..model import (
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
)
from ..stats import style_key as style_key_fn, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
from ..tokens import (
Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, avg_char_width as avg_char_width_fn, trie_full_match, first_anchor_span, is_char_token, is_word_token,
)
# --------------------------------------------------------------------------- #
# Heading candidate wrapper #
# --------------------------------------------------------------------------- #
def _viewport_y_fraction(viewport_box, rot: int, user_x: float, user_y: float) -> float:
"""Return viewport-normalized y coordinate for a PDF user-space point. Applies the same page ``/Rotate`` and the unrotated view box to a user-space point, then normalises the y component by the viewport height."""
x_min, y_min, x_max, y_max = viewport_box
center_x = (x_max + x_min) / 2.0
center_y = (y_max + y_min) / 2.0
rotation = rot % 360
if rotation < 0:
rotation += 360
if rotation == 90:
x_axis_scale, y_axis_scale = 1, 0
x_axis_sign = 0
elif rotation == 180:
x_axis_scale, y_axis_scale = 0, 1
x_axis_sign = -1
elif rotation == 270:
x_axis_scale, y_axis_scale = -1, 0
x_axis_sign = 0
else:
x_axis_scale, y_axis_scale = 0, -1
x_axis_sign = 1
if x_axis_sign == 0:
viewport_offset = abs(center_x - x_min)
height = abs(x_max - x_min)
else:
viewport_offset = abs(center_y - y_min)
height = abs(y_max - y_min)
# transform[1]=b, transform[3]=d, transform[5]=off_y - b*cx - d*cy;
# the viewport y-coordinate = b*x + d*y + transform[5].
viewport_y = x_axis_scale * user_x + y_axis_scale * user_y + (viewport_offset - x_axis_scale * center_x - y_axis_scale * center_y)
return viewport_y / (height or 1.0)
class HeadingCandidate:
"""One heading candidate. It stores the candidate type, page, underlying block, optional anchor block, numbering array, optional prefix tokens, title tokens, structural-numbering flag, prominence flag, dominant script family, and vertical page position."""
__slots__ = ("type", "page", "group_slot", "tertiary_slot", "numbering", "secondary_slot", "primary_slot", "has_numbering", "is_prominent", "state_slot", "auxiliary_slot")
def __init__(self, type_, page, group_value, anchor, numbering_value, tokens, title_tokens, has_numbering_flag, prominent_flag):
self.type = type_
self.page = page
self.group_slot = group_value
self.tertiary_slot = anchor
self.numbering = numbering_value or []
self.secondary_slot = tokens
self.primary_slot = title_tokens
self.has_numbering = has_numbering_flag
self.is_prominent = prominent_flag
# Compute the dominant script family over prefix and title tokens.
acc = ScriptHistogram()
if tokens is not None:
for token_value in tokens:
tally_scripts(acc, token_value.str)
if title_tokens is not None:
for token_value in title_tokens:
tally_scripts(acc, token_value.str)
self.state_slot = dominant_script_family(acc)
# Compute vertical fraction on page. The viewport applies the page
# /Rotate and view box; when that metadata is absent, fall back to the
# origin-0 upright shortcut.
viewport_box_value = getattr(page, "viewport_box", None)
if viewport_box_value is not None:
self.auxiliary_slot = _viewport_y_fraction(viewport_box_value, getattr(page, "rot", 0) or 0, group_value.left_edge(), group_value.top_edge())
else:
page_height = page.bounds.bbox_height() or 1.0
self.auxiliary_slot = (page.bounds.top_edge() - group_value.top_edge()) / page_height
def __repr__(self) -> str: # diagnostic
return f"<HeadingCandidate t={self.type} M={self.numbering} G={block_text(self.group_slot)[:30]!r}>"
# --------------------------------------------------------------------------- #
# Outline node #
# --------------------------------------------------------------------------- #
class OutlineNode:
"""Heading plus child outline nodes."""
__slots__ = ("heading", "child_nodes")
def __init__(self, heading: HeadingCandidate):
self.heading = heading
self.child_nodes: list["OutlineNode"] = []
# --------------------------------------------------------------------------- #
# Page and reading-position comparator.
# --------------------------------------------------------------------------- #
def compare_heading_order(heading_candidate: HeadingCandidate, other_heading_candidate: HeadingCandidate) -> float:
"""Order by page, then block reading position."""
if heading_candidate.page.page_index != other_heading_candidate.page.page_index:
return heading_candidate.page.page_index - other_heading_candidate.page.page_index
return _compare_block_order(heading_candidate.group_slot, other_heading_candidate.group_slot)
def _compare_block_order(block: Block, other_block: Block) -> float:
"""Compare by column index first, then by reading position."""
from ..model import cmp_reading_order
from ..stats import column_index_of as _column_index
heading_anchor = _column_index(block)
other_column_index = _column_index(other_block)
if heading_anchor != other_column_index:
return heading_anchor - other_column_index
return cmp_reading_order(block, other_block)
def heading_order_key(heading_candidate: HeadingCandidate) -> tuple:
from ..stats import column_index_of as _column_index
return (heading_candidate.page.page_index, _column_index(heading_candidate.group_slot), -heading_candidate.group_slot.top_edge(), -heading_candidate.group_slot.bottom_edge(), heading_candidate.group_slot.left_edge(), heading_candidate.group_slot.right_edge())
# --------------------------------------------------------------------------- #
# Candidate compatibility and style-cluster helpers #
# --------------------------------------------------------------------------- #
def is_script_compatible(number: int, other_heading_candidate: HeadingCandidate) -> bool:
"""Return whether candidate script/type is compatible with prior context. Args: a: integer previous script/type context b: heading candidate """
candidate_item = other_heading_candidate.state_slot
if candidate_item == 0 or candidate_item == 2 or candidate_item == 10:
return True
if number == candidate_item:
return False
if other_heading_candidate.type == 5:
return False
if other_heading_candidate.is_prominent:
return False
if len(other_heading_candidate.numbering) > 0:
return False
if number == 3 and candidate_item == 9:
return False
if number == 9 and candidate_item == 3:
return False
if number == 7 and candidate_item == 5:
return False
if number == 6 and candidate_item == 3:
return False
return True
def heading_signature(heading_candidate: HeadingCandidate) -> str:
"""Return a full heading signature including numbering or text."""
if len(heading_candidate.numbering) > 0:
# Numbering arrays are serialized as comma-joined values, not Python
# list representations.
return f"{heading_candidate.type}|{','.join(map(str, heading_candidate.numbering))}"
heading = f"{heading_candidate.type}|"
if heading_candidate.primary_slot is not None:
for token in heading_candidate.primary_slot:
if is_char_token(token):
heading += token.str.lower()
return heading
def parent_signature(heading_candidate: HeadingCandidate) -> str:
"""Return the signature of the candidate's parent numbering prefix."""
secondary_item = f"{heading_candidate.type}|"
for candidate_item in range(len(heading_candidate.numbering) - 1):
if candidate_item > 0:
secondary_item += ","
secondary_item += str(heading_candidate.numbering[candidate_item])
return secondary_item
def cached_signature(primary_item: "StyleCluster", other_heading_candidate: HeadingCandidate) -> str:
"""Cached heading-signature lookup. Keyed by the candidate object itself, not by object id, because addresses can be reused after a discarded object is collected."""
candidate_item = primary_item.auxiliary_slot.get(other_heading_candidate)
if candidate_item is not None:
return candidate_item
candidate_item = heading_signature(other_heading_candidate)
primary_item.auxiliary_slot[other_heading_candidate] = candidate_item
return candidate_item
def is_in_oo_range(primary_item: "StyleCluster", other_heading_candidate: HeadingCandidate) -> bool:
"""Return True if the candidate lies within a style cluster's order range."""
if primary_item.primary_slot is None or primary_item.tertiary_slot is None:
return False
return compare_heading_order(other_heading_candidate, primary_item.primary_slot) >= 0 and compare_heading_order(other_heading_candidate, primary_item.tertiary_slot) <= 0
def has_style_neighbor(style: "StyleCluster", other_heading_candidate: HeadingCandidate, candidate_item: float) -> bool:
"""Return True if a candidate is close to a compatible style neighbor."""
candidate_score = heading_score(other_heading_candidate.group_slot)
def cmp_target():
return {"z": candidate_score, "HeadingCandidate": other_heading_candidate}
matched = [False]
def fcheck(item):
if abs(candidate_score - heading_score(item.group_slot)) >= candidate_item:
return True
# Within tolerance, check signature match:
measure_item = other_heading_candidate.group_slot
line_value = item.group_slot
if abs(heading_score(measure_item) - heading_score(line_value)) >= candidate_item:
state_item = False
elif len(other_heading_candidate.numbering) > 0 and len(item.numbering) > 0:
state_item = (other_heading_candidate.type == item.type and len(other_heading_candidate.numbering) == len(item.numbering))
elif (len(other_heading_candidate.numbering) <= 0 and len(item.numbering) > 1) or (len(item.numbering) <= 0 and len(other_heading_candidate.numbering) > 1):
state_item = False
else:
block = measure_item.isolated_centered
other_centered = line_value.isolated_centered
if block or other_centered:
state_item = (block == other_centered)
elif dominant_style_of(measure_item) == dominant_style_of(line_value):
state_item = True
else:
if first_span_of(measure_item).font_style() != first_span_of(line_value).font_style():
state_item = False
else:
state_item = abs(dominant_font_size(measure_item) - dominant_font_size(line_value)) < candidate_item
if state_item:
matched[0] = True
return True
return False
# Walk sibling candidates in both directions from the candidate's page position
if style.secondary_slot is None:
return False
# SortedKeyList walk
target_key = (candidate_score, heading_order_key(other_heading_candidate))
idx = style.secondary_slot.bisect_right(other_heading_candidate)
for scan_index in range(idx, len(style.secondary_slot)):
if fcheck(style.secondary_slot[scan_index]):
break
if not matched[0]:
for scan_index in range(idx - 1, -1, -1):
if fcheck(style.secondary_slot[scan_index]):
break
return matched[0]
+403
View File
@@ -0,0 +1,403 @@
"""Keyword cliques, clique trees, body-heading detection, and candidate partitioning."""
from __future__ import annotations
from typing import Any, Callable, Optional
from ..model import (
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
)
from ..stats import style_key as style_key_fn, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
from ..tokens import (
Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, avg_char_width as avg_char_width_fn, trie_full_match, first_anchor_span, is_char_token, is_word_token,
)
# --------------------------------------------------------------------------- #
# Numbering-pattern clique selection.
# --------------------------------------------------------------------------- #
# Section-keyword trie shared with outline filtering.
from ..outline import SECTION_KEYWORD_TRIE
from .candidates import (
HeadingCandidate,
OutlineNode,
compare_heading_order,
heading_order_key,
has_style_neighbor,
)
from .style_context import (
StyleCluster,
is_compatible_with_context,
OutlineContext,
)
def find_keyword_clique(heading_candidates: list[HeadingCandidate]) -> Optional[StyleCluster]:
"""Find the largest clique of section-keyword headings sharing a font signature."""
buckets: dict[str, StyleCluster] = {}
for candidate_item in heading_candidates:
if candidate_item.primary_slot is None:
continue
if not trie_full_match(SECTION_KEYWORD_TRIE, candidate_item.primary_slot):
continue
first = first_token(candidate_item.primary_slot)
if first is None or not first.anchor_ranges:
continue
font_size = first_anchor_span(first).font_style()
style_cluster = buckets.get(font_size)
if style_cluster is not None:
if style_cluster.has_nearby_duplicate(candidate_item):
return None # conflict -> abort
if has_style_neighbor(style_cluster, candidate_item, 2.0):
style_cluster.add(candidate_item)
else:
style_cluster = StyleCluster()
buckets[font_size] = style_cluster
style_cluster.add(candidate_item)
winner: Optional[StyleCluster] = None
max_size = 0
for style_cluster in buckets.values():
if style_cluster.size() > max_size:
winner = style_cluster
max_size = style_cluster.size()
if winner is None or max_size <= 1:
return None
for entry_item in heading_candidates:
if winner.contains(entry_item):
continue
if has_style_neighbor(winner, entry_item, 0.5):
winner.add(entry_item)
return winner
# --------------------------------------------------------------------------- #
# Clique-based clusters #
# --------------------------------------------------------------------------- #
class CliqueTreeNode:
"""Tree node used by clique-based heading filtering. Each node holds a heading candidate, parent pointer, child list, and sibling links. ``next`` walks the in-order successor."""
__slots__ = ("heading", "parent", "primary_slot", "secondary_slot", "tertiary_slot")
def __init__(self, heading, parent):
self.heading = heading
self.parent = parent if parent is not None else self
self.primary_slot: list = []
self.secondary_slot = None
self.tertiary_slot = None
def next(self):
if self.primary_slot:
return self.primary_slot[0]
if self.secondary_slot is not None:
return self.secondary_slot
return find_ancestor_next_sibling(self.parent)
def find_ancestor_next_sibling(primary_item: CliqueTreeNode):
"""walk up parents until we find one with a next sibling."""
if primary_item.parent is primary_item:
return None
return primary_item.secondary_slot or find_ancestor_next_sibling(primary_item.parent)
def descend_to_deepest_last(primary_item: CliqueTreeNode) -> CliqueTreeNode:
"""descend to deepest last-child."""
while primary_item.primary_slot:
primary_item = primary_item.primary_slot[-1]
return primary_item
def append_tree_child(ao_tree, parent_node: CliqueTreeNode, heading) -> None:
"""Append a new clique-tree child and advance the builder cursor."""
new_node = CliqueTreeNode(heading, parent_node)
last = parent_node.primary_slot[-1] if parent_node.primary_slot else None
if last is not None:
last.secondary_slot = new_node
new_node.tertiary_slot = last
parent_node.primary_slot.append(new_node)
ao_tree.primary_slot = new_node
class CliqueTreeBuilder:
"""(class at table entry). Builds a clique-tree from a heading list using a comparator. Each heading is placed by walking the cursor up/down based on comparator result. Depth capped at 8. """
__slots__ = ("root", "primary_slot")
def __init__(self, headings: list[HeadingCandidate], compare):
self.root = CliqueTreeNode(None, None)
self.primary_slot = self.root
depth = 0
for height in headings:
while True:
if self.primary_slot is self.root:
append_tree_child(self, self.primary_slot, height)
depth += 1
break
comparison = compare(self.primary_slot.heading, height)
if comparison < 0:
self.primary_slot = self.primary_slot.parent
depth -= 1
else:
if comparison > 0 and depth < 8:
append_tree_child(self, self.primary_slot, height)
depth += 1
else:
append_tree_child(self, self.primary_slot.parent, height)
break
def block_style_signature(block) -> str:
"""Return a block-style signature combining dominant style and caps-heavy state."""
from ..model import dominant_style_of, is_caps_heavy
# The boolean portion is lower-case because the signature is used as an
# opaque stable key.
return f"{dominant_style_of(block)} {'true' if is_caps_heavy(block) else 'false'}"
def is_member_of_tree(doc, block, target_sig: str, sentence_like: bool, node: CliqueTreeNode) -> bool:
"""Return whether the target block is already represented by an ancestor in the candidate tree, using heading signature, body-text weight, and recursive parent traversal."""
from ..model import is_sentence_like
from ..stats import info_weight
if node is None or node.parent is node:
return False
tree_parent_candidate = node.heading
if tree_parent_candidate is None or tree_parent_candidate.type == 5 or tree_parent_candidate.is_prominent:
return False
if len(tree_parent_candidate.numbering) > 0:
return is_member_of_tree(doc, block, target_sig, sentence_like, node.parent)
parent_block = tree_parent_candidate.group_slot
if target_sig != block_style_signature(parent_block) or (sentence_like and is_sentence_like(parent_block)):
return is_member_of_tree(doc, block, target_sig, sentence_like, node.parent)
if info_weight(block.char_stats) >= max(100, 4 * info_weight(parent_block.char_stats)):
return is_member_of_tree(doc, block, target_sig, sentence_like, node.parent)
return True
def can_share_heading_style(heading, other_heading, neighbor_map) -> bool:
"""Return whether two blocks can share a heading-style assignment after checking overlap, style signature, neighboring ambiguity, and predecessor consistency."""
from ..model import y_overlaps, dominant_style_of
from ..heading_detection import neighbor_right, neighbor_above
if other_heading is None or not y_overlaps(heading, other_heading) or dominant_style_of(heading) != dominant_style_of(other_heading):
return False
heading_above = neighbor_above(neighbor_map, heading)
other_above = neighbor_above(neighbor_map, other_heading)
heading_right = neighbor_right(neighbor_map, heading)
other_right = neighbor_right(neighbor_map, other_heading)
if (heading_above is not None and heading_above.marker_slot != 0
or other_above is not None and other_above.marker_slot != 0
or heading_right is not None and heading_right.marker_slot != 0
or other_right is not None and other_right.marker_slot != 0):
return True
if (heading_right is not other_right
and (heading_right is not None and heading_right.is_body_paragraph)
and (other_right is not None and other_right.is_body_paragraph)):
return False
return True
def compare_block_order(left_value, right_value) -> float:
"""Compare blocks or lines by column index first, then reading position."""
from ..model import cmp_reading_order
from ..stats import column_index_of
left_column_index = column_index_of(left_value)
right_column_index = column_index_of(right_value)
if left_column_index != right_column_index:
return left_column_index - right_column_index
return cmp_reading_order(left_value, right_value)
def heading_precedes_line(line_heading_candidate: HeadingCandidate, page, line) -> bool:
"""Return whether the heading candidate sorts before the given page/line position."""
if line_heading_candidate.page.page_index < page.page_index:
return True
if line_heading_candidate.page.page_index != page.page_index:
return False
return compare_block_order(line_heading_candidate.group_slot, line) < 0
class CliqueFilterContext:
"""State for clique-based body-heading discovery."""
__slots__ = ("auxiliary_slot", "state_slot", "tertiary_slot", "measure_slot", "secondary_slot", "option_slot", "primary_slot", "candidates", "compare")
def __init__(self, doc, candidates: list[HeadingCandidate], compare):
self.auxiliary_slot = doc
self.state_slot: set = set()
self.tertiary_slot: dict = {}
for reference_item in candidates:
self.state_slot.add(reference_item.group_slot)
if reference_item.has_numbering:
continue
if len(reference_item.numbering) > 0:
continue
sig = block_style_signature(reference_item.group_slot)
self.tertiary_slot[sig] = self.tertiary_slot.get(sig, 0) + 1
self.measure_slot = CliqueTreeBuilder(candidates, compare)
self.secondary_slot = self.measure_slot.root
self.option_slot = CliqueTreeBuilder(list(reversed(candidates)), compare)
self.primary_slot = self.option_slot.primary_slot
self.candidates = candidates
self.compare = compare
def detect_body_headings(filter_context: CliqueFilterContext) -> list[HeadingCandidate]:
"""Discover body headings by comparing unvisited blocks against clique trees."""
from ..model import style_key, dominant_style_of, last_span, last_line_of, first_span_of
from ..heading_detection import neighbor_right, neighbor_above, closest_body_neighbor_above, PageNeighborMap as _bo_class, is_cover_page
from ..tokens import first_token, tokenize_block
out: list[HeadingCandidate] = []
if not filter_context.candidates:
return out
# Reset cursors to root of forward tree / deepest of reversed tree.
filter_context.secondary_slot = filter_context.measure_slot.root
filter_context.primary_slot = filter_context.option_slot.primary_slot
for page in filter_context.auxiliary_slot.primary_slot:
if is_cover_page(filter_context.auxiliary_slot, page):
continue
all_blocks = page.output_slot
if len(all_blocks) <= 0:
continue
neighbor_cache = _bo_class(page)
for block in page.secondary_slot:
# Advance the forward tree cursor while the next node is before
# the current page and block in reading order.
while True:
next_item = filter_context.secondary_slot.next()
if (next_item is None
or next_item.heading is None
or not heading_precedes_line(next_item.heading, page, block)):
break
filter_context.secondary_slot = next_item
# Advance the reverse tree cursor while the predecessor is before
# cursor's heading is still before the current page and block.
while filter_context.primary_slot.heading is not None and heading_precedes_line(filter_context.primary_slot.heading, page, block):
left_sib = filter_context.primary_slot.tertiary_slot
filter_context.primary_slot = descend_to_deepest_last(left_sib) if left_sib is not None else filter_context.primary_slot.parent
if filter_context.primary_slot is filter_context.option_slot.root:
break
if block in filter_context.state_slot:
continue
if filter_context.secondary_slot.heading is None:
continue
# Body-heading filters.
if (block.char_count() <= 0 or block.skew_frac() > 1
or (block.char_count() <= 1 and block.char_stats.secondary_slot != 4)
or block.line_count() >= 5
or block.type != 0
or block.marker_slot != 0
or (block.char_stats.primary_slot[2] <= 0 and block.char_stats.primary_slot[4] <= 0)):
continue
if block.measure_slot:
continue
value = block.bold_frac()
if 0.1 < value < 0.9:
continue
block_style = dominant_style_of(block)
first_tok = first_token(tokenize_block(block))
# Compare against the dominant style, first span, last token
# anchor, and last span. The last anchor matters for wrapped tokens.
anchor = first_tok.anchor_ranges[-1].anchor_span if (first_tok is not None and first_tok.anchor_ranges) else None
if (block_style != style_key(first_span_of(block))
and (anchor is None or block_style != style_key(anchor))
and block_style != style_key(last_span(last_line_of(block)))):
continue
if block_style == page.primary_slot.auxiliary_slot:
continue
above = neighbor_above(neighbor_cache, block)
if (above is not None
and above.bottom_edge() - block.top_edge() < 0.3 * block.avg_font_size()
and block.line_count() > 1):
continue
if above is not None and above.type == 3:
continue
sig = block_style_signature(block)
pred_neigh = neighbor_right(neighbor_cache, block)
# Reject when the block repeats the style signature of a close
# vertical or right-side neighbor.
if above is not None and sig == block_style_signature(above):
continue
if pred_neigh is not None and sig == block_style_signature(pred_neigh):
continue
previous_block = all_blocks[block.orig_index - 1] if 0 <= block.orig_index - 1 < len(all_blocks) else None
next_block = all_blocks[block.orig_index + 1] if 0 <= block.orig_index + 1 < len(all_blocks) else None
if can_share_heading_style(block, previous_block, neighbor_cache):
continue
if can_share_heading_style(block, next_block, neighbor_cache):
continue
if filter_context.tertiary_slot.get(sig, 0) < 3:
continue
# Sentence-like flag: enough long lowercase-leading word tokens make
# a block look like body text rather than a heading.
tok_total = 0
tok_g3 = 0
for token in tokenize_block(block):
if token.type != 2 or len(token.str) < 5:
continue
tok_total += 1
if token.primary_slot == 3:
tok_g3 += 1
sentence_like = tok_g3 >= max(2, tok_total / 2)
# A block must fit either the forward or reverse clique cursor.
if not (is_member_of_tree(filter_context, block, sig, sentence_like, filter_context.secondary_slot)
or is_member_of_tree(filter_context, block, sig, sentence_like, filter_context.primary_slot)):
continue
body_heading_candidate = HeadingCandidate(
0, page, block,
closest_body_neighbor_above(neighbor_cache, block),
[], None, tokenize_block(block),
False, False,
)
out.append(body_heading_candidate)
return out
# --------------------------------------------------------------------------- #
# Partition candidates and interleave clusters #
# --------------------------------------------------------------------------- #
def partition_candidates(heading_candidates: list[HeadingCandidate], other_outline_nodes: list[OutlineNode]) -> dict:
"""Partition candidates into labeled-compatible and remaining groups."""
labeled_headings = [entry_item.heading for entry_item in other_outline_nodes]
accepted_context = OutlineContext(labeled_headings)
remaining: list[HeadingCandidate] = []
for entry_item in heading_candidates:
if is_compatible_with_context(accepted_context, entry_item):
other_outline_nodes.append(OutlineNode(entry_item))
accepted_context.add(entry_item)
else:
remaining.append(entry_item)
other_outline_nodes.sort(key=lambda sort_node: heading_order_key(sort_node.heading))
return {"remaining": remaining, "labeled": other_outline_nodes}
def interleave_clusters(heading_candidates: list[HeadingCandidate], other_outline_nodes: list[OutlineNode]) -> list[dict]:
"""Interleave general candidates between successive labeled headings. Returns clusters with the labeled heading and intervening candidates. """
out: list[dict] = []
index = 0
previous: Optional[OutlineNode] = None
acc: list[HeadingCandidate] = []
for labeled_outline_node in other_outline_nodes:
boundary_candidate = labeled_outline_node.heading
while index < len(heading_candidates) and compare_heading_order(heading_candidates[index], boundary_candidate) < 0:
acc.append(heading_candidates[index])
index += 1
if acc or previous is not None:
out.append({"labeled_anchor": previous, "cluster_candidates": acc})
acc = []
previous = labeled_outline_node
while index < len(heading_candidates):
acc.append(heading_candidates[index])
index += 1
out.append({"labeled_anchor": previous, "cluster_candidates": acc})
return out
@@ -0,0 +1,385 @@
"""Heading rejection rules, hierarchy stack, and sub/top-level heading extraction."""
from __future__ import annotations
import math
from typing import Any, Callable, Optional
from ..model import (
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
)
from ..tokens import (
Token, TokenView, wrap_tokens, enumerate_tokens, last_token, trie_prefix_match, first_token, set_case_fold, TrieConfig, build_trie, tokenize_block, avg_char_width as avg_char_width_fn, trie_full_match, first_anchor_span, is_char_token, is_word_token,
)
from .candidates import (
HeadingCandidate,
OutlineNode,
heading_signature,
parent_signature,
is_in_oo_range,
has_style_neighbor,
)
from .style_context import (
StyleCluster,
count_sibling_numberings,
OutlineState,
compare_heading_depth,
)
# --------------------------------------------------------------------------- #
# cp / bp -- state mutators (,) #
# --------------------------------------------------------------------------- #
def min_font_distance(state: OutlineState, other_heading_candidate: HeadingCandidate) -> float:
"""minimum font-distance between b and any other heading in the same fontStyle bucket within b's line."""
min_value = math.inf
line = other_heading_candidate.group_slot.line()
for token_list in (other_heading_candidate.secondary_slot, other_heading_candidate.primary_slot):
if token_list is None:
continue
for token in token_list:
if token.type != 2:
continue
for anchor in token.anchor_ranges:
if anchor.line is not line:
return min_value
span = anchor.anchor_span
tree = state.state_slot.get(span.font_style())
if tree is None:
continue
for entry in tree:
if entry["heading"] is other_heading_candidate:
continue
diff = abs(span.font_size - entry["size"])
if diff < min_value:
min_value = diff
if diff <= 0:
return 0
return min_value
def should_reject_heading(state: OutlineState, other_heading_candidate: HeadingCandidate) -> bool:
"""should we REJECT heading b given current state? True = reject."""
if other_heading_candidate.type == 0:
for previous in state.style_slot:
if previous is None:
continue
if compare_heading_depth(previous, other_heading_candidate) == 1:
continue
style_cluster = state.marker_slot.get(parent_signature(previous))
if style_cluster is not None and is_in_oo_range(style_cluster, other_heading_candidate):
return True
if state.primary_slot is not None and state.primary_slot.is_prominent and other_heading_candidate.type == 0:
count = 0
for candidate_token in tokenize_block(other_heading_candidate.group_slot):
if is_word_token(candidate_token) or candidate_token.type == 1:
count += 1
if count >= 3:
break
if count >= 3:
return True
if (
other_heading_candidate.type == 1 and len(other_heading_candidate.numbering) <= 1
and (
(0 if (other_heading_candidate.type != 1 or len(other_heading_candidate.numbering) <= 0) else count_sibling_numberings(state.cache_slot, other_heading_candidate, 0)) <= 1
)
):
return True
if other_heading_candidate.type in (1, 5, 9, 10, 7):
reject = False
else:
distance = min_font_distance(state, other_heading_candidate)
if distance <= 0.9:
reject = False
elif distance >= math.inf:
reject = True
else:
reject = not (is_caps_heavy(other_heading_candidate.group_slot) and other_heading_candidate.tertiary_slot is not None and other_heading_candidate.group_slot.bottom_edge() - other_heading_candidate.tertiary_slot.top_edge() < 5 * other_heading_candidate.group_slot.bbox_height())
if reject:
return True
if other_heading_candidate.type == 1:
first = other_heading_candidate.numbering[0]
if (first < state.secondary_slot and first < state.tertiary_slot) or (state.secondary_slot > 0 and first > state.secondary_slot + 2):
return True
if len(other_heading_candidate.numbering) == 1 and state.auxiliary_slot is not None:
if first == state.tertiary_slot:
return True
existing = state.auxiliary_slot.group_slot
candidate_style = style_key(first_anchor_span(first_token(other_heading_candidate.primary_slot))) if other_heading_candidate.primary_slot is not None and first_token(other_heading_candidate.primary_slot) is not None else ""
state_style = style_key(first_anchor_span(first_token(state.auxiliary_slot.primary_slot))) if state.auxiliary_slot.primary_slot is not None and first_token(state.auxiliary_slot.primary_slot) is not None else ""
if candidate_style != state_style:
# Bold-fraction comparison uses exact half-up integer rounding;
# Python f-string rounding is half-even.
if abs(dominant_font_size(other_heading_candidate.group_slot) - dominant_font_size(existing)) > 0.5 or _round_half_up_to_int(other_heading_candidate.group_slot.bold_frac()) != _round_half_up_to_int(existing.bold_frac()):
return True
if (
state.primary_slot is not None
and other_heading_candidate.type == 4 and state.primary_slot.type == 4
and len(state.primary_slot.numbering) > 0 and len(other_heading_candidate.numbering) > 0
and (state.primary_slot.numbering[0] > other_heading_candidate.numbering[0] or (len(other_heading_candidate.numbering) == 1 and state.primary_slot.numbering[0] == other_heading_candidate.numbering[0]))
):
return True
if (state.primary_slot is not None and state.primary_slot.type == 8 and len(other_heading_candidate.numbering) <= 0):
from ..model import _strip_diacritics
candidate_tokens = other_heading_candidate.primary_slot or []
tokens = state.primary_slot.primary_slot or []
if len(candidate_tokens) == len(tokens):
same = True
for heading in range(len(candidate_tokens)):
token = candidate_tokens[heading] if heading < len(candidate_tokens) else None
state_token = tokens[heading] if heading < len(tokens) else None
if token is None or state_token is None:
same = False
break
if _strip_diacritics(token.str.lower()) != _strip_diacritics(state_token.str.lower()):
same = False
break
if same:
return True
return False
def push_heading_to_state(state: OutlineState, other_heading_candidate: HeadingCandidate) -> None:
"""Push a heading into the outline state and update level trackers."""
if len(other_heading_candidate.numbering) > 0:
# Ensure S is long enough
while len(state.style_slot) < len(other_heading_candidate.numbering):
state.style_slot.append(None)
state.style_slot[len(other_heading_candidate.numbering) - 1] = other_heading_candidate
if other_heading_candidate.type == 1:
first = other_heading_candidate.numbering[0]
state.secondary_slot = max(state.secondary_slot, first)
state.tertiary_slot = max(state.tertiary_slot, first)
if len(other_heading_candidate.numbering) == 1:
state.auxiliary_slot = other_heading_candidate
elif other_heading_candidate.type in (8, 9):
state.tertiary_slot = 0
state.primary_slot = other_heading_candidate
# --------------------------------------------------------------------------- #
# Hierarchy-walk stack #
# --------------------------------------------------------------------------- #
class HierarchyStack:
"""Tree-walk stack of currently open outline nodes."""
__slots__ = ("auxiliary_slot", "primary_slot", "secondary_slot", "tertiary_slot")
def __init__(self, anchor):
self.auxiliary_slot = anchor
self.primary_slot: list[OutlineNode] = []
self.secondary_slot = False
self.tertiary_slot = False
def pop(self) -> Optional[OutlineNode]:
return self.primary_slot.pop() if self.primary_slot else None
def push(self, other_outline_node: OutlineNode) -> None:
self.primary_slot.append(other_outline_node)
self.secondary_slot = self.secondary_slot or other_outline_node.heading.type == 4
self.tertiary_slot = self.tertiary_slot or other_outline_node.heading.is_prominent
def find_parent_heading(stack: HierarchyStack, other_heading_candidate: HeadingCandidate) -> Optional[OutlineNode]:
"""Pop entries from the stack until a parent for the candidate is found."""
heading: Optional[HeadingCandidate] = None
while stack.primary_slot:
stack_outline_node = stack.primary_slot[-1]
state_candidate = stack_outline_node.heading
if other_heading_candidate.is_prominent and len(other_heading_candidate.numbering) <= 1 and state_candidate.type != 8:
stack.pop()
heading = state_candidate
continue
if state_candidate.is_prominent and other_heading_candidate.type == 5:
stack.pop()
heading = state_candidate
continue
cmp = compare_heading_depth(state_candidate, other_heading_candidate, stack.auxiliary_slot)
if cmp != -1:
if cmp == 1:
return stack_outline_node
# Appendix and Roman/letter headings can nest under the current
# parent only when the numbering sequence remains coherent.
if (state_candidate.type != other_heading_candidate.type and other_heading_candidate.type in (4, 2) and not stack.tertiary_slot
and is_appendix_nesting_ok(stack, other_heading_candidate, heading)):
first_number = other_heading_candidate.numbering[0] if other_heading_candidate.numbering else 0
if heading is None:
if first_number == 1:
return stack_outline_node
else:
# Empty numbering on the previous heading cannot establish
# an increasing appendix sequence.
if other_heading_candidate.type == heading.type and other_heading_candidate.numbering and heading.numbering and first_number > heading.numbering[0]:
return stack_outline_node
stack.pop()
heading = state_candidate
return None
def is_appendix_nesting_ok(stack: HierarchyStack, other_heading_candidate: HeadingCandidate, candidate_heading_candidate: Optional[HeadingCandidate]) -> bool:
"""Return whether an appendix candidate may be nested under the current stack state. Non-appendix headings always pass; appendix headings pass when the stack is already in appendix mode, has no numbering context, or starts at appendix depth 1..3."""
if other_heading_candidate.type != 4:
return True
if stack.secondary_slot:
return True
# Last heading info
if not stack.primary_slot:
return True
entry_item = stack.primary_slot[-1].heading
if len(entry_item.numbering) <= 0:
return True
return entry_item.numbering[0] <= 3
# --------------------------------------------------------------------------- #
# Sub-headings within a cluster #
# --------------------------------------------------------------------------- #
def extract_sub_headings(doc, state: OutlineState, parent_node: Optional[OutlineNode], cluster_candidates: list[HeadingCandidate]) -> list[OutlineNode]:
"""Walk a cluster's candidate list and emit subheadings. The input list is consumed in place so later passes do not reprocess headings already assigned to this cluster."""
if not cluster_candidates:
return []
# Content cap: walk from the parent page to the first candidate page and
# abort the cluster if accumulated body-block text exceeds 1000.
from ..stats import info_weight as _info_weight
first = cluster_candidates[0]
page_index = (parent_node.heading.page.page_index - 1) if parent_node is not None else 0
acc = 0
end_pg = min(first.page.page_index, len(doc.primary_slot))
while page_index < end_pg:
heading_page = doc.primary_slot[page_index]
if getattr(heading_page, "state_slot", False):
for block in heading_page.output_slot:
if page_index >= first.page.page_index - 1 and block.reading_order_index >= first.group_slot.reading_order_index:
break
if getattr(block, "is_body_paragraph", None):
acc += _info_weight(block.char_stats)
if acc >= 1000:
return []
page_index += 1
out: list[OutlineNode] = []
parent_anchor = parent_node if (parent_node is not None and parent_node.heading.type == 5) else None
seen_signatures: set[str] = set()
style_cluster = StyleCluster()
saw_numbered = False
index = 0
while index < len(cluster_candidates):
cluster_candidate = cluster_candidates[index]
if not (
cluster_candidate.type == 5
or cluster_candidate.type == 6
or (cluster_candidate.type == 11 and cluster_candidate.has_numbering and parent_anchor is not None and index <= 1)
):
next_item = cluster_candidates[index + 1] if index + 1 < len(cluster_candidates) else None
if next_item and next_item.type == 5 and next_item.page is cluster_candidate.page and next_item.tertiary_slot is cluster_candidate.tertiary_slot:
index += 1
continue
break
candidate_signature = heading_signature(cluster_candidate)
if candidate_signature in seen_signatures:
index += 1
continue
if should_reject_heading(state, cluster_candidate):
index += 1
continue
push_heading_to_state(state, cluster_candidate)
seen_signatures.add(candidate_signature)
if cluster_candidate.has_numbering:
saw_numbered = True
elif saw_numbered:
break
if parent_anchor is None:
parent_anchor = OutlineNode(cluster_candidate)
out.append(parent_anchor)
style_cluster.add(cluster_candidate)
index += 1
continue
anchor_heading_candidate = parent_anchor.heading
if cluster_candidate.page.page_index > anchor_heading_candidate.page.page_index:
break
cmp = compare_heading_depth(anchor_heading_candidate, cluster_candidate)
if cmp != 1:
if not has_style_neighbor(style_cluster, cluster_candidate, 1.0):
break
parent_anchor = OutlineNode(cluster_candidate)
out.append(parent_anchor)
style_cluster.add(cluster_candidate)
index += 1
# Remove processed items so the outline loop does not reprocess them.
del cluster_candidates[:index]
if (
len(out) >= 3
or (len(out) == 2 and out[0].heading.has_numbering and out[1].heading.has_numbering)
) and out[0].heading.type != 5:
return []
return out
# --------------------------------------------------------------------------- #
# Flatten outline to top-level headings #
# --------------------------------------------------------------------------- #
def extract_top_level_headings(item_list: list[OutlineNode]) -> list[OutlineNode]:
"""Walk the outline and emit top-level prominent headings."""
out: list[OutlineNode] = []
saw_prominent = False
for heading in item_list:
if heading.heading.is_prominent:
if not saw_prominent:
out.append(heading)
saw_prominent = True
else:
saw_prominent = False
out.extend(extract_top_level_headings(heading.child_nodes))
return out
# --------------------------------------------------------------------------- #
# Outline validation.
# --------------------------------------------------------------------------- #
def is_outline_valid(doc, item_list: list[OutlineNode]) -> bool:
"""Return True when top-level headings span a meaningful fraction of the document."""
top = extract_top_level_headings(item_list)
if len(top) < 3:
return False
if len(top) >= 5:
return True
last_page = 1
for top_node in top:
line = top_node.heading.page.page_index
if line - last_page > 0.5 * len(doc.primary_slot):
return False
last_page = line
return True
def is_chapter_outline_valid(doc, item_list: list[OutlineNode]) -> bool:
"""Secondary validity check based on chapter count and inter-chapter span."""
chapters = 0
span = 0
previous = -1
for chapter_outline_node in item_list:
chapter_page = chapter_outline_node.heading.page.page_index
if previous >= 0:
span += chapter_page - previous
previous = -1
if chapter_outline_node.heading.type == 8:
chapters += 1
previous = chapter_page
if previous >= 0:
span += len(doc.primary_slot) - previous + 1
return (
chapters >= 3
and span >= 0.7 * len(doc.primary_slot)
and span / max(1, chapters) < 100
)
@@ -0,0 +1,365 @@
"""Style clusters, outline context/state, numbering trie, and depth comparison."""
from __future__ import annotations
import math
from typing import Any, Callable, Optional
from sortedcontainers import SortedKeyList
from ..model import (
style_key, left_aligned, right_aligned, center_aligned, x_aligned, rect_union,
Rect, last_span, avg_char_width, raw_text_of_line, heading_score, numbering_text, numbering_value, numbering_kind,
reading_order_key, left_edge_key, _trim_unicode_ws, _round_half_up_to_int, Line, last_line_of, first_span_of, block_text, deaccented_text, letter_count, dominant_style_of, info_weight, dominant_font_size, is_upper_dominant, is_caps_heavy, alignment_code, Block,
)
from .candidates import (
HeadingCandidate,
compare_heading_order,
heading_order_key,
parent_signature,
cached_signature,
is_in_oo_range,
has_style_neighbor,
)
# --------------------------------------------------------------------------- #
# Font/style-clustered heading group #
# --------------------------------------------------------------------------- #
class StyleCluster:
"""Group of headings sharing a font/style signature."""
__slots__ = ("auxiliary_slot", "state_slot", "secondary_slot", "primary_slot", "tertiary_slot")
def __init__(self):
self.auxiliary_slot: dict[HeadingCandidate, str] = {} # candidate -> signature
self.state_slot: dict[str, HeadingCandidate] = {} # signature -> candidate
self.secondary_slot: SortedKeyList = SortedKeyList(
key=lambda sort_node: (heading_score(sort_node.group_slot), heading_order_key(sort_node))
)
self.primary_slot: Optional[HeadingCandidate] = None # min by heading order
self.tertiary_slot: Optional[HeadingCandidate] = None # max by heading order
def size(self) -> int:
return len(self.secondary_slot)
def contains(self, other_heading_candidate: HeadingCandidate) -> bool:
# Containment is key-based, not object identity. SortedKeyList's ``in``
# tests identity among equal-key elements, so compare the sort keys.
idx = self.secondary_slot.bisect_left(other_heading_candidate)
return idx < len(self.secondary_slot) and self.secondary_slot.key(self.secondary_slot[idx]) == self.secondary_slot.key(other_heading_candidate)
def add(self, other_heading_candidate: HeadingCandidate) -> None:
self.state_slot[cached_signature(self, other_heading_candidate)] = other_heading_candidate
# Keep set semantics over the sort key: equal-key elements are dropped,
# while the signature and min/max heading-order state still update.
idx = self.secondary_slot.bisect_left(other_heading_candidate)
if idx >= len(self.secondary_slot) or self.secondary_slot.key(self.secondary_slot[idx]) != self.secondary_slot.key(other_heading_candidate):
self.secondary_slot.add(other_heading_candidate)
if self.primary_slot is None or compare_heading_order(other_heading_candidate, self.primary_slot) < 0:
self.primary_slot = other_heading_candidate
if self.tertiary_slot is None or compare_heading_order(other_heading_candidate, self.tertiary_slot) > 0:
self.tertiary_slot = other_heading_candidate
def has_nearby_duplicate(self, other_heading_candidate: HeadingCandidate) -> bool:
""""have we seen a nearby matching signature within +/- 20 pages?"."""
existing = self.state_slot.get(cached_signature(self, other_heading_candidate))
return existing is not None and abs(other_heading_candidate.page.page_index - existing.page.page_index) < 20
# --------------------------------------------------------------------------- #
# Outline-context style-bucket operations #
# --------------------------------------------------------------------------- #
def pick_style_bucket(outline_context: "OutlineContext", other_heading_candidate: HeadingCandidate) -> StyleCluster:
"""Pick the right style bucket for a candidate."""
if other_heading_candidate.type == 10:
return outline_context.secondary_slot
if other_heading_candidate.type == 8:
return outline_context.primary_slot
if len(other_heading_candidate.numbering) > 0:
return outline_context.auxiliary_slot
return outline_context.tertiary_slot
def has_conflict_in_context(outline_context: "OutlineContext", other_heading_candidate: HeadingCandidate) -> bool:
"""Return True if a candidate conflicts with the existing outline context."""
if other_heading_candidate.type != 10 and is_in_oo_range(outline_context.secondary_slot, other_heading_candidate):
return True
if other_heading_candidate.type != 8 and is_in_oo_range(outline_context.primary_slot, other_heading_candidate):
return True
if len(other_heading_candidate.numbering) <= 0 and is_in_oo_range(outline_context.auxiliary_slot, other_heading_candidate):
return True
if other_heading_candidate.type != 8 and len(other_heading_candidate.numbering) > 0 and outline_context.primary_slot.size() > 0:
return True
return False
def is_compatible_with_context(outline_context: "OutlineContext", other_heading_candidate: HeadingCandidate) -> bool:
"""Return True iff a candidate can be added to the outline context."""
if has_conflict_in_context(outline_context, other_heading_candidate):
return False
# Find the nearest predecessor by heading order.
text: Optional[HeadingCandidate] = None
for item in outline_context.state_slot:
if compare_heading_order(item, other_heading_candidate) <= 0:
if text is None or compare_heading_order(item, text) > 0:
text = item
else:
break
if text is not None:
candidate_block = other_heading_candidate.group_slot
previous_block = text.group_slot
if previous_block.isolated_centered and not candidate_block.isolated_centered:
return False
if not other_heading_candidate.is_prominent and heading_score(previous_block) > heading_score(candidate_block) + 0.5:
return False
if text.is_prominent and not other_heading_candidate.is_prominent and heading_score(previous_block) > heading_score(candidate_block) - 0.5:
return False
style_cluster = pick_style_bucket(outline_context, other_heading_candidate)
if not style_cluster.has_nearby_duplicate(other_heading_candidate) and has_style_neighbor(style_cluster, other_heading_candidate, 1.0):
return True
if len(other_heading_candidate.numbering) == 1 and has_style_neighbor(outline_context.tertiary_slot, other_heading_candidate, 1.0):
return True
return False
# --------------------------------------------------------------------------- #
# Outline-context group.
# --------------------------------------------------------------------------- #
class OutlineContext:
"""Bundles style clusters for chapter, appendix, numbered, and general headings."""
__slots__ = ("secondary_slot", "primary_slot", "auxiliary_slot", "tertiary_slot", "state_slot")
def __init__(self, headings: list[HeadingCandidate]):
self.secondary_slot = StyleCluster() # type == 10
self.primary_slot = StyleCluster() # type == 8
self.auxiliary_slot = StyleCluster() # has M (numbered)
self.tertiary_slot = StyleCluster() # everything else
# The ordered heading list is set-like by heading-order key, with the
# first equal-key candidate retained.
self.state_slot: list[HeadingCandidate] = []
seen_keys: set = set()
for secondary_item in headings:
self.add(secondary_item)
key_value = heading_order_key(secondary_item)
if key_value not in seen_keys:
seen_keys.add(key_value)
self.state_slot.append(secondary_item)
self.state_slot.sort(key=heading_order_key)
def add(self, other_heading_candidate: HeadingCandidate) -> None:
pick_style_bucket(self, other_heading_candidate).add(other_heading_candidate)
def has_nearby_duplicate(self, other_heading_candidate: HeadingCandidate) -> bool:
return pick_style_bucket(self, other_heading_candidate).has_nearby_duplicate(other_heading_candidate)
# --------------------------------------------------------------------------- #
# Numbering-prefix tree.
# --------------------------------------------------------------------------- #
class NumberingTrie:
"""a recursive map for numbering prefixes."""
__slots__ = ("primary_slot", "secondary_slot")
def __init__(self):
self.primary_slot: dict[int, "NumberingTrie"] = {}
self.secondary_slot = 0
def insert_numbering(primary_item: NumberingTrie, other_heading_candidate: HeadingCandidate, index: int) -> None:
"""Insert the candidate numbering suffix into the trie."""
if index == len(other_heading_candidate.numbering):
primary_item.secondary_slot += 1
return
reference_item = primary_item.primary_slot.get(other_heading_candidate.numbering[index])
if reference_item is None:
reference_item = NumberingTrie()
primary_item.primary_slot[other_heading_candidate.numbering[index]] = reference_item
insert_numbering(reference_item, other_heading_candidate, index + 1)
def count_sibling_numberings(primary_item: NumberingTrie, other_heading_candidate: HeadingCandidate, index: int) -> int:
"""Count sibling numbering branches at the target depth."""
if index >= len(other_heading_candidate.numbering) - 1:
count = 0
for reference_item in primary_item.primary_slot.values():
if reference_item.secondary_slot > 0:
count += 1
return count
reference_item = primary_item.primary_slot.get(other_heading_candidate.numbering[index])
if reference_item is None:
return 0
return count_sibling_numberings(reference_item, other_heading_candidate, index + 1)
# --------------------------------------------------------------------------- #
# Global outline state.
# --------------------------------------------------------------------------- #
class OutlineState:
"""Global state for outline assembly walks."""
__slots__ = ("state_slot", "cache_slot", "marker_slot", "previous_slot", "option_slot", "measure_slot", "style_slot", "secondary_slot", "auxiliary_slot", "tertiary_slot", "primary_slot")
def __init__(self, clusters: list[dict]):
self.state_slot: dict = {}
self.cache_slot = NumberingTrie()
self.marker_slot: dict[str, StyleCluster] = {}
self.previous_slot = math.inf
self.option_slot = -math.inf
self.measure_slot = False
clusters_by_parent_signature: dict[str, list[StyleCluster]] = {}
for cluster in clusters:
heading_branch = cluster.get("labeled_anchor")
cluster_candidates = cluster.get("cluster_candidates", [])
if heading_branch is not None:
_apply_heading_to_state(self, heading_branch.heading)
for state_candidate in cluster_candidates:
_apply_heading_to_state(self, state_candidate)
if len(state_candidate.numbering) <= 0:
continue
key = parent_signature(state_candidate)
item_list = clusters_by_parent_signature.get(key)
if item_list is None:
item_list = []
clusters_by_parent_signature[key] = item_list
placed = None
for style_cluster in item_list:
if not style_cluster.has_nearby_duplicate(state_candidate) and has_style_neighbor(style_cluster, state_candidate, 1.0):
placed = style_cluster
break
if placed is None and len(item_list) < 3:
placed = StyleCluster()
item_list.append(placed)
if placed is not None:
placed.add(state_candidate)
for key, group in clusters_by_parent_signature.items():
group.sort(key=lambda bucket_group: -bucket_group.size())
best = group[0]
if best.size() <= 2:
continue
self.marker_slot[key] = best
self.style_slot: list[Optional[HeadingCandidate]] = []
self.secondary_slot = 0
self.auxiliary_slot: Optional[HeadingCandidate] = None
self.tertiary_slot = 0
self.primary_slot: Optional[HeadingCandidate] = None
def _apply_heading_to_state(state: OutlineState, other_heading_candidate: HeadingCandidate) -> None:
"""Add a heading to the per-font tree and update document-level outline state."""
seen: set[str] = set()
line = other_heading_candidate.group_slot.line()
for token_list in (other_heading_candidate.secondary_slot, other_heading_candidate.primary_slot):
if token_list is None:
continue
for token in token_list:
if token.line() is not line:
break
if token.type != 2:
continue
for anchor in token.anchor_ranges:
if anchor.line is not line:
break
span = anchor.anchor_span
style = style_key(span)
if style in seen:
continue
seen.add(style)
font_size = span.font_style()
tree = state.state_slot.get(font_size)
if tree is None:
tree = SortedKeyList(
key=lambda heading: (heading["size"], heading_order_key(heading["heading"]))
)
state.state_slot[font_size] = tree
# Each per-font-size bucket is set-like by (size, heading-order)
# key, retaining the first equal-key entry.
entry = {"size": span.font_size, "heading": other_heading_candidate}
idx = tree.bisect_left(entry)
if idx >= len(tree) or tree.key(tree[idx]) != tree.key(entry):
tree.add(entry)
if other_heading_candidate.type == 1 and len(other_heading_candidate.numbering) > 0:
insert_numbering(state.cache_slot, other_heading_candidate, 0)
state.previous_slot = min(state.previous_slot, other_heading_candidate.page.page_index)
state.option_slot = max(state.option_slot, other_heading_candidate.page.page_index)
if not state.measure_slot:
state.measure_slot = other_heading_candidate.is_prominent
# --------------------------------------------------------------------------- #
# Pairwise heading-depth comparator #
# --------------------------------------------------------------------------- #
def compare_heading_depth(heading_candidate: HeadingCandidate, other_heading_candidate: HeadingCandidate, clique: Optional[StyleCluster] = None) -> int:
"""Compare two heading candidates for relative nesting depth. Returns ``-1`` when the first candidate should be shallower, ``1`` when it should be deeper, and ``0`` when both candidates should share a level. The decision combines special heading types, numbering depth, structural numbering, style prominence, centered layout, clique membership, and bold weight. """
special = heading_candidate.type in (8, 9, 10)
other_special = other_heading_candidate.type in (8, 9, 10)
if special and other_special:
return 0
if special or other_special:
return 1 if special else -1
if (heading_candidate.type == 1 and other_heading_candidate.type == 1) or (heading_candidate.type == 4 and other_heading_candidate.type == 4):
left_length = len(heading_candidate.numbering)
right_length = len(other_heading_candidate.numbering)
if left_length == right_length:
return 0
return 1 if left_length < right_length else -1
if heading_candidate.type == 2 and other_heading_candidate.type == 2:
return 0
if heading_candidate.type == 11 and len(other_heading_candidate.numbering) == 1:
return -1
heading_block = heading_candidate.group_slot
block = other_heading_candidate.group_slot
if heading_candidate.has_numbering != other_heading_candidate.has_numbering:
return -1 if heading_candidate.has_numbering else 1
if heading_candidate.has_numbering and other_heading_candidate.has_numbering and abs(first_span_of(heading_block).font_size - first_span_of(block).font_size) < 0.9:
return 0
score = heading_score(heading_block)
other_score = heading_score(block)
heading_in_clique = clique is not None and clique.contains(heading_candidate)
in_value = clique is not None and clique.contains(other_heading_candidate)
# Z-based major-gap return
if abs(score - other_score) > 1.9 or (abs(score - other_score) > 0.9 and (not heading_in_clique or not in_value)):
return 1 if score > other_score else -1
# uppercase-dominant comparison
heading_caps_heavy = is_caps_heavy(heading_block)
caps_heavy = is_caps_heavy(block)
if heading_caps_heavy != caps_heavy:
return 1 if heading_caps_heavy else -1
# paragraph-end / isolated-centered comparison
centered_flag = heading_block.isolated_centered
if centered_flag != block.isolated_centered:
return 1 if centered_flag else -1
# skew (rotation) comparison -- skipped if both type 5
if not (heading_candidate.type == 5 and other_heading_candidate.type == 5):
heading_skewed = heading_block.previous_slot > 0.99
skew = block.previous_slot > 0.99
if heading_skewed != skew:
return -1 if heading_skewed else 1
# uppercase or both in clique -> tied
if heading_caps_heavy or (heading_in_clique and in_value):
return 0
# clique containment asymmetric
if (heading_in_clique and other_heading_candidate.type != 4) or (in_value and heading_candidate.type != 4):
return 1 if heading_in_clique else -1
# Bold comparison.
bold = heading_block.bold_frac() > 0.5
other_bold = block.bold_frac() > 0.5
if bold != other_bold:
return 1 if bold else -1
return 0
@@ -0,0 +1,171 @@
"""PDFium-backed text-item reconstruction via textpage chars and bbox-mapped font handles.
The parser reconstructs content-stream text items from rendered characters while
preserving the geometry needed by downstream line clustering and heading
detection. The merge thresholds operate on glyph advance, font size, text
matrix scale, and spacing introduced by char spacing, text-position operators,
and ``TJ`` adjustments.
Per page, the reconstruction uses rendered character origins, glyph widths,
font bbox containment, effective font size, text-item merging, baseline-anchored
character boxes, and the minimum font size derived in each emitted chunk. Those
calibrations keep small caps, math glyphs, ligatures, Type 3 fonts, rotated
text, and vertical writing stable enough for layout statistics.
"""
import bisect
import ctypes
import difflib
import json
import math
import re
import unicodedata
from collections import Counter
from io import BytesIO
from pathlib import Path
from typing import Union
import pypdfium2 as pdfium
import pypdfium2.raw as pdfium_c
# Raw PDF object access (ToUnicode CMaps, content streams, font dicts, /WMode)
# that PDFium does not expose, read via PyPDF2 -- already a project dependency and
# permissively licensed. A thin adapter exposes the small raw-object API the
# helpers below need, so their calibrated logic stays unchanged.
import PyPDF2 as _pypdf2 # declared dependency (also imported by pageindex.utils/client)
from PyPDF2.generic import (
IndirectObject as PdfIndirectRef, NameObject as PdfName, NumberObject as PdfNumber,
FloatObject as PdfFloat, BooleanObject as PdfBoolean,
DictionaryObject as PdfDictionary, ArrayObject as PdfArray,
)
from ..model import Span, Rect
from .pdf_objects import (
_pdf_tok,
_pdf_obj_str,
_pdf_typed,
_PdfPage,
_PdfDoc,
_PDF_WHITESPACE_BYTES,
_PDF_DELIMITER_BYTES,
_PDF_STRING_ESCAPE_BYTES,
_decode_pdf_name,
)
from .text_normalize import (
_DROP_CHARS,
_NORMALIZED_UNICODES,
_normalize_unicodes,
TRACKING_SPACE_FACTOR,
NON_SPACE_GAP_FACTOR,
NEGATIVE_SPACE_FACTOR,
SPACE_IN_FLOW_MIN_FACTOR,
SPACE_IN_FLOW_MAX_FACTOR,
_WHITESPACE_CODEPOINTS,
_is_whitespace,
_is_zero_width_diacritic,
_is_invisible_format_mark,
_BIDI_BASE_TYPES,
_BIDI_ARABIC_TYPES,
_apply_bidi_reordering,
_rtl_sign,
_reverse_if_rtl,
_read_end,
_read_gap,
)
from .content_stream import (
_FLUSH_OPS,
_SHOW_OPS,
_OP_LEX_PREFIX,
_OP_OPERAND_COUNTS,
_tokenize_show_operators,
_assign_vertical_tags,
_assign_show_tz,
_page_vertical_resource_names,
)
from .glyph_tables import (
_GLYPHLIST_PATH,
_cached_glyphs,
_cached_encodings,
_load_glyph_tables,
_get_unicode_for_glyph,
_from_char_code,
)
from .cmap_parse import (
_utf16be_units_to_str,
_NUM_DECIMAL_RE,
_NUM_INFINITY_RE,
_NUM_HEX_RE,
_NUM_OCTAL_RE,
_NUM_BINARY_RE,
_WHITESPACE_STRIP,
_ieee_div,
_compute_skew,
_to_number,
_parse_int,
_cmap_str_to_int,
_parse_tounicode_cmap,
)
from .font_unicode import (
_TYPE1_SPECIAL_BYTES,
_TYPE1_WHITESPACE_BYTES,
_type1_builtin_encoding,
_simple_font_to_unicode,
_font_unicode_map,
)
from .code_walk import (
_resource_dict_xrefs,
_page_show_codes,
_char_category,
_walk_codes,
)
from .unicode_apply import (
_apply_font_unicode,
_synthesize_dropped_glyphs,
)
from .geometry import (
_obj_rotation,
_xf_point,
_compose_mtx,
_IDENT_MTX,
_collect_text_objs,
_build_obj_index,
_char_render_fs,
_find_obj_for_char,
)
from .char_extract import (
_extract_raw_chars,
_accumulate_type3_extents,
_type3_size_by_font,
_apply_type3_sizes,
_finalize_chars,
_inherited_box,
_page_view_rect,
_off_page,
)
from .merge import _merge_text_items
from .remerge import (
_start_rot_span,
_grow_rot_span,
_merge_rotated_one,
_remerge_rotated,
_new_oblique_span,
_close_oblique,
_oblique_space,
_merge_oblique_one,
_remerge_oblique,
_start_vert_span,
_close_vert_span,
_merge_vertical_one,
_grow_vert_span,
_remerge_vertical,
)
from .pipeline import (
_page_pass1,
_page_pass2,
_page_spans,
parse_charlevel_meta,
parse_charlevel,
)
__all__ = ["parse_charlevel", "parse_charlevel_meta"]
@@ -0,0 +1,393 @@
"""Raw textpage char extraction, Type3 sizing, and page viewport handling."""
from __future__ import annotations
import ctypes
import pypdfium2.raw as pdfium_c
from .text_normalize import (
_is_whitespace,
_is_zero_width_diacritic,
_is_invisible_format_mark,
)
from .geometry import (
_collect_text_objs,
_build_obj_index,
_find_obj_for_char,
)
def _extract_raw_chars(page, text_page) -> tuple[list[dict], list[dict]]:
"""First pass: walk textpage chars and attach font info via the bbox-containing text-object lookup. Returns ``(raw_chars, objects)``; glyph widths and the identity-matrix Type-3 size override are applied later, after document-wide Type-3 extents are known."""
objects = _collect_text_objs(page, text_page)
if not objects:
return [], []
obj_index = _build_obj_index(objects)
# First pass: collect raw textpage chars with their host obj.
count_item = pdfium_c.FPDFText_CountChars(text_page)
font_name_buffer = (ctypes.c_char * 256)()
flags = ctypes.c_int(0)
field = ctypes.c_float(0)
# Per-char FFI out-buffers and entry points, hoisted: each is overwritten
# by its call (buffers whose call result is unchecked are re-zeroed below,
# so a failed call reads back 0 exactly as a fresh buffer would).
char_origin_x = ctypes.c_double(0); char_origin_y = ctypes.c_double(0)
char_left_box = ctypes.c_double(0); char_right_box = ctypes.c_double(0)
value = ctypes.c_double(0); char_top_box = ctypes.c_double(0)
loose_box = pdfium_c.FS_RECTF(0, 0, 0, 0)
u32 = ctypes.c_uint32(0)
fs32 = ctypes.c_float(0)
byref = ctypes.byref
ox_ref = byref(char_origin_x); oy_ref = byref(char_origin_y)
l_ref = byref(char_left_box); r_ref = byref(char_right_box)
b_ref = byref(value); t_ref = byref(char_top_box)
loose_ref = byref(loose_box)
w_ref = byref(field)
flags_ref = byref(flags)
get_unicode = pdfium_c.FPDFText_GetUnicode
is_generated = pdfium_c.FPDFText_IsGenerated
get_char_origin = pdfium_c.FPDFText_GetCharOrigin
get_char_box = pdfium_c.FPDFText_GetCharBox
get_loose_box = pdfium_c.FPDFText_GetLooseCharBox
get_font_info = pdfium_c.FPDFText_GetFontInfo
get_glyph_width = pdfium_c.FPDFFont_GetGlyphWidth
js_is_ws = _is_whitespace
name_cache: dict[bytes, str] = {}
raw_chars: list[dict] = []
last_obj: dict | None = None
for index_value in range(count_item):
codepoint = get_unicode(text_page, index_value)
if codepoint < 0:
continue
# u == 0 (PDFium found no unicode for the glyph) is KEPT as '\x00':
# text extraction emits the raw charcode for unmapped codes, so its items
# really contain chr(0) for extension-font pieces at code 0, and the
# textpage char carries normal geometry. Skipping it lost the char AND desynced
# the unicode walk's object pairing around it.
ch_str = chr(codepoint)
is_ws = js_is_ws(codepoint)
# FPDFText_IsGenerated returns a c_int: 1 generated, 0 real, -1 error.
# Only a POSITIVE 1 may mark a char generated -- the -1 has to read the
# same way here as it does in the page-mode unicode walk, or the two
# char sets disagree and that walk desyncs.
is_gen = is_generated(text_page, index_value) == 1
# PDFium inserts is_generated chars as layout placeholders for
# Td/Tm jumps with no literal content-stream char (typically
# " ", "\r", "\n"). Dropping them outright leaves an
# unexplained advance gap that the merger then turns into a
# fake-space chunk, splitting e.g. "2.1 Computing the EMD"
# into three spans (2.1, " ", Computing the EMD) that pipeline
# treats as a numeric prefix alone (not a heading). Keep
# generated whitespace so the merger's whitespace branch fires
# save_last_char without emitting, letting the next visible
# glyph compute a tracking-size in-flow advance. Drop only
# non-whitespace generated chars (very rare).
if is_gen and not is_ws:
continue
char_origin_x.value = 0.0; char_origin_y.value = 0.0
get_char_origin(text_page, index_value, ox_ref, oy_ref)
ox_v = char_origin_x.value; oy_v = char_origin_y.value
# Fetch char bbox first so we can use its center for the obj
# lookup — origin alone fails when adjacent obj bboxes nearly
# touch (e.g. math-heavy page "(", math italic font \x01, ")" all on the same line
# with sub-pt gaps, where origin x falls inside the wrong obj's
# tolerance window). Using bbox center gives unambiguous
# containment.
char_left_box.value = 0.0; char_right_box.value = 0.0; value.value = 0.0; char_top_box.value = 0.0
get_char_box(text_page, index_value, l_ref, r_ref, b_ref, t_ref)
char_left, char_right, char_top, char_bottom = char_left_box.value, char_right_box.value, char_top_box.value, value.value
# Tight (ink) box center -> font-object disambiguation only.
center_x = (char_left + char_right) / 2 if char_right > char_left else ox_v
center_y = (char_top + char_bottom) / 2 if char_top > char_bottom else oy_v
# Horizontal extent for the SPAN comes from the LOOSE char box (the
# glyph's full advance cell), not the tight ink box. the PDF text-item
# widths are advance-based; the ink box undershoots each glyph's right
# edge by its side bearing (e.g. "]" ink-right 274.0 vs advance 275.2,
# as expected for advance-based text items). Using the ink box cumulatively under-fills
# display-math gaps so the column detector mis-reads them as gutters
# and splits a line ("E[x] = μ" -> "E[x]" fragment). Fall back to the
# ink box if the loose box is unavailable/degenerate.
# (_loose is only READ when the call succeeded, so the hoisted struct
# never leaks a previous char's values.)
if (get_loose_box(text_page, index_value, loose_ref)
and loose_box.right > loose_box.left):
loose_left, loose_right = loose_box.left, loose_box.right
# Vertical edges of the loose (advance-cell) box. For vertical-
# writing (Identity-V / WMode 1) text PDFium builds this cell by
# advancing -y from the PEN, so its upper edge IS the pen y and
# its extent IS the per-char vertical advance (W2/DW2 applied by
# PDFium itself). PDFium fills top/bottom in flow order here, so
# they arrive inverted (top < bottom); keep both raw edges.
cell_top, cell_bottom = loose_box.top, loose_box.bottom
else:
loose_left, loose_right = char_left, char_right
cell_top, cell_bottom = char_top, char_bottom
# Character font size disambiguates overlapping objects, such as large
# figure labels sharing a y range with smaller heading text.
# text-page and character-index lookup read the true per-char rendered size
# (FPDFText_GetMatrix) and the reported font size (FPDFText_GetFontSize)
# lazily, only to break a multi-object containment tie — see
# _find_obj_for_char.
obj = _find_obj_for_char(
obj_index, center_x, center_y, tol=1.0, char_fs=None, text_page=text_page, char_idx=index_value
)
if obj is None:
obj = (
_find_obj_for_char(obj_index, ox_v, oy_v, tol=1.0,
char_fs=None, text_page=text_page, char_idx=index_value)
or _find_obj_for_char(obj_index, ox_v, oy_v, tol=5.0,
char_fs=None, text_page=text_page, char_idx=index_value)
or last_obj
)
if obj is None:
continue
last_obj = obj
name = get_font_info(text_page, index_value, font_name_buffer, 256, flags_ref)
if name > 1:
raw_name = font_name_buffer[:name]
char_font_name = name_cache.get(raw_name)
if char_font_name is None:
char_font_name = raw_name.decode(
"latin-1", errors="replace").rstrip("\x00")
name_cache[raw_name] = char_font_name
else:
char_font_name = obj["font_name"]
# Use baseline (oy) as bbox bottom and baseline + fs_eff as top.
# the span anchoring rule uses matrix.f (= baseline y) for both top/
# bottom anchors of its span, so chars of the same line all
# land at the same bottom even when their ink extends below
# baseline ("(", "g", "y" with descenders) or above ("\x01"
# math glyphs). This is what the heading heuristics' tokenizer assumes when
# checking |c1.C - c2.C| < 1 to decide whether two spans are on
# the same line.
baseline_y = oy_v
char_top = baseline_y + obj["fs_eff"]
# Capture the raw glyph advance now, while this page (and thus the
# font handle) is alive. The fs_eff-dependent scaling happens later
# in _finalize_chars, after the document-wide Type-3 size is known,
# so deferring the call would require keeping every page open just to
# keep font handles valid (PDFium frees the font when the page is
# closed -> dangling handle).
u32.value = codepoint
fs32.value = obj["fs_raw"]
get_glyph_width(obj["font"], u32, fs32, w_ref)
raw_chars.append({
"i": index_value, "ch": ch_str, "u": codepoint,
"is_gen": is_gen,
"is_ws": is_ws,
"is_mn": _is_zero_width_diacritic(codepoint),
"is_cf": _is_invisible_format_mark(codepoint),
"ox": ox_v, "oy": oy_v,
"left": loose_left, "right": loose_right,
"top": char_top, "bottom": baseline_y,
"box_top": char_top,
"box_bottom": char_bottom,
"cell_top": cell_top, "cell_bot": cell_bottom,
"w_raw": field.value,
"obj": obj, "font_name": char_font_name,
})
return raw_chars, objects
def _accumulate_type3_extents(raw_chars: list[dict], acc: dict) -> None:
"""Accumulate document-wide per-font glyph-bbox extents for identity-matrix Type-3 fonts. These fonts use a synthesized font bbox from the union of CharProc glyph boxes and render every glyph at that uniform height. PDFium reports a constant font size and identity CTM for these fonts, but its char box returns each glyph's declared bounds exactly, so box-top/bottom relative to the baseline reveal the rendered glyph extents. Aggregating across the whole document makes the font sizing coverage-independent; a per-page union would drift with sparse page content. Scoped to the identity-matrix Type-3 branch so normal and scaled-matrix fonts are untouched."""
for candidate_item in raw_chars:
item_value = candidate_item["obj"]
if item_value["fs_raw"] >= 1.5 or item_value["scale_y"] >= 1.5 or candidate_item["is_ws"]:
continue
top = candidate_item["box_top"] - candidate_item["oy"]
bot = candidate_item["box_bottom"] - candidate_item["oy"]
if top <= bot: # degenerate glyph box (text extraction skips d1 i==0)
continue
_xref_key = item_value["font_key"]
entry_item = acc.get(_xref_key)
if entry_item is None:
acc[_xref_key] = [top, bot]
else:
if top > entry_item[0]:
entry_item[0] = top
if bot < entry_item[1]:
entry_item[1] = bot
def _type3_size_by_font(acc: dict) -> dict:
"""font handle -> rendered font.bbox height = max ascent - min descent, i.e. span merger ``a = font.bbox[3] - font.bbox[1]`` in page units. Snap to the shortest decimal (PDFium float32 vs span merger float64) for clean knife-edge size comparisons downstream (the page-median gate)."""
out: dict = {}
for _xref_key, (top, bot) in acc.items():
if top > bot:
out[_xref_key] = float(f"{top - bot:.6g}")
return out
def _apply_type3_sizes(raw_chars: list[dict], size_by_font: dict) -> None:
"""Override fs_eff with the document-wide Type-3 size and reset each char's span top to baseline + that size."""
if not size_by_font:
return
for candidate_item in raw_chars:
item_value = candidate_item["obj"]
if item_value["fs_raw"] >= 1.5 or item_value["scale_y"] >= 1.5:
continue
font_size_value = size_by_font.get(item_value["font_key"])
if font_size_value:
item_value["fs_eff"] = font_size_value
candidate_item["top"] = candidate_item["oy"] + font_size_value
def _finalize_chars(raw_chars: list[dict]) -> list[dict]:
"""Second pass: compute glyph_w per char and emit the merged-ready dicts. The right glyph width definition depends on how PDFium reports the font's metrics: (a) Normal Type 1 fonts (fs_raw >= 1.5, scale.a ~= 1): FPDFFont_GetGlyphWidth(font, code, fs_raw) returns the advance in page units. Use as-is x matrix.a. (b) Scaled-matrix Type 3 (fs_raw < 1.5 but matrix scale >= 1.5, e.g. vector-heavy page's a scaled Type-3 subset with scale=36.49): GetGlyphWidth at fs_raw=0.19 gives font-natural-unit width; x matrix scale recovers page units. (c) Identity-matrix Type 3 (fs_raw < 1.5, matrix.a ~= 1, e.g. identity-matrix Type-3 sample an identity-matrix Type-3 font): GetGlyphWidth's output is wrong by an unknown FontMatrix factor (PDFium doesn't fold this for these fonts). Fall back to neighbor-step fallback (next_char.ox - this_char.ox within same obj). """
out: list[dict] = []
for key_value, candidate_item in enumerate(raw_chars):
if candidate_item.get("drop"):
# Folded into the previous char by _apply_font_unicode (PDFium's
# decomposition of a glyph text extraction emits as ONE precomposed char).
continue
obj = candidate_item["obj"]
# w_raw = FPDFFont_GetGlyphWidth(font, code, fs_raw), captured in the
# first pass while the page/font handle was alive.
raw = candidate_item["w_raw"]
if "w_synth" in candidate_item:
# Synthesized glyph (PDFium font-layer drop): the advance was
# computed from the surviving neighbors' pen gap.
glyph_w = candidate_item["w_synth"]
elif obj["fs_raw"] >= 1.5 or obj["scale_y"] >= 1.5:
# Cases (a) and (b): GetGlyphWidth + matrix scaling works.
glyph_w = raw * obj["scale_x"]
else:
# Case (c): Identity-matrix Type 3 — derive from neighbor.
nxt = raw_chars[key_value + 1] if key_value + 1 < len(raw_chars) else None
if (
nxt is not None
and nxt["obj"] is obj
and abs(nxt["oy"] - candidate_item["oy"]) < 0.5
and nxt["ox"] > candidate_item["ox"]
):
glyph_w = nxt["ox"] - candidate_item["ox"]
else:
# Last char in obj or new line — scale by fs_eff/fs_raw.
scale = (obj["fs_eff"] / obj["fs_raw"]) if obj["fs_raw"] > 0 else 1.0
glyph_w = raw * scale
# NOTE: glyph_w is PDFium's FPDFFont_GetGlyphWidth, used by the extraction advance model
# the font's glyph width; this is the advance model. For some RTL
# (Hebrew/Arabic) fonts PDFium's GetGlyphWidth does not match the actual
# rendered char spacing, which leaves spurious intra-word spaces; that is
# a PDF backend DATA LIMITATION (PDFium's hmtx/advance reporting),
# not a condition to compensate for here (any positional override conflates
# glyph advance with TJ word-gaps and breaks shaped Arabic). Left as-is.
reference_item = {
"ch": candidate_item["ch"],
"is_ws": candidate_item["is_ws"],
"is_mn": candidate_item["is_mn"],
"is_cf": candidate_item["is_cf"],
"ox": candidate_item["ox"], "oy": candidate_item["oy"],
"glyph_w": glyph_w,
"fs": obj["fs_eff"],
"fs_x": obj["fs_raw"] * obj["scale_x"] if obj["scale_x"] > 0 else obj["fs_eff"],
# Left edge from the text-positioning pen origin (ox), matching
# span merger, not the glyph ink box: the ink-box left drifts ~0.1pt by
# first-glyph side bearing, which trips the column-alignment gate
# gate (tol 0.1) and over-splits double-spaced blocks. Right stays
# ink-box (pen-right via glyph_w is unreliable for Type-3 fonts).
"left": candidate_item["ox"], "right": candidate_item["right"],
"top": candidate_item["top"], "bottom": candidate_item["bottom"],
"font_name": candidate_item["font_name"],
# Unique per-font identity (the PDFium font handle, == span merger'
# loaded font identity). The merger splits chunks on this, not on font_name:
# identity-matrix Type-3 fonts (an identity-matrix Type-3 font) all report an
# empty name, so a name-based split can't separate a 12pt body run
# from an inline 11pt code word ("...of expressions..."). span merger
# emits a separate text item per font, so the body keeps fs=12 and
# the code word fs=11.16 instead of the whole run collapsing to the
# smaller fs_min.
"font_key": obj["font_key"],
"weight": obj["weight"],
"obj": obj, # host text object (Tj/show-text)
}
if obj.get("vertical"):
# Vertical-writing pen model, from the loose advance cell (probe-
# for Identity-V: cell upper edge == pen y, cell extent ==
# the per-char vertical advance with W2/DW2 applied by PDFium, and
# the cell is horizontally centred on the pen x because the default
# vertical origin vx is w/2 -- the default vertical-origin convention when the
# font has no per-char vmetric).
pen_y = max(candidate_item["cell_top"], candidate_item["cell_bot"])
reference_item["v_pen_x"] = (candidate_item["left"] + candidate_item["right"]) / 2.0
reference_item["v_pen_y"] = pen_y
# pen y after this glyph's advance (text extraction previous glyph transform[5])
reference_item["v_after"] = min(candidate_item["cell_top"], candidate_item["cell_bot"])
out.append(reference_item)
return out
def _inherited_box(pdf_doc, page_idx: int, name: str):
"""span merger ``inherited page-box lookup`` definition: MediaBox/CropBox resolved through the page-tree ``/Parent`` chain (page-tree inheritance lookup). PDFium's FPDFPage_Get*Box does NOT inherit (pdfium bug 1786), so inherited boxes must come from the PyPDF2 channel. Returns a raw 4-tuple or None (absent / not a 4-number array, matching span merger length gate)."""
try:
xref_cursor = pdf_doc.page_xref(page_idx)
for _ in range(32):
token_value, value = pdf_doc.xref_get_key(xref_cursor, name)
if token_value != "null":
if token_value != "array":
return None
box_tokens = value.strip().lstrip("[").rstrip("]").split()
if len(box_tokens) != 4:
return None # span merger: array check and length == 4
try:
return tuple(float(box_token) for box_token in box_tokens)
except ValueError:
return None
parent_key_type, position_value = pdf_doc.xref_get_key(xref_cursor, "Parent")
if parent_key_type != "xref":
return None
xref_cursor = int(position_value.split()[0])
except Exception:
return None
return None
def _page_view_rect(page, med_raw=None, crop_raw=None) -> tuple[float, float, float, float] | None:
"""span merger ``normalized page view`` : rectangle normalization'd CropBox clamped to the rectangle normalization'd MediaBox. Differing boxes are intersected (rectangle intersection); an empty or zero-area intersection, and a degenerate CropBox, fall back to the MediaBox (a degenerate MediaBox falls back to US-Letter, text extraction US-Letter fallback media box). ``med_raw``/``crop_raw`` are the INHERITED boxes from ``_inherited_box`` (None = absent/no reader); the PDFium getters below are the non-inheriting fallback."""
def norm(secondary_item):
if secondary_item is None:
return None
box_x_min, box_y_min, box_x_max, box_y_max = secondary_item
count_item = (min(box_x_min, box_x_max), min(box_y_min, box_y_max), max(box_x_min, box_x_max), max(box_y_min, box_y_max))
return count_item if (count_item[2] - count_item[0] > 0 and count_item[3] - count_item[1] > 0) else None
med = norm(med_raw)
if med is None:
try:
med = norm(tuple(page.get_mediabox()))
except Exception:
med = None
if med is None:
med = (0.0, 0.0, 612.0, 792.0)
crop = norm(crop_raw)
if crop is None:
try:
crop = norm(tuple(page.get_cropbox()))
except Exception:
crop = None
if crop is None or crop == med:
return med
x_min, y_min = max(crop[0], med[0]), max(crop[1], med[1])
x_max, y_max = min(crop[2], med[2]), min(crop[3], med[3])
if x_max - x_min <= 0 or y_max - y_min <= 0:
return med
return (x_min, y_min, x_max, y_max)
def _off_page(mapping: dict, view_box) -> bool:
"""Position-comparison view box test: a non-diacritic glyph whose text origin is outside the page view box is dropped. The check compares ``pos - view box origin`` against the raw x1/y1 upper bounds, not width/height. ``view_box`` is the normalized page view as ``(x0, y0, x1, y1)``; None disables the test."""
if view_box is None:
return False
origin_offset_x = mapping["ox"] - view_box[0]
origin_offset_y = mapping["oy"] - view_box[1]
return origin_offset_x < 0 or origin_offset_x > view_box[2] or origin_offset_y < 0 or origin_offset_y > view_box[3]
@@ -0,0 +1,341 @@
"""PostScript number parsing and ToUnicode CMap interpretation."""
from __future__ import annotations
import math
import re
from .pdf_objects import (
_PDF_WHITESPACE_BYTES,
_PDF_DELIMITER_BYTES,
_PDF_STRING_ESCAPE_BYTES,
)
from .text_normalize import _WHITESPACE_CODEPOINTS
def _utf16be_units_to_str(units: list[int]) -> str:
"""Decode UTF-16BE token bytes into text. Odd trailing bytes pair with 0. A unit can exceed 0xFF during range carry, and no byte mask is applied before surrogate handling, so a composed value may exceed 0xFFFF and become an astral character."""
if len(units) % 2:
units = units + [0]
out: list[int] = []
key_value = 0
while key_value < len(units):
width_one = (units[key_value] << 8) | units[key_value + 1]
key_value += 2
if (width_one & 0xF800) != 0xD800:
out.append(width_one)
continue
width_two = 0
if key_value < len(units):
width_two = (units[key_value] << 8) | units[key_value + 1]
key_value += 2
out.append(((width_one & 0x3FF) << 10) + (width_two & 0x3FF) + 0x10000)
return "".join(chr(candidate_item) for candidate_item in out)
# ASCII-only numeric grammar used for PDF numeric-name heuristics. It uses the
# same decimal grammar as model.to_number but without NFKC normalization. Trim set is the
# Unicode WhiteSpace + LineTerminator set, not Python's str.strip set.
_NUM_DECIMAL_RE = re.compile(r"^[+-]?(?:[0-9]+\.?[0-9]*|\.[0-9]+)(?:[eE][+-]?[0-9]+)?$")
_NUM_INFINITY_RE = re.compile(r"^[+-]?Infinity$")
_NUM_HEX_RE = re.compile(r"^0[xX][0-9a-fA-F]+$")
_NUM_OCTAL_RE = re.compile(r"^0[oO][0-7]+$")
_NUM_BINARY_RE = re.compile(r"^0[bB][01]+$")
_WHITESPACE_STRIP = "".join(chr(unit_value) for unit_value in _WHITESPACE_CODEPOINTS)
def _ieee_div(value: float, other_item: float) -> float:
"""IEEE-754 division, no ZeroDivisionError (``0/0-> NaN, ``x/±0-> ±Inf with the usual sign rules)."""
if other_item != 0.0:
return value / other_item
if value == 0.0 or value != value:
return math.nan
return math.inf if (value > 0.0) == (math.copysign(1.0, other_item) > 0.0) else -math.inf
def _compute_skew(mtx: tuple) -> float:
"""Return the text matrix skew score for an item. transform's rotation/shear ratios, no zero guard (cardinal rotation -> Inf, upright -> 0). Degenerate case: the matrix-size path folds font size into the transform, so ``Tf 0`` text gives 0/0 = NaN there; the PDFium object matrix keeps font size separate and yields finite ratios (degenerate invisible text only)."""
primary_item, secondary_item, candidate_item, reference_item = mtx
quad_one = _ieee_div(secondary_item, primary_item)
quad_two = _ieee_div(candidate_item, reference_item)
return quad_one * quad_one + quad_two * quad_two
def _to_number(text: str) -> float:
"""/ ``numeric conversion`` (no NFKC): trim parser whitespace, ``""-> 0, then the numeric literal grammar (decimal/exponent, ``0x``/``0o``/``0b``, ``+-Infinity``); anything else -> NaN."""
token_value = text.strip(_WHITESPACE_STRIP)
if token_value == "":
return 0.0
if _NUM_INFINITY_RE.match(token_value):
return -math.inf if token_value[0] == "-" else math.inf
if _NUM_HEX_RE.match(token_value):
return float(int(token_value[2:], 16))
if _NUM_OCTAL_RE.match(token_value):
return float(int(token_value[2:], 8))
if _NUM_BINARY_RE.match(token_value):
return float(int(token_value[2:], 2))
if _NUM_DECIMAL_RE.match(token_value):
return float(token_value)
return math.nan
def _parse_int(text: str, radix: int) -> float:
"""skip leading parser whitespace, an optional sign, an optional ``0x`` prefix when ``radix == 16``, then the leading run of radix digits. Returns ``NaN`` (as in the heading heuristics) when no digit is consumed."""
token_value = text.lstrip(_WHITESPACE_STRIP)
index_value = 0
neg = False
if index_value < len(token_value) and token_value[index_value] in "+-":
neg = token_value[index_value] == "-"
index_value += 1
if radix == 16 and token_value[index_value:index_value + 2] in ("0x", "0X"):
index_value += 2
digits = "0123456789abcdefghijklmnopqrstuvwxyz"[:radix]
start = index_value
val = 0
while index_value < len(token_value) and token_value[index_value].lower() in digits:
val = val * radix + digits.index(token_value[index_value].lower())
index_value += 1
if index_value == start:
return math.nan
return float(-val if neg else val)
def _cmap_str_to_int(seq) -> int:
"""Accumulate CMap definition-code bytes with 32-bit unsigned wrap."""
primary_item = 0
for codepoint in seq:
primary_item = ((primary_item << 8) | codepoint) & 0xFFFFFFFF
return primary_item
def _parse_tounicode_cmap(data: bytes) -> dict[int, str]:
"""CMap reader for ToUnicode streams, following text extraction CMap parsing + ToUnicode parsing: bfchar/bfrange with hex, literal-string, and (bfrange dst / array elements) integer tokens, plus cidchar/cidrange (numeric entries -> code-point conversion, the numeric-CID class). Structural junk is contained per block like CMap parsing's warn-and-continue catch (the block is dropped, the map survives); only decode-level errors (chr on a code-point conversion-invalid value) propagate so the caller reaches span merger ToUnicode parsing rejection path (-> no included map)."""
tokens: list = []
index_value, count_item = 0, len(data)
while index_value < count_item:
candidate_item = data[index_value]
if candidate_item in _PDF_WHITESPACE_BYTES:
index_value += 1
elif candidate_item == 0x25: # comment
while index_value < count_item and data[index_value] not in b"\r\n":
index_value += 1
elif candidate_item == 0x3C: # << dict-open (skip) or <hex>
if index_value + 1 < count_item and data[index_value + 1] == 0x3C:
index_value += 2
continue
state_item = data.find(b">", index_value)
if state_item < 0:
break # unterminated hex string: stop and keep tokens already read
hex_values = "".join(chr(secondary_item) for secondary_item in data[index_value + 1:state_item]
if chr(secondary_item) in "0123456789abcdefABCDEF")
if len(hex_values) % 2:
hex_values = hex_values[:-1] # drop a lone trailing hex digit
tokens.append(("hex", tuple(bytes.fromhex(hex_values))))
index_value = state_item + 1
elif candidate_item == 0x3E: # >> dict-close (skip)
index_value += 2 if (index_value + 1 < count_item and data[index_value + 1] == 0x3E) else 1
elif candidate_item in b"[]":
tokens.append(("delim", chr(candidate_item)))
index_value += 1
elif candidate_item == 0x2F: # /name
state_item = index_value + 1
while state_item < count_item and data[state_item] not in _PDF_WHITESPACE_BYTES and data[state_item] not in _PDF_DELIMITER_BYTES:
state_item += 1
tokens.append(("name", data[index_value + 1:state_item].decode("latin-1")))
index_value = state_item
elif candidate_item == 0x28: # (string) -- literal-string lexer code units (dst values)
depth = 0
unicode_scalar: list[int] = []
while index_value < count_item:
byte_value = data[index_value]
if byte_value == 0x5C:
if index_value + 1 >= count_item:
index_value += 1
break
entry_item = data[index_value + 1]
if entry_item in _PDF_STRING_ESCAPE_BYTES:
unicode_scalar.append(_PDF_STRING_ESCAPE_BYTES[entry_item])
index_value += 2
elif 0x30 <= entry_item <= 0x37:
state_item = index_value + 1
val = 0
while state_item < count_item and state_item - index_value <= 3 and 0x30 <= data[state_item] <= 0x37:
val = (val << 3) | (data[state_item] - 0x30)
state_item += 1
unicode_scalar.append(val)
index_value = state_item
elif entry_item in (0x0D, 0x0A):
index_value += 2
if entry_item == 0x0D and index_value < count_item and data[index_value] == 0x0A:
index_value += 1
else:
unicode_scalar.append(entry_item)
index_value += 2
continue
if byte_value == 0x28:
if depth:
unicode_scalar.append(byte_value)
depth += 1
elif byte_value == 0x29:
depth -= 1
if depth == 0:
index_value += 1
break
unicode_scalar.append(byte_value)
else:
unicode_scalar.append(byte_value)
index_value += 1
tokens.append(("hex", tuple(unicode_scalar)))
else:
state_item = index_value
while state_item < count_item and data[state_item] not in _PDF_WHITESPACE_BYTES and data[state_item] not in _PDF_DELIMITER_BYTES:
state_item += 1
word = data[index_value:state_item].decode("latin-1")
if (0x30 <= data[index_value] <= 0x39) or data[index_value] in b"+-.":
try:
numeric_value = float(word)
except ValueError:
numeric_value = 0.0
tokens.append(("num", numeric_value))
else:
tokens.append(("op", word))
index_value = state_item
out: dict[int, str] = {}
def codepoint_to_string(numeric_value: float) -> str:
# ToUnicode parsing numeric entry: code-point conversion(token) -- its
# RangeError (non-integer / out of range) kills the whole map, so
# chr's ValueError propagate.
codepoint = int(numeric_value)
if codepoint != numeric_value:
raise ValueError("code-point conversion non-integer")
return chr(codepoint)
def is_int(numeric_value: float) -> bool:
# The integer test that guards the numeric-entry check and selects the
# destination branch rejects +-Infinity, NaN AND any fractional value.
return math.isfinite(numeric_value) and numeric_value == int(numeric_value)
def map_range_units(range_start: int, range_end: int, units: list[int]) -> None:
# text extraction CMap.bf-range mapping : ``last byte`` is FIXED to
# the ORIGINAL dst length-1; only THAT byte index is incremented. On
# 0xFF overflow it carries into byte last byte-1 (byte-to-character conversion ToUint16
# == the & 0xFFFF) and sets the tail to 0x00; the next non-overflow
# step is substring(0,last byte)+chr(next), so a 1-byte dst collapses
# back to ONE byte. A 1-byte 0xFF overflow gives "\x00\x00"
# Empty destinations yield "" for the first code and "\x00" for each
# subsequent code after carry.
last_byte = len(units) - 1
for code in range(range_start, range_end + 1):
out[code] = _utf16be_units_to_str(units)
if last_byte < 0:
units = [0x00]
continue
cur = units[last_byte] if last_byte < len(units) else 0
nxt = cur + 1
if nxt > 0xFF:
if last_byte - 1 >= 0:
units = (units[:last_byte - 1]
+ [(units[last_byte - 1] + 1) & 0xFFFF, 0x00])
else:
units = [0x00, 0x00]
else:
units = units[:last_byte] + [nxt]
key_value = 0
while key_value < len(tokens):
kind, val = tokens[key_value]
if kind == "op" and val == "beginbfchar":
key_value += 1
while key_value + 1 < len(tokens) and tokens[key_value][0] == "hex":
src = _cmap_str_to_int(tokens[key_value][1])
if tokens[key_value + 1][0] != "hex":
# the heading heuristics string-operand check throws -> CMap parsing catch drops the
# rest of the block, map survives.
key_value += 2
break
out[src] = _utf16be_units_to_str(list(tokens[key_value + 1][1]))
key_value += 2
elif kind == "op" and val == "beginbfrange":
key_value += 1
while (key_value + 1 < len(tokens) and tokens[key_value][0] == "hex"
and tokens[key_value + 1][0] == "hex"):
src_start = _cmap_str_to_int(tokens[key_value][1])
src_end = _cmap_str_to_int(tokens[key_value + 1][1])
key_value += 2
if src_end - src_start > 0xFFFFFF:
# The range-limit throw is raised from INSIDE the bf-range
# mapping itself, i.e. from inside the call that CMap
# parsing wraps, so the rest of the block goes with it (the
# destination has already been lexed -- for an array, up to
# and including the "]").
if key_value < len(tokens) and tokens[key_value] == ("delim", "["):
while key_value < len(tokens) and tokens[key_value] != ("delim", "]"):
key_value += 1
key_value += 1
elif key_value < len(tokens) and tokens[key_value][0] in ("hex", "num"):
key_value += 1
break
if key_value < len(tokens) and tokens[key_value] == ("delim", "["):
key_value += 1
code = src_start
# The array form stores EVERY lexed object up to "]" or end
# of input; the UTF-16BE walk over a value that has no
# length (a name, an operator) runs zero times and yields
# the empty string.
while key_value < len(tokens) and tokens[key_value] != ("delim", "]"):
if code <= src_end:
dst_token = tokens[key_value]
if dst_token[0] == "hex":
out[code] = _utf16be_units_to_str(list(dst_token[1]))
elif dst_token[0] == "num":
out[code] = codepoint_to_string(dst_token[1])
else:
out[code] = ""
code += 1
key_value += 1
if key_value < len(tokens):
key_value += 1
elif key_value < len(tokens) and tokens[key_value][0] == "hex":
units = list(tokens[key_value][1])
key_value += 1
map_range_units(src_start, src_end, units)
elif key_value < len(tokens) and tokens[key_value][0] == "num" and is_int(tokens[key_value][1]):
# Integer destinations are one UTF-16 unit, then the normal
# increment walk applies. A non-integer number is neither an
# integer nor a string nor "[", so it falls through to the
# `else` arm below.
units = [int(tokens[key_value][1]) & 0xFFFF]
key_value += 1
map_range_units(src_start, src_end, units)
else:
break # parse error -> contained: drop the block
elif kind == "op" and val == "begincidchar":
key_value += 1
while (key_value + 1 < len(tokens) and tokens[key_value][0] == "hex"
and tokens[key_value + 1][0] == "num"):
if not is_int(tokens[key_value + 1][1]):
# The integer check throws -> the CMap parsing catch drops
# the rest of the block, map survives.
key_value += 2
break
out[_cmap_str_to_int(tokens[key_value][1])] = codepoint_to_string(tokens[key_value + 1][1])
key_value += 2
elif kind == "op" and val == "begincidrange":
key_value += 1
while (key_value + 2 < len(tokens) and tokens[key_value][0] == "hex"
and tokens[key_value + 1][0] == "hex" and tokens[key_value + 2][0] == "num"):
src_start = _cmap_str_to_int(tokens[key_value][1])
src_end = _cmap_str_to_int(tokens[key_value + 1][1])
start = tokens[key_value + 2][1]
key_value += 3
if not is_int(start):
break # the integer check precedes CID-range mapping: block dropped
if src_end - src_start > 0xFFFFFF:
break # CID-range range-limit: the block is dropped too
for code in range(src_start, src_end + 1):
out[code] = codepoint_to_string(start + (code - src_start))
else:
key_value += 1
return out
@@ -0,0 +1,249 @@
"""Resource-dictionary xref walking and per-page show-code enumeration."""
from __future__ import annotations
import re
import unicodedata
from PyPDF2.generic import (
IndirectObject as PdfIndirectRef, NameObject as PdfName, NumberObject as PdfNumber,
FloatObject as PdfFloat, BooleanObject as PdfBoolean,
DictionaryObject as PdfDictionary, ArrayObject as PdfArray,
)
from .pdf_objects import _decode_pdf_name
from .text_normalize import (
_normalize_unicodes,
_WHITESPACE_CODEPOINTS,
_is_whitespace,
)
from .content_stream import _tokenize_show_operators
def _resource_dict_xrefs(pdf_doc, owner_xref: int, sub: str) -> dict[bytes, int]:
"""{canonical resname bytes: xref} for /Redefinitions/<sub> of a page or Form XObject dict, following indirection; for pages, /Redefinitions may be inherited through the /Parent chain."""
val = ("null", "null")
xref_cursor = owner_xref
for _ in range(32): # /Parent chain (pages); XObjects never recurse here
val = pdf_doc.xref_get_key(xref_cursor, f"Resources/{sub}")
if val[0] != "null":
break
if pdf_doc.xref_get_key(xref_cursor, "Resources")[0] != "null":
break # Redefinitions exists but lacks <sub>
parent_key_type, parent_xref_value = pdf_doc.xref_get_key(xref_cursor, "Parent")
if parent_key_type != "xref":
break
xref_cursor = int(parent_xref_value.split()[0])
if val[0] == "xref":
body = pdf_doc.xref_object(int(val[1].split()[0]), compressed=True)
elif val[0] == "dict":
body = val[1]
else:
return {}
out: dict[bytes, int] = {}
for measure_item in re.finditer(r"/([^\s/\[\]<>()]+)\s+(\d+)\s+\d+\s+R", body):
out[_decode_pdf_name(measure_item.group(1).encode("latin-1"))] = int(measure_item.group(2))
# DIRECT (inline) sub-dict entries carry no `N G R` for the regex; span merger
# reference resolution resolves them all the same, so register each as a virtual
# pseudo-xref and the normal integer-keyed pipeline address it.
try:
node = pdf_doc._resolve_object(xref_cursor)
for part in ("Resources", sub):
if isinstance(node, PdfIndirectRef):
node = node.get_object()
node = node["/" + part] if (node is not None and "/" + part in node) else None
if node is not None:
if isinstance(node, PdfIndirectRef):
node = node.get_object()
for key_value in node.keys():
raw = node.raw_get(key_value)
if isinstance(raw, PdfIndirectRef):
continue # indirect: the regex pass covered it
if not hasattr(raw, "raw_get"):
continue # not a dict (malformed entry)
name = _decode_pdf_name(key_value.lstrip("/").encode("latin-1"))
if name not in out:
out[name] = pdf_doc.register_virtual(raw)
except Exception:
pass
return out
def _page_show_codes(
pdf_doc, page_idx: int,
) -> list[tuple[int | None, tuple[int, ...], float]] | None:
"""Every show op the page paints, in paint order, as ``(font_xref | None, charcode units, horizontal scale)-- including text inside Form XObjects, spliced at their ``Do`` position with the XObject's own font redefinitions (span merger text-content extraction recurses the same way; PDFium's textpage flattens them inline). The recursion runs on a CLONE of the live text state, so a form inherits both the active font and the horizontal scale; a Tz inside the form REPLACES it and never leaks back out. None when the page can't be read."""
page_xref = pdf_doc.page_xref(page_idx)
def walk(stream: bytes, fonts_res: dict[bytes, int],
xobjs_res: dict[bytes, int], cur_font: int | None, cur_tz: float,
visited: frozenset, depth: int,
out: list[tuple[int | None, tuple[int, ...], float]]) -> None:
if depth > 8:
return
flush_ids, fonts, show_text_units, horizontal_scales, xobject_paints = _tokenize_show_operators(stream, cur_tz)
dict_index = 0
for key_value in range(len(show_text_units) + 1):
while dict_index < len(xobject_paints) and xobject_paints[dict_index][0] == key_value:
paint_position, xname, font_at_do, tz_at_do = xobject_paints[dict_index]
dict_index += 1
xobject_ref = xobjs_res.get(xname) # lexer names arrive #XX-parsed
if xobject_ref is None or xobject_ref in visited:
continue
state_values, string_value = pdf_doc.xref_get_key(xobject_ref, "Subtype")
if state_values != "name" or string_value.lstrip("/") != "Form":
continue
sub_fonts = _resource_dict_xrefs(pdf_doc, xobject_ref, "Font") or fonts_res
sub_xobjs = _resource_dict_xrefs(pdf_doc, xobject_ref, "XObject") or xobjs_res
inherited = (fonts_res.get(font_at_do)
if font_at_do is not None else None)
try:
sub_stream = pdf_doc.xref_stream(xobject_ref)
except Exception:
# span merger: "XObject should be a stream" -> recovery mode skips
# THIS Do and keeps walking the page (a direct dict posing
# as /Form has no stream; must not kill the whole page).
continue
walk(sub_stream, sub_fonts, sub_xobjs,
inherited, tz_at_do, visited | {xobject_ref}, depth + 1, out)
if key_value < len(show_text_units):
resource_font_name = fonts[key_value]
resource_font_index = (fonts_res.get(resource_font_name)
if resource_font_name is not None else cur_font)
out.append((resource_font_index, show_text_units[key_value], horizontal_scales[key_value]))
try:
out: list[tuple[int | None, tuple[int, ...], float]] = []
walk(
pdf_doc[page_idx].read_contents(),
_resource_dict_xrefs(pdf_doc, page_xref, "Font"),
_resource_dict_xrefs(pdf_doc, page_xref, "XObject"),
None, 1.0, frozenset(), 0, out, # the initial text state starts at scale 1
)
return out
except Exception:
return None
def _char_category(text: str) -> tuple[bool, bool, bool]:
"""text extraction glyph Unicode category classification over a (possibly multi-char) glyph Unicode string: first match of /^(\\s)|(\\p{Mn})|(\\p{Cf})$/u decides (isWhitespace, zero-width diacritic classification, invisible format-mark classification)."""
for pos, char in enumerate(text):
codepoint = ord(char)
if pos == 0 and codepoint in _WHITESPACE_CODEPOINTS:
return True, False, False
cat = unicodedata.category(char)
if cat == "Mn":
return False, True, False
if cat == "Cf" and pos == len(text) - 1:
return False, False, True
return False, False, False
def _walk_codes(
chars: list[tuple[int, str]],
targets: list[str],
allow_skips: bool = False,
) -> tuple[list[tuple[int, str]], list[int], list[tuple[int, int]],
list[tuple[int, int]]] | None:
"""Walk one run of font-resolved per-code targets against the PDFium chars emitted for the same run; return (patches, drops, consumed, skips) or None on desync. ``consumed`` maps each consumed char's textpage index to the target index that consumed it, which lets the group re-walk repair char-to-object attribution. ``skips`` records each skipped target as (target index, char position it belongs before) for glyph re-synthesis. PDFium's emission per code is unknowable a priori: it may match the font target, fall back to the raw code, or expand a glyph into several chars. Consumption is resolved per code by candidate match: the font target, then its normalized expansions (fixed unicode substitution table, NFKC, NFKD, NFD). On an expansion match where the final span text still converges, the chars are left alone; otherwise the first char is patched to the target unicode and the rest of the run is dropped so the single glyph still carries the advance. The run is valid only if both streams end in sync. """
text = "".join(target_char for _, target_char in chars)
line_value = len(text)
pos = 0
patches: list[tuple[int, str]] = []
drops: list[int] = []
consumed: list[tuple[int, int]] = [] # (char textpage index, target index)
skips: list[tuple[int, int]] = [] # (target index, char position)
skip_until = -1
shift_run = 0
for index_value, token_value in enumerate(targets):
if index_value <= skip_until:
continue # part of an anchored skip run recorded below
if pos >= line_value:
if allow_skips:
# Chars exhausted with targets left: the walk arrived here in
# sync, so every remaining target is a glyph PDFium never
# emitted (the both-exhaust gate in reverse).
skips.append((index_value, pos))
continue
return None # codes left over: desync
target_len = len(token_value)
if text[pos:pos + target_len] == token_value:
consumed.extend((chars[query_value][0], index_value) for query_value in range(pos, pos + target_len))
pos += target_len
shift_run = 0
continue
matched = False
for normalized_text in (_normalize_unicodes(token_value),
unicodedata.normalize("NFKC", token_value),
unicodedata.normalize("NFKD", token_value),
unicodedata.normalize("NFD", token_value)):
if normalized_text != token_value and text[pos:pos + len(normalized_text)] == normalized_text:
if _normalize_unicodes(token_value) != normalized_text:
patches.append((chars[pos][0], token_value))
drops.extend(chars[query_value][0] for query_value in range(pos + 1, pos + len(normalized_text)))
consumed.extend((chars[query_value][0], index_value) for query_value in range(pos, pos + len(normalized_text)))
pos += len(normalized_text)
matched = True
break
if matched:
shift_run = 0
continue
# Anchored drop-skip (LAST-RESORT mode only: the window re-walk has
# already ruled out the stolen-edge-glyph hypothesis): when PDFium
# genuinely never emitted the glyph (font-layer drop -- dense math-heavy page's
# 4 α, math-heavy page's scanned-page '~', both absent from the textpage AND
# FPDFTextObj_GetText), the target has no char anywhere. Skip it
# WITHOUT consuming, but only when the next two unskipped targets
# literally anchor on the upcoming chars, so a mis-decode (which
# needs the 1-char patch below instead) can't be eaten as a skip.
# Drops can be CONSECUTIVE (OCR pages drop runs of glyphs), so scan
# forward for the smallest run i..i+m-1 whose following pair
# anchors; a wrong run leaves chars unconsumed and the exhaust gate
# below still rolls everything back.
if allow_skips:
skip_run_length = 0
for skip_len in range(1, len(targets) - index_value + 1):
anchor = targets[index_value + skip_len:index_value + skip_len + 2]
if not anchor or not all(len(primary_item) == 1 for primary_item in anchor):
break
str_value = "".join(anchor)
if text[pos:pos + len(str_value)] == str_value:
skip_run_length = skip_len
break
if skip_run_length:
skips.extend((query_value, pos) for query_value in range(index_value, index_value + skip_run_length))
skip_until = index_value + skip_run_length - 1
shift_run = 0
continue
# PDFium's textpage COLLAPSES space runs: a whitespace target facing
# a non-whitespace char means the space's char simply does not exist
# in the textpage (it can never be a re-decode of the current char).
# Desync rather than mis-patch the neighbouring glyph into a space;
# table rows can contain real star glyphs adjacent to synthetic spaces.
if (all(_is_whitespace(ord(unit_char)) for unit_char in token_value)
and not _is_whitespace(ord(text[pos]))):
return None
# Off-by-one guard for the 1-char assumption below: when an edge
# glyph was mis-attributed to a neighbouring object, every pair
# mismatches with the streams shifted by one, and a ligature
# expansion elsewhere can re-balance the counts so the exhaust gate
# alone would COMMIT the shifted alignment and attach punctuation to
# the wrong run. The shift has a literal signature --
# the NEXT target equals the current char(s), or the current target
# equals the NEXT char(s) -- which legitimate decode mismatches
# (text extraction symbol vs PDFium control char) never produce. Two
# consecutive hits = systematic shift -> desync, letting the
# adjacent-run group re-walk re-align both objects cleanly.
nxt = targets[index_value + 1] if index_value + 1 < len(targets) else None
if ((nxt is not None and text[pos:pos + len(nxt)] == nxt)
or text[pos + 1:pos + 1 + target_len] == token_value):
shift_run += 1
if shift_run >= 2:
return None
else:
shift_run = 0
patches.append((chars[pos][0], token_value))
consumed.append((chars[pos][0], index_value))
pos += 1
if pos != line_value:
return None # chars left over: desync
return patches, drops, consumed, skips
@@ -0,0 +1,424 @@
"""Content-stream show-operator tokenization and per-page operator tagging."""
from __future__ import annotations
from .pdf_objects import (
_PDF_WHITESPACE_BYTES,
_PDF_DELIMITER_BYTES,
_PDF_STRING_ESCAPE_BYTES,
_decode_pdf_name,
)
# Text items start at font/size changes, positional line breaks or gaps, and
# content-stream flush operators (q/Q, Do, gs-/Font, marked content). PDFium's
# flattened FPDF_PAGEOBJ_TEXT objects can be one-per-glyph for per-glyph Tj
# streams, so this merger keeps a strict per-object split unless a later
# whitespace-aware rule proves a prose continuation.
#
# q/Q flush grouping is intentionally disabled. It requires fragile ordinal
# alignment between flattened PDFium objects and content-stream show operators,
# while the merger only needs the content stream for per-show-op font names
# (vertical-font flags) and Unicode-map reconstruction. setFont/gs-font are not
# flush scopes here; font_key/fs changes carry the style-boundary split.
_FLUSH_OPS = frozenset({b"q", b"Q", b"Do", b"BDC", b"BMC", b"EMC"})
_SHOW_OPS = frozenset({b"Tj", b"TJ", b"'", b'"'})
# content stream tokenizer content operator table: {op: (operand count, variable operand count)}.
# Commands NOT in this table are span merger "Unknown command" -- warned and skipped
# with the accumulated args PRESERVED (not cleared).
# content stream tokenizer operator table's null-value entries: pure lexer aids so object parser's
# longest-known-command walk can pass through prefixes of longer commands
# (B -> BM -> BMC, f -> false, n -> null). Not operators.
_OP_LEX_PREFIX = frozenset({
b"BM", b"BD", b"true", b"fa", b"fal", b"fals", b"false",
b"nu", b"nul", b"null",
})
_OP_OPERAND_COUNTS: dict[bytes, tuple[int, bool]] = {
b"w": (1, False), b"J": (1, False), b"j": (1, False), b"M": (1, False),
b"d": (2, False), b"ri": (1, False), b"i": (1, False), b"gs": (1, False),
b"q": (0, False), b"Q": (0, False), b"cm": (6, False), b"m": (2, False),
b"l": (2, False), b"c": (6, False), b"v": (4, False), b"y": (4, False),
b"h": (0, False), b"re": (4, False), b"S": (0, False), b"s": (0, False),
b"f": (0, False), b"F": (0, False), b"f*": (0, False), b"B": (0, False),
b"B*": (0, False), b"b": (0, False), b"b*": (0, False), b"n": (0, False),
b"W": (0, False), b"W*": (0, False), b"BT": (0, False), b"ET": (0, False),
b"Tc": (1, False), b"Tw": (1, False), b"Tz": (1, False), b"TL": (1, False),
b"Tf": (2, False), b"Tr": (1, False), b"Ts": (1, False), b"Td": (2, False),
b"TD": (2, False), b"Tm": (6, False), b"T*": (0, False), b"Tj": (1, False),
b"TJ": (1, False), b"'": (1, False), b'"': (3, False), b"d0": (2, False),
b"d1": (6, False), b"CS": (1, False), b"cs": (1, False), b"SC": (4, True),
b"SCN": (33, True), b"sc": (4, True), b"scn": (33, True), b"G": (1, False),
b"g": (1, False), b"RG": (3, False), b"rg": (3, False), b"K": (4, False),
b"k": (4, False), b"sh": (1, False), b"BI": (0, False), b"ID": (0, False),
b"EI": (1, False), b"Do": (1, False), b"MP": (1, False), b"DP": (2, False),
b"BMC": (1, False), b"BDC": (2, False), b"EMC": (0, False),
b"BX": (0, False), b"EX": (0, False),
}
def _tokenize_show_operators(
content_bytes: bytes, init_tz: float = 1.0,
) -> tuple[list[int], list[bytes | None], list[tuple[int, ...]], list[float],
list[tuple[int, bytes, bytes | None, float]]]:
"""Tokenize a PDF page content stream. For each text-showing operator, records the active flush scope, font redefinition name, raw charcode units, and horizontal scaling (starting at ``init_tz`` on stream entry, saved and restored by q/Q), plus every Form XObject paint position with the font and horizontal scaling live at that paint. The tokenizer is deliberately tolerant of malformed operators: it skips bad or short operands, preserves unknown-command operands, and emits an empty string for a show operator with the wrong string operand type."""
flush_ids: list[int] = []
fonts: list[bytes | None] = []
show_text_units: list[tuple[int, ...]] = []
xobject_paints: list[tuple[int, bytes, bytes | None, float]] = []
horizontal_scales: list[float] = []
flush_id = 0
cur_font: bytes | None = None
font_stack: list[bytes | None] = []
cur_tz = init_tz
tz_stack: list[float] = []
opnds: list[tuple[str, object]] = []
frames: list[tuple[str, list]] = [] # open [ / << collectors
non_processed: list[tuple[str, object]] = []
bi_mark: int | None = None
def push(kind: str, val: object) -> None:
(frames[-1][1] if frames else opnds).append((kind, val))
index_value = 0
count_item = len(content_bytes)
while index_value < count_item:
byte_value = content_bytes[index_value]
if byte_value in _PDF_WHITESPACE_BYTES:
index_value += 1
elif byte_value == 0x25: # % comment -> end of line
while index_value < count_item and content_bytes[index_value] not in b"\r\n":
index_value += 1
elif byte_value == 0x28: # ( literal string: decode per PDF 7.3.4.2
depth = 0
out: list[int] = []
while index_value < count_item:
literal_byte = content_bytes[index_value]
if literal_byte == 0x5c: # backslash escape
if index_value + 1 >= count_item:
index_value += 1
break
escape_byte = content_bytes[index_value + 1]
if escape_byte in _PDF_STRING_ESCAPE_BYTES:
out.append(_PDF_STRING_ESCAPE_BYTES[escape_byte])
index_value += 2
elif 0x30 <= escape_byte <= 0x37: # \ddd octal, 1-3 digits
token_end = index_value + 1
val = 0
while token_end < count_item and token_end - index_value <= 3 and 0x30 <= content_bytes[token_end] <= 0x37:
val = (val << 3) | (content_bytes[token_end] - 0x30)
token_end += 1
# text extraction literal-string lexer pushes byte-to-character conversion with NO
# byte mask: \400..\777 stay 256..511 (the PDF-spec
# high-order-overflow mask is deliberately absent).
out.append(val)
index_value = token_end
elif escape_byte in (0x0D, 0x0A): # \<EOL> line continuation
index_value += 2
if escape_byte == 0x0D and index_value < count_item and content_bytes[index_value] == 0x0A:
index_value += 1
else: # \x -> x
out.append(escape_byte)
index_value += 2
continue
# NOTE: bare CR/LF inside a literal string fall through to the
# raw push below: literal strings keep bare CR/LF as-is here
# rather than applying PDF-spec "treat as 0x0A" normalization
# is deliberately absent there; no CRLF collapsing either).
if literal_byte == 0x28:
if depth:
out.append(literal_byte)
depth += 1
elif literal_byte == 0x29:
depth -= 1
if depth == 0:
index_value += 1
break
out.append(literal_byte)
else:
out.append(literal_byte)
index_value += 1
push("str", tuple(out))
elif byte_value == 0x3c: # < : << dict-open, else <hex>
if index_value + 1 < count_item and content_bytes[index_value + 1] == 0x3c:
frames.append(("dict", []))
index_value += 2
else:
index_value += 1
nib: list[int] = []
while index_value < count_item and content_bytes[index_value] != 0x3e:
hex_byte = content_bytes[index_value]
if 0x30 <= hex_byte <= 0x39:
nib.append(hex_byte - 0x30)
elif 0x41 <= hex_byte <= 0x46:
nib.append(hex_byte - 0x37)
elif 0x61 <= hex_byte <= 0x66:
nib.append(hex_byte - 0x57)
index_value += 1
index_value += 1
if len(nib) % 2:
nib.pop() # drop a lone trailing hex digit
push("str", tuple(
(nib[key_value] << 4) | nib[key_value + 1] for key_value in range(0, len(nib), 2)
))
elif byte_value == 0x3e: # >> dict-close (or stray >)
if index_value + 1 < count_item and content_bytes[index_value + 1] == 0x3e:
index_value += 2
if frames and frames[-1][0] == "dict":
frames.pop()
push("dict", None)
# stray >> : text extraction command token -> unknown command -> tally preserved
else:
index_value += 1
elif byte_value == 0x5b: # [ -- one array operand (text extraction parser builds an array)
frames.append(("arr", []))
index_value += 1
elif byte_value == 0x5d: # ]
index_value += 1
if frames and frames[-1][0] == "arr":
items = frames.pop()[1]
# TJ semantics: only direct string elements show; numbers are
# kern adjustments and nested non-strings are ignored.
push("arr", tuple(codepoint for kerning_delta, vertical_value in items if kerning_delta == "str"
for codepoint in vertical_value)) # type: ignore[union-attr]
# stray ] : text extraction command token(']') -> unknown command -> tally preserved
elif byte_value in b"{}":
index_value += 1 # text extraction command token -> not in operator table -> "Unknown command", preserved
elif byte_value == 0x2f: # /name operand
index_value += 1
token_end = index_value
while token_end < count_item and content_bytes[token_end] not in _PDF_WHITESPACE_BYTES and content_bytes[token_end] not in _PDF_DELIMITER_BYTES:
token_end += 1
# Decode #XX escapes while lexing names so consumers receive the
# canonical name and do not re-decode downstream.
push("name", _decode_pdf_name(content_bytes[index_value:token_end]))
index_value = token_end
else: # number, keyword operand, or operator
first_char = content_bytes[index_value]
if (0x30 <= first_char <= 0x39) or first_char in b"+-.":
# Number token: consume the whole run to whitespace/delimiter.
# Malformed numeric runs are zeroed by the downstream parser.
token_end = index_value
while token_end < count_item and content_bytes[token_end] not in _PDF_WHITESPACE_BYTES and content_bytes[token_end] not in _PDF_DELIMITER_BYTES:
token_end += 1
else:
# text extraction command lexing: once the accumulated run IS a
# known command, stop extending as soon as the next char
# would break that -- 'q1' lexes as command token 'q' + number 1
# (real-world PDFs; text extraction built known commands for them).
token_end = index_value
known = False
while token_end < count_item and content_bytes[token_end] not in _PDF_WHITESPACE_BYTES and content_bytes[token_end] not in _PDF_DELIMITER_BYTES:
cand = content_bytes[index_value:token_end + 1]
if (known and cand not in _OP_OPERAND_COUNTS
and cand not in _OP_LEX_PREFIX):
break
token_end += 1
known = (content_bytes[index_value:token_end] in _OP_OPERAND_COUNTS
or content_bytes[index_value:token_end] in _OP_LEX_PREFIX)
operator_token = content_bytes[index_value:token_end]
index_value = token_end
if not operator_token:
index_value += 1
continue
if (0x30 <= first_char <= 0x39) or first_char in b"+-.":
# text extraction number lexer accepts only digit/sign/dot/exponent
# runs; Python float would also take "-inf"/"nan" tokens,
# which must not poison the operand (or the Tz state).
try:
num_val = float(operator_token)
except ValueError:
num_val = 0.0
if num_val != num_val or num_val in (float("inf"), -float("inf")):
num_val = 0.0
push("num", num_val)
continue
if operator_token in (b"true", b"false"):
push("other", None) # text extraction booleans -> operands
continue
if operator_token == b"null":
continue # content operator evaluator: `if (obj != null) args append`
if frames:
# text extraction builds arrays/dicts by recursive object parser: a command
# token inside an open [ / << becomes an ELEMENT, never an op.
frames[-1][1].append(("other", None))
continue
if operator_token == b"BI":
# text extraction object parser intercepts BI (inline-image parser): the
# image never reaches the operator table protocol; it becomes ONE arg.
bi_mark = len(opnds)
continue
spec = _OP_OPERAND_COUNTS.get(operator_token)
if spec is None:
continue # span merger: warn "Unknown command", tally PRESERVED
if operator_token == b"ID": # inline image data (text extraction inline-image parser)
# Filter-specific ender first (content stream tokenizer dispatch): DCT scans
# for the FFD9 EOI, ASCII85 for '~>', ASCIIHex for '>'; then
# the 'EI' marker whose FOLLOWING byte is SPACE/LF/CR
# (inline-image end search -- there is NO whitespace
# requirement BEFORE the marker: inline-image data can touch it).
# span merger extra 10-byte-lookahead / lookahead false-EI checks
# are not reproduced (light version).
filt = b""
if bi_mark is not None:
for kerning_delta, vertical_value in opnds[bi_mark:]:
if kerning_delta == "name" and vertical_value in (
b"DCTDecode", b"DCT", b"ASCII85Decode",
b"A85", b"ASCIIHexDecode", b"AHx"):
filt = vertical_value
break
key_value = index_value + 1
if filt in (b"DCTDecode", b"DCT"):
measure_item = content_bytes.find(b"\xff\xd9", key_value)
if measure_item >= 0:
key_value = measure_item + 2
elif filt in (b"ASCII85Decode", b"A85"):
measure_item = content_bytes.find(b"~>", key_value)
if measure_item >= 0:
key_value = measure_item + 2
elif filt in (b"ASCIIHexDecode", b"AHx"):
measure_item = content_bytes.find(b">", key_value)
if measure_item >= 0:
key_value = measure_item + 1
while key_value < count_item - 1:
if (content_bytes[key_value] == 0x45 and content_bytes[key_value + 1] == 0x49
and (key_value + 2 >= count_item or content_bytes[key_value + 2] in b" \n\r")):
index_value = key_value + 2
break
key_value += 1
else:
index_value = count_item # EOF recovery (text extraction inline-image end recovery)
if bi_mark is not None:
del opnds[bi_mark:] # the BI..ID dict guts
bi_mark = None
opnds.append(("other", None)) # the InlineImage operand
# text extraction then executes a synthetic command token EI (operand count 1):
# pre-BI dangles shift into deferred-operand, the image
# operand is consumed, args end empty.
while len(opnds) > 1:
non_processed.append(opnds.pop(0))
opnds.clear()
else:
# Stray ID without BI: text extraction dispatches it via operator table
# (operand count 0), shifting every pending arg into the
# deferred-operand stack before the no-op executes.
non_processed.extend(opnds)
opnds.clear()
continue
need, variable = spec
if not variable and len(opnds) != need:
while len(opnds) > need:
non_processed.append(opnds.pop(0))
while len(opnds) < need and non_processed:
opnds.insert(0, non_processed.pop())
if len(opnds) < need:
# Detail: "Skipping command ...: expected N args" + args
# cleared; the op has NO side effect (no flush, no Tf).
opnds.clear()
continue
if operator_token in _FLUSH_OPS:
flush_id += 1
if operator_token == b"q":
font_stack.append(cur_font)
tz_stack.append(cur_tz)
elif operator_token == b"Q":
if font_stack:
cur_font = font_stack.pop()
if tz_stack:
cur_tz = tz_stack.pop()
elif operator_token == b"Do" and opnds[0][0] == "name":
xobject_paints.append((len(flush_ids), opnds[0][1], cur_font, cur_tz)) # type: ignore[arg-type]
elif operator_token == b"Tf":
if opnds[0][0] == "name":
cur_font = opnds[0][1] # type: ignore[assignment]
else:
# text extraction REPLACES the font either way: a non-name slot
# loads undefined -> fallback/fallback font, so the previous
# font is gone. None = "no usable resname" here.
cur_font = None
elif operator_token == b"Tz":
# content stream tokenizer horizontal-scale operator: the text state's horizontal scale = args[0]/100
# (any type, ToNumber-coerced). We track only numeric operands:
# the divisor must account for what PDFIUM folded into the object
# matrix, and PDFium's own parser rejects non-numeric Tz --
# following span merger coercion here would break consistency with the horizontal font scale.
if opnds[0][0] == "num":
cur_tz = opnds[0][1] / 100.0 # type: ignore[operator]
elif operator_token in _SHOW_OPS:
if operator_token == b"TJ":
primary_item = opnds[0]
# the spaced-text show operator iterates elements by .length/.at, which a
# plain STRING also satisfies -- its chars all show.
units = primary_item[1] if primary_item[0] in ("arr", "str") else ()
elif operator_token == b'"':
primary_item = opnds[2]
units = primary_item[1] if primary_item[0] == "str" else ()
else: # Tj, '
primary_item = opnds[0]
units = primary_item[1] if primary_item[0] == "str" else ()
# A wrong-typed slot or an empty string yields ZERO glyphs in
# span merger (glyph conversion -> no item pushed) and no PDFium text
# object either -- emit no show entry, so both ordinal
# alignments (objects <-> show ops) stay tight.
if units:
flush_ids.append(flush_id)
fonts.append(cur_font)
horizontal_scales.append(cur_tz)
show_text_units.append(units) # type: ignore[arg-type]
opnds.clear() # executed op consumes its args (caller resets)
return flush_ids, fonts, show_text_units, horizontal_scales, xobject_paints
def _assign_vertical_tags(
objects: list[dict],
show_fonts: list[bytes | None] | None = None,
vertical_resnames: set[bytes] | None = None,
) -> None:
"""Tag each text object (paint order) with the vertical-font flag from its matching show-text operator. Ordinal alignment is valid when object and show-op counts agree, such as ligature-free pages and per-glyph CJK Tj streams. On a count mismatch the tag stays False and vertical runs fall back to per-glyph handling. Objects keep ``flush_id=None`` so the merger always splits per object."""
if not objects or not vertical_resnames or not show_fonts:
return
if len(show_fonts) != len(objects):
return
for item_value, font_name_value in zip(objects, show_fonts):
if font_name_value is not None and font_name_value in vertical_resnames:
item_value["vertical"] = True
def _assign_show_tz(objects: list[dict], show_tzs: list[float]) -> None:
"""Tag each text object with its show-op's text horizontal scale (Tz/100) by the same ordinal alignment as ``_assign_vertical_tags``; on a count mismatch every object keeps tz=1.0 (thresholds behave as before)."""
if not objects or not show_tzs or len(show_tzs) != len(objects):
return
for item_value, timezone_value in zip(objects, show_tzs):
item_value["tz"] = timezone_value
def _page_vertical_resource_names(pdf_doc, page_idx: int) -> set[bytes]:
"""Font resource names (``F4`` of ``/F4 14 Tf``) on this page whose encoding is a vertical CMap: a predefined ``*-V`` name (Identity-V, UniJIS-UCS2-V, ...) or an embedded CMap stream with ``/WMode 1``. This derives the vertical-font flag used by the item merger, read from the same PyPDF2 document already opened for content streams. Returns an empty set on any failure, which leaves vertical handling disabled for that page."""
names: set[bytes] = set()
try:
for rec in pdf_doc[page_idx].get_fonts(full=True):
xref, font_extension, font_type, _basefont, resname, enc = rec[:6]
# Predefined vertical CMaps: every shipped vertical bcmap ends in
# "-V" EXCEPT the bare Adobe-Japan1 "V" (bcmaps/V.bcmap, header
# bit 1 set -- content stream tokenizer reads verticality from that bit).
if isinstance(enc, str) and (enc == "V" or enc.endswith("-V")):
names.add(resname.encode("latin-1", "replace"))
continue
# Embedded CMap: /Encoding is an indirect stream; vertical iff its
# dict carries /WMode 1.
page, resource_names = pdf_doc.xref_get_key(xref, "Encoding")
if page == "xref":
width_type, width_value_local = pdf_doc.xref_get_key(int(resource_names.split()[0]), "WMode")
if width_type in ("int", "real"):
try:
wmode_number = float(width_value_local.split()[0])
except ValueError:
wmode_number = float("nan")
# Only integer-valued nonzero numbers enable the vertical
# font flag, so ``/WMode 1.0`` still counts.
if wmode_number.is_integer() and wmode_number != 0:
names.add(resname.encode("latin-1", "replace"))
except Exception:
return set()
return names
@@ -0,0 +1,394 @@
"""Simple-font encoding resolution and per-font Unicode map construction."""
from __future__ import annotations
import re
from .glyph_tables import (
_load_glyph_tables,
_get_unicode_for_glyph,
_from_char_code,
)
from .cmap_parse import (
_to_number,
_parse_int,
_parse_tounicode_cmap,
)
_TYPE1_SPECIAL_BYTES = b"/[]{}()"
# content stream tokenizer tokenises with PDF parser whitespace = {SP, TAB, CR, LF}
# ONLY -- narrower than the content-stream/CMap lexer's whitespace-byte set (no 0x0C, no 0x00).
_TYPE1_WHITESPACE_BYTES = frozenset(b" \t\r\n")
def _type1_builtin_encoding(font_file: bytes):
"""content stream tokenizer font-header extraction's /Encoding case, run over the cleartext segment of an embedded Type1 font file. Returns ("named", encoding-name) | ("array", {code: glyphname}) | None."""
end = font_file.find(b"eexec")
head = font_file[: end if end >= 0 else len(font_file)]
header_tokens: list[bytes] = []
index_value, count_item = 0, len(head)
while index_value < count_item:
candidate_item = head[index_value]
if candidate_item in _TYPE1_WHITESPACE_BYTES:
index_value += 1
elif candidate_item == 0x25: # % comment runs to EOL (PDF token reader's comment eater)
while index_value < count_item and head[index_value] not in b"\r\n":
index_value += 1
elif candidate_item in _TYPE1_SPECIAL_BYTES:
header_tokens.append(head[index_value:index_value + 1])
index_value += 1
else:
state_item = index_value
while state_item < count_item and head[state_item] not in _TYPE1_WHITESPACE_BYTES and head[state_item] not in _TYPE1_SPECIAL_BYTES:
state_item += 1
header_tokens.append(head[index_value:state_item])
index_value = state_item
def _header_token(number: int) -> bytes | None:
return header_tokens[number] if number < len(header_tokens) else None
# font-header extraction consumes "/"+name PAIRS and keeps scanning after each
# case, so a LATER /Encoding overwrites an earlier one (last wins), and a
# "//Encoding" pair is consumed whole (its bare "Encoding" never matches).
result: tuple | None = None
page_value = 0
while page_value < len(header_tokens):
if header_tokens[page_value] != b"/":
page_value += 1
continue
name_tok = _header_token(page_value + 1)
page_value += 2 # the name scanner advances past the slash unconditionally after a '/'
if name_tok != b"Encoding":
continue
arg = _header_token(page_value)
if arg is None:
# Detail: encoding lookup(null) -> null assigned to built-in encoding.
result = None
break
if not arg.isdigit():
# named encoding: encoding lookup(name) -- null when unknown
# Overwrite any previous result; later encoding declarations win.
glyph_name, encs = _load_glyph_tables()
name = arg.decode("latin-1")
result = ("named", name) if name in encs else None
page_value += 1
continue
# Decimal integer count is parsed through float64, then coerced to int32.
# Huge digit strings may round or overflow to Infinity before coercion.
array_size_float = float(arg)
size = 0 if array_size_float == float("inf") else ((int(array_size_float) + 2**31) % 2**32) - 2**31
page_value += 1 # at 'array'
enc: dict[int, str] = {}
for _ in range(size):
token_value = _header_token(page_value)
while token_value is not None and token_value not in (b"dup", b"def"):
page_value += 1
token_value = _header_token(page_value)
if token_value is None:
# Invalid headers abort the scan and keep any previous encoding.
return result
if token_value == b"def":
break
page_value += 1 # past 'dup'
# Malformed integer tokens coerce to 0 and do not abort the entry.
token_value = _header_token(page_value)
try:
value = _parse_int(token_value.decode("latin-1"), 10) if token_value is not None else 0.0
except OverflowError:
value = float("inf") # huge digit run
if value != value or value == float("inf") or value == -float("inf"):
value = 0.0 # ToInt32(NaN / ±Infinity) = 0
idx = ((int(value) + 2**31) % 2**32) - 2**31
page_value += 1
page_value += 1 # '/' slot consumed blindly
group_value = _header_token(page_value)
page_value += 1
if group_value is not None:
enc[idx] = group_value.decode("latin-1")
page_value += 1 # 'put' slot consumed blindly
result = ("array", enc) # keep scanning: a later /Encoding wins
return result
def _simple_font_to_unicode(
default_enc: list[str],
base_encoding_name: str | None,
differences: dict[int, str],
force_glyphs: bool = False,
) -> dict[int, str]:
"""content stream tokenizer simple-font Unicode-map construction, detailed behavior (including the byte-to-character conversion 16-bit truncation on glyphlist hits, the Gxx/g00xx/Cdd/cdd/u heuristics, the base encoding correction branch, and the forced glyph-name pass re-parse when a Cdd name turns out hexadecimal)."""
glyphs, encs = _load_glyph_tables()
encoding: dict[int, str] = {font: glyph_name_value for font, glyph_name_value in enumerate(default_enc)}
for font, glyph_name_value in differences.items():
if glyph_name_value == ".notdef":
continue # text extraction skips .notdef (.notdef entries)
encoding[font] = glyph_name_value
to_unicode: dict[int, str] = {}
for charcode in sorted(encoding):
glyph_name = encoding[charcode]
if glyph_name == "":
continue
codepoint = glyphs.get(glyph_name)
if codepoint is not None:
to_unicode[charcode] = _from_char_code(codepoint)
continue
code = 0
glyph_prefix = glyph_name[0]
if glyph_prefix == "G": # Gxx
if len(glyph_name) == 3:
parsed_integer = _parse_int(glyph_name[1:], 16)
code = int(parsed_integer) if parsed_integer == parsed_integer else 0 # pi==pi: not NaN
elif glyph_prefix == "g": # g00xx
if len(glyph_name) == 5:
parsed_integer = _parse_int(glyph_name[1:], 16)
code = int(parsed_integer) if parsed_integer == parsed_integer else 0
elif glyph_prefix in ("C", "c"): # Cdd{d} / cdd{d}
if 3 <= len(glyph_name) <= 4:
code_str = glyph_name[1:]
if force_glyphs:
parsed_integer = _parse_int(code_str, 16)
code = int(parsed_integer) if parsed_integer == parsed_integer else 0
else:
# First try the full numeric grammar. Only when that is NaN
# and tolerant base-16 parsing succeeds do we re-parse the
# whole encoding as base-16. Non-integer numeric values pass
# through and then fail the integer gate below.
num = _to_number(code_str)
if num != num: # NaN
parsed_integer = _parse_int(code_str, 16)
if parsed_integer == parsed_integer:
return _simple_font_to_unicode(
default_enc, base_encoding_name,
differences, force_glyphs=True)
code = 0
elif num.is_integer():
code = int(num)
else:
code = 0
elif glyph_prefix == "u":
unicode_unit = _get_unicode_for_glyph(glyph_name, glyphs)
if unicode_unit != -1:
code = unicode_unit
if 0 < code <= 0x10FFFF:
# Prefer the base encoding glyph when code == charcode
if base_encoding_name and code == charcode:
base = encs.get(base_encoding_name)
# the heading heuristics base encoding[charcode] for charcode > 255 is undefined
# (falsy) -- fall through instead of IndexError.
if base and 0 <= charcode < len(base) and base[charcode]:
to_unicode[charcode] = _from_char_code(
glyphs.get(base[charcode], 0))
continue
to_unicode[charcode] = chr(code) # code-point conversion
return to_unicode
def _font_unicode_map(pdf_doc, xref: int) -> tuple[int, dict[int, str]] | None:
"""Return the final per-charcode glyph-unicode map for one font as ``(bytes_per_code, {charcode: unicode})``. Simple fonts use 1-byte codes; Identity-H/V composite fonts use 2-byte codes with the included ToUnicode map. ``None`` means uncovered input such as non-Identity composite CMaps or unreadable dictionaries; callers then skip the page patch walk and keep PDFium's output."""
glyphs, encs = _load_glyph_tables()
def _xref_key(number: int, other_text: str) -> tuple[str, str]:
return pdf_doc.xref_get_key(number, other_text)
pdf_value_type, pdf_value = _xref_key(xref, "Subtype")
subtype = pdf_value.lstrip("/") if pdf_value_type == "name" else ""
if subtype == "Type0":
pdf_value_type, pdf_value = _xref_key(xref, "Encoding")
if pdf_value_type != "name" or pdf_value.lstrip("/") not in ("Identity-H", "Identity-V"):
return None
# text extraction reads ToUnicode from the DESCENDANT dict first, then the
# Type0 dict (the composite-font prepass uses the descendant for composites).
desc_xref = 0
delta_top, delta_value = _xref_key(xref, "DescendantFonts")
if delta_top == "xref":
delta_value = pdf_doc.xref_object(int(delta_value.split()[0]), compressed=True)
delta_top = "array"
if delta_top == "array":
delta_matrix = re.search(r"(\d+)\s+\d+\s+R", delta_value)
if delta_matrix:
desc_xref = int(delta_matrix.group(1))
pdf_value_type, pdf_value = ("null", "null")
if desc_xref:
pdf_value_type, pdf_value = _xref_key(desc_xref, "ToUnicode")
if pdf_value_type != "xref":
pdf_value_type, pdf_value = _xref_key(xref, "ToUnicode")
tu_map: dict[int, str] | None = None
if pdf_value_type == "xref":
try:
tu_map = _parse_tounicode_cmap(
pdf_doc.xref_stream(int(pdf_value.split()[0])))
except Exception:
tu_map = None # ToUnicode parsing rejects -> no ToUnicode map
# The "font carries a ToUnicode map" flag is set only for a present,
# accepted and NON-EMPTY map. A missing, rejected or empty ToUnicode all
# leave it false, so all three take the composite branch below.
if tu_map:
return 2, tu_map
# No usable ToUnicode: predefined-collection Unicode-map construction
# maps Adobe-{GB1,CNS1,Japan1,Korea1} CIDSystemInfo through the shipped
# Adobe-XX-UCS2 bcmap (real unicode per cid) -- not implemented.
# Returning identity chr(cid) would actively CORRUPT PDFium's
# table-driven decode for that class, so keep the guarded None (PDFium
# output). Every other registry/ordering IS the identity fallback.
if desc_xref:
right_type, right_value_local = _xref_key(desc_xref, "CIDSystemInfo/Registry")
other_type, other_value_local = _xref_key(desc_xref, "CIDSystemInfo/Ordering")
reg = re.sub(r"[()\s]", "", right_value_local) if right_type != "null" else ""
ordering = re.sub(r"[()\s]", "", other_value_local) if other_type != "null" else ""
if reg == "Adobe" and ordering in ("GB1", "CNS1", "Japan1", "Korea1"):
return None
return 2, {} # identity Unicode map: unicode == chr(cid)
pdf_value_type, pdf_value = _xref_key(xref, "BaseFont")
base_font = pdf_value.lstrip("/") if pdf_value_type == "name" else ""
flags = 0
fd_xref = 0
has_descriptor = False
pdf_value_type, pdf_value = _xref_key(xref, "FontDescriptor")
if pdf_value_type == "xref":
fd_xref = int(pdf_value.split()[0])
has_descriptor = True
font_token, font_value = _xref_key(fd_xref, "Flags")
if font_token == "int":
flags = int(font_value)
elif pdf_value_type == "dict":
has_descriptor = True
flags_match = re.search(r"/Flags\s+([+-]?\d+)", pdf_value)
if flags_match:
flags = int(flags_match.group(1))
if not has_descriptor and subtype != "Type3":
# font loading's simulated descriptor (span merger `if (!descriptor)`,
# non-Type3 branch): flags come from the BaseFont name with the style
# suffix stripped -- Symbol/Dingbats/ZapfDingbats get Symbolic, all
# else Nonsymbolic. (the heading heuristics also sets Serif/FixedPitch there; nothing in
# this implementation consults those bits, so they are not simulated.) A missing
# BaseFont makes the heading heuristics throw parse error -> fallback font, i.e. text extraction DROPS
# that font's text entirely; returning None keeps PDFium's decode
# instead -- the implementation's conservative boundary, not the same branch. Type3
# takes the OTHER the heading heuristics arm:
# a barebones descriptor with NO flags and NO BaseFont requirement
# (dvips bitmap fonts have neither), so flags stay 0 there.
if not base_font:
return None
base_wo_style = re.sub(r"[,_]", "-", base_font).split("-")[0]
flags = 4 if base_wo_style in ("Symbol", "Dingbats", "ZapfDingbats") else 32
file_key = None
if fd_xref:
for char_code in ("FontFile", "FontFile2", "FontFile3"):
font_token, font_value = _xref_key(fd_xref, char_code)
if font_token == "xref":
file_key = (char_code, int(font_value.split()[0]))
break
# --- encoding and Differences extraction: /Encoding -> base encodingName + differences
differences: dict[int, str] = {}
base_encoding_name: str | None = None
pdf_value_type, pdf_value = _xref_key(xref, "Encoding")
enc_obj: str | None = None
if pdf_value_type == "name":
base_encoding_name = pdf_value.lstrip("/")
elif pdf_value_type == "xref":
enc_obj = pdf_doc.xref_object(int(pdf_value.split()[0]), compressed=True)
elif pdf_value_type == "dict":
enc_obj = pdf_value
if enc_obj is not None:
flags_match = re.search(r"/BaseEncoding\s*/([^\s/\[\]<>()]+)", enc_obj)
if flags_match:
base_encoding_name = flags_match.group(1)
flags_match = re.search(r"/Differences\s*\[", enc_obj)
if flags_match:
depth = 1
scan_index = flags_match.end()
while scan_index < len(enc_obj) and depth:
if enc_obj[scan_index] == "[":
depth += 1
elif enc_obj[scan_index] == "]":
depth -= 1
scan_index += 1
idx = 0
for token_match in re.findall(r"/([^\s/\[\]<>()]+)|(\d+)", enc_obj[flags_match.end():scan_index - 1]):
if token_match[1]:
idx = int(token_match[1])
else:
name_value = re.sub(
r"#([0-9a-fA-F]{2})",
lambda encoding_key: chr(int(encoding_key.group(1), 16)), token_match[0])
differences[idx] = name_value
idx += 1
# Table 114: a named base encoding must be one of these three.
if base_encoding_name not in ("MacRomanEncoding", "MacExpertEncoding",
"WinAnsiEncoding"):
base_encoding_name = None
if base_encoding_name:
default_name = base_encoding_name
else:
symbolic = bool(flags & 4)
nonsymbolic = bool(flags & 32)
default_name = "StandardEncoding"
if subtype == "TrueType" and not nonsymbolic:
default_name = "WinAnsiEncoding"
if symbolic:
default_name = "MacRomanEncoding"
if file_key is None:
if re.search(r"Symbol", base_font, re.IGNORECASE):
default_name = "SymbolSetEncoding"
elif re.search(r"Dingbats|Wingdings", base_font, re.IGNORECASE):
default_name = "ZapfDingbatsEncoding"
default_enc = encs[default_name]
has_encoding = bool(base_encoding_name) or bool(differences)
included: dict[int, str] | None = None
pdf_value_type, pdf_value = _xref_key(xref, "ToUnicode")
if pdf_value_type == "xref":
try:
included = _parse_tounicode_cmap(pdf_doc.xref_stream(int(pdf_value.split()[0])))
except Exception:
included = None # ToUnicode parsing error path: treat as absent
# Detail: included ToUnicode-map flag = !!toUnicode and toUnicode.length > 0. An
# empty-but-valid ToUnicode (parsed to {}) is treated as ABSENT, so fall
# through to _simple_font_to_unicode + the Type1 builtin amend below
# (Type 1 Unicode-map repair), while preserving the existing item-boundary semantics.
if included:
final = dict(included)
if has_encoding: # predefined collection Unicode-map construction -> fallback Unicode map gap fill
for font, glyph_name in _simple_font_to_unicode(
default_enc, base_encoding_name, differences).items():
if font not in final:
final[font] = glyph_name
return 1, final
final = _simple_font_to_unicode(default_enc, base_encoding_name, differences)
# Type 1 Unicode-map repair: amend from the embedded Type1 program's builtin
# encoding (codes not already fixed by the dict's Encoding entry).
if file_key is not None and file_key[0] == "FontFile" and subtype in (
"Type1", "MMType1"):
try:
builtin = _type1_builtin_encoding(pdf_doc.xref_stream(file_key[1]))
except Exception:
builtin = None
if builtin is not None:
kind, payload = builtin
# `built-in encoding == properties.defaultEncoding` (same module
# array object) -- true iff both name the same predefined encoding.
if not (kind == "named" and payload == default_name):
items: list[tuple[int, str]] = (
list(enumerate(encs[payload])) if isinstance(payload, str)
else sorted(payload.items()))
for font, name_value in items:
if has_encoding and (base_encoding_name or font in differences):
continue
if not name_value:
continue
codepoint = _get_unicode_for_glyph(name_value, glyphs)
if codepoint != -1:
final[font] = _from_char_code(codepoint) # amend overwrites
return 1, final
@@ -0,0 +1,236 @@
"""Transform matrices, text-object collection, and char-to-object mapping."""
from __future__ import annotations
import ctypes
import math
import pypdfium2.raw as pdfium_c
def _obj_rotation(value: float, other_item: float, candidate_item: float, reference_item: float) -> int:
"""Classify a text-object matrix as upright, cardinal rotation, or oblique. Near-cardinal matrices snap to the cardinal bucket; genuinely oblique matrices use the baseline remerge path."""
x_scale = math.hypot(value, other_item)
y_scale = math.hypot(candidate_item, reference_item)
if x_scale < 1e-9 or y_scale < 1e-9:
return 0
eps = 1e-3
if abs(other_item) < eps * x_scale and abs(candidate_item) < eps * y_scale:
return 0 if value >= 0 else 180
if abs(value) < eps * x_scale and abs(reference_item) < eps * y_scale:
return 90 if other_item > 0 else 270
return -1
def _xf_point(items: tuple, other_item: float, candidate_item: float) -> tuple[float, float]:
"""Apply an (a,b,c,d,e,f) PDF matrix to a point (row-vector convention)."""
return (items[0] * other_item + items[2] * candidate_item + items[4], items[1] * other_item + items[3] * candidate_item + items[5])
def _compose_mtx(first_matrix: tuple, second_matrix: tuple) -> tuple:
"""Matrix product applying ``m1`` first, then ``m2``."""
return (
first_matrix[0] * second_matrix[0] + first_matrix[1] * second_matrix[2],
first_matrix[0] * second_matrix[1] + first_matrix[1] * second_matrix[3],
first_matrix[2] * second_matrix[0] + first_matrix[3] * second_matrix[2],
first_matrix[2] * second_matrix[1] + first_matrix[3] * second_matrix[3],
first_matrix[4] * second_matrix[0] + first_matrix[5] * second_matrix[2] + second_matrix[4],
first_matrix[4] * second_matrix[1] + first_matrix[5] * second_matrix[3] + second_matrix[5],
)
_IDENT_MTX = (1.0, 0.0, 0.0, 1.0, 0.0, 0.0)
def _collect_text_objs(page, text_page) -> list[dict]:
"""Per-page list of (font_handle, fs_raw, matrix_scale_*, bbox, ...) for each text object. Used for bbox-containment lookup. Walks Form XObjects manually in stream order, composing each ancestor form's matrix. Without the composition a scaled or shifted chart's text objects land at the wrong page position and every chart glyph fails the bbox-containment lookup."""
objects: list[dict] = []
sz_field = ctypes.c_float(0)
matrix = pdfium_c.FS_MATRIX()
font_name_buffer = (ctypes.c_char * 256)()
bounds_left = ctypes.c_float(0)
value = ctypes.c_float(0)
bounds_right = ctypes.c_float(0)
bounds_top = ctypes.c_float(0)
def iter_text_objs(parent, anc_mtx, depth):
"""Yield (raw_text_obj, ancestor_matrix) in stream order."""
object_count = (pdfium_c.FPDFFormObj_CountObjects(parent) if parent is not None
else pdfium_c.FPDFPage_CountObjects(page.raw))
for text in range(object_count):
raw = (pdfium_c.FPDFFormObj_GetObject(parent, text) if parent is not None
else pdfium_c.FPDFPage_GetObject(page.raw, text))
if not raw:
continue
typ = pdfium_c.FPDFPageObj_GetType(raw)
if typ == pdfium_c.FPDF_PAGEOBJ_TEXT:
yield raw, anc_mtx
elif typ == pdfium_c.FPDF_PAGEOBJ_FORM and depth < 10:
pdfium_c.FPDFPageObj_GetMatrix(raw, matrix)
font_matrix = (matrix.a, matrix.b, matrix.c, matrix.d, matrix.e, matrix.f)
yield from iter_text_objs(raw, _compose_mtx(font_matrix, anc_mtx), depth + 1)
for raw_obj, anc_mtx in iter_text_objs(None, _IDENT_MTX, 0):
font = pdfium_c.FPDFTextObj_GetFont(raw_obj)
if not font:
continue
pdfium_c.FPDFTextObj_GetFontSize(raw_obj, ctypes.byref(sz_field))
fs_raw = sz_field.value
pdfium_c.FPDFPageObj_GetMatrix(raw_obj, matrix)
# Effective (page-space) matrix: the object's own matrix composed with
# its ancestor forms' -- text extraction folds that ancestor chain into the text matrix.
matrix_a, matrix_b, matrix_c, matrix_d, _, _ = _compose_mtx(
(matrix.a, matrix.b, matrix.c, matrix.d, matrix.e, matrix.f), anc_mtx)
scale_x = math.sqrt(matrix_a * matrix_a + matrix_b * matrix_b) or 1.0
scale_y = math.sqrt(matrix_c * matrix_c + matrix_d * matrix_d) or 1.0
if not pdfium_c.FPDFPageObj_GetBounds(
raw_obj, ctypes.byref(bounds_left), ctypes.byref(value),
ctypes.byref(bounds_right), ctypes.byref(bounds_top)):
continue
# Bounds include the object's own matrix but not its ancestors'; map
# the four corners into page space.
x00, y00 = _xf_point(anc_mtx, bounds_left.value, value.value)
x01, y01 = _xf_point(anc_mtx, bounds_left.value, bounds_top.value)
x10, y10 = _xf_point(anc_mtx, bounds_right.value, value.value)
x11, y11 = _xf_point(anc_mtx, bounds_right.value, bounds_top.value)
object_left = min(x00, x01, x10, x11)
object_right = max(x00, x01, x10, x11)
text = min(y00, y01, y10, y11)
object_top = max(y00, y01, y10, y11)
ink_height = max(0.0, object_top - text)
# text extraction folds Tfs (text font size) + FontMatrix into the text transform
# so ``hypot(transform[2], transform[3])`` always gives the
# rendered font size. PDFium splits these and doesn't fold non-identity
# FontMatrix back. Rendered-font-size fallback chain:
# raw >= 1.5 and scale > 0 -> raw * scale (normal text)
# scale >= 1.5 -> scale (Type 3: raw=0.1, ctm=N)
# raw >= 1.5 -> raw (no scale info)
# else -> ink_h (Type 3 inside identity ctm)
if anc_mtx is not _IDENT_MTX and fs_raw > 0 and scale_y > 0:
# Inside a Form XObject, span merger font size = hypot(trm[2],trm[3])
# with the form CTM folded in = Tfs * composed scale, exactly
# (scaled vector-figure case: Tf 0.167 * 72 * form 0.5722 =
# 6.88 == the heading heuristics' item height; the placeholder chain below
# would misread it as Type-3-with-fs-in-ctm and emit 41pt boxes
# that swallow the neighbouring "2.2" heading). The chain stays
# for top-level objects, for top-level objects.
fs_eff = fs_raw * scale_y
elif fs_raw >= 1.5 and scale_y > 0:
fs_eff = fs_raw * scale_y
elif scale_y >= 1.5:
fs_eff = scale_y
elif fs_raw >= 1.5:
fs_eff = fs_raw
else:
fs_eff = max(1.0, ink_height)
# PDFium's FS_MATRIX is float32, so a size authored as 9.9pt arrives as
# 9.89999962; text extraction parses the content stream in float64 and keeps 9.9.
# Snap back to the shortest decimal so knife-edge font-size comparisons
# match the content-stream value.
fs_eff = float(f"{fs_eff:.6g}")
name = pdfium_c.FPDFFont_GetFontName(font, font_name_buffer, 256)
font_name = (
bytes(font_name_buffer[:name]).decode("latin-1", errors="replace").rstrip("\x00")
if name > 1 else ""
)
weight = int(pdfium_c.FPDFFont_GetWeight(font))
objects.append({
"font": font,
# Handle address as a hashable per-document font identity; computed
# once here so per-char consumers never re-cast.
"font_key": ctypes.cast(font, ctypes.c_void_p).value,
"fs_raw": fs_raw,
"scale_x": scale_x,
"scale_y": scale_y,
"fs_eff": fs_eff,
"l": object_left, "r": object_right, "b": text, "t": object_top,
"area": max(0.0, (object_right - object_left) * (object_top - text)),
"font_name": font_name,
"weight": weight,
# Rotation class of this text object (0/90/180/270, or -1 oblique).
# text extraction normalises it inside position comparison; the charlevel
# merger is horizontal-only, so cardinal runs (rotated-sidebar sidebar stamp,
# chart axis labels) shatter per-glyph and are re-merged by
# _remerge_rotated; oblique objects go to _remerge_oblique (needs the
# matrix below for the inverse-rotation projection baseline projection).
"rot": _obj_rotation(matrix_a, matrix_b, matrix_c, matrix_d),
"mtx": (matrix_a, matrix_b, matrix_c, matrix_d),
# Paint (content-stream) order. text extraction emits items in stream order but
# PDFium's textpage reorders vertical-writing chars page-wide, so
# _remerge_vertical needs this to restore text extraction item order.
"page_order": len(objects),
# True iff this object's show-op used a vertical-CMap (-V / WMode 1)
# font -- span merger vertical-font flag. Set by _assign_vertical_tags.
"vertical": False,
# Show-op text horizontal scale (Tz/100). text extraction keeps Tz OUT of the space
# thresholds (base = raw font size) while PDFium folds it into the
# object matrix (hence into fs_x); open_chunk divides it back out.
# Set by _assign_show_tz via the same ordinal alignment as
# ``vertical``; stays 1.0 on a count mismatch.
"tz": 1.0,
})
return objects
def _build_obj_index(objects: list[dict]) -> dict[int, list[dict]]:
"""Bucket text objects by integer y so per-character lookup scans only nearby baselines. Each object is inserted into padded y-buckets that form a superset for the exact containment check."""
index: dict[int, list[dict]] = {}
for item_value in objects:
lower_bound = int(math.floor(item_value["b"])) - 6
upper_bound = int(math.ceil(item_value["t"])) + 6
for text_key in range(lower_bound, upper_bound + 1):
index.setdefault(text_key, []).append(item_value)
return index
def _char_render_fs(text_page, char_idx: int) -> float:
"""True per-char rendered size: ``FPDFText_GetMatrix`` folds Tfs and FontMatrix into the rendered text matrix, so ``sqrt(c^2+d^2)`` is the text-item height. Returns 0.0 when the call is unavailable. Read lazily, only when a char is contained by more than one object, since the FFI call is expensive and most chars have a single, unambiguous host object."""
current_matrix = pdfium_c.FS_MATRIX()
if pdfium_c.FPDFText_GetMatrix(text_page, char_idx, ctypes.byref(current_matrix)):
return math.sqrt(current_matrix.c * current_matrix.c + current_matrix.d * current_matrix.d)
return 0.0
def _find_obj_for_char(
obj_index: dict[int, list[dict]], query_origin_x: float, query_origin_y: float, tol: float = 1.0,
char_fs: float | None = None, text_page=None, char_idx: int | None = None,
) -> dict | None:
"""Bbox containment lookup. When a char falls inside more than one text object, pick the candidate whose effective rendered size matches the char's true per-char matrix size from ``FPDFText_GetMatrix``. That folds Tfs and FontMatrix into the same glyph-to-font attribution used by the text-item reconstruction. This disambiguates overlapping objects such as a large figure-axis label drawn over a smaller heading, and avoids selecting tiny ghost objects that share the same raw textpage font size. Falls back to the PDFium ``fs_raw`` textpage font size and finally to smallest area."""
first: dict | None = None
cands: list[dict] | None = None
for item_value in obj_index.get(int(round(query_origin_y)), ()):
if (item_value["l"] - tol) <= query_origin_x <= (item_value["r"] + tol) and\
(item_value["b"] - tol) <= query_origin_y <= (item_value["t"] + tol):
if first is None:
first = item_value
elif cands is None:
cands = [first, item_value]
else:
cands.append(item_value)
if first is None:
return None
if cands is None:
return first
char_render = (
_char_render_fs(text_page, char_idx) if text_page is not None and char_idx is not None
else 0.0
)
if char_render > 0:
# Match the per-char rendered size (== text extraction font size); area tiebreak.
return min(
cands,
key=lambda item_value: (abs(item_value["fs_eff"] - char_render), item_value["area"]),
)
if char_fs is None and text_page is not None and char_idx is not None:
# Deferred FPDFText_GetFontSize: only this rare branch (multi-candidate
# AND no per-char matrix) consumes it, so the caller no longer pays the
# FFI call on every char.
char_fs = pdfium_c.FPDFText_GetFontSize(text_page, char_idx)
if char_fs is not None and char_fs > 0:
# Sort by absolute fs diff first, then smallest area as tiebreak.
return min(
cands,
key=lambda item_value: (abs(item_value["fs_raw"] - char_fs) / max(char_fs, 0.01), item_value["area"]),
)
return min(cands, key=lambda item_value: item_value["area"])
@@ -0,0 +1,82 @@
"""Bundled glyph-name and encoding tables with cached lazy loading."""
from __future__ import annotations
import json
from pathlib import Path
from .cmap_parse import _parse_int
# ---------------------------------------------------------------------------
# Font Unicode-map construction.
#
# PDFium's per-character Unicode can diverge when a simple font's ToUnicode CMap
# is missing or incomplete. The repair path resolves the charcode through the
# font's encoding (dictionary /Encoding BaseEncoding+Differences, or an embedded
# Type1 program's builtin encoding) to a glyph name, maps that name through the
# bundled glyph table, and otherwise falls back to the raw charcode. The map is
# rebuilt from the PDF's own font dictionaries via the PyPDF2 xref channel
# (font metadata only, no text decode), then applied where PDFium's output
# disagrees.
#
# Covered rules: encoding and Differences extraction, simple-font Unicode-map
# construction, predefined collection Unicode-map construction, ToUnicode
# parsing, fallback Unicode-map repair, Type 1 Unicode-map repair, and glyph
# mapping as the included ToUnicode value when present, otherwise the raw charcode.
# /Encoding extraction from an embedded Type1 file
# Glyph-name Unicode lookup
# glyph names and standard encodings are bundled in data/glyph_name_table.json
# (kept deliberately conservative
#
# Boundaries (documented, all conservative -- no map entry means no patch):
# - composite (Type0) fonts: separate path, never patched here;
# - CFF (FontFile3) builtin encodings: not parsed here; dict-encoding-based
# mapping still applies;
# - symbolic-TrueType WinAnsi inference (content stream tokenizer TrueType Unicode-map repair):
# needs the TTF name records, not implemented.
# ---------------------------------------------------------------------------
_GLYPHLIST_PATH = Path(__file__).parent.parent / "data" / "glyph_name_table.json"
_cached_glyphs: dict[str, int] | None = None
_cached_encodings: dict[str, list[str]] | None = None
def _load_glyph_tables() -> tuple[dict[str, int], dict[str, list[str]]]:
global _cached_glyphs, _cached_encodings
glyphs, encodings = _cached_glyphs, _cached_encodings
if glyphs is None or encodings is None:
data = json.loads(_GLYPHLIST_PATH.read_text(encoding="utf-8"))
glyphs = _cached_glyphs = data["glyphs"]
encodings = _cached_encodings = data["encodings"]
return glyphs, encodings
def _get_unicode_for_glyph(name: str, glyphs: dict[str, int]) -> int:
"""Resolve a glyph name through glyphlist lookup and uppercase-hex recovery patterns."""
codepoint = glyphs.get(name)
if codepoint is not None:
return codepoint
if not name:
return -1
if name[0] == "u":
glyph_name_length = len(name)
if glyph_name_length == 7 and name[1] == "n" and name[2] == "i":
hex_str = name[3:]
elif 5 <= glyph_name_length <= 7:
hex_str = name[1:]
else:
return -1
if hex_str == hex_str.upper():
# Tolerant base-16 parsing trims Unicode whitespace and accepts an
# optional sign / 0X prefix. NaN fails the >= 0 gate; "-0" passes it.
u16 = _parse_int(hex_str, 16)
if u16 >= 0:
return int(u16)
return -1
def _from_char_code(number: int) -> str:
"""Return the UTF-16 code unit after ToUint16 truncation."""
return chr(number & 0xFFFF)
@@ -0,0 +1,526 @@
"""Joins page glyphs into text runs with spacing and style thresholds."""
from __future__ import annotations
from .text_normalize import (
TRACKING_SPACE_FACTOR,
NON_SPACE_GAP_FACTOR,
NEGATIVE_SPACE_FACTOR,
SPACE_IN_FLOW_MIN_FACTOR,
SPACE_IN_FLOW_MAX_FACTOR,
_rtl_sign,
_read_end,
_read_gap,
)
from .char_extract import _off_page
def _merge_text_items(chars: list[dict], view_box=None) -> list[dict]:
"""exact text extraction position comparison + synthetic-space insertion + last-character buffer."""
items: list[dict] = []
chunk: dict | None = None
two_last = [" ", " "]
two_last_pos = [0]
# text extraction active text item.previous glyph transform: set only by a glyph with a real
# advance (`if (scaled advance)`), NEVER reset by text-item flush/setFont
# -- it survives across item flushes for the whole page. (None, None)
# until the first real glyph.
last_ref: tuple = (None, None)
def _object_merge_id(mapping: dict):
# Per-object merge id: the boundary test below hard-splits between
# different objects (the per-object split). q/Q grouping would require
# fragile object/show-op ordinal alignment, so this is just the object's
# identity.
return id(mapping["obj"])
def reset_last_chars() -> None:
two_last[0] = " "
two_last[1] = " "
two_last_pos[0] = 0
def save_last_char(char: str) -> bool:
next_pos = (two_last_pos[0] + 1) % 2
ret = (two_last[two_last_pos[0]] != " " and two_last[next_pos] == " ")
two_last[two_last_pos[0]] = char
two_last_pos[0] = next_pos
return ret
def flush() -> None:
nonlocal chunk
if chunk is not None and chunk["str"]:
items.append(chunk)
chunk = None
def open_chunk(mapping: dict) -> None:
nonlocal chunk, last_ref
sign = _rtl_sign(mapping["ch"])
# |text horizontal scale|: fs_x = |matrix scale| carries |Tz|, so the divisor is
# the magnitude (a negative Tz uses the same text but scales it by |Tz|).
# Tz == 0 keeps 1.0 (moot: PDFium emits no textpage chars for a
# degenerate x-column). Boundary conditions:
# PDFium's SYNTHESIZED layout spaces derive from the unscaled
# text-space gap (~0.135em), so compressed Tz < ~75 can inject
# spaces text extraction would not; negative-Tz runs re-merge via the
# 180-degree pass with different item structure than span merger'
# text orientation=-1 model; anisotropic CTM x rotated Tm differs
# (norm-of-product vs span merger product-of-norms text advance scale).
horizontal_scale_factor = abs(mapping["obj"].get("tz", 1.0))
if not (horizontal_scale_factor > 0):
horizontal_scale_factor = 1.0
fs_x_tz = mapping["fs_x"] / horizontal_scale_factor
chunk = {
"str": [],
"sign": sign, # +1 LTR, -1 RTL (signed x-axis)
"obj": mapping["obj"], # host text object (Tj/show-text)
# text extraction fixes item transform at the item's FIRST glyph
# (item initialization) and never updates it mid-item, while
# chunk["obj"] re-points to the LAST appended glyph's object (the
# flush/prose bookkeeping needs that). Snapshot the opening
# object's matrix so the emitted skew reads first-glyph geometry.
"mtx0": mapping["obj"]["mtx"],
"flush_id": _object_merge_id(mapping), # per-object merge id (was q/Q flush scope)
"left": mapping["left"], "right": mapping["right"],
"top": mapping["top"], "bottom": mapping["bottom"],
"fs": mapping["fs"],
"fs_min": mapping["fs"],
# Glyph advance (FPDFFont_GetGlyphWidth*scale). Unused by the
# horizontal merger (it reads prev_text_x); carried only so
# _remerge_rotated can run a direct 1-D position comparison
# (gap = next_origin - (cur_origin + glyph_w)) along the rotation axis.
"glyph_w": mapping.get("glyph_w", 0.0),
"font_name": mapping["font_name"],
"font_key": mapping["font_key"],
"weight": mapping["weight"],
# Per-char style tallies for majority-vote at span emission.
# span merger text item records only the first char's font name;
# the heading heuristics' heading detection ends up marking paragraph
# lead-ins like **Bold prefix.** Regular continuation as
# "bold lines" because of that. Tally per-char so we can
# emit the dominant font/weight instead.
"font_tally": {mapping["font_name"]: 1},
"weight_tally": {mapping["weight"]: 1},
# ``prev_text_x`` tracks where the next glyph would land if
# charSpacing=0 — i.e. text matrix.e after this glyph's emit.
# For ligature components, PDFium reports them at the same
# origin but with bbox spanning the full ligature, so taking
# max(ox+glyph_w, bbox.right) makes the next non-ligature
# char see a small positive advance instead of a big gap.
# (_read_end uses the same this for an RTL chunk.)
"prev_text_x": _read_end(mapping, sign),
"prev_oy": mapping["oy"],
# text extraction threshold base is text state.font size WITHOUT Tz
# (item initialization: Tz enters only the pen advance, not
# text advance scale). PDFium folds Tz into the object matrix, so
# fs_x carries it; divide the show-op's text horizontal scale back out.
"tracking": fs_x_tz * TRACKING_SPACE_FACTOR,
"not_a_space": fs_x_tz * NON_SPACE_GAP_FACTOR,
"negative": fs_x_tz * NEGATIVE_SPACE_FACTOR,
"flow_min": fs_x_tz * SPACE_IN_FLOW_MIN_FACTOR,
"flow_max": fs_x_tz * SPACE_IN_FLOW_MAX_FACTOR,
"height": mapping["fs"],
# True once a real whitespace glyph follows the last visible glyph in
# this chunk; gates whether an object boundary is a prose word-break
# (merge) or a layout jump (hard split). See the is_ws handler.
"ws_pending": False,
}
if "v_pen_y" in mapping:
# Vertical-writing pen state for _remerge_vertical (set only for
# vertical-CMap objects).
chunk["v_pen_x"] = mapping["v_pen_x"]
chunk["v_pen_y"] = mapping["v_pen_y"]
chunk["v_after"] = mapping["v_after"]
chunk["v_last_x"] = mapping["v_pen_x"]
if mapping["is_mn"]:
# A zero-width diacritic has scaled advance == 0, so it does NOT
# establish the advance reference; the chunk INHERITS the
# page-surviving one (text extraction previous glyph transform persists across
# flushes; (None, None) only until the page's first real glyph).
chunk["prev_text_x"], chunk["prev_oy"] = last_ref
else:
last_ref = (chunk["prev_text_x"], chunk["prev_oy"])
def emit_fake_space(gap: float) -> None:
"""Emit an out-of-flow synthetic space after the current chunk. The synthetic item uses the previous glyph transform, not the next glyph, so the space remains attached to the line it trails. Its height stays zero; otherwise vertical-alignment checks can attach the space to a neighboring line and create a spurious line merge. """
assert chunk is not None
reset_last_chars() # text extraction synthetic-space insertion standalone path resets first
page_x, baseline = chunk["prev_text_x"], chunk["prev_oy"]
# WIDTH = abs(gap). Synthetic out-of-flow spaces use ``width: abs(e)``,
# where e is the out-of-flow advance (the gap it
# spans), height 0 for horizontal text. The box therefore runs from the
# previous glyph's pen end (px) forward by the gap: [px, px+gap] LTR,
# [px-gap, px] RTL. In-flow spaces are handled by pushing " " into the
# current item; standalone spaces use this separate geometry. Height
# stays 0, so vertical-alignment guards are
# untouched and the outline is unaffected.
width_value = abs(gap)
if chunk["sign"] >= 0:
sp_left, sp_right = page_x, page_x + width_value
else:
sp_left, sp_right = page_x - width_value, page_x
meta = (chunk["obj"], chunk["fs"], chunk["font_name"],
chunk["font_key"], chunk["weight"])
flush()
items.append({
"str": [" "], "sign": 1, "obj": meta[0],
"left": sp_left, "right": sp_right,
"top": baseline, "bottom": baseline, # HEIGHT 0
"fs": meta[1], "fs_min": meta[1],
"font_name": meta[2], "font_key": meta[3], "weight": meta[4],
"font_tally": {meta[2]: 1}, "weight_tally": {meta[4]: 1},
})
def extend_chunk(mapping: dict, leading_space: bool) -> None:
nonlocal last_ref
assert chunk is not None
if leading_space:
chunk["str"].append(" ")
chunk["str"].append(mapping["ch"])
chunk["left"] = min(chunk["left"], mapping["left"])
chunk["right"] = max(chunk["right"], mapping["right"])
# Text-item box accumulation: appending a glyph only grows the item's
# width. The item's vertical box is
# fixed at item creation -- transform[5] = first-glyph baseline, and
# height = font size (== the item's em). A per-glyph baseline offset
# within the item (e.g. a lowered character inside a mixed-baseline
# logo, or any sub/superscript not split into its own item) is therefore
# absorbed: it does NOT extend the item box. Do not expand top/bottom
# here; they stay at the open glyph's [oy, oy+fs]. Expanding them would
# let inline baseline offsets distort downstream line-height gates.
chunk["prev_text_x"] = _read_end(mapping, chunk["sign"])
chunk["prev_oy"] = mapping["oy"]
last_ref = (chunk["prev_text_x"], chunk["prev_oy"])
chunk["glyph_w"] = mapping.get("glyph_w", 0.0) # last glyph's advance (for _remerge_rotated)
# When a real-space word-break us merge across a text-object boundary
# (prose case), the chunk must adopt the new object so the rest of that
# word's glyphs (same object, no space before them) don't re-trigger the
# object hard-split mid-word. span merger line item has no per-glyph object.
chunk["obj"] = mapping["obj"]
chunk["flush_id"] = _object_merge_id(mapping)
chunk["font_tally"][mapping["font_name"]] = chunk["font_tally"].get(mapping["font_name"], 0) + 1
chunk["weight_tally"][mapping["weight"]] = chunk["weight_tally"].get(mapping["weight"], 0) + 1
# Track min fs within the chunk so small-caps headings ("A"+
# "BSTRACT") expose the body-text fs of the small-cap part
# rather than the leading full-cap fs. The downstream big-font
# check then doesn't false-positive on inline math labels like
# "LEMMA 1" whose small-cap fs is below body size.
if mapping["fs"] > 0:
chunk["fs_min"] = min(chunk["fs_min"], mapping["fs"])
if "v_pen_y" in mapping and "v_pen_y" in chunk:
chunk["v_after"] = mapping["v_after"]
chunk["v_last_x"] = mapping["v_pen_x"]
for text in chars:
# text extraction text-item box accumulation char loop order (span merger+):
# invisible format-mark classification is skipped entirely BEFORE the whitespace test.
if text["is_cf"]:
# The format-mark skip sits ahead of the scaled-advance and
# char-spacing block, so the mark moves neither the text matrix nor
# the previous-position reference: the reference pen never sees it.
# PDFium's char origins DO include its advance, so carry the
# reading-direction reference past it; otherwise that advance
# reappears as a gap and the next glyph gets an in-flow or
# standalone " " with no counterpart. (Char spacing, also skipped
# here, is not separable from PDFium's origins.) With no chunk open
# the page's previous-position reference is still unset, so the
# position comparison is unconditionally true and there is no gap
# to correct.
if chunk is not None and chunk["prev_text_x"] is not None:
chunk["prev_text_x"] += chunk["sign"] * text.get("glyph_w", 0.0)
last_ref = (chunk["prev_text_x"], chunk["prev_oy"])
continue
if text["is_ws"]:
save_last_char(" ")
# Remember a real whitespace glyph bridged the gap. PDFium fragments a
# flowing prose line into per-word text-objects (each with a trailing
# space glyph); text extraction keeps the whole line as one Tj item. A real space
# at an object boundary marks a prose word-break -> merge across it.
# A positional (spaceless) object change marks a layout jump (table
# cell, separate Tj) -> keep the hard object split.
if chunk is not None:
chunk["ws_pending"] = True
continue
# Zero-width diacritics append without a position-based flush: they do NOT call
# position comparison (no position-based flush) and uses
# scaled advance=0 (no advance) -- it just appends the mark to the
# current item. Append it without touching prev_text_x.
# BUT a Tf style flush is an operator-level split that already closed
# the previous item before the glyph loop ran, so a mark arriving in a
# different font/size (for example, a math accent over an italic letter,
# letter, each its own Tf'd show op) opens its OWN item, with its own
# raised origin and em height. Only
# the position-based flush is skipped for diacritics, never the style
# flush, so the mark passes the same font_key/fs/object boundary test
# as any visible glyph.
if text["is_mn"]:
if chunk is not None and (
chunk["font_key"] != text["font_key"]
or abs(text["fs"] - chunk["fs"]) > 1e-6
or (_object_merge_id(text) != chunk["flush_id"] and not chunk["ws_pending"])
):
flush()
if chunk is None:
# open_chunk inherits the page-surviving advance reference
# (text extraction previous glyph transform persists across flushes; None only at
# page start -- see the prev_text_x-is-None guard below).
open_chunk(text)
assert chunk is not None
lead = save_last_char(text["ch"])
# the heading heuristics pushes the last-character buffer lead into the fresh mark item
# (a Tf flush does not reset the last-character buffer).
if lead:
chunk["str"].append(" ")
chunk["str"].append(text["ch"])
else:
lead = save_last_char(text["ch"])
if lead:
chunk["str"].append(" ")
chunk["str"].append(text["ch"])
# span merger: a zero-width diacritic has scaled advance=0, so it neither
# moves the text matrix NOR updates previous glyph transform (span merger
# `if (scaled advance)` is false). The next glyph's line-break /
# dy test therefore compares against the last VISIBLE glyph's
# baseline -> leave BOTH prev_text_x and prev_oy untouched here
# (the box also stays the open glyph's -- see extend_chunk note).
continue
# span merger: a non-diacritic glyph whose origin is off the page view box is
# skipped (position comparison returns false only off-page). cf/ws
# were handled above and diacritics (is_mn) never reach here, matching
# span merger `!zero-width diacritic classification and !position comparison`.
if _off_page(text, view_box):
continue
if chunk is None:
open_chunk(text)
save_last_char(text["ch"])
assert chunk is not None
chunk["str"].append(text["ch"])
continue
# Style boundary: split on font-identity change (font_key, the PDFium
# font handle == span merger per-font loaded font identity) OR ANY font size change.
# span merger emit a separate text item on every setFont
# (Tf) operator -- i.e. on any font OR size change. Represent
# that with an exact effective-fs compare (the 1e-6 is only to absorb
# float noise in the snapped fs). An earlier 10% tolerance under-split
# small-caps runs; exact is intentional here.
# font_key/fs is a proxy for the Tf flush, not a literal replay of every
# content-stream flush boundary. Text-item boundaries are driven mostly
# by position comparison; PDFium exposes final glyph coordinates, so the
# per-object + font_key/fs proxy gives the heading pipeline the intended
# span structure without overfitting to partial operator state. The
# remaining boundary cases, such as missing-glyph fallback handles or
# rendered-size jitter under scaled Type-3 matrices, are limited to span
# boundaries.
# The heading heuristics' downstream heading detector then
# treats a chunk's first-char style as the whole chunk's style:
#
# * font split: a paragraph lead-in like "**Inflation.**
# Consumer price..." would otherwise be a single bold chunk
# and false-detect as a heading on every paragraph. (font_name
# alone can't separate identity-matrix Type-3 fonts, whose names
# are all empty, so a 12pt body run and an inline 11pt code word
# would merge and collapse to the smaller fs_min.)
# * fs split: inline math labels like "LEMMA 1 (...) ..."
# (first-cap large + small-cap rest + body) would otherwise
# merge into a single chunk that pipeline accepts as a
# heading; splitting forces the small-cap rest into its own
# chunk where the heading heuristics' short-text/type checks reject it. Same
# guard also helps math-heavy page/identity-matrix Type-3 sample exercise items
# ("X.Y www") and section headings stay detectable —
# without it they collapse into the surrounding body chunk.
#
# Trade-off: small-caps "ABSTRACT" / "ECONOMIC ANALYSIS"
# don't merge across the cap-to-small-cap fs step. The
# downstream tokenizer relaxation (LineTokenizer.add_line below)
# joins them at token level instead.
ws_bridge = chunk["ws_pending"]
chunk["ws_pending"] = False
if chunk["prev_text_x"] is None:
# The item was opened by a zero-width diacritic AT PAGE START (no
# real glyph has set the page's advance reference yet, so span merger'
# previous glyph transform is still null): position comparison returns
# true unconditionally -- no positional boundary, no fake space,
# no line break. Only the style flush (the Tf proxy) still
# applies; otherwise the glyph appends plainly and, being a real
# advance, establishes the reference via extend_chunk.
if (chunk["font_key"] != text["font_key"]
or abs(text["fs"] - chunk["fs"]) > 1e-6):
flush()
open_chunk(text)
lead = save_last_char(text["ch"])
assert chunk is not None
if lead: # Detail: no reset on this path, the lead survives
chunk["str"].append(" ")
chunk["str"].append(text["ch"])
else:
extend_chunk(text, save_last_char(text["ch"]))
continue
if (
chunk["font_key"] != text["font_key"]
or abs(text["fs"] - chunk["fs"]) > 1e-6
# Hard-split at a text-object boundary -- but ONLY when no real
# whitespace glyph bridged it. the object merge id groups consecutive per-glyph show operators
# objects (PDFium emits one FPDF_PAGEOBJ_TEXT per glyph when the PDF
# draws glyphs individually) into one id, so a CJK title set as N
# per-glyph Tj does NOT shatter into N single-glyph items -- the
# positional logic below merges it / line-breaks it like the item merger.
# Every normal (multi-glyph) object keeps its own id, so this stays
# the per-object split for Latin text: on dense justified tables (2023
# dense table document) each fragment is its own object -> hard split, exact
# with the item merger. On flowing prose PDFium may split
# per word with a real space glyph between words, where text extraction keeps
# the whole line as one item; a real space at the boundary (ws_bridge)
# marks the prose case -> fall through to the in-flow/out-of-flow gap
# logic, which merges the word-objects into one line item like span merger.
or (_object_merge_id(text) != chunk["flush_id"] and not ws_bridge)
):
# span merger position comparison runs synthetic-space insertion for EVERY glyph,
# including the first glyph of a new item/Tj. So an out-of-flow gap
# across an item boundary still gets a standalone height-0 " "
# (this is the trailing space after math-heavy page's "...y)" before the next
# equation-number object). An in-flow / adjacent boundary does not.
# text extraction position comparison order: a line break (|advance-y| >
# height -> line-break emission) or backward jump takes precedence over the
# space logic; only a same-line gap past tracking-space threshold emits the
# standalone " " (covering BOTH the in-flow empty-item case and
# the out-of-flow synthetic-space insertion case, which are identical here).
boundary_gap = _read_gap(chunk["prev_text_x"], text, chunk["sign"])
_same_line = abs(text["oy"] - chunk["prev_oy"]) <= chunk["height"]
if _same_line and boundary_gap > chunk["tracking"]:
emit_fake_space(boundary_gap)
keep_lead = False # the heading heuristics synthetic-space insertion reset the last-character buffer
else:
# the heading heuristics resets the two-char buffer on every positional branch
# (line-break emission / negative / non-space) but NOT in the tracking
# window (non-space, tracking-space threshold] -- a thin real space
# just before a Tf-style flush survives into the new item.
keep_lead = (_same_line
and chunk["not_a_space"] < boundary_gap <= chunk["tracking"])
flush()
open_chunk(text)
lead = save_last_char(text["ch"])
assert chunk is not None
if lead and keep_lead:
chunk["str"].append(" ")
chunk["str"].append(text["ch"])
continue
advance = _read_gap(chunk["prev_text_x"], text, chunk["sign"])
line_delta_y = text["oy"] - chunk["prev_oy"]
height = chunk["height"]
# Ligature decomposition: PDFium reports consecutive ligature
# components at the same x origin (e.g. "fi" -> 'f' and 'i' at the
# identical origin). prev_text_x was set to prev.ox +
# prev.glyph_w, so we see advance ≈ -prev.glyph_w. Glyph widths
# of typical Latin chars are in [0.2*fs, 0.9*fs]. Detect this
# case (negative advance whose magnitude is in that range) and
# silently merge — matches span merger on same-origin
# ligature components. ONLY within one text object: decomposition
# is per-glyph, so both components always share the show op. A
# cross-object negative advance is a real content-stream back-jump
# that text extraction itself sees and breaks on (TeX standalone accents:
# math-heavy page 'accented name stem'+'¨'+'lkopf' is three show ops, '¨' jumps back -0.41fs;
# text extraction raw-categorizes U+00A8 as a normal glyph -- category comes
# from glyph Unicode BEFORE the normalized Unicode expansion -- so
# position comparison flushes and ' ̈lkopf' opens a new item).
if (text["obj"] is chunk["obj"] and abs(line_delta_y) < 0.1 * height
and -0.9 * chunk["fs"] <= advance < -0.2 * chunk["fs"]):
lead = save_last_char(text["ch"])
# Don't extend prev_text_x backwards; ligature component
# shares position with prev, so prev_text_x stays the same.
if lead:
chunk["str"].append(" ")
chunk["str"].append(text["ch"])
chunk["left"] = min(chunk["left"], text["left"])
chunk["right"] = max(chunk["right"], text["right"])
# Ligature component shares the open glyph's item box; only width
# grows (see extend_chunk note -- text extraction never expands the item's
# vertical extent on append).
chunk["prev_oy"] = text["oy"]
last_ref = (chunk["prev_text_x"], chunk["prev_oy"])
continue
# text extraction position comparison compares advance-x against
# ``text orientation * threshold`` (text orientation = sign(item.width)).
# We get the same result for HORIZONTAL text by normalising the gap into
# READING DIRECTION up front: ``advance`` (= _read_gap with chunk["sign"]
# from _rtl_sign) is already signed so that "forward" is positive for BOTH
# LTR and RTL, hence the thresholds below are compared UNMULTIPLIED.
# * LTR (sign=+1): intentional to the literal layout-classifier form.
# * horizontal RTL (Hebrew/Arabic, sign=-1): handled via the x-axis
# paired logic in _read_end/_read_gap (added in 528f958).
# VERTICAL text (span merger vertical-font flag / advance-y branch) is NOT handled
# here: a vertical column shatters per-glyph below and is re-merged by
# the gated _remerge_vertical post-pass (detection: vertical-CMap fonts
# via _page_vertical_resnames; matched against the item merger's
# vertical-text rules.
if advance < chunk["negative"]:
if abs(line_delta_y) > 0.5 * height:
# the heading heuristics line-break emission calls reset the last-character buffer before flushing.
reset_last_chars()
flush()
open_chunk(text)
save_last_char(text["ch"])
assert chunk is not None
chunk["str"].append(text["ch"])
else:
reset_last_chars()
flush()
open_chunk(text)
save_last_char(text["ch"])
assert chunk is not None
chunk["str"].append(text["ch"])
continue
if abs(line_delta_y) > height:
# the heading heuristics line-break emission calls reset the last-character buffer before flushing.
reset_last_chars()
flush()
open_chunk(text)
save_last_char(text["ch"])
assert chunk is not None
chunk["str"].append(text["ch"])
continue
if advance <= chunk["not_a_space"]:
reset_last_chars()
if advance <= chunk["tracking"]:
lead = save_last_char(text["ch"])
extend_chunk(text, lead)
continue
if chunk["flow_min"] <= advance <= chunk["flow_max"]:
reset_last_chars()
chunk["str"].append(" ")
lead = save_last_char(text["ch"])
extend_chunk(text, lead)
continue
# OUT-OF-FLOW gap (advance > in-flow space threshold): text extraction synthetic-space insertion
# flushes the current item and pushes a
# STANDALONE " " item with height 0, then a new item begins at this glyph.
# The zero height is load-bearing: the heading heuristics' vertical-alignment test
# can't align this inter-run space with a neighbouring line, so it doesn't
# cause a spurious line merge (the math-heavy page inline math heading heading drop). The
# standalone " " also keeps the word separator in the joined line text Yf
# so a positionally-spaced title like "3 The section heading"
# (Type-3 fonts, no real space glyphs) does not collapse to
# "3TheStaticSemantics" and lose its section number to the _Tf regex.
reset_last_chars()
emit_fake_space(advance) # standalone height-0 " ", width=abs(gap) (exact span merger)
open_chunk(text)
save_last_char(text["ch"])
assert chunk is not None
chunk["str"].append(text["ch"])
flush()
return items
@@ -0,0 +1,191 @@
"""Raw PDF object access (PyPDF2-backed) and PDF lexical primitives."""
from __future__ import annotations
from PyPDF2.generic import (
IndirectObject as PdfIndirectRef, NameObject as PdfName, NumberObject as PdfNumber,
FloatObject as PdfFloat, BooleanObject as PdfBoolean,
DictionaryObject as PdfDictionary, ArrayObject as PdfArray,
)
def _pdf_tok(value) -> str:
"""Serialise one PDF value back to content-syntax (for xref_object's regex)."""
if isinstance(value, PdfIndirectRef):
return f"{value.idnum} {value.generation} R"
if isinstance(value, PdfName):
return str(value)
if isinstance(value, PdfBoolean):
return "true" if value.value else "false"
if isinstance(value, PdfDictionary):
return _pdf_obj_str(value)
if isinstance(value, PdfArray):
return "[ " + " ".join(_pdf_tok(array_item) for array_item in value) + " ]"
return str(value)
def _pdf_obj_str(obj) -> str:
"""Serialize an object body as a PDF-syntax string."""
if isinstance(obj, PdfIndirectRef):
obj = obj.get_object()
if isinstance(obj, PdfDictionary):
parts = ["<<"]
for key_value, val in obj.items():
parts.append(str(key_value))
parts.append(_pdf_tok(val))
parts.append(">>")
return " ".join(parts)
if isinstance(obj, PdfArray):
return "[ " + " ".join(_pdf_tok(array_item) for array_item in obj) + " ]"
return _pdf_tok(obj)
def _pdf_typed(value):
"""Return ``(type, value-string)`` for a raw, unresolved PDF value."""
if value is None:
return ("null", "null")
if isinstance(value, PdfIndirectRef):
return ("xref", f"{value.idnum} {value.generation} R")
if isinstance(value, PdfName):
return ("name", str(value))
if isinstance(value, PdfBoolean):
return ("bool", "true" if value.value else "false")
if isinstance(value, PdfFloat):
return ("real", str(value))
if isinstance(value, PdfNumber):
return ("int", str(int(value)))
if isinstance(value, PdfDictionary):
return ("dict", _pdf_obj_str(value))
if isinstance(value, PdfArray):
return ("array", _pdf_obj_str(value))
try:
return ("string", str(value))
except Exception:
return ("null", "null")
class _PdfPage:
__slots__ = ("_page_object",)
def __init__(self, page):
self._page_object = page
def read_contents(self) -> bytes:
candidate_item = self._page_object.get_contents()
if candidate_item is None:
return b""
if isinstance(candidate_item, PdfIndirectRef):
candidate_item = candidate_item.get_object()
if hasattr(candidate_item, "get_data"):
return candidate_item.get_data()
# /Contents is an array of streams; concatenate them with a single
# space (intentional); join the raw decompressed data the same.
return b" ".join(text.get_object().get_data() for text in candidate_item)
def get_fonts(self, full: bool = True):
out: list = []
res = self._page_object.get("/Resources")
if res is None:
return out
fonts = res.get_object().get("/Font")
if fonts is None:
return out
for _xref_key, ref in fonts.get_object().items():
idnum = ref.idnum if isinstance(ref, PdfIndirectRef) else 0
filter_context = ref.get_object()
subtype = str(filter_context.get("/Subtype", "")).lstrip("/")
basefont = str(filter_context.get("/BaseFont", "")).lstrip("/")
enc_raw = filter_context.raw_get("/Encoding") if "/Encoding" in filter_context else None
enc = str(enc_raw).lstrip("/") if isinstance(enc_raw, PdfName) else ""
out.append((idnum, "", subtype, basefont, str(_xref_key).lstrip("/"), enc))
return out
class _PdfDoc:
"""PyPDF2-backed adapter for raw object and stream access PDFium cannot expose."""
__slots__ = ("_reader", "_virtual")
def __init__(self, reader):
self._reader = reader
# Negative pseudo-xrefs for DIRECT (inline) dicts that have no object
# number -- text extraction reference resolution treats direct and indirect values alike,
# so inline font dicts must be addressable by the same integer-keyed
# pipeline (_redefinition_dict_xrefs registers them).
self._virtual: dict[int, object] = {}
def register_virtual(self, obj) -> int:
vid = -(len(self._virtual) + 1)
self._virtual[vid] = obj
return vid
@property
def page_count(self) -> int:
return len(self._reader.pages)
def __getitem__(self, idx):
return _PdfPage(self._reader.pages[idx])
def page_xref(self, idx: int) -> int:
return self._reader.pages[idx].indirect_reference.idnum
def _resolve_object(self, xref: int):
if xref < 0:
return self._virtual.get(xref)
return PdfIndirectRef(xref, 0, self._reader).get_object()
def xref_get_key(self, xref: int, _xref_key: str):
cur = self._resolve_object(xref)
parts = _xref_key.split("/")
for index_value, part in enumerate(parts):
if cur is None:
return ("null", "null")
if isinstance(cur, PdfIndirectRef):
cur = cur.get_object()
if not hasattr(cur, "raw_get"):
return ("null", "null")
name = "/" + part
if name not in cur:
return ("null", "null")
if index_value == len(parts) - 1:
return _pdf_typed(cur.raw_get(name))
cur = cur[name]
return _pdf_typed(cur)
def xref_stream(self, xref: int) -> bytes:
return self._resolve_object(xref).get_data()
def xref_object(self, xref: int, compressed: bool = True) -> str:
return _pdf_obj_str(self._resolve_object(xref))
def close(self) -> None:
try:
self._reader.stream.close()
except Exception:
pass
_PDF_WHITESPACE_BYTES = frozenset({0x20, 0x09, 0x0d, 0x0a, 0x0c, 0x00})
_PDF_DELIMITER_BYTES = frozenset(b"()<>[]{}/%")
_PDF_STRING_ESCAPE_BYTES = {0x6E: 0x0A, 0x72: 0x0D, 0x74: 0x09, 0x62: 0x08, 0x66: 0x0C,
0x28: 0x28, 0x29: 0x29, 0x5C: 0x5C}
def _decode_pdf_name(raw: bytes) -> bytes:
"""Decode #XX escapes in a PDF name token to its canonical bytes."""
if b"#" not in raw:
return raw
out = bytearray()
index_value = 0
while index_value < len(raw):
if raw[index_value] == 0x23 and index_value + 2 < len(raw):
try:
out.append(int(raw[index_value + 1:index_value + 3], 16))
index_value += 3
continue
except ValueError:
pass
out.append(raw[index_value])
index_value += 1
return bytes(out)
@@ -0,0 +1,295 @@
"""Whole-document parse drivers assembling per-page charlevel metadata."""
from __future__ import annotations
from io import BytesIO
from pathlib import Path
from typing import Union
import pypdfium2 as pdfium
# Raw PDF object access (ToUnicode CMaps, content streams, font dicts, /WMode)
# that PDFium does not expose, read via PyPDF2 -- already a project dependency and
# permissively licensed. A thin adapter exposes the small raw-object API the
# helpers below need, so their calibrated logic stays unchanged.
import PyPDF2 as _pypdf2 # declared dependency (also imported by pageindex.utils/client)
from ..model import Span, Rect
from .pdf_objects import _PdfDoc
from .text_normalize import (
_DROP_CHARS,
_NORMALIZED_UNICODES,
_apply_bidi_reordering,
_reverse_if_rtl,
)
from .content_stream import (
_tokenize_show_operators,
_assign_vertical_tags,
_assign_show_tz,
_page_vertical_resource_names,
)
from .cmap_parse import _compute_skew
from .code_walk import _page_show_codes
from .unicode_apply import _apply_font_unicode
from .char_extract import (
_extract_raw_chars,
_accumulate_type3_extents,
_type3_size_by_font,
_apply_type3_sizes,
_finalize_chars,
_inherited_box,
_page_view_rect,
)
from .merge import _merge_text_items
from .remerge import (
_remerge_rotated,
_remerge_oblique,
_remerge_vertical,
)
def _page_pass1(pdf, pdf_doc, page_idx: int, type3_ext: dict, font_map_cache: dict):
"""Pass-1 body for ONE page: extract raw chars, tag objects, accumulate
Type-3 extents into ``type3_ext``. Returns ``(page, raw_chars, page_vb,
page_rot)``; the PAGE is returned still open — the caller owns closing it
(the sequential driver must keep every page open until pass 2's Type-3
size lookups are done; see keep_pages in ``parse_charlevel_meta``)."""
page = pdf[page_idx]
text_page = page.get_textpage()
raw_chars, objects = _extract_raw_chars(page, text_page.raw)
try:
media_box_raw = _inherited_box(pdf_doc, page_idx, "MediaBox") if pdf_doc is not None else None
crop_box_raw = _inherited_box(pdf_doc, page_idx, "CropBox") if pdf_doc is not None else None
page_vb = _page_view_rect(page, media_box_raw, crop_box_raw) # (x0, y0, x1, y1) page space
except Exception:
page_vb = None # no box -> off-page test disabled
try:
page_rot = int(page.get_rotation()) # PDFium /Rotate (0/90/180/270)
except Exception:
page_rot = 0
show_fonts: list[bytes | None] = []
show_tzs: list[float] = []
vert_names: set[bytes] = set()
if pdf_doc is not None and page_idx < pdf_doc.page_count:
try:
# show-op flush ids (q/Q flush scope) are no longer used -- the merge-id
# grouping was removed; only show_fonts (per-op font resname)
# feeds vertical tagging.
show_flush_ids, show_fonts, show_text_units, horizontal_scales, xobject_paints = _tokenize_show_operators(
pdf_doc[page_idx].read_contents())
except Exception:
show_fonts = []
vert_names = _page_vertical_resource_names(pdf_doc, page_idx)
# Tz follows the text into Form XObjects (the whole text state is
# cloned for the recursion), so the per-show-op horizontal-scale
# list has to come from the SAME form-descending walk as the codes:
# the page's own stream alone under-counts every form page and the
# ordinal gate below would then drop the tag for the whole page.
try:
show_codes = _page_show_codes(pdf_doc, page_idx)
if show_codes:
show_tzs = [horizontal_scale for _fx, _s, horizontal_scale in show_codes]
# Patch per-char unicode to span merger glyph Unicode where
# PDFium's decode differs (guarded: any failure keeps
# PDFium's output).
if raw_chars:
_apply_font_unicode(
text_page.raw, raw_chars, objects, show_codes, pdf_doc,
font_map_cache)
except Exception:
pass
_assign_vertical_tags(objects, show_fonts, vert_names)
_assign_show_tz(objects, show_tzs)
_accumulate_type3_extents(raw_chars, type3_ext)
text_page.close()
return page, raw_chars, page_vb, page_rot
def _page_pass2(raw_chars: list[dict], page_vb, size_by_font: dict) -> list[dict]:
"""Pass-2 body for ONE page: apply the document-wide Type-3 sizes,
restore paint order, finalize glyph widths, run the text merger."""
_apply_type3_sizes(raw_chars, size_by_font)
# text extraction emits glyphs in CONTENT-STREAM (paint) order; PDFium's textpage
# reorders whole segments page-wide (math-heavy page margin labels 'margin label' /
# 'Section N' arrive at a different point of the char stream than their
# show ops). obj["page_order"] is the object's stream position (objects
# parse sequentially, incl. the Form XObject walk), so sorting real
# glyphs by it restores span merger processing order for the merger.
# GENERATED chars (PDFium's synthetic layout whitespace -- no span merger
# counterpart, pure merger bookkeeping) keep no position of their own:
# their geometric obj lookup can land on the WRONG object (the
# multi-column "4 | Super | vision" heading puts the '4'->'S' gap
# space inside the 'vision' object, which would re-emit it mid-word as
# "Super vision"), so each one stays glued behind the real glyph that
# precedes it in textpage order. Character-level ordering's
# own items on the reordered pages.
keys: list[tuple] = [()] * len(raw_chars)
last_key = None
lead_gens: list[int] = []
for key_value, candidate_item in enumerate(raw_chars):
if candidate_item["is_gen"]:
if last_key is None:
lead_gens.append(key_value)
else:
keys[key_value] = (last_key[0], last_key[1], 1, key_value)
else:
last_key = (candidate_item["obj"]["page_order"], candidate_item["i"])
keys[key_value] = (last_key[0], last_key[1], 0, key_value)
for key_value in lead_gens:
keys[key_value] = (-1, -1, 1, key_value)
raw_chars[:] = [raw_chars[key_value] for key_value in sorted(range(len(raw_chars)),
key=keys.__getitem__)]
fin = _finalize_chars(raw_chars)
merged = _merge_text_items(fin, page_vb)
merged = _remerge_rotated(merged) # collapse cardinal-rotated per-glyph shards
merged = _remerge_vertical(merged) # collapse vertical-writing per-glyph shards
return _remerge_oblique(merged, fin) # oblique objects: inverse-rotation projection re-merge
def _page_spans(raw: list[dict]) -> list[Span]:
"""Final emission for ONE page: merged chunks -> ``Span`` objects."""
spans: list[Span] = []
for item in raw:
# the heading heuristics pushes normalized glyph Unicode = the normalized-Unicode table[u] or u
# per glyph, a WHOLE-string lookup. Each r["str"] piece is one glyph's
# unicode (or a synthesized space), so look up per piece -- a
# multi-codepoint ToUnicode value is left intact when the whole-string
# lookup misses, instead of decomposing a table-key char inside it.
# span merger: normalized glyph Unicode = RTL ligature reversal(the normalized-Unicode table
# [u] or u) -- the table lookup is then wrapped in RTL ligature reversal, which
# reverses a multi-char Arabic/Hebrew ligature value (span merger
#). Apply per piece (each r["str"] piece is one glyph's unicode).
joined = "".join(
_reverse_if_rtl(_NORMALIZED_UNICODES.get(page_value, page_value)) for page_value in item["str"] # type: ignore[arg-type]
)
# text extraction text-item flush -> bidirectional transform: the joined item
# text runs the bidi pass ON TOP of the per-glyph RTL ligature reversal
# above (both layers exist in span merger). Pass-through for LTR text
# and vertical items (dir 'ttb').
joined = _apply_bidi_reordering(joined, -1, bool(item["obj"].get("vertical")))
text = joined.translate(_DROP_CHARS)
if not text:
continue
# font_size = hypot(text matrix[2], text matrix[3])
# taken once at the item's open glyph, i.e. the chunk's first-char
# fs. The merger breaks a chunk on any fs change (exact compare;
# see the font_key/fs guard above) and never lowers fs mid-chunk, so
# chunk["fs"] (set in open_chunk from the first char) is exactly
# that value. Emit it rather than the per-chunk minimum.
fs_emit = item["fs"]
spans.append(
Span(
bbox=Rect(item["left"], item["right"], item["top"], item["bottom"]),
text=text,
font_name_raw=item["font_name"],
font_size=fs_emit,
# the heading heuristics bold is name-regex only (the font-name bold regex,
# OR'd into the emitted span). span merger bold detector ignores the descriptor
# ForceBold flag and numeric weight, so we must NOT inject a
# weight-based bold here — that over-bolds Demi/Medium/bold math font
# faces (weight 665-675) text extraction treats as regular.
bold=False,
italic=False,
# Span skew score: P = (f[1]/f[0])² + (f[2]/f[3])² from the item
# transform (IEEE: cardinal rotation -> Inf, upright -> 0).
# The owning object's PDFium matrix has the same
# rotation/shear structure as span merger item transform.
# mtx0 = the FIRST glyph's object matrix (text extraction fixes the
# item transform at open); standalone fake-space items
# carry no mtx0 and fall back to their obj (= the previous
# glyph's object == text extraction previous glyph transform for that space).
skew=_compute_skew(item.get("mtx0") or item["obj"]["mtx"]),
)
)
return spans
def parse_charlevel_meta(doc_handle: Union[str, Path, BytesIO]) -> tuple[list[list[Span]], list]:
if isinstance(doc_handle, (str, Path)):
pdf = pdfium.PdfDocument(str(doc_handle))
elif isinstance(doc_handle, BytesIO):
pdf = pdfium.PdfDocument(doc_handle)
else:
pdf = doc_handle
# Open the same document in PyPDF2 (already a project dependency) to read the
# page content streams: span merger item-flush operators (q/Q save/restore, marked
# content, XObject) live there and PDFium's flattened object model cannot expose
# them. Optional/guarded -- any failure leaves flush_id unset so the merger
# keeps its per-object split (the fallback behavior). A separate bytes copy
# avoids racing pypdfium2's read of the same BytesIO.
pdf_doc = None
if _pypdf2 is not None:
try:
if isinstance(doc_handle, (str, Path)):
pdf_doc = _PdfDoc(_pypdf2.PdfReader(str(doc_handle)))
elif isinstance(doc_handle, BytesIO):
# Read a copy so we never race pdfium's read of the same buffer.
pdf_doc = _PdfDoc(_pypdf2.PdfReader(BytesIO(doc_handle.getvalue())))
except Exception:
pdf_doc = None
# Pass 1: extract raw chars for every page (including each glyph's raw
# advance) and accumulate per-font identity-matrix Type-3 glyph-bbox
# extents document-wide, so each Type-3 font is sized once over every
# glyph it renders anywhere (coverage-independent), matching span merger
# synthesizing font.bbox once from the CharProcs. Font handles are only
# stable per document while their pages stay open (see keep_pages below).
per_page: list[list[dict]] = []
page_view_boxes: list = [] # parallel to per_page: text extraction page view box per page
page_rotations: list = [] # parallel: PDFium page /Rotate in degrees per page
type3_ext: dict = {}
font_map_cache: dict = {}
# Hold every page open until pass 2's Type-3 size lookups are done.
# type3_ext / size_by_font key on the raw FPDF_FONT pointer VALUE, and
# PDFium frees a font once the last page using it closes -- a later
# page's (different) font can then be allocated at the same address,
# silently merging two fonts' extent bins. Which addresses get reused
# depends on the process's prior malloc state, so the output could vary
# with whatever ran earlier in the process. Keeping the pages alive makes
# the handle a true per-document
# font identity (PDFium's document-level font cache returns one handle
# per font redefinition).
keep_pages = []
for page_idx in range(len(pdf)):
page, raw_chars, page_vb, page_rot = _page_pass1(
pdf, pdf_doc, page_idx, type3_ext, font_map_cache)
keep_pages.append(page)
per_page.append(raw_chars)
page_view_boxes.append(page_vb)
page_rotations.append(page_rot)
if pdf_doc is not None and pdf_doc is not doc_handle:
try:
pdf_doc.close()
except Exception:
pass
size_by_font = _type3_size_by_font(type3_ext)
# Pass 2: apply the document-wide Type-3 sizes, finalize glyph widths,
# then run text extraction text merger.
raw_pages: list[list[dict]] = []
for page_view_index, raw_chars in enumerate(per_page):
raw_pages.append(_page_pass2(raw_chars, page_view_boxes[page_view_index], size_by_font))
for page_handle in keep_pages:
try:
page_handle.close()
except Exception:
pass
keep_pages.clear()
out: list[list[Span]] = []
for raw in raw_pages:
out.append(_page_spans(raw))
pdf.close()
# Per-page viewport metadata (text extraction normalized page view = cropbox clamped to the
# mediabox, via _page_view_rect, + /Rotate) parallel to out, so heading
# coordinates can apply span merger viewport-coordinate transform.
return out, list(zip(page_view_boxes, page_rotations))
def parse_charlevel(doc_handle: Union[str, Path, BytesIO]) -> list[list[Span]]:
"""Per-page span entry: per-page spans only (drops viewport meta). the high-level TOC pipeline uses ``parse_charlevel_meta`` to also get the per-page (view box, /Rotate) for heading coordinates; every other caller just wants the spans. """
return parse_charlevel_meta(doc_handle)[0]
@@ -0,0 +1,327 @@
"""Re-merges rotated, oblique, and vertical spans after the first join pass."""
from __future__ import annotations
import math
from .text_normalize import (
TRACKING_SPACE_FACTOR,
NEGATIVE_SPACE_FACTOR,
SPACE_IN_FLOW_MIN_FACTOR,
SPACE_IN_FLOW_MAX_FACTOR,
)
def _start_rot_span(chunk: dict) -> dict:
"""A fresh single-glyph rotated span = a deep-enough copy of the merger chunk (keeps fs/font/obj/sign so the downstream span conversion is unchanged)."""
span = dict(chunk)
span["str"] = list(chunk["str"])
span["font_tally"] = dict(chunk.get("font_tally", {}))
span["weight_tally"] = dict(chunk.get("weight_tally", {}))
return span
def _grow_rot_span(cur: dict, chunk: dict) -> None:
"""Extend a rotated span with the next glyph: append text, union the page box (left/right/top/bottom stay in page coords -> output box is exact), merge the per-char style tallies."""
cur["str"].extend(chunk["str"])
cur["left"] = min(cur["left"], chunk["left"])
cur["right"] = max(cur["right"], chunk["right"])
cur["top"] = max(cur["top"], chunk["top"])
cur["bottom"] = min(cur["bottom"], chunk["bottom"])
for span, count in chunk.get("font_tally", {}).items():
cur["font_tally"][span] = cur["font_tally"].get(span, 0) + count
for span, count in chunk.get("weight_tally", {}).items():
cur["weight_tally"][span] = cur["weight_tally"].get(span, 0) + count
def _merge_rotated_one(group: list[dict], rot: int) -> list[dict]:
"""1-D position comparison along the rotation axis for one cardinally rotated text object. ``read_origin`` is the glyph origin in reading order (90 reads up +y, 270 down -y, 180 left -x); the pen advances by glyph_w, so the inter-glyph gap is ``next_origin - (cur_origin + glyph_w)``. In-flow gaps join, larger gaps start a new item, and the box remains the page-space AABB required by downstream layout. Cardinal rotation intentionally does less than the oblique path: its box convention cannot match the oblique item-box convention, and the extra out-of-flow/cross-axis branches are not useful for these short rotated labels."""
def read_origin(chunk: dict) -> float:
if rot == 90:
return chunk["bottom"]
if rot == 270:
return -chunk["top"]
if rot == 180:
return -chunk["right"]
return chunk["left"]
ordered = sorted(group, key=read_origin)
spans: list[dict] = []
cur: dict | None = None
pen = 0.0
for chunk in ordered:
font_size = chunk.get("fs", 0.0) or 0.0
glyph_width = chunk.get("glyph_w", 0.0) or 0.0
origin = read_origin(chunk)
if cur is None:
cur = _start_rot_span(chunk)
pen = origin + glyph_width
continue
gap = origin - pen
if gap <= font_size * SPACE_IN_FLOW_MAX_FACTOR:
if gap > font_size * TRACKING_SPACE_FACTOR:
cur["str"].append(" ")
_grow_rot_span(cur, chunk)
else:
spans.append(cur)
cur = _start_rot_span(chunk)
pen = origin + glyph_width
if cur is not None:
spans.append(cur)
return spans
def _remerge_rotated(items: list[dict]) -> list[dict]:
"""Re-merge the per-glyph chunks of each rotated text object into text items along the rotation axis. Upright text is untouched; merged spans keep the first chunk position for reading order."""
rot_groups: dict[int, list[dict]] = {}
for item in items:
obj = item.get("obj")
if isinstance(obj, dict) and obj.get("rot") in (90, 180, 270):
rot_groups.setdefault(id(obj), []).append(item)
if not rot_groups:
return items
merged_for = {
oid: _merge_rotated_one(group, group[0]["obj"]["rot"])
for oid, group in rot_groups.items()
}
out: list[dict] = []
emitted: set[int] = set()
for item in items:
obj = item.get("obj")
if isinstance(obj, dict) and obj.get("rot") in (90, 180, 270):
oid = id(obj)
if oid not in emitted:
emitted.add(oid)
out.extend(merged_for[oid])
else:
out.append(item)
return out
def _new_oblique_span(glyph: dict, baseline_pos: float, cross_pos: float, glyph_width: float) -> dict:
"""Open an oblique item at its first reading-order glyph. Records the glyph's page-space pen origin, along-baseline start, cross-axis position, and running pen so the gap logic can compare the next glyph."""
return {
"str": [glyph["ch"]],
"obj": glyph["obj"],
"fs": glyph["fs"],
"font_name": glyph["font_name"],
"_ox0": glyph["ox"], "_oy0": glyph["oy"],
"_u0": baseline_pos, "_uend": baseline_pos + glyph_width, "_pen": baseline_pos + glyph_width, "_vlast": cross_pos,
"_lox": glyph["ox"], "_loy": glyph["oy"], "_lgw": glyph_width,
}
def _close_oblique(cur: dict) -> dict:
"""Finalize an oblique item's box. The item merger is rotation-agnostic -- it turns ANY text extraction item into a span via left=transform[4], right=+width, bottom=transform[5], top=+height -- so an oblique item's box is upright at its pen origin, with width = the along-baseline advance (text extraction item.width, NOT the diagonal x-extent the horizontal merger would compute) and height = font size."""
width = cur["_uend"] - cur["_u0"]
cur["left"] = cur["_ox0"]
cur["right"] = cur["_ox0"] + width
cur["bottom"] = cur["_oy0"]
cur["top"] = cur["_oy0"] + cur["fs"]
return cur
def _oblique_space(cur: dict, adv: float, baseline_unit_x: float, baseline_unit_y: float, scale: float) -> dict:
"""span merger ``synthetic-space insertion`` out-of-flow item: a STANDALONE " " at the previous glyph's pen (previous glyph transform), width=|advance-x|, height 0 (horizontal). The pen sits at the last glyph's origin advanced by its width along the baseline unit direction ``(ux,uy)``. span merger ``advance-x`` is ``(posX-lastPosX)/text advance scale``, so the width is normalised by the matrix scale (== text advance scale here); on identity CTM scale==1 so this is a no-op, but under a scaled CTM it matters. Output box = left=pen_x, right=+width, bottom=top=pen_y."""
pen_x = cur["_lox"] + cur["_lgw"] * baseline_unit_x
pen_y = cur["_loy"] + cur["_lgw"] * baseline_unit_y
width_value = abs(adv) / scale
return {
"str": [" "], "obj": cur["obj"], "fs": cur["fs"], "font_name": cur["font_name"],
"left": pen_x, "right": pen_x + width_value, "bottom": pen_y, "top": pen_y,
}
def _merge_oblique_one(chs: list[dict]) -> list[dict]:
"""text extraction position comparison (inverse-rotation projection path) for ONE oblique text object's glyphs -- the explicit horizontal-branch implementation. ``inverse-rotation projection(x,y,m) = [(m0*x+m1*y)/s, (m2*x+m3*y)/s]`` (s=hypot(m0,m1)); component 0 is the reading-order (baseline) coordinate, component 1 the cross axis. Projecting each glyph's pen origin onto these gives advance-x (along, the gap beyond the prev glyph's advance) and advance-y (cross). Then apply the item split thresholds: advance-x<backward-jump threshold (back-jump) or |advance-y|>height -> split; advance-x<=tracking-space threshold -> join no space; <=in-flow space threshold -> in-flow space in str; else synthetic-space insertion -> a STANDALONE " " item then split. Items carry the item-box convention box (see _close_oblique)."""
matrix_a, matrix_b, matrix_c, matrix_d = chs[0]["obj"]["mtx"]
scale = math.hypot(matrix_a, matrix_b) or 1.0
baseline_unit_x, baseline_unit_y = matrix_a / scale, matrix_b / scale # baseline unit direction (page space)
def along(glyph: dict) -> float:
return (matrix_a * glyph["ox"] + matrix_b * glyph["oy"]) / scale
def cross(glyph: dict) -> float:
return (matrix_c * glyph["ox"] + matrix_d * glyph["oy"]) / scale
ordered = sorted(chs, key=along)
spans: list[dict] = []
cur: dict | None = None
for glyph in ordered:
if glyph.get("is_ws"):
# Skip whitespace glyphs entirely (== main span merger skips whitespace,
# no pen update): text extraction never pushes a raw space glyph to str; the gap
# they leave is re-synthesised by the in-flow/out-of-flow logic below
# for the next visible glyph. This collapses runs of spaces to one and
# trims trailing/leading spaces using the last-character buffer.
continue
font_size = glyph.get("fs", 0.0) or 0.0
glyph_width = glyph.get("glyph_w", 0.0) or 0.0
baseline_pos = along(glyph)
cross_pos = cross(glyph)
if cur is None:
cur = _new_oblique_span(glyph, baseline_pos, cross_pos, glyph_width)
continue
baseline_gap = baseline_pos - cur["_pen"] # along-baseline gap beyond prev advance
cross_shift = cross_pos - cur["_vlast"] # cross-axis shift
if baseline_gap < font_size * NEGATIVE_SPACE_FACTOR or abs(cross_shift) > font_size:
# back-jump (backward-jump threshold) or cross-axis line break: span merger
# flush/line-break emission -- either way the item merger just starts a new item.
spans.append(_close_oblique(cur))
cur = _new_oblique_span(glyph, baseline_pos, cross_pos, glyph_width)
continue
if baseline_gap <= font_size * TRACKING_SPACE_FACTOR:
cur["str"].append(glyph["ch"]) # join, no space
elif baseline_gap <= font_size * SPACE_IN_FLOW_MAX_FACTOR:
cur["str"].append(" ") # in-flow space
cur["str"].append(glyph["ch"])
else:
spans.append(_close_oblique(cur)) # out-of-flow:
spans.append(_oblique_space(cur, baseline_gap, baseline_unit_x, baseline_unit_y, scale)) # standalone " "
cur = _new_oblique_span(glyph, baseline_pos, cross_pos, glyph_width)
continue
cur["_uend"] = baseline_pos + glyph_width
cur["_pen"] = baseline_pos + glyph_width
cur["_vlast"] = cross_pos
cur["_lox"], cur["_loy"], cur["_lgw"] = glyph["ox"], glyph["oy"], glyph_width
if cur is not None:
spans.append(_close_oblique(cur))
return spans
def _remerge_oblique(items: list[dict], fin_chars: list[dict]) -> list[dict]:
"""Rebuild oblique text objects by re-merging per-glyph chunks along the baseline and emitting item-box-convention boxes. Upright and cardinal text are untouched."""
groups: dict[int, list[dict]] = {}
for glyph in fin_chars:
obj = glyph.get("obj")
if isinstance(obj, dict) and obj.get("rot") == -1:
groups.setdefault(id(obj), []).append(glyph)
if not groups:
return items
merged_for = {oid: _merge_oblique_one(chs) for oid, chs in groups.items()}
out: list[dict] = []
emitted: set[int] = set()
for item in items:
obj = item.get("obj")
if isinstance(obj, dict) and obj.get("rot") == -1:
oid = id(obj)
if oid not in emitted:
emitted.add(oid)
out.extend(merged_for[oid])
else:
out.append(item)
return out
def _start_vert_span(chunk: dict) -> dict:
"""Create a vertical item from its first chunk. Vertical items use the rendered font size as width, accumulate height per glyph, and keep the first glyph's pen as the item transform. The item-to-span conversion reads the style's vertical flag and flips the sign of the height offset, so a vertical item's box runs DOWN from the pen where a horizontal one runs up. The span box reproduces that convention rather than the ink AABB."""
span = dict(chunk)
span["str"] = list(chunk["str"])
span["font_tally"] = dict(chunk.get("font_tally", {}))
span["weight_tally"] = dict(chunk.get("weight_tally", {}))
span["v_height"] = chunk["v_pen_y"] - chunk["v_after"] # first glyph's advance
return span
def _close_vert_span(mapping: dict) -> dict:
"""Finalize the item merger-convention box of a vertical item."""
mapping["left"] = mapping["v_pen_x"]
mapping["right"] = mapping["v_pen_x"] + mapping["fs"]
mapping["top"] = mapping["v_pen_y"]
mapping["bottom"] = mapping["v_pen_y"] - abs(mapping["v_height"])
return mapping
def _merge_vertical_one(group: list[dict]) -> list[dict]:
"""Apply vertical-writing position comparison over one text object's per-glyph chunks in stream order. The previous pen-after-advance and current pen define the along-axis gap; x shift is the cross-axis break signal. Small gaps join, in-flow gaps insert a space, out-of-flow gaps emit a standalone zero-width space item, and backward or cross-axis jumps start a new item. Whitespace glyphs are consumed by the span merger, so their advance arrives here as an in-flow gap."""
spans: list[dict] = []
cur: dict | None = None
after = 0.0 # text extraction previous glyph transform[5]: pen y after the previous glyph
last_x = 0.0 # text extraction previous glyph transform[4]
for chunk in group:
font_size = chunk.get("fs", 0.0) or 0.0
if cur is None:
cur = _start_vert_span(chunk)
after, last_x = chunk["v_after"], chunk["v_pen_x"]
continue
vertical_gap = after - chunk["v_pen_y"]
x_shift = chunk["v_pen_x"] - last_x
direction_sign = 1.0 if cur["v_height"] >= 0 else -1.0
width = cur["fs"]
if vertical_gap < direction_sign * NEGATIVE_SPACE_FACTOR * font_size or abs(x_shift) > width:
# backward jump or cross-axis break: text extraction line-break emission/flush -- both
# end the item (we don't model line-break marker, and the item merger ignores it).
spans.append(_close_vert_span(cur))
cur = _start_vert_span(chunk)
elif vertical_gap <= direction_sign * TRACKING_SPACE_FACTOR * font_size:
cur["v_height"] += vertical_gap + (chunk["v_pen_y"] - chunk["v_after"])
_grow_vert_span(cur, chunk)
elif direction_sign * SPACE_IN_FLOW_MIN_FACTOR * font_size <= vertical_gap <= direction_sign * SPACE_IN_FLOW_MAX_FACTOR * font_size:
cur["str"].append(" ")
cur["v_height"] += vertical_gap + (chunk["v_pen_y"] - chunk["v_after"])
_grow_vert_span(cur, chunk)
else:
# out-of-flow: standalone " " at previous glyph transform, width 0, height |e|
# (vertical synthetic spaces store the gap as height and leave width at zero).
meta = cur
spans.append(_close_vert_span(cur))
spans.append({
"str": [" "], "sign": 1, "obj": meta["obj"],
"left": last_x, "right": last_x, # WIDTH 0
# A vertical style flips the height offset: the box runs DOWN
# from the previous pen, like _close_vert_span's.
"top": after, "bottom": after - abs(vertical_gap),
"fs": meta["fs"], "fs_min": meta["fs"],
"font_name": meta["font_name"], "font_key": meta["font_key"],
"weight": meta["weight"],
"font_tally": {meta["font_name"]: 1},
"weight_tally": {meta["weight"]: 1},
})
cur = _start_vert_span(chunk)
after, last_x = chunk["v_after"], chunk["v_pen_x"]
if cur is not None:
spans.append(_close_vert_span(cur))
return spans
def _grow_vert_span(cur: dict, chunk: dict) -> None:
"""Append a glyph to a vertical item: text + style tallies. The box is NOT unioned here -- it is derived from the first pen + accumulated v_height in _close_vert_span, with transform fixed at the first glyph and height accumulated."""
cur["str"].extend(chunk["str"])
for span, count in chunk.get("font_tally", {}).items():
cur["font_tally"][span] = cur["font_tally"].get(span, 0) + count
for span, count in chunk.get("weight_tally", {}).items():
cur["weight_tally"][span] = cur["weight_tally"].get(span, 0) + count
def _remerge_vertical(items: list[dict]) -> list[dict]:
"""Re-merge the per-glyph chunks of each vertical-writing (Identity-V / WMode 1) text object into PDF content tokenizer style items. Uses the same _remerge_rotated: the horizontal merger is untouched (it shatters a vertical column because the glyphs stack along its line-break axis) and this gated post-pass rewrites only vertical-object chunks. One extra wrinkle vs the rotated pass: text extraction emits items in content-stream order, but PDFium's textpage reorders vertical chars page-wide (its own column heuristic), so the merged groups are reassigned to the vertical slot positions in object paint order."""
groups: dict[int, list[dict]] = {}
obj_of: dict[int, dict] = {}
for item in items:
obj = item.get("obj")
if (isinstance(obj, dict) and obj.get("vertical") and not obj.get("rot")
and "v_pen_y" in item):
oid = id(obj)
groups.setdefault(oid, []).append(item)
obj_of[oid] = obj
if not groups:
return items
merged_for = {oid: _merge_vertical_one(group_value) for oid, group_value in groups.items()}
paint_order = sorted(groups, key=lambda oid: obj_of[oid]["page_order"])
out: list[dict] = []
slot = 0 # next paint-order group to emit at the next vertical slot
seen: set[int] = set()
for item in items:
obj = item.get("obj")
oid = id(obj) if isinstance(obj, dict) else None
if oid in groups:
if oid not in seen:
seen.add(oid)
out.extend(merged_for[paint_order[slot]])
slot += 1
else:
out.append(item)
return out
@@ -0,0 +1,288 @@
"""Unicode normalization tables, whitespace classes, spacing factors, and bidi reordering."""
from __future__ import annotations
import json
import unicodedata
from pathlib import Path
_DROP_CHARS = str.maketrans({
# U+FFFE is PDFium's "no unicode mapping" textpage sentinel. The
# patch pipeline (_apply_font_unicode) replaces it with decoded text
# wherever the map walk succeeds; a REMAINING U+FFFE means the guarded
# walk gave up for that run, so deleting it keeps PDFium noise out of the
# spans. A pathological ToUnicode map that intentionally emits literal
# U+FFFE is indistinguishable from this sentinel here and is dropped.
"￾": None,
"\t": " ",
"\n": " ",
"\r": " ",
# text extraction maps a glyph whose unicode lands on U+00AD to U+002D. (An
# earlier "\x02" -> "-" entry here compensated PDFium decoding
# re-encoded hyphens (charcode 2, ToUnicode gap) as U+0002; that decode
# is now handled by _apply_font_unicode: mapped soft hyphens emit '-' where the
# font's Differences name the glyph, and keeps the raw \x02 where they
# don't, e.g. math-heavy page body ligature codes.
"­": "-",
})
# The normalized Unicode table is a fixed, sparse per-glyph
# lookup table; a char absent from it is emitted unchanged. This is NOT Unicode
# NFKC: NFKC over-normalises (fullwidth→ASCII, superscripts→digits, ohm→omega,
# nbsp→space) exactly where this table leaves the glyph untouched. Apply the
# table per code point.
_NORMALIZED_UNICODES: dict[str, str] = json.loads(
(Path(__file__).parent.parent / "data" / "normalized_unicodes.json")
.read_text(encoding="utf-8")
)
def _normalize_unicodes(text: str) -> str:
"""Apply the per-glyph normalized-Unicode substitution table to a text item. The table is keyed by single code points and never introduces table keys, so applying it to the already-joined LTR item string preserves per-glyph substitution after the text item is joined."""
unit_count = _NORMALIZED_UNICODES
if not any(candidate_item in unit_count for candidate_item in text):
return text
return "".join(unit_count.get(candidate_item, candidate_item) for candidate_item in text)
# span merger
TRACKING_SPACE_FACTOR = 0.1
NON_SPACE_GAP_FACTOR = 0.03
NEGATIVE_SPACE_FACTOR = -0.2
SPACE_IN_FLOW_MIN_FACTOR = 0.1
SPACE_IN_FLOW_MAX_FACTOR = 0.6
# Whitespace classification uses the Unicode WhiteSpace + LineTerminator set.
# Python's str.isspace is not the same set: it omits U+FEFF and adds
# U+001C-U+001F and U+0085. Use the explicit code points so the
# whitespace-skip branch fires on the intended glyphs.
_WHITESPACE_CODEPOINTS = frozenset({
0x9, 0xA, 0xB, 0xC, 0xD, 0x20, 0xA0, 0x1680,
0x2000, 0x2001, 0x2002, 0x2003, 0x2004, 0x2005, 0x2006, 0x2007,
0x2008, 0x2009, 0x200A, 0x2028, 0x2029, 0x202F, 0x205F, 0x3000,
0xFEFF,
})
def _is_whitespace(number: int) -> bool:
"""Return whether a glyph code point is classified as whitespace."""
return number in _WHITESPACE_CODEPOINTS
# Character classification checks whitespace before marks/formats, so a code
# point such as U+FEFF that is also Cf is treated as whitespace, not as an
# invisible format mark.
def _is_zero_width_diacritic(number: int) -> bool:
"""text extraction zero-width diacritic classification (group 2 = ``\\p{Mn}``)."""
return number not in _WHITESPACE_CODEPOINTS and unicodedata.category(chr(number)) == "Mn"
def _is_invisible_format_mark(number: int) -> bool:
"""text extraction invisible format-mark classification (group 3 = ``\\p{Cf}``)."""
return number not in _WHITESPACE_CODEPOINTS and unicodedata.category(chr(number)) == "Cf"
# Bidirectional character-type tables. base bidi type table covers
# U+0000..U+00FF; Arabic bidi type table covers U+0600..U+06FF indexed by the low byte
# (the "" at 0x1D follows the extraction rule placeholder for nonexistent U+061D).
_BIDI_BASE_TYPES = (
"BN BN BN BN BN BN BN BN BN S B S WS B BN BN BN BN BN BN BN BN BN BN BN BN "
"BN BN B B B S WS ON ON ET ET ET ON ON ON ON ON ES CS ES CS CS EN EN EN EN "
"EN EN EN EN EN EN CS ON ON ON ON ON ON L L L L L L L L L L L L L L L L L L "
"L L L L L L L L ON ON ON ON ON ON L L L L L L L L L L L L L L L L L L L L "
"L L L L L L ON ON ON ON BN BN BN BN BN BN B BN BN BN BN BN BN BN BN BN BN "
"BN BN BN BN BN BN BN BN BN BN BN BN BN BN BN BN CS ON ET ET ET ET ON ON ON "
"ON L ON ON BN ON ON ET ET EN EN ON L ON ON ON EN L ON ON ON ON ON L L L L "
"L L L L L L L L L L L L L L L L L L L ON L L L L L L L L L L L L L L L L L "
"L L L L L L L L L L L L L L ON L L L L L L L L "
).split()
assert len(_BIDI_BASE_TYPES) == 256
_BIDI_ARABIC_TYPES = [
"" if bidi_type == "~" else bidi_type for bidi_type in (
"AN AN AN AN AN AN ON ON AL ET ET AL CS AL ON ON NSM NSM NSM NSM NSM NSM "
"NSM NSM NSM NSM NSM AL AL ~ AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
"AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
"AL AL AL AL AL NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM NSM "
"NSM NSM NSM NSM NSM NSM AN AN AN AN AN AN AN AN AN AN ET AN AN AL AL AL "
"NSM AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
"AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
"AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
"AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL AL "
"AL AL AL NSM NSM NSM NSM NSM NSM NSM AN ON NSM NSM NSM NSM NSM NSM AL AL "
"NSM NSM ON NSM NSM NSM NSM AL AL EN EN EN EN EN EN EN EN EN EN AL AL AL AL "
"AL AL "
).split()
]
assert len(_BIDI_ARABIC_TYPES) == 256
def _apply_bidi_reordering(text: str, start_level: int = -1, vertical: bool = False) -> str:
"""Apply the simplified single-line UAX#9 pass used for flushed PDF text items. Empty, vertical, and purely LTR text pass through. Otherwise the pass resolves W1-W7/N1-N2/I1-I2 levels from the tables above, reverses runs, and strips literal '<'/'>'. Astral-codepoint handling follows Python strings; surrogate pairs are not corrupted because both halves classify L at equal levels and reversal spans restore the pair."""
if not text or vertical:
return text
count_item = len(text)
chars = list(text)
types: list[str] = [""] * count_item
num_bidi = 0
for index_value, char in enumerate(chars):
codepoint = ord(char)
token_value = "L"
if codepoint <= 0xFF:
token_value = _BIDI_BASE_TYPES[codepoint]
elif 0x0590 <= codepoint <= 0x05F4:
token_value = "R"
elif 0x0600 <= codepoint <= 0x06FF:
token_value = _BIDI_ARABIC_TYPES[codepoint & 0xFF]
elif 0x0700 <= codepoint <= 0x08AC:
token_value = "AL"
if token_value in ("R", "AL", "AN"):
num_bidi += 1
types[index_value] = token_value
if num_bidi == 0:
return text
if start_level == -1:
if num_bidi / count_item < 0.3 and count_item > 4:
start_level = 0
else:
start_level = 1
levels = [start_level] * count_item
entry_item = "R" if (start_level & 1) else "L"
sor = entry_item
eor = sor
# W1: NSM takes the type of the previous character (sor at run start).
last = sor
for index_value in range(count_item):
if types[index_value] == "NSM":
types[index_value] = last
else:
last = types[index_value]
# W2: EN after an AL (searching back to the first strong type) becomes AN.
last = sor
for index_value in range(count_item):
token_value = types[index_value]
if token_value == "EN":
types[index_value] = "AN" if last == "AL" else "EN"
elif token_value in ("R", "L", "AL"):
last = token_value
# W3: AL -> R.
for index_value in range(count_item):
if types[index_value] == "AL":
types[index_value] = "R"
# W4: single ES between ENs -> EN; single CS between same-type numbers.
for index_value in range(1, count_item - 1):
if types[index_value] == "ES" and types[index_value - 1] == "EN" and types[index_value + 1] == "EN":
types[index_value] = "EN"
if (types[index_value] == "CS" and types[index_value - 1] in ("EN", "AN")
and types[index_value + 1] == types[index_value - 1]):
types[index_value] = types[index_value - 1]
# W5: ET runs adjacent to EN -> EN.
for index_value in range(count_item):
if types[index_value] == "EN":
for state_item in range(index_value - 1, -1, -1):
if types[state_item] != "ET":
break
types[state_item] = "EN"
for state_item in range(index_value + 1, count_item):
if types[state_item] != "ET":
break
types[state_item] = "EN"
# W6: WS/ES/ET/CS -> ON.
for index_value in range(count_item):
if types[index_value] in ("WS", "ES", "ET", "CS"):
types[index_value] = "ON"
# W7: EN after an L (searching back to the first strong type) -> L.
last = sor
for index_value in range(count_item):
token_value = types[index_value]
if token_value == "EN":
types[index_value] = "L" if last == "L" else "EN"
elif token_value in ("R", "L"):
last = token_value
# N1: neutrals between same-direction strongs take that direction
# (numbers count as R); N2: leftovers take the embedding direction.
index_value = 0
while index_value < count_item:
if types[index_value] == "ON":
end = index_value + 1
while end < count_item and types[end] == "ON":
end += 1
before = types[index_value - 1] if index_value > 0 else sor
after = types[end + 1] if end + 1 < count_item else eor
if before != "L":
before = "R"
if after != "L":
after = "R"
if before == after:
for state_item in range(index_value, end):
types[state_item] = before
index_value = end - 1
index_value += 1
for index_value in range(count_item):
if types[index_value] == "ON":
types[index_value] = entry_item
# I1/I2: level bumps.
for index_value in range(count_item):
token_value = types[index_value]
if levels[index_value] % 2 == 0:
if token_value == "R":
levels[index_value] += 1
elif token_value in ("AN", "EN"):
levels[index_value] += 2
else:
if token_value in ("L", "AN", "EN"):
levels[index_value] += 1
#: reverse contiguous runs from the highest level down to the lowest
# odd level.
highest = -1
lowest_odd = 99
for layout_value in levels:
if layout_value > highest:
highest = layout_value
if layout_value < lowest_odd and (layout_value & 1):
lowest_odd = layout_value
for level in range(highest, lowest_odd - 1, -1):
start = -1
for index_value in range(count_item):
if levels[index_value] < level:
if start >= 0:
chars[start:index_value] = chars[start:index_value][::-1]
start = -1
elif start < 0:
start = index_value
if start >= 0:
chars[start:count_item] = chars[start:count_item][::-1]
# text extraction final loop: literal '<' and '>' are dropped (numBidi > 0 only).
return "".join("" if char in "<>" else char for char in chars)
def _rtl_sign(char: str) -> int:
"""+1 for LTR runs, -1 for a strong right-to-left char (bidi class R/AL, e.g. Hebrew/Arabic). PDFium reports RTL text in logical order with decreasing char origins, so the LTR ``advance = ox - prev_text_x`` model (prev_text_x = ox+glyph_w, a right edge) yields a large negative advance. For RTL chunks the x-axis is signed with ``sign*ox`` so the reading-direction advance is positive and the existing LTR merge logic applies unchanged."""
return -1 if unicodedata.bidirectional(char) in ("R", "AL") else 1
def _reverse_if_rtl(chars: str) -> str:
"""span merger ``RTL ligature reversal`` : reverse a multi-char (Arabic/Hebrew ligature) value when its FIRST code unit is in the Hebrew ``[0x0590,0x05ff)`` or Arabic ``[0x0600,0x06ff)`` range (Unicode range table[11]/ [13], ``right-to-left range test`` uses ``>= begin and < end``, so the range end is EXCLUSIVE). text extraction wraps every glyph's ``normalized Unicode`` in this, so a table value like "\u0626\u062c" emitted for U+FC00 is reversed to "\u062c\u0626"; a single-char value (length <= 1) is returned as-is."""
if len(chars) <= 1:
return chars
first_codepoint = ord(chars[0])
if (0x0590 <= first_codepoint < 0x05FF) or (0x0600 <= first_codepoint < 0x06FF):
return chars[::-1]
return chars
def _read_end(mapping: dict, sign: int) -> float:
"""The reading-direction FAR edge of a glyph (the edge facing the next char). PDFium reports the origin (ox) as the glyph's LEFT edge in both directions; the glyph extends RIGHT by glyph_w. So: * LTR (reading right): far edge = right edge = max(ox+glyph_w, ink right). * RTL (reading left): far edge = LEFT edge = ox (the origin itself). The next char's gap is then measured to its NEAR edge -- ox for LTR, ox+glyph_w for RTL -- in ``_read_gap`` below. (Earlier this added glyph_w on the RTL side too, which used the PREVIOUS glyph's width and injected spurious spaces.)"""
if sign > 0:
return max(mapping["ox"] + mapping["glyph_w"], mapping["right"])
return mapping["ox"]
def _read_gap(prev_far: float, other_mapping: dict, sign: int) -> float:
"""Reading-direction gap between the previous glyph's far edge and the current glyph's NEAR edge. LTR near edge = ox (left); RTL near edge = ox+glyph_w (right). ==0 for adjacent glyphs, >0 for a word gap, <0 for a backward jump."""
if sign > 0:
return other_mapping["ox"] - prev_far
return prev_far - (other_mapping["ox"] + other_mapping["glyph_w"])
@@ -0,0 +1,375 @@
"""Applies per-font Unicode maps to page chars and synthesizes dropped glyphs."""
from __future__ import annotations
import bisect
import difflib
from collections import Counter
import pypdfium2.raw as pdfium_c
from .text_normalize import _is_whitespace
from .font_unicode import _font_unicode_map
from .code_walk import (
_char_category,
_walk_codes,
)
def _apply_font_unicode(
text_page,
raw_chars: list[dict],
objects: list[dict],
show_codes: list[tuple[int | None, tuple[int, ...], float]],
pdf_doc,
map_cache: dict,
) -> None:
"""Patch each char's unicode to span merger glyph Unicode (`map.get(code) or chr(code)`, content stream tokenizer glyph mapping) where PDFium's decode disagrees. Two granularities, both gated by _walk_codes' both-streams-exhaust rule: - object mode (when PDFium's text objects pair consistent with the page's show ops, the _assign_flush_ids precondition): each object's chars are walked against its own show op's codes. This is immune to PDFium's textpage segment reordering (e.g. math-heavy page margin labels emitted at a different page position than paint order) because chars keep stream order WITHIN an object; a desync rolls back only that object. - page mode (counts differ, e.g. PDFium splitting a TJ into several objects): all non-generated textpage chars are walked against all show ops' codes in paint order; any desync rolls back the whole page. """
if not show_codes:
return
def targets_for(font_xref: int | None, other_numbers: tuple[int, ...]) -> list[str] | None:
if font_xref is None:
return None
if font_xref not in map_cache:
try:
map_cache[font_xref] = _font_unicode_map(pdf_doc, font_xref)
except Exception:
map_cache[font_xref] = None
entry = map_cache[font_xref]
if entry is None:
return None
next_block, measure_item = entry
if next_block == 1:
return [measure_item.get(code) or chr(code) for code in other_numbers]
return [measure_item.get((other_numbers[key_value] << 8) | other_numbers[key_value + 1]) or chr((other_numbers[key_value] << 8) | other_numbers[key_value + 1])
for key_value in range(0, len(other_numbers) - 1, 2)]
def apply(patches: list[tuple[int, str]], drops: list[int],
chars_by_index: dict[int, dict]) -> None:
for index_value, token_value in patches:
candidate_item = chars_by_index.get(index_value)
if candidate_item is None:
continue # char was dropped at extraction; nothing to patch
candidate_item["ch"] = token_value
candidate_item["is_ws"], candidate_item["is_mn"], candidate_item["is_cf"] = _char_category(token_value)
for index_value in drops:
candidate_item = chars_by_index.get(index_value)
if candidate_item is not None:
candidate_item["drop"] = True
chars_by_index = {raw_char["i"]: raw_char for raw_char in raw_chars}
if len(objects) == len(show_codes):
# Object mode: pair text objects with show ops ordinally (both are in
# content-stream paint order) and walk each pair independently.
chars_by_obj: dict[int, list[tuple[int, str]]] = {}
for raw_char in raw_chars:
if raw_char["is_gen"]:
continue
chars_by_obj.setdefault(id(raw_char["obj"]), []).append((raw_char["i"], raw_char["ch"]))
desynced: list[int] = []
failed_windows: list[list[int]] = []
synth_sites: list[dict] = []
targets_by_object_index: dict[int, list[str] | None] = {}
for object_index, (obj, (font_index, encoded_text, _tz)) in enumerate(zip(objects, show_codes)):
target_text_items = targets_for(font_index, encoded_text)
targets_by_object_index[object_index] = target_text_items
if target_text_items is None:
continue # uncovered font: this object keeps PDFium's output
res = _walk_codes(chars_by_obj.get(id(obj), []), target_text_items)
if res is None:
# Desync: often a boundary-attribution error (the geometric
# char->object lookup parks a show op's edge glyph in the
# NEIGHBOURING object's list: punctuation at a run boundary can
# land in the previous object, and heavily overlapped chart
# labels can park a leading glyph in the wrong object. Record for the
# window re-walk below; a genuine mismatch stays rolled back
# there too.
desynced.append(object_index)
continue
apply(res[0], res[1], chars_by_index)
# Re-walk each window of desynced objects (bridging up to 2 covered,
# successfully-walked objects between them) as one unit: boundary-
# attribution errors cancel inside the window (the page-mode walk
# scoped to the ambiguous region) and the exhaust-in-sync gate still
# rejects anything else. On commit, REASSIGN each consumed char to
# the object whose show op consumed it -- the stream-side ownership --
# repairing the geometric attribution for the paint-order sort, the
# merger's font/fs identity and the Type-3 sizing alike.
def _rewalk_window(window: list[int]) -> bool:
char_value = sorted(
(pair for state_item in window for pair in chars_by_obj.get(id(objects[state_item]), [])))
text_transform: list[str] = []
owner: list[int] = []
for state_item in window:
target_text_items = targets_by_object_index[state_item]
assert target_text_items is not None
text_transform.extend(target_text_items)
owner.extend([state_item] * len(target_text_items))
def _commit(res) -> bool:
if res is None:
return False
apply(res[0], res[1], chars_by_index)
for char_index, text_index in res[2]:
candidate_item = chars_by_index.get(char_index)
if candidate_item is not None and candidate_item["obj"] is not objects[owner[text_index]]:
candidate_item["obj"] = objects[owner[text_index]]
# Skipped targets are glyphs PDFium never emitted; record
# each with its show op and surviving stream neighbours so
# _synthesize_dropped_glyphs can re-emit it (text extraction does).
for text_index, pos in res[3]:
synth_sites.append({
"t": text_transform[text_index], "owner": objects[owner[text_index]],
"prev_i": char_value[pos - 1][0] if pos > 0 else None,
"next_i": char_value[pos][0] if pos < len(char_value) else None,
})
return True
if len(window) >= 2 and _commit(_walk_codes(char_value, text_transform)):
return True
# Pure-displacement fallback: PDFium's textpage can also REORDER a
# char across the window (TeX accents again: 'accented word stem'+'´'+'es'
# arrives as '...ilites´', and the 't' sits in the 'es' object),
# which the linear walk above can never align. When the chars are
# EXACTLY the targets as a multiset (no decode work left -- only
# placement is wrong), align via SequenceMatcher and repair
# OWNERSHIP alone: equal blocks map positionally, the few
# displaced chars (<=4) map by literal value. Single-char targets
# only, so target index == string position.
def _displacement_repair() -> bool:
if any(len(token_value) != 1 for token_value in text_transform):
return False
chs = "".join(candidate_item for _, candidate_item in char_value)
tts = "".join(text_transform)
deficit = len(tts) - len(chs)
if (chs == tts or deficit < 0 or deficit > 8
or (Counter(chs) - Counter(tts))):
return False
state_map = difflib.SequenceMatcher(None, tts, chs, autojunk=False)
char_to_tgt: dict[int, int] = {}
loose_target_indexes: list[int] = []
loose_char_indexes: list[int] = []
for tag, index_one, index_two, char_start, char_end in state_map.get_opcodes():
if tag == "equal":
for reference_item in range(index_two - index_one):
char_to_tgt[char_start + reference_item] = index_one + reference_item
else:
loose_target_indexes.extend(range(index_one, index_two))
loose_char_indexes.extend(range(char_start, char_end))
if len(loose_char_indexes) > 24:
return False
used_targets: set[int] = set()
for char_index in loose_char_indexes:
cdict = chars_by_index.get(char_value[char_index][0])
cands = [target_index for target_index in loose_target_indexes
if target_index not in used_targets and tts[target_index] == chs[char_index]]
if not cands:
return False # a displaced char with no equal target
if cdict is not None and len(cands) > 1:
# Identical glyphs (the 21 scattered 'α' labels):
# pick the candidate whose OBJECT box sits closest
# to the char -- the one signal that distinguishes
# equal-valued slots.
origin_x, origin_y = cdict["ox"], cdict["oy"]
def _object_distance_sq(target_index: int) -> float:
item_value = objects[owner[target_index]]
delta_x = max(item_value["l"] - origin_x, 0.0, origin_x - item_value["r"])
delta_y = max(item_value["b"] - origin_y, 0.0, origin_y - item_value["t"])
return delta_x * delta_x + delta_y * delta_y
cands.sort(key=_object_distance_sq)
char_to_tgt[char_index] = cands[0]
used_targets.add(cands[0])
# Leftover loose TARGETS = glyphs PDFium never emitted (the
# font-layer drop class). Record each between its
# nearest MAPPED neighbours for re-synthesis.
leftover = [target_index for target_index in loose_target_indexes if target_index not in used_targets]
if leftover:
tgt_to_char = {target_index: char_index for char_index, target_index in char_to_tgt.items()}
mapped_tis = sorted(tgt_to_char)
for target_index in leftover:
page_value = bisect.bisect_left(mapped_tis, target_index)
point_value = mapped_tis[page_value - 1] if page_value > 0 else None
normalized_token = mapped_tis[page_value] if page_value < len(mapped_tis) else None
synth_sites.append({
"t": tts[target_index], "owner": objects[owner[target_index]],
"prev_i": char_value[tgt_to_char[point_value]][0] if point_value is not None else None,
"next_i": char_value[tgt_to_char[normalized_token]][0] if normalized_token is not None else None,
})
for char_index, target_index in char_to_tgt.items():
candidate_item = chars_by_index.get(char_value[char_index][0])
if candidate_item is not None and candidate_item["obj"] is not objects[owner[target_index]]:
candidate_item["obj"] = objects[owner[target_index]]
return True
if _displacement_repair():
return True
# Final resort: the same walk with anchored drop-skips, for
# windows containing glyphs PDFium never emitted (font-layer
# drops). The rest of the window still gets its patches and
# stream-side ownership; the dropped glyphs are recorded for
# synthesis.
if not _commit(_walk_codes(char_value, text_transform, allow_skips=True)):
failed_windows.append(list(window))
return False
return True
# A window that resolves only by DECLARING drops (recording synth
# sites) has trusted its local char census; when chars were stolen
# ACROSS window boundaries that census lies (a starved window
# "drops" a glyph whose char sits, surplus, in another failed
# window). Track those windows so the mega pass below can supersede
# their local verdicts.
synth_windows: list[tuple[list[int], int, int]] = []
def _run_window(window: list[int]) -> None:
before = len(synth_sites)
if _rewalk_window(window) and len(synth_sites) > before:
synth_windows.append((list(window), before, len(synth_sites)))
window: list[int] = []
for object_index in desynced:
if window:
gap = range(window[-1] + 1, object_index)
if (len(gap) <= 2
and all(targets_by_object_index.get(bridge_index) is not None for bridge_index in gap)):
window.extend(gap)
window.append(object_index)
continue
_run_window(window)
window = [object_index]
if window:
_run_window(window)
# Page-scope last resort: scattered same-glyph labels (dense math-heavy page's
# 21 'α' show ops over a vector figure) defeat per-window walks --
# the geometric attribution piles several chars on some ops and
# leaves others empty ACROSS window boundaries (donor ops hold a
# stolen surplus char, starved ops none). Merge every failed AND
# every drop-declaring window into one final window so the
# displacement/skip repairs see the whole cluster at once: the
# surplus cancels the deficit, stolen chars are reassigned to their
# true ops, and only the genuine font-layer drops remain as synth
# sites. The locally-recorded sites are dropped first (the mega
# re-records with full context) and restored if the mega fails.
cand = failed_windows + [window for window, _, _ in synth_windows]
if len(cand) >= 2:
stash = synth_sites[:]
for _, font, window_end in reversed(synth_windows):
del synth_sites[font:window_end]
failed_windows = []
mega = sorted({mega_index for window in cand for mega_index in window})
if not _rewalk_window(mega):
synth_sites[:] = stash # mega failed: keep local verdicts
if synth_sites:
# Census gate: a recorded site is a REAL font-layer drop only if
# the PAGE-WIDE multiset still misses that value (covered ops'
# target codepoints minus PDFium's final chars). A window-local
# repair can otherwise declare a glyph dropped whose char simply
# sits, mis-attributed, in an op that walked clean -- the a clipped-cell table
# with star glyphs: 7 star codes, 7 star chars page-wide, but the
# clip-overlapped cells starve two ops, and the donors never
# fail so the mega pass can't see them. WHITESPACE is never
# synthesized: a missing space char is PDFium's textpage
# space-run normalization (text extraction runs its own space
# normalization, already implemented in the merger), not a font-layer
# drop.
census: Counter = Counter()
for text_adjustment in targets_by_object_index.values():
if text_adjustment is not None:
for target_text in text_adjustment:
census.update(target_text)
for raw_char in raw_chars:
if not raw_char["is_gen"] and not raw_char.get("drop"):
census.subtract(raw_char["ch"])
kept: list[dict] = []
for encoded_text in synth_sites:
if all(_is_whitespace(ord(ch_)) for ch_ in encoded_text["t"]):
continue
if all(census[ch_] > 0 for ch_ in encoded_text["t"]):
for ch_ in encoded_text["t"]:
census[ch_] -= 1
kept.append(encoded_text)
if kept:
_synthesize_dropped_glyphs(kept, raw_chars, chars_by_index)
return
# Page mode.
seq: list[tuple[int, str]] = []
char_count = pdfium_c.FPDFText_CountChars(text_page)
for char_index in range(char_count):
if pdfium_c.FPDFText_IsGenerated(text_page, char_index) == 1:
continue
codepoint = pdfium_c.FPDFText_GetUnicode(text_page, char_index)
seq.append((char_index, chr(codepoint) if codepoint > 0 else "\x00"))
targets: list[str] = []
for font_index, encoded_text, _tz in show_codes:
if not encoded_text:
continue
text_state = targets_for(font_index, encoded_text)
if text_state is None:
return # uncovered font used on this page: no patch
targets.extend(text_state)
res = _walk_codes(seq, targets)
if res is None:
return
apply(res[0], res[1], chars_by_index)
def _synthesize_dropped_glyphs(
sites: list[dict], raw_chars: list[dict], chars_by_index: dict[int, dict],
) -> None:
"""Re-emit glyphs PDFium's font layer never produced, even though the content stream contains them. Geometry comes from the pen model rather than a guess: PDFium still advances the pen over the missing glyph when placing surviving neighbours, so a dropped glyph starts at the previous survivor's advance-cell right edge and its advance is the gap to the next survivor's origin. With no surviving neighbour on a side, the advance is unknowable; emit zero-width there so presence and stream order are preserved without inserting a synthetic gap."""
groups: list[list[dict]] = []
for site in sites:
if (groups and groups[-1][0]["prev_i"] == site["prev_i"]
and groups[-1][0]["next_i"] == site["next_i"]
and groups[-1][0]["owner"] is site["owner"]):
groups[-1].append(site)
else:
groups.append([site])
for group_value in groups:
owner = group_value[0]["owner"]
prev = chars_by_index.get(group_value[0]["prev_i"]) if group_value[0]["prev_i"] is not None else None
nxt = chars_by_index.get(group_value[0]["next_i"]) if group_value[0]["next_i"] is not None else None
text = "".join(site["t"] for site in group_value) # one char per target codepoint
count_item = len(text)
if not count_item:
continue
if prev is not None:
pen, baseline_y = prev["right"], prev["oy"]
elif nxt is not None:
pen, baseline_y = nxt["ox"], nxt["oy"]
else:
# Whole show op dropped: park at the object box's pen start.
pen, baseline_y = owner["l"], owner["b"]
total = 0.0
if (prev is not None and nxt is not None
and abs(nxt["oy"] - baseline_y) < 0.5 and nxt["ox"] > pen):
total = nxt["ox"] - pen
adv = total / count_item
# Textpage index: fractional, slotted against the owner's own chars
# so the paint-order sort keys (page_order, i) place the run in
# stream position; only order WITHIN the owner object matters.
if prev is not None and prev["obj"] is owner:
base, sgn = prev["i"], 1.0
elif nxt is not None and nxt["obj"] is owner:
base, sgn = nxt["i"], -1.0
elif prev is not None:
base, sgn = prev["i"], 1.0
elif nxt is not None:
base, sgn = nxt["i"], -1.0
else:
base, sgn = -1.0, 1.0
for key_value, char in enumerate(text):
is_ws, is_mn, is_cf = _char_category(char)
glyph_left = pen + adv * key_value
step = (key_value + 1) if sgn > 0 else (count_item - key_value)
raw_chars.append({
"i": base + sgn * step * 1e-3,
"ch": char, "u": ord(char),
"is_gen": False, "synth": True,
"is_ws": is_ws, "is_mn": is_mn, "is_cf": is_cf,
"ox": glyph_left, "oy": baseline_y,
"left": glyph_left, "right": glyph_left + adv,
"top": baseline_y + owner["fs_eff"], "bottom": baseline_y,
# Degenerate ink box: PDFium reports no ink box for the glyph
# (this also keeps it out of the Type-3 extent union).
"box_top": baseline_y, "box_bottom": baseline_y,
"cell_top": baseline_y, "cell_bot": baseline_y,
"w_raw": 0.0, "w_synth": adv,
"obj": owner, "font_name": owner["font_name"],
})
+139
View File
@@ -0,0 +1,139 @@
"""Per-page parallel driver for the charlevel parser.
Wraps the UNMODIFIED per-page pipeline (``_page_pass1`` / ``_page_pass2`` /
``_page_spans``) in a process pool. PDFium's FFI is not thread-safe and its
handles are process-local, so parallelism uses processes, each opening its
own copy of the document.
Parity contract: per-page processing depends on no cross-page state
except the document-wide identity-matrix Type-3 extent union. An empty union
makes ``_apply_type3_sizes`` a no-op, so per-page == whole-document exactly.
Workers run pass 1 + pass 2 per page assuming the union stays empty and
poison the run the moment any page accumulates an extent; the driver then
discards the parallel attempt and reruns the document on the sequential
path, which is the source of truth. Any other worker failure falls back the
same way, so this entry can only ever return sequential-identical output.
Worker startup pays the full package import chain plus its own document
open; ``min_pages`` routes documents too small to amortize that to the
sequential path directly.
"""
from __future__ import annotations
import multiprocessing
import os
from concurrent.futures import ProcessPoolExecutor
from io import BytesIO
from pathlib import Path
from typing import Union
import pypdfium2 as pdfium
import PyPDF2 as _pypdf2 # declared dependency (also imported by pageindex.utils/client)
from .model import Span
from .parser_pdfium_charlevel import (
parse_charlevel_meta,
_PdfDoc,
_page_pass1,
_page_pass2,
_page_spans,
)
_MIN_PARALLEL_PAGES = 64
class _Type3Detected(Exception):
"""A page accumulated an identity-matrix Type-3 extent: the document
needs the cross-page font sizing only the sequential path performs."""
# Per-worker state, set once by _init_worker in each spawned process.
_worker_pdf = None
_worker_pdf_doc = None
_worker_font_maps: dict = {}
def _init_worker(kind: str, payload) -> None:
global _worker_pdf, _worker_pdf_doc, _worker_font_maps
# Open the document exactly as parse_charlevel_meta does, including
# the guarded PyPDF2 open and its separate bytes copy.
if kind == "path":
_worker_pdf = pdfium.PdfDocument(payload)
else:
_worker_pdf = pdfium.PdfDocument(BytesIO(payload))
_worker_pdf_doc = None
if _pypdf2 is not None:
try:
if kind == "path":
_worker_pdf_doc = _PdfDoc(_pypdf2.PdfReader(payload))
else:
_worker_pdf_doc = _PdfDoc(_pypdf2.PdfReader(BytesIO(payload)))
except Exception:
_worker_pdf_doc = None
_worker_font_maps = {}
def _run_page(page_idx: int):
type3_ext: dict = {}
page, raw_chars, page_vb, page_rot = _page_pass1(
_worker_pdf, _worker_pdf_doc, page_idx, type3_ext, _worker_font_maps)
try:
if type3_ext:
raise _Type3Detected(page_idx)
merged = _page_pass2(raw_chars, page_vb, {})
spans = _page_spans(merged)
finally:
page.close()
return spans, (page_vb, page_rot)
def parse_charlevel_meta_parallel(
doc_handle: Union[str, Path, BytesIO],
workers: int | None = None,
min_pages: int = _MIN_PARALLEL_PAGES,
) -> tuple[list[list[Span]], list]:
"""Parallel-when-possible variant of ``parse_charlevel_meta``.
Returns the same ``(pages, page_meta)`` with identical content for
every input. ``workers`` caps the pool size (default: CPU count - 1).
"""
if isinstance(doc_handle, (str, Path)):
src = ("path", str(doc_handle))
elif isinstance(doc_handle, BytesIO):
src = ("bytes", doc_handle.getvalue())
else:
# An already-open PdfDocument cannot be reopened per worker.
return parse_charlevel_meta(doc_handle)
probe = pdfium.PdfDocument(BytesIO(src[1]) if src[0] == "bytes" else src[1])
n_pages = len(probe)
probe.close()
max_w = max(1, (os.cpu_count() or 2) - 1)
w = max(1, min(workers if workers is not None else max_w, max_w, n_pages))
if w <= 1 or n_pages < min_pages:
return parse_charlevel_meta(doc_handle)
executor = ProcessPoolExecutor(
max_workers=w,
mp_context=multiprocessing.get_context("spawn"),
initializer=_init_worker,
initargs=src,
)
try:
results = list(executor.map(_run_page, range(n_pages)))
except Exception:
# _Type3Detected or any worker/pool failure. Cancel what is queued
# and rerun sequentially; in-flight pages finish in their workers
# and are discarded (separate processes, no shared PDFium state).
executor.shutdown(wait=False, cancel_futures=True)
return parse_charlevel_meta(doc_handle)
executor.shutdown()
out = [spans for spans, _meta_entry in results]
meta = [meta_entry for _spans, meta_entry in results]
return out, meta
__all__ = ["parse_charlevel_meta_parallel"]
+50
View File
@@ -0,0 +1,50 @@
"""Per-page pipeline orchestration. For each page, the extractor builds initial lines, computes page statistics,
detects columns, reclusters lines with column awareness, removes line-number
artifacts, recomputes statistics, and assigns reading order.
"""
import math
from typing import Optional
from sortedcontainers import SortedKeyList
from ..clustering import LinesContainer, cluster_lines, build_initial_lines
from ..columns import detect_columns, ColumnDetectionContext, columns_to_x_bounds
from ..model import (
Span,
left_aligned,
right_aligned,
center_aligned,
x_centers_close,
to_number,
Rect,
append_span,
avg_char_width,
Line,
info_weight,
)
from ..stats import column_index_of, PageStats, compute_page_stats
from .page_view import (
PageView,
assign_reading_order,
process_page,
)
from .line_numbers import (
LineNumberCluster,
init_line_number_cluster,
nearest_cluster,
validate_line_number_cluster,
strip_line_numbers,
)
__all__ = [
"assign_reading_order",
"LineNumberCluster",
"init_line_number_cluster",
"nearest_cluster",
"validate_line_number_cluster",
"strip_line_numbers",
"PageView",
"process_page",
]
+162
View File
@@ -0,0 +1,162 @@
"""Line-number column detection and stripping."""
from __future__ import annotations
import math
from typing import Optional
from sortedcontainers import SortedKeyList
from ..model import (
Span,
left_aligned,
right_aligned,
center_aligned,
x_centers_close,
to_number,
Rect,
append_span,
avg_char_width,
Line,
info_weight,
)
# --------------------------------------------------------------------------- #
# Line-number stripper #
# --------------------------------------------------------------------------- #
class LineNumberCluster:
"""Drop-cap or line-number cluster used to detect removable line numbers."""
__slots__ = ("lines", "left", "secondary_slot", "primary_slot", "is_valid_sequence")
def __init__(self, line: Line, candidate_item: float, valid_sequence_flag: bool):
self.lines: list = [line]
self.left: float = line.left_edge()
self.secondary_slot: float = avg_char_width(line)
self.primary_slot: float = candidate_item
self.is_valid_sequence: bool = valid_sequence_flag
def init_line_number_cluster(line: Line) -> LineNumberCluster:
"""Build an initial line-number cluster for a candidate line."""
line_number = to_number(line.primary_slot[0].state_slot)
is_valid_integer = (line_number > 0 and line_number < 1e4 and not math.isnan(line_number) and line_number == math.floor(line_number))
return LineNumberCluster(line, line_number, is_valid_integer)
def nearest_cluster(line: LineNumberCluster, other_line: Optional[LineNumberCluster], candidate_line: Optional[LineNumberCluster]) -> Optional[LineNumberCluster]:
"""Pick the nearer left or right cluster within two character heights."""
distance = (line.left - other_line.left) if other_line is not None else math.inf
candidate_distance = (candidate_line.left - line.left) if candidate_line is not None else math.inf
tol = 2 * line.secondary_slot
if distance > tol and candidate_distance > tol:
return None
return other_line if distance < candidate_distance else candidate_line
def validate_line_number_cluster(rect: Rect, other_lines: list[Line], candidate_line: LineNumberCluster) -> bool:
"""validate a candidate cluster (>= 5 lines, near left edge, bulk of body weight overlapping the cluster's vertical span)."""
if len(candidate_line.lines) < 5:
return False
if candidate_line.left < 0.05 * rect.bbox_width():
return True
empty_line_count = 0
flag = False
top = -math.inf
bot = math.inf
min_gap = math.inf
max_gap = -math.inf
prev: Optional[Line] = None
for cluster_line in candidate_line.lines:
if cluster_line.char_count() - cluster_line.char_stats.primary_slot[1] <= 0:
empty_line_count += 1
first = cluster_line.alignment_slot
if first and first.char_stats.secondary_slot == 3:
flag = True
top = max(top, cluster_line.top_edge())
bot = min(bot, cluster_line.bottom_edge())
if prev is not None:
gap = prev.bottom_edge() - cluster_line.bottom_edge()
min_gap = min(min_gap, gap)
max_gap = max(max_gap, gap)
prev = cluster_line
# Preserve IEEE-754 division for the spacing-ratio test.
if min_gap != 0:
gap_ratio = max_gap / min_gap
elif max_gap != 0:
gap_ratio = math.copysign(math.inf, max_gap)
else:
gap_ratio = math.nan
if empty_line_count < len(candidate_line.lines) / 2 and (not flag or gap_ratio > 1.3):
return False
total = 0.0
covered = 0.0
for line in other_lines:
block_weight = info_weight(line.char_stats)
total += block_weight
if line.bottom_edge() < top and line.top_edge() > bot:
covered += block_weight
return covered >= 0.8 * total
def strip_line_numbers(rect: Rect, other_lines: list[Line]) -> list[Line]:
"""Detect a column of line numbers and strip it. Returns the original lines if no line-numbering pattern is detected."""
# Cluster candidates by ``left`` x-position. SortedKeyList by left.
cluster_tree: SortedKeyList = SortedKeyList(key=lambda line_key: line_key.left)
for line in other_lines:
if len(line.primary_slot) == 0 or len(line.primary_slot[0].state_slot) == 0:
continue
if line.left_edge() > 0.15 * rect.bbox_width():
continue
candidate_cluster = init_line_number_cluster(line)
if not candidate_cluster.is_valid_sequence:
continue
# Equal-left clusters must merge, so predecessor/successor lookup is
# inclusive: successor = first left >= current, predecessor = last left <=
# current. Strict bisect would fragment a fixed-x line-number column.
idx_succ = cluster_tree.bisect_left(candidate_cluster)
successor_cluster: Optional[LineNumberCluster] = (
cluster_tree[idx_succ] if idx_succ < len(cluster_tree) else None
) # type: ignore[assignment]
idx_pred = cluster_tree.bisect_right(candidate_cluster)
neighbor: Optional[LineNumberCluster] = (
cluster_tree[idx_pred - 1] if idx_pred > 0 else None
) # type: ignore[assignment]
match = nearest_cluster(candidate_cluster, neighbor, successor_cluster)
if match is not None:
if match.is_valid_sequence:
match.is_valid_sequence = (candidate_cluster.primary_slot == match.primary_slot + 1)
match.lines.append(line)
match.primary_slot = candidate_cluster.primary_slot
else:
cluster_tree.add(candidate_cluster)
# Find largest valid (Ua) cluster
best: Optional[LineNumberCluster] = None
for cluster in cluster_tree:
if cluster.is_valid_sequence and (best is None or len(cluster.lines) > len(best.lines)):
best = cluster
if best is None or not validate_line_number_cluster(rect, other_lines, best):
return other_lines
# Build output: for each affected line, drop its first span
affected = set(id(line) for line in best.lines)
out: list[Line] = []
for source_line in other_lines:
if id(source_line) not in affected:
out.append(source_line)
continue
new_line = Line()
first_span = source_line.primary_slot[0]
for span in source_line:
if span is first_span:
continue
append_span(new_line, span)
if new_line.char_count() <= 0:
continue
new_line.measure_slot = source_line.measure_slot
out.append(new_line)
return out
+123
View File
@@ -0,0 +1,123 @@
"""Per-page processing driver and reading-order assignment."""
from __future__ import annotations
from typing import Optional
from ..clustering import LinesContainer, cluster_lines, build_initial_lines
from ..columns import detect_columns, ColumnDetectionContext, columns_to_x_bounds
from ..model import (
Span,
left_aligned,
right_aligned,
center_aligned,
x_centers_close,
to_number,
Rect,
append_span,
avg_char_width,
Line,
info_weight,
)
from ..stats import column_index_of, PageStats, compute_page_stats
from .line_numbers import strip_line_numbers
# --------------------------------------------------------------------------- #
# Per-page reading order and paragraph-break flagging #
# --------------------------------------------------------------------------- #
class PageView:
"""Per-page mutable state carried through layout classification."""
__slots__ = (
"bounds", "output_slot", "secondary_slot", "measure_slot", "page_index", "primary_slot", "tertiary_slot", "lines", "blocks",
"text", "previous_slot", "annotations",
"auxiliary_slot", "state_slot", "style_slot", "option_slot", "viewport_box", "rot",
)
def __init__(self, page_num: int, page_bbox: Rect):
self.bounds: Rect = page_bbox
self.output_slot: list = []
self.secondary_slot: list = []
self.measure_slot: bool = False # set when a labeled section appears
self.page_index: int = page_num
self.primary_slot: Optional[PageStats] = None
self.tertiary_slot: list = [] # column rects
self.lines: list = []
self.blocks: list = []
self.text: Optional[list] = None # raw text items reconstructed by parser
self.previous_slot = 0.0
self.annotations = []
# per-page fields used by heading detection and outline assembly:
self.auxiliary_slot: bool = False # marked as references page
self.state_slot: bool = False # has substantive body
self.style_slot: set = set() # set of body-style hashes (sh)
self.option_slot = None # reserved, unused here
# Page viewport for heading coordinates: unrotated view box + /Rotate.
# None -> fallback to the origin-0 upright shortcut.
self.viewport_box: Optional[tuple] = None
self.rot: int = 0
def assign_reading_order(primary_item: PageView, other_items: list) -> None:
"""Assign reading order and paragraph-break flags for a page. The column-aware path expects blocks, not raw lines, because the sort key reads the first child line's column index. Passing raw lines would read a different flag from the first span."""
primary_item.output_slot = other_items
for candidate_item in range(len(other_items)):
setattr(other_items[candidate_item], "orig_index", candidate_item)
primary_item.secondary_slot = list(other_items)
primary_item.secondary_slot.sort(key=lambda sort_block: (column_index_of(sort_block), -sort_block.top_edge(), -sort_block.bottom_edge(), sort_block.left_edge(), sort_block.right_edge()))
# Assign sorted index and paragraph/end-isolated flags to each item.
for idx in range(len(primary_item.secondary_slot)):
candidate_item = primary_item.secondary_slot[idx]
candidate_item.reading_order_index = idx
reference_item = primary_item.secondary_slot[idx + 1] if idx + 1 < len(primary_item.secondary_slot) else None
# Isolated-centered is true when the item is centered on the page and
# either has no successor, is vertically separated from it, or is not
# left/right aligned with it. Non-page-centered items can still be
# isolated if they are centered relative to a page-centered successor.
if candidate_item.alignment_slot and x_centers_close(primary_item.bounds, candidate_item):
candidate_item.isolated_centered = (not reference_item) or (reference_item.top_edge() > candidate_item.bottom_edge()) or (not left_aligned(candidate_item, reference_item, 1) and not right_aligned(candidate_item, reference_item, 1))
else:
candidate_item.isolated_centered = bool(
candidate_item.alignment_slot and reference_item
and not left_aligned(candidate_item, reference_item, 1) and not right_aligned(candidate_item, reference_item, 1)
and center_aligned(candidate_item, reference_item, candidate_item.bbox_width() / 10) and x_centers_close(primary_item.bounds, reference_item)
)
# --------------------------------------------------------------------------- #
# Per-page orchestrator #
# --------------------------------------------------------------------------- #
def process_page(spans: list[Span], page_num: int, page_bbox: Rect) -> PageView:
"""Run the full per-page pipeline on flat span input."""
page = PageView(page_num, page_bbox)
# Raw parser items are kept before clustering
# so document statistics can accumulate the script-family histogram over them (the lines
# below are merged + line-number-stripped, a different character multiset).
page.text = spans
# 1) Build initial lines.
container = LinesContainer()
container.primary_slot = build_initial_lines(spans, page_bbox)
# 2) First clustering pass: no column info yet.
cluster_lines(container, 0.75, [])
# 3) Compute first-pass per-page stats.
page.primary_slot = compute_page_stats(page_bbox, container.primary_slot)
# 4) Detect column rectangles and assign each line's column index.
column_context = ColumnDetectionContext(page_bbox, page.primary_slot, container.primary_slot)
page.tertiary_slot = detect_columns(column_context)
# 5) Second clustering pass: tighter tolerance with column info.
cols = columns_to_x_bounds(page.tertiary_slot)
cluster_lines(container, 0.5, cols)
# 6) Strip line-number column if present.
container.primary_slot = strip_line_numbers(page_bbox, container.primary_slot)
# 7) Recompute stats on cleaned lines.
page.primary_slot = compute_page_stats(page_bbox, container.primary_slot)
page.lines = container.primary_slot
return page
+48
View File
@@ -0,0 +1,48 @@
"""Page-level and document-level layout statistics. The statistics layer computes weighted percentiles, dominant styles, script
families, page spacing measures, and document-wide recurrence signals used by
classification and outline assembly.
"""
import functools
import json
import math
from pathlib import Path
from typing import Optional
from ..model import Span, _format_half_up_one_decimal, Line, info_weight, _max_nan_propagating
from .scripts import (
_SCRIPT_BUCKET_TABLE_PATH,
SCRIPT_BUCKET_TABLE,
char_script_bucket,
SCRIPT_FAMILY_WEIGHTS,
ScriptHistogram,
tally_scripts,
dominant_script_family,
)
from .aggregates import (
_percentile_sample_cmp,
weighted_percentile,
style_key,
PageStats,
compute_page_stats,
DocStats,
compute_doc_stats,
column_index_of,
)
__all__ = [
"weighted_percentile",
"style_key",
"PageStats",
"compute_page_stats",
"DocStats",
"compute_doc_stats",
"column_index_of",
"char_script_bucket",
"tally_scripts",
"dominant_script_family",
"ScriptHistogram",
"SCRIPT_FAMILY_WEIGHTS",
"SCRIPT_BUCKET_TABLE",
]
+313
View File
@@ -0,0 +1,313 @@
"""Page-level and document-level statistics aggregation."""
from __future__ import annotations
import functools
import math
from typing import Optional
from ..model import Span, _format_half_up_one_decimal, Line, info_weight, _max_nan_propagating
from .scripts import (
ScriptHistogram,
tally_scripts,
dominant_script_family,
)
# --------------------------------------------------------------------------- #
# Weighted percentile.
# --------------------------------------------------------------------------- #
def _percentile_sample_cmp(values: tuple[float, float], other_values: tuple[float, float]) -> float:
"""Comparator for weighted percentile samples. NaN comparison results are treated as equal so insertion order is preserved for NaN-valued samples."""
if values[0] != other_values[0]:
return values[0] - other_values[0]
return values[1] - other_values[1]
def weighted_percentile(values: list[tuple[float, float]], other_item: float) -> float:
"""Weighted percentile over ``(value, weight)`` samples. Returns ``NaN`` for empty input or an out-of-range percentile. Ties at the target weight return the average of current and previous values; overshoots return the current value."""
if len(values) <= 0 or other_item < 0 or other_item > 100:
return float("nan")
samples = sorted(values, key=functools.cmp_to_key(_percentile_sample_cmp)) # type: ignore[arg-type]
total = sum(page_value[1] for page_value in samples)
target = total * other_item / 100.0
candidate_item = 0.0
reference_item: Optional[float] = None
for value, weight in samples:
if candidate_item == target:
return value if reference_item is None else (reference_item + value) / 2.0
reference_item = value
candidate_item += weight
if candidate_item > target:
return value
return float("nan") if reference_item is None else reference_item
# --------------------------------------------------------------------------- #
# Per-span style hash #
# --------------------------------------------------------------------------- #
def style_key(span: Span) -> str:
"""Return ``"<fontStyle> <size rounded to 0.1>"`` for same-style span histograms."""
return f"{span.font_style()} {_format_half_up_one_decimal(span.font_size)}"
# --------------------------------------------------------------------------- #
# Per-page statistics #
# --------------------------------------------------------------------------- #
class PageStats:
"""Per-page layout statistics used by column detection and classification."""
__slots__ = ("line_count", "secondary_slot", "tertiary_slot", "previous_slot", "style_slot", "cache_slot", "option_slot", "primary_slot", "measure_slot", "state_slot", "auxiliary_slot")
def __init__(
self,
valid_line_count: int,
total_line_weight: float,
median_overlap_gap: float,
median_line_width: float,
median_char_count: float,
median_area_metric: float,
median_center_y: float,
median_font_size: float,
average_char_width: float,
dominant_font: str,
dominant_style: str,
):
self.line_count = valid_line_count
self.secondary_slot = total_line_weight
self.tertiary_slot = median_overlap_gap
self.previous_slot = median_line_width
self.style_slot = median_char_count
self.cache_slot = median_area_metric
self.option_slot = median_center_y
self.primary_slot = median_font_size
self.measure_slot = average_char_width
self.state_slot = dominant_font
self.auxiliary_slot = dominant_style
# --------------------------------------------------------------------------- #
# Per-page statistics.
# --------------------------------------------------------------------------- #
def compute_page_stats(page, other_lines: list[Line]) -> PageStats:
"""Compute weighted medians plus dominant font/style for one page."""
overlap_gap_samples: list[tuple[float, float]] = [] # bucket-overlap samples
line_width_samples: list[tuple[float, float]] = [] # line-width samples
char_count_samples: list[tuple[float, float]] = [] # line-char-count samples
area_metric_samples: list[tuple[float, float]] = [] # line.U samples
array: list[tuple[float, float]] = [] # y-center samples
font_size_samples: list[tuple[float, float]] = [] # font-size samples
font: dict[str, float] = {} # font-name histogram (weighted)
style: dict[str, float] = {} # style-hash histogram (weighted)
chars = 0 # total char count across spans
width = 0.0 # total width across spans
total = 0.0 # total line weight (sum of tf)
valid = 0 # valid line count
bucket_size = page.bbox_width() / 20.0 # page width / 20 buckets
buckets: list[Optional[Line]] = [None] * 21 # 20 buckets, +1 guard
for line in other_lines:
if line.skew_frac() > 1: # rotated/skewed line: skip
continue
valid += 1
for span in line: # for each span t in line u
span_weight = info_weight(span.char_stats)
span_weight = span_weight * span_weight * span.bbox_height() # weight = tf^2 * height
font[span.font_name] = font.get(span.font_name, 0.0) + span_weight
sty = style_key(span)
style[sty] = style.get(sty, 0.0) + span_weight
chars += span.char_count()
width += span.bbox_width()
line_weight = info_weight(line.char_stats)
total += line_weight
sample = line_weight * line.avg_font_size() # weight = line_weight * font_size
font_size_samples.append((line.avg_font_size(), sample))
line_width_samples.append((line.bbox_width(), sample))
char_count_samples.append((line.char_count(), sample))
area_metric_samples.append((line.cache_slot, line.area())) # NB: this one is weighted by area
array.append((line.center_y(), sample))
# Vertical overlap with the most recent occupant of each horizontal
# bucket. Infinite sentinel boxes skip overlap sampling.
left = line.left_edge()
right = line.right_edge()
if not (left < float("inf") and right > float("-inf")):
continue
if bucket_size <= 0:
# Degenerate zero-width pages skip overlap sampling; downstream
# statistics still include font and line-width samples.
continue
# Clamp infinite sentinels before converting bucket indexes to integers.
bucket_left = 0 if left == float("-inf") else max(0, int(left / bucket_size))
bucket_right = 20 if right == float("inf") else min(20, math.ceil(right / bucket_size))
best_gap = float("inf")
best_prev: Optional[Line] = None
idx = bucket_left
while idx < bucket_right:
prev_in_bucket = buckets[idx]
buckets[idx] = line
idx += 1
if prev_in_bucket is None:
continue
gap = max(prev_in_bucket.bottom_edge(), line.top_edge()) - line.bottom_edge()
if gap < best_gap:
best_gap = gap
best_prev = prev_in_bucket
if best_gap < float("inf") and best_prev is not None:
overlap_gap_samples.append((best_gap, info_weight(best_prev.char_stats) * line_weight))
# Dominant values update only on strictly greater positive weight. This
# keeps the empty value for all-zero pages and preserves first-seen ties.
dominant_font = ""
dominant_font_weight = 0.0
for font_name, weight in font.items():
if weight > dominant_font_weight:
dominant_font_weight = weight
dominant_font = font_name
dominant_style = ""
dominant_style_weight = 0.0
for style_name, weight in style.items():
if weight > dominant_style_weight:
dominant_style_weight = weight
dominant_style = style_name
return PageStats(
valid_line_count=valid,
total_line_weight=total,
median_overlap_gap=weighted_percentile(overlap_gap_samples, 50),
median_line_width=weighted_percentile(line_width_samples, 50),
median_char_count=weighted_percentile(char_count_samples, 50),
median_area_metric=weighted_percentile(area_metric_samples, 50),
median_center_y=weighted_percentile(array, 50),
median_font_size=weighted_percentile(font_size_samples, 50),
# Average char width with IEEE edge cases: no characters with positive
# width yields +inf, and no characters with no width yields NaN.
average_char_width=(width / chars) if chars != 0
else (float("inf") if width > 0 else float("nan")),
dominant_font=dominant_font,
dominant_style=dominant_style,
)
# --------------------------------------------------------------------------- #
# Document-level statistics #
# --------------------------------------------------------------------------- #
class DocStats:
"""Document-level layout statistics: dominant script family, landscape-page count, total valid lines, total line weight, max page line weight, median page total weight, width/height percentiles, center statistic, and median body font size."""
__slots__ = ("tertiary_slot", "style_slot", "cache_slot", "state_slot", "previous_slot", "secondary_slot", "option_slot", "auxiliary_slot", "measure_slot", "primary_slot")
def __init__(
self,
dominant_script: int,
landscape_pages: int,
total_lines: int,
total_weight: float,
max_page_weight: float,
median_page_weight: float,
median_line_width: float,
upper_width_percentile: float,
median_center_y: float,
median_body_font_size: float,
):
self.tertiary_slot = dominant_script
self.style_slot = landscape_pages
self.cache_slot = total_lines
self.state_slot = total_weight
self.previous_slot = max_page_weight
self.secondary_slot = median_page_weight
self.option_slot = median_line_width
self.auxiliary_slot = upper_width_percentile
self.measure_slot = median_center_y
self.primary_slot = median_body_font_size
# --------------------------------------------------------------------------- #
# Document-level statistics.
# --------------------------------------------------------------------------- #
def compute_doc_stats(pages: list) -> DocStats:
"""Compute document-wide recurrence and script statistics from page records."""
script = ScriptHistogram()
landscape = 0
total_lines = 0
total_weight = 0.0
max_weight = 0.0
total_weights: list[tuple[float, float]] = []
bucket_overlaps: list[tuple[float, float]] = []
widths: list[tuple[float, float]] = []
char_counts: list[tuple[float, float]] = []
upper_samples: list[tuple[float, float]] = []
centers: list[tuple[float, float]] = []
font_sizes: list[tuple[float, float]] = []
for query_value in pages:
# Accumulate the script-family histogram over raw parser text items,
# capped at 100k chars. Merged line text can omit line-number spans.
if script.secondary_slot < 100_000:
for span in (query_value.text or []):
if script.secondary_slot >= 100_000:
break
tally_scripts(script, span.text)
if query_value.bounds.bbox_width() > query_value.bounds.bbox_height():
landscape += 1
stats: PageStats = query_value.primary_slot
total_lines += stats.line_count
weight = stats.secondary_slot
total_weight += weight
max_weight = _max_nan_propagating(max_weight, weight)
if stats.line_count <= 0 or weight <= 0:
continue
per_page = min(100.0, weight / stats.line_count)
total_weights.append((weight, per_page))
bucket_overlaps.append((stats.tertiary_slot, per_page))
widths.append((stats.previous_slot, per_page))
char_counts.append((stats.style_slot, per_page))
upper_samples.append((stats.cache_slot, per_page))
centers.append((stats.option_slot, per_page))
font_sizes.append((stats.primary_slot, per_page))
return DocStats(
dominant_script=dominant_script_family(script), # dominant script family
landscape_pages=landscape,
total_lines=total_lines,
total_weight=total_weight,
max_page_weight=max_weight,
median_page_weight=weighted_percentile(total_weights, 50),
# Width and uppercase samples are the document-wide outputs used later.
median_line_width=weighted_percentile(widths, 50),
upper_width_percentile=weighted_percentile(upper_samples, 80), # NB: 80th percentile, not 50
median_center_y=weighted_percentile(centers, 50),
median_body_font_size=weighted_percentile(font_sizes, 50),
)
# --------------------------------------------------------------------------- #
# Column-index accessor #
# --------------------------------------------------------------------------- #
def column_index_of(line: Line) -> int:
"""Return the column index stored on the first child line/span. In normal use this receives a block, so the first child is a line and its stored column index is returned. If a raw line is passed, the same field access still succeeds but reads a different flag; the block-clustering pipeline avoids that path for column-aware ordering."""
if not line.primary_slot:
return -1
first = line.primary_slot[0]
# If ``first`` is another container (block.g[0] is a line) consult its H.
value = getattr(first, "measure_slot", None)
return -1 if value is None else value
+85
View File
@@ -0,0 +1,85 @@
"""Script bucket tables and script histogram helpers."""
from __future__ import annotations
import json
from pathlib import Path
# --------------------------------------------------------------------------- #
# Script-family detector and bucket table.
# --------------------------------------------------------------------------- #
_SCRIPT_BUCKET_TABLE_PATH = Path(__file__).parent.parent / "data" / "script_bucket_table.json"
SCRIPT_BUCKET_TABLE: list[int] = json.loads(_SCRIPT_BUCKET_TABLE_PATH.read_text(encoding="utf-8"))
def char_script_bucket(text: str) -> int:
"""Return the script-bucket id for a single character: empty/multi-character, ASCII punctuation/digit, control, ASCII letter, or a table-driven non-Latin script bucket."""
if not text:
return 0
if len(text) != 1:
return 10
# Astral code points are classified as the multi-unit script bucket.
if ord(text) > 0xFFFF:
return 10
if ("a" <= text <= "z") or ("A" <= text <= "Z"):
return 3
if "\x00" < text < " ":
return 2
if text < "€":
return 1
idx = ord(text) >> 4
if 0 <= idx < len(SCRIPT_BUCKET_TABLE):
return SCRIPT_BUCKET_TABLE[idx]
return 0
# map: each output category -> contributing bucket weights.
# Fixed weights for collapsing script buckets into document script families.
SCRIPT_FAMILY_WEIGHTS: dict[int, list[tuple[int, int]]] = {
2: [(2, 10)],
0: [(0, 1), (2, 1)],
3: [(3, 1), (4, -3), (5, -3), (6, -3), (7, -3), (8, -3), (9, -10)],
4: [(4, 1)],
5: [(5, 1), (6, -10), (7, -10)],
6: [(6, 1)],
7: [(7, 1)],
8: [(8, 1)],
9: [(9, 1)],
10: [(10, 1)],
}
class ScriptHistogram:
"""Script-bucket accumulator with total character count and per-bucket histogram."""
__slots__ = ("secondary_slot", "primary_slot")
def __init__(self):
self.secondary_slot: int = 0
self.primary_slot: list[int] = [0] * 11
def tally_scripts(primary_item: ScriptHistogram, other_text: str) -> None:
"""feed a string into the bucket accumulator."""
for candidate_item in other_text:
primary_item.primary_slot[char_script_bucket(candidate_item)] += 1
primary_item.secondary_slot += 1
def dominant_script_family(primary_item: ScriptHistogram) -> int:
"""best-scoring script-family for the accumulator. Returns the output category 0..10 with the highest weighted score. """
secondary_item = 0
candidate_item = 0
for reference_item in range(11):
entry_item = SCRIPT_FAMILY_WEIGHTS.get(reference_item)
if not entry_item:
continue
score_value = 0
for (script_index, weight) in entry_item:
score_value += weight * primary_item.primary_slot[script_index]
if score_value > candidate_item:
secondary_item = reference_item
candidate_item = score_value
return secondary_item
+54
View File
@@ -0,0 +1,54 @@
"""Document-title detection. The scoring formula is the heart of title detection: score is a product of layout, recurrence, label, script, width, numbering, punctuation, alignment, and page-position factors. Each factor is in roughly ``[0.1, 3.0]-- the product can grow to a few
thousand for a strong title candidate. The factors are documented in the
scoring body. The multilingual title-keyword and institution-word sets are stored in
``data/dictionaries.json`` as ``title`` and ``institution_words``.
"""
import json
import math
import unicodedata
from pathlib import Path
from typing import Optional
from ..model import (
_trim_unicode_ws,
left_aligned,
right_aligned,
center_aligned,
Rect,
last_span,
heading_score,
Line,
last_line_of,
first_span_of,
block_text,
deaccented_text,
letter_count,
dominant_style_of,
info_weight,
is_upper_dominant,
alignment_code,
Block,
)
from ..stats import DocStats, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
from ..tokens import is_superscript_adjacent, clamp_value, enumerate_tokens, jenkins_hash, trie_prefix_match, set_case_fold, TrieConfig, build_trie, tokenize_block, _de_norm, BuiltTrie, is_word_token
from .dicts import (
_DICT_PATH,
_normalize_text_key,
_load_dicts,
INSTITUTION_WORDS,
TITLE_LABEL_TRIE,
)
from .scoring import (
TitleCandidate,
is_cover_like_page,
is_title_candidate_block,
score_title_candidate,
)
from .detect import (
TitleSearchState,
detect_title,
)
__all__ = ["is_title_candidate_block", "score_title_candidate", "TitleSearchState", "TitleCandidate", "detect_title", "is_cover_like_page", "TITLE_LABEL_TRIE", "INSTITUTION_WORDS"]
+140
View File
@@ -0,0 +1,140 @@
"""Document title search over early pages."""
from __future__ import annotations
from typing import Optional
from ..model import (
_trim_unicode_ws,
left_aligned,
right_aligned,
center_aligned,
Rect,
last_span,
heading_score,
Line,
last_line_of,
first_span_of,
block_text,
deaccented_text,
letter_count,
dominant_style_of,
info_weight,
is_upper_dominant,
alignment_code,
Block,
)
from ..tokens import is_superscript_adjacent, clamp_value, enumerate_tokens, jenkins_hash, trie_prefix_match, set_case_fold, TrieConfig, build_trie, tokenize_block, _de_norm, BuiltTrie, is_word_token
from .scoring import (
TitleCandidate,
is_cover_like_page,
is_title_candidate_block,
score_title_candidate,
)
# --------------------------------------------------------------------------- #
# Title detection state.
# --------------------------------------------------------------------------- #
class TitleSearchState:
"""Title-detection state: document, visited blocks, and current best candidate."""
__slots__ = ("tertiary_slot", "primary_slot", "secondary_slot")
def __init__(self, doc):
self.tertiary_slot = doc
self.primary_slot: set = set()
self.secondary_slot: Optional[TitleCandidate] = None
# --------------------------------------------------------------------------- #
# Title-detection driver.
# --------------------------------------------------------------------------- #
def detect_title(doc) -> Optional[TitleCandidate]:
"""Iterate early pages, score title-like block groups, and return the best candidate."""
state = TitleSearchState(doc)
has_seen_da = False # "broke into body" flag
for page in doc.primary_slot:
# Special branch: landscape cover document
if (
doc.secondary_slot.style_slot > len(doc.primary_slot) / 2
and page.page_index <= 1
and page.bounds.bbox_width() > page.bounds.bbox_height()
and page.primary_slot.secondary_slot < 500
):
for idx, block in enumerate(page.secondary_slot):
if (
is_title_candidate_block(block) and id(block) not in state.primary_slot
and heading_score(block) > page.primary_slot.primary_slot - 0.1
):
score_title_candidate(state, page, idx)
break
if (
is_cover_like_page(doc, page)
or (page.page_index <= 1 and len(doc.primary_slot) >= 10 and page.primary_slot.secondary_slot < 0.8 * doc.secondary_slot.secondary_slot)
):
# Cover / front-matter page
for idx, block in enumerate(page.secondary_slot):
if not is_title_candidate_block(block) or id(block) in state.primary_slot:
continue
score = heading_score(block)
if (
(score > doc.secondary_slot.primary_slot + 0.1 and score > page.primary_slot.primary_slot + 0.1)
or (score > doc.secondary_slot.primary_slot + 2 and score > page.primary_slot.primary_slot - 0.1)
or (score > doc.secondary_slot.primary_slot - 0.1 and score > page.primary_slot.primary_slot - 0.1 and block.isolated_centered)
or (score > doc.secondary_slot.primary_slot - 0.1 and score > page.primary_slot.primary_slot - 0.1
and page.page_index <= 1 and page.primary_slot.secondary_slot < 500)
):
score_title_candidate(state, page, idx)
else:
# Body page: only consider initial blocks until we hit body text
local_done = False
for idx, block in enumerate(page.secondary_slot):
if id(block) in state.primary_slot:
continue
score = heading_score(block)
# Block clearly larger than body
size_trigger = (
is_title_candidate_block(block) and (
(score > doc.secondary_slot.primary_slot + 0.1 and score > page.primary_slot.primary_slot + 0.1)
or (score > doc.secondary_slot.primary_slot + 2 and score > page.primary_slot.primary_slot - 0.1)
or (block.isolated_centered and score > page.primary_slot.primary_slot - 0.1)
or (page.page_index == 1 and score > page.primary_slot.primary_slot + 2)
)
)
if size_trigger:
score_title_candidate(state, page, idx)
elif block.is_body_paragraph and not block.isolated_centered:
# Body-break flag: stop scanning once body text is reached.
if not has_seen_da:
if (block.bottom_edge() - page.bounds.bottom_edge() < 2 * page.bounds.bbox_height() / 3):
has_seen_da = False
elif block.line_count() >= 3 and alignment_code(block) == 4:
has_seen_da = True
else:
digit_or_period = 0
tokens = tokenize_block(block)
for token in tokens:
if is_word_token(token) or token.type == 1:
digit_or_period += 1
has_seen_da = digit_or_period >= len(tokens) / 3
has_seen_da = not has_seen_da
if has_seen_da:
local_done = True
break
local_done = True
# Once body text is seen, the flag stays sticky so a later
# body block on this page breaks immediately.
has_seen_da = True
if local_done:
break
if doc.secondary_slot.secondary_slot < 400:
break
return state.secondary_slot
+35
View File
@@ -0,0 +1,35 @@
"""Dictionary tables for title detection."""
from __future__ import annotations
import json
import unicodedata
from pathlib import Path
from ..tokens import is_superscript_adjacent, clamp_value, enumerate_tokens, jenkins_hash, trie_prefix_match, set_case_fold, TrieConfig, build_trie, tokenize_block, _de_norm, BuiltTrie, is_word_token
# --------------------------------------------------------------------------- #
# Load title-label and institution dictionaries #
# --------------------------------------------------------------------------- #
_DICT_PATH = Path(__file__).parent.parent / "data" / "dictionaries.json"
def _normalize_text_key(text: str) -> str:
"""NFKC + strip + collapse-whitespace + lowercase."""
return " ".join(unicodedata.normalize("NFKC", text).strip().split()).lower()
def _load_dicts() -> tuple[BuiltTrie, set[str]]:
raw = json.loads(_DICT_PATH.read_text(encoding="utf-8"))
title_label_trie = build_trie(raw.get("title", []), set_case_fold(TrieConfig(), True))
# institution words use normalized single-token set membership.
# The title-label dictionary stays a trie because it handles the
# multi-token "Title:" match; institution words are single-token only.)
institution_words = set(raw.get("institution_words", []))
return title_label_trie, institution_words
TITLE_LABEL_TRIE, INSTITUTION_WORDS = _load_dicts()
+264
View File
@@ -0,0 +1,264 @@
"""Title-candidate scoring."""
from __future__ import annotations
import math
from ..model import (
_trim_unicode_ws,
left_aligned,
right_aligned,
center_aligned,
Rect,
last_span,
heading_score,
Line,
last_line_of,
first_span_of,
block_text,
deaccented_text,
letter_count,
dominant_style_of,
info_weight,
is_upper_dominant,
alignment_code,
Block,
)
from ..stats import DocStats, column_index_of, tally_scripts, dominant_script_family, ScriptHistogram
from ..tokens import is_superscript_adjacent, clamp_value, enumerate_tokens, jenkins_hash, trie_prefix_match, set_case_fold, TrieConfig, build_trie, tokenize_block, _de_norm, BuiltTrie, is_word_token
from .dicts import (
INSTITUTION_WORDS,
TITLE_LABEL_TRIE,
)
# --------------------------------------------------------------------------- #
# Title candidate state container #
# --------------------------------------------------------------------------- #
class TitleCandidate:
"""Best title candidate so far: page, contributing blocks, and score."""
__slots__ = ("page", "output_slot", "score")
def __init__(self, page, blocks: list[Block], score_value: float):
self.page = page
self.output_slot = blocks
self.score = score_value
def to_string(self) -> str:
"""Join contributing blocks into the displayed title string, inserting one inter-block space only after the accumulator is non-empty."""
primary_item = ""
for block in self.output_slot:
if primary_item:
primary_item += " "
primary_item += _trim_unicode_ws(tokenize_block(block).to_string())
return primary_item
def __str__(self) -> str:
return self.to_string()
# --------------------------------------------------------------------------- #
# Cover-like page predicate #
# --------------------------------------------------------------------------- #
def is_cover_like_page(doc, page) -> bool:
"""Return whether a page is sparse enough to behave like a cover page."""
if getattr(page, "measure_slot", False):
return False
threshold = 0.5 * min(doc.secondary_slot.secondary_slot, 5e3)
if page.page_index <= 1 and page.primary_slot.secondary_slot < threshold:
return True
early_limit = 1 + min(15, len(doc.primary_slot) / 5)
return page.page_index < early_limit and page.primary_slot.secondary_slot < 0.8 * threshold
# --------------------------------------------------------------------------- #
# xp: candidate-block filter #
# --------------------------------------------------------------------------- #
def is_title_candidate_block(block: Block) -> bool:
"""Return whether ``block`` can be considered as a document-title candidate."""
return (
letter_count(block.char_stats) > 0
and block.skew_frac() < 1
and block.type == 0
and block.char_count() < 400
and block.bbox_height() < 2 * block.bbox_width()
)
# --------------------------------------------------------------------------- #
# yp: multiplicative scoring for a candidate group #
# --------------------------------------------------------------------------- #
def score_title_candidate(zp_state, page, index: int) -> None:
"""Score a candidate block group and update the title-search state."""
doc = zp_state.tertiary_slot
blocks = page.secondary_slot # sorted blocks
title_block = blocks[index]
title_group: list[Block] = [title_block]
# Try to extend with next block if alignment / style / vertical proximity match
if index + 1 < len(blocks):
next_item = blocks[index + 1]
title_score = heading_score(title_block)
height = title_block.avg_font_size()
# Two acceptance conditions:
if (
(abs(title_score - heading_score(next_item)) < 0.1
and dominant_style_of(title_block) == dominant_style_of(next_item)
and title_block.bottom_edge() - next_item.top_edge() < height)
or (
title_score > doc.secondary_slot.primary_slot + 5
and title_score > page.primary_slot.primary_slot + 1
and abs(height - next_item.avg_font_size()) < 0.1
and title_block.bottom_edge() - next_item.top_edge() < 0.5 * height
)
):
tolerance = 0.1 * height
align_value = alignment_code(title_block)
next_alignment = alignment_code(next_item)
if (
(left_aligned(title_block, next_item, tolerance)
and align_value in (1, 2) and next_alignment in (1, 2))
or (right_aligned(title_block, next_item, tolerance)
and align_value in (1, 4) and next_alignment in (1, 4))
or (center_aligned(title_block, next_item, tolerance)
and title_block.alignment_slot and next_item.alignment_slot)
):
title_group.append(next_item)
group = title_group
for measure_item in group:
zp_state.primary_slot.add(id(measure_item))
doc_state = zp_state.tertiary_slot
previous_block = blocks[index - 1] if index - 1 >= 0 else None
# Accumulate statistics over the title group
max_heading_score = 0
max_width = 0.0
consecutive = 0
max_consecutive = 0
bracket_count = 0
total_tokens = 0
right_pen = 1.0
email_count = 0
for result_value in group:
max_heading_score = max(max_heading_score, heading_score(result_value))
max_width = max(max_width, result_value.bbox_width())
title_tokens_view = tokenize_block(result_value)
for entry in enumerate_tokens(title_tokens_view):
sample_item = entry["token"]
total_tokens += 1
if is_word_token(sample_item):
consecutive += 1
max_consecutive = max(max_consecutive, consecutive)
if sample_item.boundary_slot:
bracket_count += 1
# email detection: "@" followed by word "." word (4 tokens)
if sample_item.str == "@" and entry["index"] + 3 < title_tokens_view.length:
next_token = title_tokens_view.token_at(entry["index"] + 1)
dot = title_tokens_view.token_at(entry["index"] + 2)
after = title_tokens_view.token_at(entry["index"] + 3)
if (
next_token is not None and dot is not None and after is not None
and next_token.type == 2 and dot.str == "." and after.type == 2
):
email_count += 1
else:
consecutive = 0
if alignment_code(result_value) == 4:
# The line count is structurally positive here. Keep the fallback so
# a degenerate line cannot raise during title scoring.
right_pen /= result_value.line_count() or 1
if total_tokens <= 0:
return
# Multiplicative factors
len_value = clamp_value(total_tokens * total_tokens / 16.0, 0.5, 1.0)
# Page width should be positive. Keep IEEE-style Infinity/NaN behavior for
# degenerate pages instead of raising during scoring.
width_ratio_sq = (max_width / page.bounds.bbox_width()) if page.bounds.bbox_width() else (math.inf if max_width > 0 else math.nan)
width_ratio_sq *= width_ratio_sq
bracket = bracket_count / total_tokens
bracket_factor = max(0.1, 1 - 9 * bracket * bracket) / max(1, max_consecutive - 2)
page_pos = max(0.1, 1 - 2 * (page.page_index - 1) / max(1, len(doc_state.primary_slot)))
# Doc-wide height is positive in normal inputs. The epsilon prevents a
# degenerate input from raising and still yields the minimum density factor.
page_density_ratio = page.primary_slot.secondary_slot / max(1e-6, doc_state.secondary_slot.secondary_slot)
density_factor = max(0.5, 1 - page_density_ratio * page_density_ratio) * (1 + clamp_value((0.25 - page_density_ratio) / 0.15, 0, 1))
# Page top/height is positive in normal inputs. Degenerate pages take the
# minimum top-position factor instead of raising.
top = max(0.1, group[0].top_edge() / page.bounds.top_edge()) if page.bounds.top_edge() else 0.1
# Abbreviation penalty: count adjacent single-char + delimiter pairs
abbrev = 0
for block in group:
tokens = tokenize_block(block)
previous = None
for token in tokens:
if previous is not None and len(token.str) <= 1 and is_superscript_adjacent(previous, token):
abbrev += 1
previous = token
factor = clamp_value(1.0 / max(1, abbrev), 0.3, 1.0)
# Recurrence penalty: first block's normalized text appears how often?
# The histogram uses the same normalized text hash as the document-wide
# ghost-text map.
norm_text = jenkins_hash(deaccented_text(group[0]))
recurrence_count = doc_state.tertiary_slot.get(norm_text, 0) if hasattr(doc_state, "tertiary_slot") and isinstance(doc_state.tertiary_slot, dict) else 0
ratio = recurrence_count / max(1, len(doc_state.primary_slot))
adj = total_tokens - 3
recurrence_factor = 1 - 0.5 * clamp_value(ratio / 0.3, 0, 1) * (1 / max(1, adj * adj))
# Institution-word penalty (non-first-page)
institution = 1.0
if is_cover_like_page(doc_state, page):
inst_hits = 0
for block in group:
for token in tokenize_block(block):
# single-token
# Match using the same lowercase + diacritic-stripped form as
# the institution-word set.
if _de_norm(token.str, True) in INSTITUTION_WORDS:
inst_hits += 1
institution = 1.0 / (1 + inst_hits)
# "Title:" label bonus from previous block
label = 1.0
if previous_block is not None:
prev_tokens = tokenize_block(previous_block)
if prev_tokens.length <= 3 and trie_prefix_match(TITLE_LABEL_TRIE, prev_tokens) is not None:
label = 3.0
# Email penalty
email = 1.0 / ((1 + email_count) ** 2)
# Script-family match: build a script histogram over the candidate group's text and
# compare the candidate script family against the document script family.
script_acc = ScriptHistogram()
for result_value in group:
tally_scripts(script_acc, block_text(result_value))
script = 1.0 if dominant_script_family(script_acc) == doc_state.secondary_slot.tertiary_slot else 0.5
score = (
max_heading_score * len_value * width_ratio_sq * right_pen * bracket_factor * page_pos
* density_factor * top * factor * recurrence_factor * institution * label
* email * script
)
if zp_state.secondary_slot is None or score > zp_state.secondary_slot.score:
zp_state.secondary_slot = TitleCandidate(page, group, score)
+105
View File
@@ -0,0 +1,105 @@
"""Tokenizer subsystem. The tokenizer is character-driven: it walks each character of each line,
uses category-transition tolerances to decide when the current token can
extend, and closes tokens when script, punctuation, or spacing transitions
require a boundary. Tokens keep back-references to the contributing line and span offsets so the
visible text can be reconstructed and cross-line tokens, such as a word broken
by a hyphen across two lines, can be stitched.
"""
import unicodedata
from typing import Any, Iterable, Iterator, Optional
from ..model import (
_strip_diacritics,
avg_char_width2,
intervals_overlap,
to_number,
rect_union,
EMPTY_RECT,
avg_char_width,
Line,
char_category,
is_word_category,
is_punct_category,
letter_count,
punct_count,
info_weight,
Block,
)
from .token_types import (
SCRIPT_FAMILY_MAP,
_build_gap_tolerance_grid,
GAP_TOLERANCE_GRID,
can_extend_token,
TokenAnchor,
last_token_anchor,
first_anchor_span,
is_char_token,
is_word_token,
is_trimmable_token,
token_numeric_value,
Token,
TokenView,
wrap_tokens,
enumerate_tokens,
first_token,
last_token,
)
from .tokenizer import (
LineTokenizer,
tokenize_block,
clamp_value,
is_superscript_adjacent,
)
from .tries import (
_de_norm,
TrieConfig,
BuiltTrie,
set_reverse,
set_case_fold,
TrieNode,
trie_insert_step,
trie_walk_step,
aho_corasick_match,
aho_corasick_tokens,
TrieBuilder,
_trie_insert_entry,
trie_bulk_insert,
_trie_finalize,
build_trie,
trie_prefix_match,
_trie_full_match,
trie_full_match,
strip_trie_match,
strip_leading_if_in,
COMMA_CHARS,
strip_trailing_comma,
is_comma_token,
trim_trailing_punct,
)
from .hashing import (
_FH_MASK,
_to_uint32,
_to_int32,
_int32_xor,
_int32_left_shift,
_uint32_right_shift,
_little_endian_signed_word,
_utf8_bytes_from_utf16_units,
_jenkins_mix,
jenkins_hash,
)
__all__ = [
# state machine
"GAP_TOLERANCE_GRID", "can_extend_token", "TokenAnchor", "last_token_anchor", "first_anchor_span",
"is_char_token", "is_word_token", "is_trimmable_token", "token_numeric_value", "Token",
"TokenView", "wrap_tokens", "enumerate_tokens", "first_token", "last_token",
"LineTokenizer", "tokenize_block",
"clamp_value", "is_superscript_adjacent", "jenkins_hash",
# trie
"TrieConfig", "BuiltTrie", "set_reverse", "set_case_fold", "TrieNode", "trie_insert_step", "trie_walk_step", "build_trie", "trie_prefix_match", "trie_full_match",
"strip_trie_match", "strip_leading_if_in", "strip_trailing_comma", "trim_trailing_punct", "COMMA_CHARS",
"SCRIPT_FAMILY_MAP", "GAP_TOLERANCE_GRID",
]
+139
View File
@@ -0,0 +1,139 @@
"""32-bit integer helpers and the Jenkins string hash."""
from __future__ import annotations
# --------------------------------------------------------------------------- #
# Jenkins lookup2 string hash (UTF-8 bytes -> signed 32-bit int).
# --------------------------------------------------------------------------- #
_FH_MASK = 0xFFFFFFFF
def _to_uint32(number: int) -> int:
return number & _FH_MASK
def _to_int32(number: int) -> int:
number &= _FH_MASK
return number - 0x100000000 if number >= 0x80000000 else number
def _int32_xor(number: int, other_number: int) -> int:
"""ToInt32 of the 32-bit xor of the operands' uint32 forms."""
return _to_int32(_to_uint32(number) ^ _to_uint32(other_number))
def _int32_left_shift(number: int, other_number: int) -> int:
"""(signed 32-bit result)."""
return _to_int32((_to_uint32(number) << (other_number & 31)) & _FH_MASK)
def _uint32_right_shift(number: int, other_number: int) -> int:
"""(unsigned right shift)."""
return _to_uint32(number) >> (other_number & 31)
def _little_endian_signed_word(byte_values: list, off: int) -> int:
"""Return a little-endian four-byte word with each byte sign-extended."""
def _sign_extend_byte(number: int) -> int:
return number - 256 if number > 127 else number
return (_sign_extend_byte(byte_values[off]) + (_sign_extend_byte(byte_values[off + 1]) << 8)
+ (_sign_extend_byte(byte_values[off + 2]) << 16) + (_sign_extend_byte(byte_values[off + 3]) << 24))
def _utf8_bytes_from_utf16_units(text: str) -> list[int]:
"""Encode by walking UTF-16 code units, preserving lone surrogates."""
raw = text.encode("utf-16-le", "surrogatepass")
units = [raw[index] | (raw[index + 1] << 8) for index in range(0, len(raw), 2)]
output_bytes: list[int] = []
index = 0
while index < len(units):
value = units[index]
if value < 128:
output_bytes.append(value)
elif value < 2048:
output_bytes.append((value >> 6) | 192)
output_bytes.append((value & 63) | 128)
else:
if (
(value & 0xFC00) == 0xD800
and index + 1 < len(units)
and (units[index + 1] & 0xFC00) == 0xDC00
):
index += 1
value = 0x10000 + ((value & 1023) << 10) + (units[index] & 1023)
output_bytes.append((value >> 18) | 240)
output_bytes.append(((value >> 12) & 63) | 128)
else:
output_bytes.append((value >> 12) | 224)
output_bytes.append(((value >> 6) & 63) | 128)
output_bytes.append((value & 63) | 128)
index += 1
return output_bytes
def _jenkins_mix(mix_state: list) -> int:
"""Jenkins lookup2 mix over the 3-word state ``[a, b, c]``."""
secondary_item, candidate_item, reference_item = mix_state
secondary_item = _int32_xor(secondary_item - candidate_item - reference_item, _uint32_right_shift(reference_item, 13))
candidate_item = _int32_xor(candidate_item - reference_item - secondary_item, _int32_left_shift(secondary_item, 8))
reference_item = reference_item - secondary_item
reference_item = _int32_xor(reference_item - candidate_item, _uint32_right_shift(candidate_item, 13))
secondary_item = secondary_item - candidate_item
secondary_item = secondary_item - reference_item
secondary_item = _int32_xor(secondary_item, _uint32_right_shift(reference_item, 12))
candidate_item = _int32_xor(candidate_item - reference_item - secondary_item, _int32_left_shift(secondary_item, 16))
reference_item = reference_item - secondary_item
reference_item = _int32_xor(reference_item - candidate_item, _uint32_right_shift(candidate_item, 5))
secondary_item = secondary_item - candidate_item
secondary_item = secondary_item - reference_item
secondary_item = _int32_xor(secondary_item, _uint32_right_shift(reference_item, 3))
candidate_item = _int32_xor(candidate_item - reference_item - secondary_item, _int32_left_shift(secondary_item, 10))
reference_item = reference_item - secondary_item
reference_item = _int32_xor(reference_item - candidate_item, _uint32_right_shift(candidate_item, 15))
mix_state[0], mix_state[1], mix_state[2] = secondary_item, candidate_item, reference_item
return reference_item
def jenkins_hash(text: str) -> int:
"""Encode text through the package UTF-16/UTF-8 byte path, then run Jenkins lookup2."""
byte_values = _utf8_bytes_from_utf16_units(text)
count_item = len(byte_values)
mix_state = [-1640531527, -1640531527, 314159265] # 0x9E3779B9, 0x9E3779B9, seed
off = 0
entry_item = count_item
while entry_item >= 12:
mix_state[0] = mix_state[0] + _little_endian_signed_word(byte_values, off)
mix_state[1] = mix_state[1] + _little_endian_signed_word(byte_values, off + 4)
mix_state[2] = mix_state[2] + _little_endian_signed_word(byte_values, off + 8)
_jenkins_mix(mix_state)
entry_item -= 12
off += 12
mix_state[2] = mix_state[2] + count_item
# Tail-byte mixing follows Jenkins lookup2's fall-through layout.
if entry_item >= 11:
mix_state[2] = mix_state[2] + _int32_left_shift(byte_values[off + 10], 24)
if entry_item >= 10:
mix_state[2] = mix_state[2] + ((byte_values[off + 9] & 255) << 16)
if entry_item >= 9:
mix_state[2] = mix_state[2] + ((byte_values[off + 8] & 255) << 8)
if entry_item >= 8:
mix_state[1] = mix_state[1] + _little_endian_signed_word(byte_values, off + 4)
mix_state[0] = mix_state[0] + _little_endian_signed_word(byte_values, off)
elif entry_item >= 4:
if entry_item >= 7:
mix_state[1] = mix_state[1] + ((byte_values[off + 6] & 255) << 16)
if entry_item >= 6:
mix_state[1] = mix_state[1] + ((byte_values[off + 5] & 255) << 8)
if entry_item >= 5:
mix_state[1] = mix_state[1] + (byte_values[off + 4] & 255)
mix_state[0] = mix_state[0] + _little_endian_signed_word(byte_values, off)
else:
if entry_item >= 3:
mix_state[0] = mix_state[0] + ((byte_values[off + 2] & 255) << 16)
if entry_item >= 2:
mix_state[0] = mix_state[0] + ((byte_values[off + 1] & 255) << 8)
if entry_item >= 1:
mix_state[0] = mix_state[0] + (byte_values[off] & 255)
return _jenkins_mix(mix_state)
+243
View File
@@ -0,0 +1,243 @@
"""Token types, anchors, and token-view utilities."""
from __future__ import annotations
from typing import Any, Iterable, Iterator, Optional
from ..model import (
_strip_diacritics,
avg_char_width2,
intervals_overlap,
to_number,
rect_union,
EMPTY_RECT,
avg_char_width,
Line,
char_category,
is_word_category,
is_punct_category,
letter_count,
punct_count,
info_weight,
Block,
)
# --------------------------------------------------------------------------- #
# Character-category transition table #
# --------------------------------------------------------------------------- #
# Map 12 character categories down to script-family buckets used by statistics.
SCRIPT_FAMILY_MAP = [0, 1, 2, 2, 2, 2, 3, 4, 5, 6, 7, 7, 8]
def _build_gap_tolerance_grid() -> list[list[float]]:
"""Return the fractional gap tolerance for adjacent character categories."""
token_value = [[0.0] * 12 for _ in range(12)]
# Small punctuation-to-mark transition weights.
for candidate_item in (1, 2, 3, 4):
token_value[candidate_item][5] = 0.16
token_value[candidate_item][6] = 0.16
token_value[3][2] = 0.1
token_value[6][2] = 0.1
token_value[6][3] = 0.1
token_value[8][2] = 0.1
token_value[8][3] = 0.1
return token_value
GAP_TOLERANCE_GRID = _build_gap_tolerance_grid()
def can_extend_token(number: int, other_number: int, candidate_text: str) -> bool:
"""Return whether the current token can extend with ``candidate_text``."""
from ..stats import char_script_bucket
if other_number == 4 and char_script_bucket(candidate_text) == 5:
return False
if other_number == number and not is_punct_category(other_number):
return True
if is_word_category(number) and is_word_category(other_number):
return True
return False
# --------------------------------------------------------------------------- #
# Token span anchors.
# --------------------------------------------------------------------------- #
class TokenAnchor:
"""Cross-line anchor range attached to a token."""
__slots__ = ("line", "anchor_span", "start_offset", "primary_slot")
def __init__(self, line: Line, anchor_span_value, start_offset_value: int, next_number: int):
self.line = line
self.anchor_span = anchor_span_value
self.start_offset = start_offset_value
self.primary_slot = next_number
def last_token_anchor(token: "Token") -> TokenAnchor:
"""Return the token's last cross-line anchor entry."""
return token.anchor_ranges[-1]
def first_anchor_span(token: "Token"):
"""Return the anchor span from the token's first cross-line entry."""
return token.anchor_ranges[0].anchor_span
# --------------------------------------------------------------------------- #
# Token kind predicates #
# --------------------------------------------------------------------------- #
def is_char_token(token: "Token") -> bool:
"""Return True for digit or letter tokens."""
return token.type == 1 or token.type == 2
def is_word_token(token: "Token") -> bool:
"""Return True for merged word-like tokens: word, number-word, or symbolic token kinds."""
return token.type in (3, 4, 5)
def is_trimmable_token(token: "Token") -> bool:
"""Return True for word, number-word, or colon tokens that can be trimmed from phrase edges."""
return token.type == 3 or token.type == 4 or token.str == ":"
def token_numeric_value(token: "Token") -> float:
"""numeric value of token, NaN if non-numeric."""
return to_number(token.str)
# --------------------------------------------------------------------------- #
# Token #
# --------------------------------------------------------------------------- #
class Token:
"""One token. It stores the token kind, raw text, contributing line/span anchors, bracket attachment flag, and first/last character categories."""
__slots__ = ("type", "str", "anchor_ranges", "boundary_slot", "primary_slot", "secondary_slot")
def __init__(self, type_: int, candidate_text: str, anchor_ranges_value: list[TokenAnchor], boundary_flag: bool, previous_number: int, limit_number: int):
self.type = type_
self.str = candidate_text
self.anchor_ranges = anchor_ranges_value
self.boundary_slot = boundary_flag
self.primary_slot = previous_number
self.secondary_slot = limit_number
def line(self) -> Line:
"""Line of the first origin-span back-reference."""
return self.anchor_ranges[0].line
def __repr__(self) -> str: # diagnostic
return f"<Token t={self.type} {self.str!r} ba={self.boundary_slot}>"
# --------------------------------------------------------------------------- #
# Directional token view #
# --------------------------------------------------------------------------- #
class TokenView:
"""Sliceable, directional view over a token array. Supports forward / reverse iteration via ``dir`` = +1 / -1. ``slice`` and ``reverse`` produce new views without copying. """
__slots__ = ("primary_slot", "start", "end", "dir", "length")
def __init__(self, other_tokens: list[Token], start: int, end: int, dir_: int):
self.primary_slot = other_tokens
self.start = start
self.end = end
self.dir = dir_
self.length = (end - start) // dir_ if dir_ != 0 else 0
def __iter__(self) -> Iterator[Token]:
secondary_item = self.start
while secondary_item != self.end:
yield self.primary_slot[secondary_item]
secondary_item += self.dir
def token_at(self, other_number: int) -> Optional[Token]:
if other_number < 0 or other_number >= self.length:
return None
return self.primary_slot[self.start + other_number * self.dir]
def __getitem__(self, other_number: int) -> Optional[Token]:
return self.token_at(other_number)
def __len__(self) -> int:
return self.length
def __bool__(self) -> bool:
return self.length > 0
def __str__(self) -> str:
parts: list[str] = []
for token in self:
parts.append(token.str)
if token.boundary_slot:
parts.append(" ")
return "".join(parts)
def slice(self, other_number: int = 0, candidate_number: int = 0) -> "TokenView":
"""Bounds-clamped directional slice. Args follow Unicode-compatible semantics: a > 0 -> from index a a < 0 -> from end-relative a = 0 -> from start b > 0 -> to index b b < 0 -> end-relative b = 0 -> to end """
if other_number > 0:
slice_start = self.start + other_number * self.dir
elif other_number < 0:
slice_start = self.end + other_number * self.dir
else:
slice_start = self.start
if slice_start * self.dir < self.start * self.dir:
slice_start = self.start
if slice_start * self.dir > self.end * self.dir:
slice_start = self.end
if candidate_number > 0:
slice_end = self.start + candidate_number * self.dir
elif candidate_number < 0:
slice_end = self.end + candidate_number * self.dir
else:
slice_end = self.end
if slice_end * self.dir < slice_start * self.dir:
slice_end = slice_start
if slice_end * self.dir > self.end * self.dir:
slice_end = self.end
return TokenView(self.primary_slot, slice_start, slice_end, self.dir)
def reverse(self) -> "TokenView":
return TokenView(self.primary_slot, self.end - self.dir, self.start - self.dir, -self.dir)
def to_string(self) -> str:
return str(self)
def wrap_tokens(tokens: list[Token]) -> TokenView:
"""Wrap a list of tokens as a forward token view."""
return TokenView(tokens, 0, len(tokens), 1)
def enumerate_tokens(tokens: TokenView) -> Iterator[dict]:
"""Enumerate a token view yielding indexed token records."""
token = tokens.primary_slot
start = tokens.start
end = tokens.end
step = tokens.dir
cursor = start
while cursor != end:
yield {"index": (cursor - start) // step, "token": token[cursor]}
cursor += step
def first_token(tokens: TokenView) -> Optional[Token]:
"""Return the first token, or None."""
return tokens.primary_slot[tokens.start] if tokens.length > 0 else None
def last_token(tokens: TokenView) -> Optional[Token]:
"""Return the last token, or None."""
return tokens.primary_slot[tokens.end - tokens.dir] if tokens.length > 0 else None
+271
View File
@@ -0,0 +1,271 @@
"""Line tokenization into word, char, and number tokens."""
from __future__ import annotations
import unicodedata
from typing import Any, Iterable, Iterator, Optional
from ..model import (
_strip_diacritics,
avg_char_width2,
intervals_overlap,
to_number,
rect_union,
EMPTY_RECT,
avg_char_width,
Line,
char_category,
is_word_category,
is_punct_category,
letter_count,
punct_count,
info_weight,
Block,
)
from .token_types import (
SCRIPT_FAMILY_MAP,
GAP_TOLERANCE_GRID,
can_extend_token,
TokenAnchor,
last_token_anchor,
first_anchor_span,
Token,
TokenView,
wrap_tokens,
)
# --------------------------------------------------------------------------- #
# Line tokenizer state machine #
# --------------------------------------------------------------------------- #
class LineTokenizer:
"""Line-tokenizer state machine with line/span anchors for reconstruction."""
__slots__ = ("tertiary_slot", "secondary_slot", "cache_slot", "auxiliary_slot", "option_slot", "marker_slot", "primary_slot", "previous_slot", "state_slot", "style_slot", "measure_slot")
def __init__(self):
self.tertiary_slot: list[Token] = []
self.secondary_slot = None # last anchor span
self.cache_slot: Optional[Line] = None
self.auxiliary_slot = -1
self.option_slot = -1
self.marker_slot: list[TokenAnchor] = []
self.primary_slot = ""
self.previous_slot = False
self.state_slot = 0 # last-char category
self.style_slot = 0 # first-char category
self.measure_slot = 0 # type-hint accumulator
# --- inner state ops --------------------------------------------------
def _close_anchor_range(self) -> None:
"""Close the current anchor range into the in-flight token and reset offsets."""
self.marker_slot.append(TokenAnchor(self.cache_slot, self.secondary_slot, self.auxiliary_slot, self.option_slot))
self.auxiliary_slot = self.option_slot = -1
def _close_token(self, boundary_flag: bool) -> None:
"""Close the in-flight token into the token list."""
if self.auxiliary_slot >= 0:
self._close_anchor_range()
self.tertiary_slot.append(Token(self.measure_slot, self.primary_slot, self.marker_slot, boundary_flag, self.style_slot, self.state_slot))
self.marker_slot = []
self.primary_slot = ""
self.measure_slot = 0
self.style_slot = 0
self.state_slot = 0
def _accumulate_char(self, other_text: str, candidate_number: int) -> None:
"""Append a character and update the in-flight token kind from the category map."""
if len(self.primary_slot) == 1 and self.state_slot == 5:
# If the in-flight token is a single mark, attach it before the new
# character so combining marks bind to the following letter.
self.primary_slot = other_text + self.primary_slot
self.style_slot = candidate_number
else:
if not self.primary_slot:
self.style_slot = candidate_number
self.primary_slot += other_text
self.state_slot = candidate_number
cat = SCRIPT_FAMILY_MAP[candidate_number]
if self.measure_slot == 0:
self.measure_slot = cat
elif self.measure_slot == 1 and cat != 1:
self.measure_slot = 2
self.previous_slot = False
def _advance_char(self, other_text: str, candidate_number: int) -> None:
"""Advance the tokenizer with one character. Whitespace sets the pending-boundary flag; non-whitespace either extends or closes the current token."""
reference_item = char_category(other_text)
if reference_item == 10:
# whitespace
self.previous_slot = True
return
if self.previous_slot and self.primary_slot:
# If the last non-whitespace category and the current category cannot
# belong to the same word-like token, close the current token.
if not (reference_item == 5 and is_word_category(self.state_slot)):
self._close_token(True)
# Soft-hyphen rejoin across lines: if there is no in-flight token, the
# current char is lowercase, and the previous tokens were a word plus
# "-" ending on another line, undo the split and continue that word.
if not self.primary_slot and len(self.tertiary_slot) >= 2 and reference_item == 3:
entry_item = self.tertiary_slot[-1]
token = self.tertiary_slot[-2]
if (
token.secondary_slot == 3
and not token.boundary_slot
and entry_item.str == "-"
and last_token_anchor(entry_item).line is not self.cache_slot
):
self.tertiary_slot.pop() # drop "-"
entry_item = self.tertiary_slot.pop() # pop word
self.measure_slot = entry_item.type
self.primary_slot = entry_item.str
self.marker_slot = entry_item.anchor_ranges
self.style_slot = entry_item.primary_slot
self.state_slot = entry_item.secondary_slot
self.previous_slot = False
self._accumulate_char(other_text, reference_item)
self.auxiliary_slot = self.option_slot = candidate_number
return
if self.primary_slot:
if can_extend_token(self.state_slot, reference_item, other_text):
self._accumulate_char(other_text, reference_item)
if self.auxiliary_slot < 0:
self.auxiliary_slot = candidate_number
self.option_slot = candidate_number
else:
self._close_token(False)
self._accumulate_char(other_text, reference_item)
self.auxiliary_slot = self.option_slot = candidate_number
else:
self._accumulate_char(other_text, reference_item)
self.auxiliary_slot = self.option_slot = candidate_number
# --- public API -------------------------------------------------------
def add_line(self, other_line: Line) -> "LineTokenizer":
"""Walk one line and append its token contribution."""
line = self.tertiary_slot[-1] if self.tertiary_slot else None
if self.primary_slot:
# Close in-flight; a trailing hyphen can glue to the next line only
# when the previous token was not already bracket-attached.
self._close_token(self.primary_slot != "-" or line is None or line.boundary_slot)
self.cache_slot = other_line
# Single-codepoint pending combining mark.
pending = None # type: Optional[Any]
for index in range(len(other_line.primary_slot)):
span = other_line.primary_slot[index]
if span.char_count() <= 0:
continue
# Drop solitary combining marks (last-character category is 5)
if pending is None and span.char_count() == 1 and span.char_stats.secondary_slot == 5:
pending = span
continue
# Drop the bullet-then-content kerning glitch (layout branch:
# single-character token, previous category is 11, and next span overlaps horizontally)
if (
index + 1 < len(other_line.primary_slot)
and span.char_count() == 1
and span.char_stats.secondary_slot == 11
and span.left_edge() >= other_line.primary_slot[index + 1].left_edge()
and span.center_x() < other_line.primary_slot[index + 1].right_edge()
):
continue
if self.secondary_slot is not None and self.primary_slot:
# Decide whether the new span continues the same token
if (
span.left_edge() <= self.secondary_slot.right_edge() + 0.1 * avg_char_width2(self.secondary_slot)
and (
abs(span.bottom_edge() - self.secondary_slot.bottom_edge()) < 0.1
or abs(span.center_y() - self.secondary_slot.center_y()) < 0.1
)
and self.secondary_slot.primary_slot == span.primary_slot
):
# Continue: close the current cross-line anchor entry and switch anchor.
self._close_anchor_range()
self.secondary_slot = span
self.previous_slot = False
else:
gap_tolerance = (GAP_TOLERANCE_GRID[self.secondary_slot.char_stats.tertiary_slot][span.char_stats.secondary_slot] or 0.12) * avg_char_width(self.cache_slot)
close = (
self.previous_slot
or abs(self.secondary_slot.bottom_edge() - span.bottom_edge()) > 1
or span.left_edge() < self.secondary_slot.right_edge() - 1
or span.left_edge() > self.secondary_slot.right_edge() + gap_tolerance
)
self._close_token(close)
self.secondary_slot = span
else:
self.secondary_slot = span
for char_index in range(len(span.text)):
char_value = span.text[char_index]
if (
char_index == 0
and pending is not None
and intervals_overlap(pending.left_edge(), pending.right_edge(), span.left_edge(), span.right_edge())
):
# Compose with the pending combining mark
combined = unicodedata.normalize("NFC", char_value + pending.state_slot[0])
self._advance_char(combined[0], 0)
else:
self._advance_char(char_value, char_index)
pending = None
return self
def tokens(self) -> TokenView:
"""Finalize and return a token view."""
if self.primary_slot:
self._close_token(True)
return wrap_tokens(self.tertiary_slot)
# --------------------------------------------------------------------------- #
# X(block) -- cached token list for a block #
# --------------------------------------------------------------------------- #
def tokenize_block(block: Block) -> TokenView:
"""tokenize all lines of a block, cached on the block token cache."""
if block.tokens_cache is not None:
return block.tokens_cache # type: ignore[return-value]
token = LineTokenizer()
for line in block.primary_slot:
token.add_line(line)
block.tokens_cache = token.tokens() # type: ignore[assignment]
return block.tokens_cache # type: ignore[return-value]
# --------------------------------------------------------------------------- #
# Utility helpers.
# --------------------------------------------------------------------------- #
def clamp_value(value: float, lower_bound: float, upper_bound: float) -> float:
"""Clamp a value between lower and upper bounds. The lower bound wins when the bounds are inverted, and NaN propagates."""
measure_item = upper_bound if upper_bound < value else value
return lower_bound if lower_bound > measure_item else measure_item
def is_superscript_adjacent(token: Token, other_token: Token) -> bool:
"""Return whether the next token is a raised, shorter marker on the same line."""
candidate_item = last_token_anchor(token).anchor_span
reference_item = first_anchor_span(other_token)
return (
reference_item is not candidate_item
and last_token_anchor(token).line is other_token.line()
and reference_item.bbox_height() < candidate_item.bbox_height()
and reference_item.bottom_edge() > candidate_item.bottom_edge() + 0.1 * candidate_item.bbox_height()
)
+336
View File
@@ -0,0 +1,336 @@
"""Trie construction, matching, and token trimming utilities."""
from __future__ import annotations
from typing import Any, Iterable, Iterator, Optional
from ..model import (
_strip_diacritics,
avg_char_width2,
intervals_overlap,
to_number,
rect_union,
EMPTY_RECT,
avg_char_width,
Line,
char_category,
is_word_category,
is_punct_category,
letter_count,
punct_count,
info_weight,
Block,
)
from .token_types import (
can_extend_token,
is_trimmable_token,
TokenView,
wrap_tokens,
enumerate_tokens,
first_token,
last_token,
)
# --------------------------------------------------------------------------- #
# Token trie matcher and builder.
# --------------------------------------------------------------------------- #
def _de_norm(text: str, case_fold: bool) -> str:
"""Normalize trie keys by optional case folding, NFD decomposition, combining-mark stripping, and NFC recomposition. This strips diacritics without applying compatibility normalization."""
return _strip_diacritics(text.lower() if case_fold else text)
class TrieConfig:
"""Trie configuration: reverse-match mode and case-fold mode."""
__slots__ = ("primary_slot", "secondary_slot")
def __init__(self):
self.primary_slot: bool = False
self.secondary_slot: bool = False
class BuiltTrie:
"""Built trie wrapper containing the root node and a reverse-match flag."""
__slots__ = ("secondary_slot", "primary_slot")
def __init__(self, primary_item: "TrieNode", candidate_flag: bool):
self.secondary_slot = primary_item # root node
self.primary_slot = candidate_flag # reverse-match flag
def set_reverse(primary_item: TrieConfig) -> TrieConfig:
"""set reverse flag."""
primary_item.primary_slot = True
return primary_item
def set_case_fold(primary_item: TrieConfig, other_flag: bool) -> TrieConfig:
"""set case-fold flag."""
primary_item.secondary_slot = other_flag
return primary_item
class TrieNode:
"""- trie node."""
__slots__ = ("str", "depth", "primary_slot", "children", "dict_suffix_link", "failure_link", "is_terminal", "payload")
def __init__(self, other_text: str, depth: int, case_fold: bool):
self.str = other_text
self.depth = depth
self.primary_slot = case_fold
self.children: dict[str, "TrieNode"] = {}
self.dict_suffix_link = None
self.failure_link: Optional["TrieNode"] = None
self.is_terminal = False
self.payload = None
def normalize(self, other_text: str) -> str:
return _de_norm(other_text, self.primary_slot)
def trie_insert_step(node: TrieNode, other_text: str) -> TrieNode:
"""walk one child, creating if absent."""
key = node.normalize(other_text)
child = node.children.get(key)
if child is None:
child = TrieNode(key, node.depth + 1, node.primary_slot)
node.children[key] = child
return child
def trie_walk_step(node: TrieNode, other_text: str) -> TrieNode:
"""Walk one child; if absent, fall back through failure links."""
key = node.normalize(other_text)
child = node.children.get(key)
if child is not None:
return child
if node.failure_link is not None:
return trie_walk_step(node.failure_link, other_text)
return node
def aho_corasick_match(trie: BuiltTrie, tokens) -> Optional[dict]:
"""Aho-Corasick walk over a trie. Returns the shortest earliest terminal match and its payload. Dictionary-suffix matches use the suffix depth for match length while retaining the current node payload, which is load-bearing for edge cases."""
if isinstance(tokens, list):
tokens = wrap_tokens(tokens)
if trie.primary_slot:
tokens = tokens.reverse()
matched_tokens: Optional[TokenView] = None
matched_reverse = None
earliest_start = -1
node: TrieNode = trie.secondary_slot # root node
for entry in enumerate_tokens(tokens):
index = entry["index"]
token = entry["token"]
node = trie_walk_step(node, token.str)
depth = node.depth if node.is_terminal else 0
if depth > 0 and (earliest_start < 0 or index - depth + 1 <= earliest_start):
earliest_start = index - depth + 1
matched_tokens = tokens.slice(earliest_start, index + 1)
matched_reverse = node.payload
if trie.primary_slot:
matched_tokens = matched_tokens.reverse()
kb_node = node.dict_suffix_link
kb_depth = kb_node.depth if kb_node is not None else 0
if kb_depth > 0 and (earliest_start < 0 or index - kb_depth + 1 <= earliest_start):
earliest_start = index - kb_depth + 1
matched_tokens = tokens.slice(earliest_start, index + 1)
matched_reverse = node.payload
if trie.primary_slot:
matched_tokens = matched_tokens.reverse()
# Once a match exists and the current path start has moved past the
# earliest match start, no later token can produce an earlier match.
if earliest_start >= 0 and index - node.depth + 1 > earliest_start:
break
if matched_tokens is None:
return None
return {"tokens": matched_tokens, "payload": matched_reverse}
def aho_corasick_tokens(trie: BuiltTrie, tokens) -> Optional[TokenView]:
"""Return only the matched token view from an Aho-Corasick match."""
token = aho_corasick_match(trie, tokens)
return token["tokens"] if token is not None else None
class TrieBuilder:
"""Trie builder context holding the root node and configuration."""
__slots__ = ("primary_slot", "secondary_slot")
def __init__(self, query_value: TrieConfig):
self.primary_slot = TrieNode("", 0, query_value.secondary_slot) # root node
self.secondary_slot = query_value # the config
def _trie_insert_entry(builder: TrieBuilder, entry: str, payload: Optional[Any] = None) -> None:
"""Insert one phrase into the trie after character-by-character tokenization. This keeps punctuation-attached phrases such as ``vol.`` and ``etc.`` aligned with document tokenization. The optional payload is stored only on an empty terminal payload slot."""
node = builder.primary_slot
tokens: list[str] = []
trie = ""
previous_category = 0
for char in entry:
cat = char_category(char)
if cat == 10 or (trie and not can_extend_token(previous_category, cat, char)):
if trie:
tokens.append(trie)
trie = ""
if cat != 10:
trie += char
previous_category = cat
if trie:
tokens.append(trie)
if builder.secondary_slot.primary_slot:
tokens.reverse()
for tok in tokens:
node = trie_insert_step(node, tok)
node.is_terminal = True
# Payload assignment uses truthiness: falsy payloads are skipped, and falsy
# existing payloads are overwritten. In this package payloads are non-empty
# dictionary-like objects, so the truthiness contract is stable.
if payload and not node.payload:
node.payload = payload
def trie_bulk_insert(builder: TrieBuilder, entries, payload: Optional[Any] = None) -> None:
"""Bulk-insert phrases into ``builder`` with a shared terminal payload."""
for entry in entries:
_trie_insert_entry(builder, entry, payload)
def _trie_finalize(builder: TrieBuilder) -> BuiltTrie:
"""Assign Aho-Corasick failure links and dictionary-suffix links with breadth-first traversal, then return a built trie wrapper."""
from collections import deque
root = builder.primary_slot
queue: deque = deque([root])
while queue:
node = queue.popleft()
for child in node.children.values():
queue.append(child)
# failure link: longest proper suffix that is a prefix in the trie
trie = node
while trie.failure_link is not None:
child.failure_link = trie.failure_link.children.get(trie.failure_link.normalize(child.str))
if child.failure_link is not None:
break
trie = trie.failure_link
if child.failure_link is None:
child.failure_link = root
# dictionary-suffix link: nearest failure ancestor that is terminal
trie = child.failure_link
while trie is not None:
if trie.is_terminal:
child.dict_suffix_link = trie
break
trie = trie.failure_link
return BuiltTrie(builder.primary_slot, builder.secondary_slot.primary_slot)
def build_trie(strings: Iterable[str], other_trie: Optional[TrieConfig] = None) -> BuiltTrie:
"""Build a trie from a list of phrase strings."""
if other_trie is None:
other_trie = TrieConfig()
builder = TrieBuilder(other_trie)
for trie in strings:
_trie_insert_entry(builder, trie)
return _trie_finalize(builder)
def trie_prefix_match(trie: BuiltTrie, tokens) -> Optional[TokenView]:
"""Return the longest prefix match against the token trie."""
# ``tokens`` may be a TokenView or a list; coerce.
if isinstance(tokens, list):
tokens = wrap_tokens(tokens)
if trie.primary_slot:
tokens = tokens.reverse()
matched: Optional[TokenView] = None
node: TrieNode = trie.secondary_slot # root node
for entry in enumerate_tokens(tokens):
if not node.children:
break
index = entry["index"]
token = entry["token"]
next_node = node.children.get(node.normalize(token.str))
if next_node is None:
break
node = next_node
if node.is_terminal:
slice_view = tokens.slice(0, index + 1)
if trie.primary_slot:
slice_view = slice_view.reverse()
matched = slice_view
return matched
def _trie_full_match(trie: BuiltTrie, tokens) -> bool:
"""full-match check."""
result = trie_prefix_match(trie, tokens)
if isinstance(tokens, list):
tokens_view = wrap_tokens(tokens)
else:
tokens_view = tokens
return result is not None and result.length == tokens_view.length
trie_full_match = _trie_full_match
# --------------------------------------------------------------------------- #
# Token-list strip helpers.
# --------------------------------------------------------------------------- #
def strip_trie_match(tokens: TokenView, other_trie: BuiltTrie) -> TokenView:
"""Strip a matching keyword sequence from a token view."""
trie = trie_prefix_match(other_trie, tokens)
if trie is None:
return tokens
if other_trie.primary_slot:
return tokens.slice(0, tokens.length - trie.length)
return tokens.slice(trie.length)
def strip_leading_if_in(tokens: TokenView, other_items: set) -> TokenView:
"""Strip the leading token if its text is in the provided set."""
first = first_token(tokens)
if tokens.length > 0 and first is not None and first.str in other_items:
return tokens.slice(1)
return tokens
# Six comma variants only, not general punctuation.
COMMA_CHARS: set[str] = {",", "﹐", ",", "、", "﹑", "、"}
def strip_trailing_comma(tokens: TokenView) -> TokenView:
"""Strip a trailing comma token."""
last = last_token(tokens)
if tokens.length > 0 and last is not None and last.str in COMMA_CHARS:
return tokens.slice(0, tokens.length - 1)
return tokens
def is_comma_token(token) -> bool:
"""Return True when the token string is one of the supported comma variants."""
return token is not None and token.str in COMMA_CHARS
def trim_trailing_punct(tokens: TokenView) -> TokenView:
"""Trim trailing punctuation-like tokens."""
end = tokens.length
while end > 0:
tok = tokens.token_at(end - 1)
if tok is None or not is_trimmable_token(tok):
break
end -= 1
return tokens.slice(0, end)