From d2ae65d97a981ef32ac05069ec000bbddfc1f78f Mon Sep 17 00:00:00 2001 From: Research Assistant Date: Tue, 9 Jun 2026 14:54:47 +0800 Subject: [PATCH] feat: discover OCR body family anchor --- paperforge/worker/ocr_blocks.py | 7 +- paperforge/worker/ocr_document.py | 1 + paperforge/worker/ocr_families.py | 224 +++++++++++++++++++++++++++ tests/test_ocr_blocks.py | 80 ++++++++++ tests/test_ocr_families.py | 249 ++++++++++++++++++++++++++++++ 5 files changed, 560 insertions(+), 1 deletion(-) create mode 100644 paperforge/worker/ocr_families.py create mode 100644 tests/test_ocr_families.py diff --git a/paperforge/worker/ocr_blocks.py b/paperforge/worker/ocr_blocks.py index c08c8dce..65490b35 100644 --- a/paperforge/worker/ocr_blocks.py +++ b/paperforge/worker/ocr_blocks.py @@ -4,6 +4,8 @@ import json from pathlib import Path from typing import Any +from paperforge.worker.ocr_document import DocumentStructure +from paperforge.worker.ocr_families import discover_body_family_anchor from paperforge.worker.ocr_signatures import build_block_signatures from paperforge.worker.ocr_roles import assign_block_role @@ -115,12 +117,15 @@ def build_structured_blocks( row["render_default"] = False row["index_default"] = False + body_family_anchor = discover_body_family_anchor(rows, page_count=total_pages) + doc_structure = DocumentStructure(body_family_anchor=body_family_anchor) + # Normalize document structure (backmatter boundary, role regime, tail promotion) from paperforge.worker.ocr_document import normalize_document_structure - doc_structure = None try: doc_structure, rows = normalize_document_structure(rows) + doc_structure.body_family_anchor = body_family_anchor except Exception as exc: import logging logging.getLogger(__name__).warning( diff --git a/paperforge/worker/ocr_document.py b/paperforge/worker/ocr_document.py index 32c21772..02c635bf 100644 --- a/paperforge/worker/ocr_document.py +++ b/paperforge/worker/ocr_document.py @@ -81,6 +81,7 @@ class DocumentStructure: spread_start: int | None = None spread_end: int | None = None backmatter_form: str = "flat" + body_family_anchor: dict | None = None page_layouts: dict[int, PageLayoutProfile] | None = None tail_reading_order: list[dict] | None = None reference_zones: list[dict] | None = None diff --git a/paperforge/worker/ocr_families.py b/paperforge/worker/ocr_families.py new file mode 100644 index 00000000..b8f462f7 --- /dev/null +++ b/paperforge/worker/ocr_families.py @@ -0,0 +1,224 @@ +from __future__ import annotations + +from collections import Counter, defaultdict +from typing import Any + + +_STRONG_MARKER_TYPES = { + "figure_number", + "table_number", + "reference_numeric_bracket", + "reference_numeric_dot", + "reference_numeric_parenthesis", + "reference_pattern", + "citation_line", +} + +_FRONTMATTER_MARKER_TYPES = { + "affiliation_marker", + "preproof_marker", +} + + +def discover_body_family_anchor(blocks: list[dict], page_count: int | None = None) -> dict[str, Any]: + """Discover a document-level body family anchor from middle-page evidence.""" + resolved_page_count = max(page_count or 0, max((int(b.get("page", 0) or 0) for b in blocks), default=0)) + sample_pages = _select_middle_sample_pages(blocks, resolved_page_count) + if not sample_pages: + return { + "status": "HOLD", + "family_name": "body_family", + "reason": "no_eligible_middle_pages", + "sample_pages": [], + "candidate_count": 0, + } + + family_pages: dict[tuple[Any, ...], set[int]] = defaultdict(set) + family_counts: Counter[tuple[Any, ...]] = Counter() + + for block in blocks: + page = int(block.get("page", 0) or 0) + if page not in sample_pages: + continue + if not _is_body_family_candidate(block): + continue + key = _body_family_key(block) + if key is None: + continue + family_counts[key] += 1 + family_pages[key].add(page) + + repeated_candidates = [ + (key, family_counts[key], sorted(family_pages[key])) + for key in family_counts + if len(family_pages[key]) >= 2 + ] + if not repeated_candidates: + return { + "status": "HOLD", + "family_name": "body_family", + "reason": "no_repeated_family_across_sample_pages", + "sample_pages": sorted(sample_pages), + "candidate_count": sum(family_counts.values()), + } + + repeated_candidates.sort(key=lambda item: (-len(item[2]), -item[1], item[2][0], str(item[0]))) + winning_key, winning_count, winning_pages = repeated_candidates[0] + font_family, font_size_bucket, width_bucket, x_center_bucket = winning_key + return { + "status": "ACCEPT", + "family_name": "body_family", + "reason": "dominant_repeated_middle_page_family", + "sample_pages": winning_pages, + "candidate_count": winning_count, + "font_family_norm": font_family, + "font_size_bucket": font_size_bucket, + "width_bucket": width_bucket, + "x_center_bucket": x_center_bucket, + } + + +def _select_middle_sample_pages(blocks: list[dict], page_count: int) -> set[int]: + if page_count <= 0: + return set() + + pages = {int(block.get("page", 0) or 0) for block in blocks if int(block.get("page", 0) or 0) > 0} + if not pages: + return set() + + if page_count >= 4: + tail_page_count = max(1, round(page_count * 0.2)) + tail_cutoff = max(1, page_count - tail_page_count) + start_page = max(2, int(page_count * 0.25)) + end_page = min(int(page_count * 0.7), tail_cutoff) + if end_page < start_page: + start_page = 2 + end_page = tail_cutoff + + selected = { + page for page in pages + if page != 1 and page <= tail_cutoff and start_page <= page <= end_page + } + if not selected: + selected = { + page for page in pages + if page != 1 and page <= tail_cutoff + } + if len(selected) < 2: + selected = {page for page in pages if page != 1} + else: + selected = {page for page in pages if page != 1} + + stable_pages: set[int] = set() + for page in selected: + page_blocks = [block for block in blocks if int(block.get("page", 0) or 0) == page] + if _page_is_contaminated(page_blocks): + continue + stable_pages.add(page) + return stable_pages + + +def _page_is_contaminated(page_blocks: list[dict]) -> bool: + if not page_blocks: + return True + media_like = 0 + short_like = 0 + media_area_ratio = 0.0 + reference_like = 0 + frontmatter_like = 0 + total = len(page_blocks) + for block in page_blocks: + marker_type = ((block.get("marker_signature") or {}).get("type") or "none") + if marker_type in _STRONG_MARKER_TYPES | {"figure_number", "table_number", "panel_label"}: + media_like += 1 + media_area_ratio += _block_area_ratio(block) + if marker_type in { + "reference_numeric_bracket", + "reference_numeric_dot", + "reference_numeric_parenthesis", + "reference_pattern", + "citation_line", + }: + reference_like += 1 + if marker_type in _FRONTMATTER_MARKER_TYPES: + frontmatter_like += 1 + if _word_count(block) < 12: + short_like += 1 + if media_like / total > 0.5: + return True + if media_area_ratio > 0.33: + return True + if short_like / total > 0.75: + return True + if reference_like / total >= 0.5: + return True + if frontmatter_like / total >= 0.5: + return True + return False + + +def _is_body_family_candidate(block: dict) -> bool: + if _word_count(block) < 25: + return False + marker_type = ((block.get("marker_signature") or {}).get("type") or "none") + if marker_type in _STRONG_MARKER_TYPES: + return False + span_signature = block.get("span_signature") or {} + if not span_signature.get("font_family_norm"): + return False + if span_signature.get("font_size_median") is None and span_signature.get("font_size_bucket") is None: + return False + layout_signature = block.get("layout_signature") or {} + if not layout_signature.get("width"): + return False + return True + + +def _body_family_key(block: dict) -> tuple[Any, ...] | None: + span_signature = block.get("span_signature") or {} + layout_signature = block.get("layout_signature") or {} + font_family = span_signature.get("font_family_norm") + font_size_bucket = span_signature.get("font_size_bucket") + if font_size_bucket is None: + font_size_bucket = _bucket_number(span_signature.get("font_size_median"), step=0.5) + width_bucket = layout_signature.get("width_bucket") + if width_bucket is None: + width_bucket = _bucket_number(layout_signature.get("width"), step=25) + x_center_bucket = layout_signature.get("x_center_bucket") + if x_center_bucket is None: + x_center_bucket = _bucket_number(layout_signature.get("x_center"), step=25) + if font_family is None or font_size_bucket is None or width_bucket is None or x_center_bucket is None: + return None + return ( + font_family, + font_size_bucket, + width_bucket, + x_center_bucket, + ) + + +def _word_count(block: dict) -> int: + text = str(block.get("text") or block.get("block_content") or "") + return len([token for token in text.split() if token]) + + +def _bucket_number(value: Any, step: float) -> float | None: + if value is None: + return None + return round(round(float(value) / step) * step, 2) + + +def _block_area_ratio(block: dict) -> float: + bbox = block.get("bbox") or block.get("block_bbox") or [] + if len(bbox) >= 4: + width = max(0.0, float(bbox[2]) - float(bbox[0])) + height = max(0.0, float(bbox[3]) - float(bbox[1])) + else: + layout_signature = block.get("layout_signature") or {} + width = max(0.0, float(layout_signature.get("width") or 0.0)) + height = max(0.0, float(layout_signature.get("height") or 0.0)) + page_width = float(block.get("page_width") or 0.0) + page_height = float(block.get("page_height") or 0.0) + if page_width <= 0 or page_height <= 0: + return 0.0 + return (width * height) / (page_width * page_height) diff --git a/tests/test_ocr_blocks.py b/tests/test_ocr_blocks.py index 050f9303..a5124f28 100644 --- a/tests/test_ocr_blocks.py +++ b/tests/test_ocr_blocks.py @@ -125,3 +125,83 @@ def test_role_span_profiles_written_to_output() -> None: dumped = json.dumps(profiles) assert "section_heading" in dumped assert "body_paragraph" in dumped + + +def test_build_structured_blocks_attaches_body_family_anchor() -> None: + from paperforge.worker.ocr_blocks import build_structured_blocks + + raw_blocks = [ + { + "paper_id": "KEY001", + "page": 3, + "block_id": "p3_b1", + "raw_label": "text", + "raw_order": 0, + "bbox": [110, 100, 370, 220], + "text": "Long body text A " * 8, + "page_width": 600, + "page_height": 800, + "span_metadata": [{"font": "Times", "size": 9.0, "flags": 0, "color": 0}] * 12, + }, + { + "paper_id": "KEY001", + "page": 4, + "block_id": "p4_b1", + "raw_label": "text", + "raw_order": 0, + "bbox": [112, 100, 374, 220], + "text": "Long body text B " * 8, + "page_width": 600, + "page_height": 800, + "span_metadata": [{"font": "Times", "size": 9.0, "flags": 0, "color": 0}] * 12, + }, + ] + + _rows, doc_structure = build_structured_blocks(raw_blocks) + + assert doc_structure is not None + assert doc_structure.body_family_anchor is not None + assert doc_structure.body_family_anchor["status"] == "ACCEPT" + assert doc_structure.body_family_anchor["family_name"] == "body_family" + assert doc_structure.body_family_anchor["sample_pages"] == [3, 4] + + +def test_build_structured_blocks_discovers_body_family_before_normalization() -> None: + from unittest.mock import patch + + from paperforge.worker.ocr_blocks import build_structured_blocks + + raw_blocks = [ + { + "paper_id": "KEY001", + "page": 3, + "block_id": "p3_b1", + "raw_label": "text", + "raw_order": 0, + "bbox": [110, 100, 370, 220], + "text": "Long body text A " * 8, + "page_width": 600, + "page_height": 800, + "span_metadata": [{"font": "Times", "size": 9.0, "flags": 0, "color": 0}] * 12, + }, + { + "paper_id": "KEY001", + "page": 4, + "block_id": "p4_b1", + "raw_label": "text", + "raw_order": 0, + "bbox": [112, 100, 374, 220], + "text": "Long body text B " * 8, + "page_width": 600, + "page_height": 800, + "span_metadata": [{"font": "Times", "size": 9.0, "flags": 0, "color": 0}] * 12, + }, + ] + + with patch("paperforge.worker.ocr_document.normalize_document_structure", side_effect=RuntimeError("boom")): + _rows, doc_structure = build_structured_blocks(raw_blocks) + + assert doc_structure is not None + assert doc_structure.body_family_anchor is not None + assert doc_structure.body_family_anchor["status"] == "ACCEPT" + assert doc_structure.body_family_anchor["sample_pages"] == [3, 4] diff --git a/tests/test_ocr_families.py b/tests/test_ocr_families.py new file mode 100644 index 00000000..aa16d3ff --- /dev/null +++ b/tests/test_ocr_families.py @@ -0,0 +1,249 @@ +"""Tests for OCR family anchor discovery.""" + +from __future__ import annotations + + +def test_middle_page_body_family_anchor_uses_dominant_repeated_family() -> None: + from paperforge.worker.ocr_families import discover_body_family_anchor + + blocks = [ + { + "page": 3, + "text": "Long body text A " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 260, "x_center": 240}, + }, + { + "page": 4, + "text": "Long body text B " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 262, "x_center": 242}, + }, + { + "page": 4, + "text": "Figure 1.", + "marker_signature": {"type": "figure_number"}, + "span_signature": {"font_size_median": 8.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 210, "x_center": 260}, + }, + ] + + anchor = discover_body_family_anchor(blocks, page_count=8) + + assert anchor["status"] == "ACCEPT" + assert anchor["family_name"] == "body_family" + assert anchor["sample_pages"] == [3, 4] + + +def test_body_family_anchor_excludes_last_page_for_small_documents() -> None: + from paperforge.worker.ocr_families import discover_body_family_anchor + + blocks = [ + { + "page": 2, + "text": "Core body text A " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 260, "x_center": 240}, + }, + { + "page": 3, + "text": "Core body text B " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 262, "x_center": 242}, + }, + { + "page": 4, + "text": "Tail text A " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 8.0, "font_family_norm": "Arial"}, + "layout_signature": {"width": 250, "x_center": 230}, + }, + { + "page": 5, + "text": "Tail text B " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 8.0, "font_family_norm": "Arial"}, + "layout_signature": {"width": 250, "x_center": 230}, + }, + { + "page": 5, + "text": "Tail text C " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 8.0, "font_family_norm": "Arial"}, + "layout_signature": {"width": 250, "x_center": 230}, + }, + ] + + anchor = discover_body_family_anchor(blocks, page_count=5) + + assert anchor["status"] == "ACCEPT" + assert anchor["family_name"] == "body_family" + assert anchor["sample_pages"] == [2, 3] + + +def test_body_family_anchor_skips_media_dominated_pages_by_area() -> None: + from paperforge.worker.ocr_families import discover_body_family_anchor + + blocks = [ + { + "page": 3, + "text": "Body text page three " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 260, "height": 110, "x_center": 240}, + "bbox": [100, 100, 360, 210], + "page_width": 600, + "page_height": 800, + }, + { + "page": 4, + "text": "Figure 1.", + "marker_signature": {"type": "figure_number"}, + "span_signature": {"font_size_median": 8.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 560, "height": 520, "x_center": 300}, + "bbox": [20, 80, 580, 600], + "page_width": 600, + "page_height": 800, + }, + { + "page": 4, + "text": "Body text on contaminated page " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 260, "height": 110, "x_center": 240}, + "bbox": [100, 620, 360, 730], + "page_width": 600, + "page_height": 800, + }, + { + "page": 5, + "text": "Body text page five " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 262, "height": 112, "x_center": 242}, + "bbox": [100, 100, 362, 212], + "page_width": 600, + "page_height": 800, + }, + ] + + anchor = discover_body_family_anchor(blocks, page_count=8) + + assert anchor["status"] == "ACCEPT" + assert anchor["sample_pages"] == [3, 5] + + +def test_body_family_anchor_skips_reference_like_tail_pages() -> None: + from paperforge.worker.ocr_families import discover_body_family_anchor + + blocks = [ + { + "page": 3, + "text": "Body text page three " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 260, "x_center": 240}, + }, + { + "page": 4, + "text": "[1] Smith et al. 2020 Journal entry text " * 4, + "marker_signature": {"type": "reference_numeric_bracket"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 255, "x_center": 238}, + }, + { + "page": 5, + "text": "Body text page five " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 262, "x_center": 242}, + }, + ] + + anchor = discover_body_family_anchor(blocks, page_count=8) + + assert anchor["status"] == "ACCEPT" + assert anchor["sample_pages"] == [3, 5] + + +def test_body_family_anchor_skips_frontmatter_like_pages() -> None: + from paperforge.worker.ocr_families import discover_body_family_anchor + + blocks = [ + { + "page": 2, + "text": "$^{1}$ Department of Testing, Example University", + "marker_signature": {"type": "affiliation_marker"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 280, "x_center": 260}, + }, + { + "page": 3, + "text": "Body text page three " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 260, "x_center": 240}, + }, + { + "page": 4, + "text": "Body text page four " * 8, + "marker_signature": {"type": "none"}, + "span_signature": {"font_size_median": 9.0, "font_family_norm": "Times"}, + "layout_signature": {"width": 262, "x_center": 242}, + }, + ] + + anchor = discover_body_family_anchor(blocks, page_count=8) + + assert anchor["status"] == "ACCEPT" + assert anchor["sample_pages"] == [3, 4] + + +def test_body_family_anchor_is_not_fragmented_by_line_or_length_buckets() -> None: + from paperforge.worker.ocr_families import discover_body_family_anchor + + blocks = [ + { + "page": 3, + "text": "Dense body paragraph with stable styling " * 6, + "marker_signature": {"type": "none"}, + "span_signature": { + "font_size_median": 9.0, + "font_size_bucket": 9.0, + "font_family_norm": "Times", + }, + "layout_signature": { + "width": 260, + "width_bucket": 250, + "x_center": 240, + "x_center_bucket": 250, + "line_count": 5, + }, + }, + { + "page": 4, + "text": "Dense body paragraph with a few extra words for variation " * 6, + "marker_signature": {"type": "none"}, + "span_signature": { + "font_size_median": 9.0, + "font_size_bucket": 9.0, + "font_family_norm": "Times", + }, + "layout_signature": { + "width": 262, + "width_bucket": 250, + "x_center": 242, + "x_center_bucket": 250, + "line_count": 9, + }, + }, + ] + + anchor = discover_body_family_anchor(blocks, page_count=8) + + assert anchor["status"] == "ACCEPT" + assert anchor["sample_pages"] == [3, 4]