diff --git a/paperforge/worker/ocr_figures.py b/paperforge/worker/ocr_figures.py index 5575655d..10c5de2e 100644 --- a/paperforge/worker/ocr_figures.py +++ b/paperforge/worker/ocr_figures.py @@ -136,6 +136,66 @@ def _is_body_mention(block: dict) -> bool: _PANEL_SUBCAPTION_PATTERN = re.compile(r"^\s*[a-z][\.\)]\s") +_TRUNCATED_LEGEND_ONLY_PATTERN = re.compile( + r"^(?:Figure|Fig\.?|Supplementary\s+Figure|Supplementary\s+Fig\.?|" + r"Extended\s+Data\s+Figure|Extended\s+Data\s+Fig\.?)\s+(?:S)?\d+(?:\.0+)?\.?$", + flags=re.IGNORECASE, +) + + +def _is_validation_first_legend_candidate(block: dict) -> bool: + role = str(block.get("role") or "") + marker_signature = block.get("marker_signature") or {} + marker_type = str(marker_signature.get("type") or "none") + zone = str(block.get("zone") or "") + style_family = str(block.get("style_family") or "") + + return ( + role not in {"figure_caption", "figure_caption_candidate"} + and + marker_type == "figure_number" + and zone in {"body_zone", "display_zone"} + and style_family == "legend_like" + ) + + +def _has_anchor_supported_legend_context(block: dict) -> bool: + marker_signature = block.get("marker_signature") or {} + marker_type = str(marker_signature.get("type") or "none") + style_family = str(block.get("style_family") or "") + style_family_authority = str(block.get("style_family_authority") or "") + return ( + marker_type == "figure_number" + and style_family == "legend_like" + and style_family_authority in {"figure_marker", "figure_family_anchor"} + ) + + +def _is_insufficient_legend_evidence(block: dict) -> bool: + text = str(block.get("text") or "").strip() + marker_signature = block.get("marker_signature") or {} + marker_type = str(marker_signature.get("type") or "none") + style_family = str(block.get("style_family") or "") + + if marker_type != "figure_number": + return False + if style_family != "legend_like": + return False + if not _TRUNCATED_LEGEND_ONLY_PATTERN.fullmatch(text): + return False + + return True + + +def _has_strong_explicit_caption_text(block: dict) -> bool: + text = str(block.get("text") or "").strip() + role = str(block.get("role") or "") + if role not in {"figure_caption", "figure_caption_candidate"} and not _is_validation_first_legend_candidate(block): + return False + if _is_insufficient_legend_evidence(block): + return False + words = [w for w in re.split(r"\s+", text) if w] + return len(words) >= 5 and len(text) >= 30 def _is_formal_legend(text: str, block: dict | None = None, page_width: float = 1200) -> bool: @@ -362,12 +422,14 @@ def is_embedded_figure_text(block: dict, all_blocks: list[dict], page_width: flo def build_figure_inventory(structured_blocks: list[dict], page_width: float = 1200) -> dict[str, Any]: legends: list[dict] = [] + held_figures: list[dict] = [] rejected_legends: list[dict] = [] assets: list[dict] = [] unmatched_legends: list[dict] = [] unmatched_assets: list[dict] = [] matched_figures: list[dict] = [] unresolved_clusters: list[dict] = [] + ambiguous_figures: list[dict] = [] for block in structured_blocks: if block.get("page_width"): @@ -380,7 +442,8 @@ def build_figure_inventory(structured_blocks: list[dict], page_width: float = 12 # Skip single-letter panel labels (A, B, (C), A.) in figure legends if _PANEL_LABEL_PATTERN.match(str(block.get("text", "")).strip()): continue - if role in ("figure_caption", "figure_caption_candidate"): + is_validation_first_candidate = _is_validation_first_legend_candidate(block) + if role in ("figure_caption", "figure_caption_candidate") or is_validation_first_candidate: if _is_body_mention(block): continue if role == "figure_caption_candidate" and _looks_like_figure_narrative_prose(block.get("text", "")): @@ -439,11 +502,12 @@ def build_figure_inventory(structured_blocks: list[dict], page_width: float = 12 ordered_legends = deduped_legends used_asset_indices: set[int] = set() - ambiguous_figures: list[dict] = [] for legend in ordered_legends: legend_page = legend.get("page", 0) legend_text = legend.get("text", "") fig_num = _extract_figure_number(legend_text) + is_validation_first_candidate = _is_validation_first_legend_candidate(legend) + is_weak_truncated = _is_insufficient_legend_evidence(legend) body_prose_likelihood = _looks_like_inline_figure_mention(legend_text) @@ -458,7 +522,24 @@ def build_figure_inventory(structured_blocks: list[dict], page_width: float = 12 for ai, asset in enumerate(assets): if ai in used_asset_indices or asset.get("page", 0) != legend_page: continue - match_score = score_figure_match(legend, asset, caption_score=caption_score) + family_supported = ( + is_validation_first_candidate and str(legend.get("style_family") or "") == "legend_like" + ) + zone_supported = ( + is_validation_first_candidate and str(legend.get("zone") or "") in {"body_zone", "display_zone"} + ) + caption_text_supported = _has_strong_explicit_caption_text(legend) + match_score = score_figure_match( + legend, + asset, + caption_score=caption_score, + anchor_supported=( + _has_anchor_supported_legend_context(legend) + ), + caption_text_supported=caption_text_supported, + family_supported=family_supported, + zone_supported=zone_supported, + ) if match_score["decision"] != "rejected": candidates.append((ai, asset, match_score)) candidates.sort(key=lambda item: item[2]["score"], reverse=True) @@ -478,15 +559,19 @@ def build_figure_inventory(structured_blocks: list[dict], page_width: float = 12 lcx = (legend_bb[0] + legend_bb[2]) / 2 if len(legend_bb) >= 4 else 0 best = close[0] best_col_match = False + best_delta = abs( + lcx - ((best[1].get("bbox", [0, 0, 0, 0])[0] + best[1].get("bbox", [0, 0, 0, 0])[2]) / 2) + ) for ci, ca, cs in close: ab = ca.get("bbox") or ca.get("block_bbox") or [0,0,0,0] acx = (ab[0] + ab[2]) / 2 if len(ab) >= 4 else 0 - ca_col_ok = abs(lcx - acx) < abs(lcx - (best[1].get("bbox",[0,0,0,0])[0] + best[1].get("bbox",[0,0,0,0])[2])/2) + delta = abs(lcx - acx) + ca_col_ok = delta + 20 < best_delta if ca_col_ok: best = (ci, ca, cs) + best_delta = delta best_col_match = True - break - if best_col_match: + if best_col_match and best[2].get("decision") == "matched": best_idx, best_asset, best_score = best matched_assets = [best_asset] used_asset_indices.add(best_idx) @@ -505,31 +590,20 @@ def build_figure_inventory(structured_blocks: list[dict], page_width: float = 12 matched_assets = [] else: best_idx, best_asset, best_score = candidates[0] - matched_assets = [best_asset] - used_asset_indices.add(best_idx) - region_match = {"media_blocks": [best_asset], "match_score": best_score} - - # Fallback: if no match found but legends == assets on the same page, - # assign sequentially by vertical position - if not matched_assets and fig_num is not None: - page_assets = [ - (ai, a) for ai, a in enumerate(assets) - if ai not in used_asset_indices and a.get("page", 0) == legend_page - ] - page_legends = [ - l for l in ordered_legends - if l is not legend and _extract_figure_number(l.get("text", "")) is not None - and l.get("page", 0) == legend_page - ] - if page_assets and len(page_assets) >= len(page_legends) + 1: - page_assets.sort(key=lambda item: (item[1].get("bbox",[0,0,0,0])[1] if len(item[1].get("bbox",[]))>=4 else 0)) - # Count how many matched legends already consumed assets on this page - consumed_on_page = sum(1 for i in used_asset_indices if assets[i].get("page",0) == legend_page) - asset_idx = min(consumed_on_page, len(page_assets) - 1) - best_idx, best_asset = page_assets[asset_idx] - matched_assets = [best_asset] - used_asset_indices.add(best_idx) - region_match = {"media_blocks": [best_asset], "match_score": {"score": 0.5, "decision": "matched_fallback", "evidence": ["sequential_fallback"]}} + if best_score["decision"] == "matched": + matched_assets = [best_asset] + used_asset_indices.add(best_idx) + region_match = {"media_blocks": [best_asset], "match_score": best_score} + else: + ambiguous_figures.append({ + "legend_block_id": legend.get("block_id", ""), + "page": legend_page, + "caption_score": caption_score, + "candidates": [ + {"asset_block_id": best_asset.get("block_id", ""), "match_score": best_score} + ], + }) + ambiguous = True is_legend_only = len(matched_assets) == 0 @@ -537,38 +611,87 @@ def build_figure_inventory(structured_blocks: list[dict], page_width: float = 12 unmatched_legends.append(legend) continue - fig_id = f"figure_{fig_num:03d}" if fig_num else f"unmatched_legend_{len(matched_figures):03d}" + if is_weak_truncated and is_validation_first_candidate: + held_figures.append( + { + "figure_id": f"held_figure_{len(held_figures) + 1:03d}", + "legend_block_id": legend.get("block_id", ""), + "page": legend_page, + "text": legend_text, + "figure_number": fig_num, + "hold_reason": "insufficient_legend_evidence", + "zone": legend.get("zone"), + "style_family": legend.get("style_family"), + "marker_signature": legend.get("marker_signature") or {}, + "caption_score": caption_score, + } + ) + unmatched_legends.append(legend) + continue - if not ambiguous: - match_score = region_match["match_score"] if region_match is not None else { - "score": 0.0, - "decision": "legend_only", - "evidence": ["no_asset_match"], - } - entry = { - "figure_id": fig_id, + if is_weak_truncated: + ambiguous_figures.append({ "legend_block_id": legend.get("block_id", ""), "page": legend_page, "text": legend_text, "figure_number": fig_num, - "matched_assets": [ - { - "block_id": a.get("block_id", ""), - "bbox": a.get("bbox", [0, 0, 0, 0]), - } - for a in matched_assets - ], - "confidence": match_score["score"], - "match_score": match_score, - "flags": [] if not is_legend_only else ["legend_only"], "caption_score": caption_score, - } - if region_match is not None and len(matched_assets) > 1: - entry["cluster_bbox"] = region_match["cluster_bbox"] - matched_figures.append(entry) + "candidates": [], + "hold_reason": "ambiguous_truncated_legend", + "zone": legend.get("zone"), + "style_family": legend.get("style_family"), + "marker_signature": legend.get("marker_signature") or {}, + }) + unmatched_legends.append(legend) + continue if is_legend_only: + weak_entry = { + "legend_block_id": legend.get("block_id", ""), + "page": legend_page, + "text": legend_text, + "figure_number": fig_num, + "caption_score": caption_score, + "candidates": [], + "hold_reason": ( + "ambiguous_truncated_legend" if is_weak_truncated else "no_asset_match" + ), + "zone": legend.get("zone"), + "style_family": legend.get("style_family"), + "marker_signature": legend.get("marker_signature") or {}, + } + if not ambiguous: + ambiguous_figures.append(weak_entry) unmatched_legends.append(legend) + continue + + fig_id = f"figure_{fig_num:03d}" if fig_num else f"unmatched_legend_{len(matched_figures):03d}" + match_score = region_match["match_score"] if region_match is not None else { + "score": 0.0, + "decision": "rejected", + "evidence": ["missing_region_match"], + } + entry = { + "figure_id": fig_id, + "legend_block_id": legend.get("block_id", ""), + "page": legend_page, + "text": legend_text, + "figure_number": fig_num, + "matched_assets": [ + { + "block_id": a.get("block_id", ""), + "bbox": a.get("bbox", [0, 0, 0, 0]), + } + for a in matched_assets + ], + "confidence": match_score["score"], + "match_score": match_score, + "flags": [], + "caption_score": caption_score, + } + if region_match is not None and len(matched_assets) > 1: + entry["cluster_bbox"] = region_match["cluster_bbox"] + matched_figures.append(entry) for i, asset in enumerate(assets): if i not in used_asset_indices: @@ -601,6 +724,7 @@ def build_figure_inventory(structured_blocks: list[dict], page_width: float = 12 "figure_legends": deduped_legends, "figure_assets": assets, "matched_figures": matched_figures, + "held_figures": held_figures, "ambiguous_figures": ambiguous_figures, "unmatched_legends": unmatched_legends, "unmatched_assets": unmatched_assets, diff --git a/paperforge/worker/ocr_objects.py b/paperforge/worker/ocr_objects.py index 58d4cacc..92abc4ad 100644 --- a/paperforge/worker/ocr_objects.py +++ b/paperforge/worker/ocr_objects.py @@ -166,6 +166,9 @@ def extract_and_write_objects( page_dimensions_by_page: dict[int, tuple[int, int]] | None = None, ) -> None: """Extract figure/table asset crops from PDF and write object markdown.""" + # Validation-first figure matching may retain held figures in inventory, + # but object emission remains limited to matched figures and unresolved + # media clusters until figure evidence is sufficient. figures_asset_dir = asset_root / "figures" tables_asset_dir = asset_root / "tables" orphans_asset_dir = asset_root / "orphans" diff --git a/paperforge/worker/ocr_scores.py b/paperforge/worker/ocr_scores.py index 7f2cc21e..a3390921 100644 --- a/paperforge/worker/ocr_scores.py +++ b/paperforge/worker/ocr_scores.py @@ -89,11 +89,21 @@ def score_table_match(caption: dict, asset: dict, *, is_continuation: bool = Fal return {"score": score, "matched_asset_id": asset.get("block_id", ""), "decision": decision, "evidence": evidence} -def score_figure_match(legend: dict, asset: dict, *, caption_score: dict | None = None) -> dict: +def score_figure_match( + legend: dict, + asset: dict, + *, + caption_score: dict | None = None, + anchor_supported: bool = False, + caption_text_supported: bool = False, + family_supported: bool = False, + zone_supported: bool = False, +) -> dict: legend_bbox = legend.get("bbox") or legend.get("block_bbox") or [0, 0, 0, 0] asset_bbox = asset.get("bbox") or asset.get("block_bbox") or [0, 0, 0, 0] score = 0.0 evidence: list[str] = [] + has_x_overlap = False caption_value = float((caption_score or {}).get("score", 0.0)) if caption_value < 0.4: @@ -104,6 +114,7 @@ def score_figure_match(legend: dict, asset: dict, *, caption_score: dict | None score += 0.3 evidence.append("same_page") if _bbox_x_overlap_ratio(legend_bbox, asset_bbox) >= 0.4: + has_x_overlap = True score += 0.25 evidence.append("x_overlap") if len(legend_bbox) >= 4 and len(asset_bbox) >= 4: @@ -116,7 +127,27 @@ def score_figure_match(legend: dict, asset: dict, *, caption_score: dict | None evidence.append("caption_above_or_below") score += min(0.15, caption_value * 0.15) score = max(0.0, min(1.0, score)) - decision = "matched" if score >= 0.6 else "ambiguous" if score >= 0.4 else "rejected" + if anchor_supported: + score += 0.05 + evidence.append("anchor_supported") + if caption_text_supported: + score += 0.05 + evidence.append("caption_text_supported") + if family_supported: + score += 0.03 + evidence.append("family_supported") + if zone_supported: + score += 0.02 + evidence.append("zone_supported") + score = max(0.0, min(1.0, score)) + strong_geometry = "same_page" in evidence and "nearby_y" in evidence and "caption_above_or_below" in evidence + contextual_support = anchor_supported or family_supported or zone_supported + if score >= 0.6 and (has_x_overlap or (strong_geometry and (contextual_support or caption_text_supported))): + decision = "matched" + elif score >= 0.4: + decision = "ambiguous" + else: + decision = "rejected" return {"score": score, "matched_asset_id": asset.get("block_id", ""), "decision": decision, "evidence": evidence} diff --git a/tests/test_ocr_figures.py b/tests/test_ocr_figures.py index 92318925..f10c54c5 100644 --- a/tests/test_ocr_figures.py +++ b/tests/test_ocr_figures.py @@ -236,6 +236,7 @@ def test_figure_inventory_includes_all_sections() -> None: assert "figure_legends" in inventory assert "figure_assets" in inventory assert "matched_figures" in inventory + assert "held_figures" in inventory assert "unmatched_legends" in inventory assert "unmatched_assets" in inventory assert "unresolved_clusters" in inventory @@ -390,11 +391,10 @@ def test_legend_only_figure_no_asset_match() -> None: inventory = build_figure_inventory(structured_blocks) - assert len(inventory["matched_figures"]) == 1 - assert inventory["matched_figures"][0]["figure_number"] == 2 - assert len(inventory["matched_figures"][0]["matched_assets"]) == 0 - assert "legend_only" in inventory["matched_figures"][0]["flags"] - assert inventory["matched_figures"][0]["confidence"] == inventory["matched_figures"][0]["match_score"]["score"] + assert inventory["matched_figures"] == [] + assert len(inventory.get("ambiguous_figures", [])) == 1 + assert inventory["ambiguous_figures"][0]["legend_block_id"] == "p2_b1" + assert inventory["ambiguous_figures"][0]["hold_reason"] == "no_asset_match" def test_unmatched_legends_populated() -> None: @@ -481,9 +481,10 @@ def test_legend_does_not_steal_offpage_asset() -> None: inventory = build_figure_inventory(structured_blocks) - assert len(inventory["matched_figures"]) == 1 - assert len(inventory["matched_figures"][0]["matched_assets"]) == 0 - assert "legend_only" in inventory["matched_figures"][0]["flags"] + assert inventory["matched_figures"] == [] + assert len(inventory.get("ambiguous_figures", [])) == 1 + assert inventory["ambiguous_figures"][0]["legend_block_id"] == "p1_b1" + assert inventory["ambiguous_figures"][0]["hold_reason"] == "no_asset_match" assert len(inventory["unmatched_assets"]) == 1 assert inventory["unmatched_assets"][0]["block_id"] == "p2_b1" @@ -1009,6 +1010,373 @@ def test_figure_inventory_does_not_confidently_match_low_caption_score() -> None assert len(inventory["unmatched_legends"]) == 1 +def test_figure_matching_can_hold_when_legend_is_ambiguous() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 10, + "block_id": "p10_b1", + "zone": "body_zone", + "style_family": "legend_like", + "text": "Figure 1", + "marker_signature": {"type": "figure_number", "number": 1}, + "bbox": [50, 50, 300, 90], + "page_width": 1200, + "page_height": 1600, + }, + { + "paper_id": "K001", + "page": 10, + "block_id": "p10_b2", + "zone": "body_zone", + "style_family": "body_like", + "text": "Narrative prose", + "marker_signature": {"type": "none"}, + "bbox": [50, 100, 900, 140], + "page_width": 1200, + "page_height": 1600, + }, + ] + + inv = build_figure_inventory(structured_blocks) + + assert inv["matched_figures"] == [] + assert "held_figures" in inv + assert len(inv["held_figures"]) == 1 + assert inv["held_figures"][0]["legend_block_id"] == "p10_b1" + assert inv["held_figures"][0]["hold_reason"] == "insufficient_legend_evidence" + assert inv["held_figures"][0]["figure_number"] == 1 + + +def test_validation_first_truncated_legend_variants_are_held() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + for legend_text in ("Figure 1.", "Fig. 1."): + structured_blocks = [ + { + "paper_id": "K001", + "page": 10, + "block_id": "p10_b1", + "zone": "body_zone", + "style_family": "legend_like", + "text": legend_text, + "marker_signature": {"type": "figure_number", "number": 1}, + "bbox": [50, 50, 300, 90], + "page_width": 1200, + "page_height": 1600, + }, + { + "paper_id": "K001", + "page": 10, + "block_id": "p10_b2", + "zone": "body_zone", + "style_family": "body_like", + "text": "Narrative prose", + "marker_signature": {"type": "none"}, + "bbox": [50, 100, 900, 140], + "page_width": 1200, + "page_height": 1600, + }, + ] + + inv = build_figure_inventory(structured_blocks) + + assert inv["matched_figures"] == [] + assert len(inv.get("held_figures", [])) == 1 + assert inv["held_figures"][0]["legend_block_id"] == "p10_b1" + + +def test_validation_first_truncated_legend_with_same_page_asset_still_holds() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 10, + "block_id": "p10_b1", + "zone": "body_zone", + "style_family": "legend_like", + "text": "Figure 1.", + "marker_signature": {"type": "figure_number", "number": 1}, + "bbox": [50, 420, 300, 460], + "page_width": 1200, + "page_height": 1600, + }, + { + "paper_id": "K001", + "page": 10, + "block_id": "p10_b2", + "role": "figure_asset", + "text": "", + "bbox": [50, 60, 550, 390], + "page_width": 1200, + "page_height": 1600, + }, + ] + + inv = build_figure_inventory(structured_blocks) + + assert inv["matched_figures"] == [] + assert len(inv.get("held_figures", [])) == 1 + assert inv["held_figures"][0]["legend_block_id"] == "p10_b1" + + +def test_display_zone_validation_first_candidate_enters_figure_matching() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 7, + "block_id": "p7_b1", + "zone": "display_zone", + "style_family": "legend_like", + "text": "Figure 3.", + "marker_signature": {"type": "figure_number", "number": 3}, + "bbox": [100, 420, 360, 460], + "page_width": 1200, + "page_height": 1600, + }, + { + "paper_id": "K001", + "page": 7, + "block_id": "p7_b2", + "role": "figure_asset", + "text": "", + "bbox": [80, 60, 620, 390], + "page_width": 1200, + "page_height": 1600, + }, + ] + + inv = build_figure_inventory(structured_blocks) + + assert inv["matched_figures"] == [] + assert len(inv.get("held_figures", [])) == 1 + assert inv["held_figures"][0]["legend_block_id"] == "p7_b1" + + +def test_display_zone_validation_first_full_caption_can_match() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 7, + "block_id": "p7_b1", + "zone": "display_zone", + "style_family": "legend_like", + "style_family_authority": "figure_family_anchor", + "text": "Figure 3. Quantitative analysis of migration under stimulation.", + "marker_signature": {"type": "figure_number", "number": 3}, + "bbox": [120, 420, 620, 470], + "page_width": 1200, + "page_height": 1600, + }, + { + "paper_id": "K001", + "page": 7, + "block_id": "p7_b2", + "role": "figure_asset", + "text": "", + "bbox": [100, 60, 700, 390], + "page_width": 1200, + "page_height": 1600, + }, + ] + + inv = build_figure_inventory(structured_blocks) + + assert len(inv["matched_figures"]) == 1 + assert inv["matched_figures"][0]["legend_block_id"] == "p7_b1" + assert inv["matched_figures"][0]["matched_assets"][0]["block_id"] == "p7_b2" + + +def test_truncated_legend_variant_from_existing_caption_role_is_still_held() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 10, + "block_id": "p10_b1", + "role": "figure_caption", + "zone": "body_zone", + "style_family": "legend_like", + "text": "Figure 1.", + "marker_signature": {"type": "figure_number", "number": 1}, + "bbox": [50, 50, 300, 90], + "page_width": 1200, + "page_height": 1600, + }, + { + "paper_id": "K001", + "page": 10, + "block_id": "p10_b2", + "role": "body_paragraph", + "zone": "body_zone", + "style_family": "body_like", + "text": "Narrative prose", + "marker_signature": {"type": "none"}, + "bbox": [50, 100, 900, 140], + "page_width": 1200, + "page_height": 1600, + }, + ] + + inv = build_figure_inventory(structured_blocks) + + assert inv["matched_figures"] == [] + assert inv.get("held_figures", []) == [] + assert len(inv.get("ambiguous_figures", [])) == 1 + assert inv["ambiguous_figures"][0]["legend_block_id"] == "p10_b1" + + +def test_legitimate_offset_caption_asset_pair_can_still_match_with_overlap() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 4, + "block_id": "cap1", + "role": "figure_caption", + "text": "Figure 2. Migration assay under stimulation.", + "bbox": [180, 420, 620, 470], + "page_width": 1200, + "page_height": 1600, + }, + { + "paper_id": "K001", + "page": 4, + "block_id": "asset1", + "role": "figure_asset", + "text": "", + "bbox": [100, 60, 700, 390], + "page_width": 1200, + "page_height": 1600, + }, + ] + + inv = build_figure_inventory(structured_blocks) + + assert len(inv["matched_figures"]) == 1 + assert inv["matched_figures"][0]["legend_block_id"] == "cap1" + assert inv["matched_figures"][0]["matched_assets"][0]["block_id"] == "asset1" + + +def test_explicit_figure_caption_role_is_not_diverted_into_hold() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 10, + "block_id": "p10_b1", + "role": "figure_caption", + "zone": "body_zone", + "style_family": "legend_like", + "text": "Figure 1. Migration assay under electric field stimulation.", + "marker_signature": {"type": "figure_number", "number": 1}, + "bbox": [50, 420, 550, 470], + "page_width": 1200, + "page_height": 1600, + }, + { + "paper_id": "K001", + "page": 10, + "block_id": "p10_b2", + "role": "figure_asset", + "text": "", + "bbox": [50, 60, 550, 400], + "page_width": 1200, + "page_height": 1600, + }, + ] + + inv = build_figure_inventory(structured_blocks) + + assert inv.get("held_figures", []) == [] + assert len(inv["matched_figures"]) == 1 + assert inv["matched_figures"][0]["legend_block_id"] == "p10_b1" + + +def test_weak_single_candidate_match_is_not_forced_by_fallback() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 4, + "block_id": "p4_b1", + "role": "figure_caption", + "text": "Figure 1. Brief caption.", + "bbox": [50, 500, 250, 530], + "page_width": 1200, + "page_height": 1600, + }, + { + "paper_id": "K001", + "page": 4, + "block_id": "p4_b2", + "role": "figure_asset", + "text": "", + "bbox": [700, 60, 1100, 420], + "page_width": 1200, + "page_height": 1600, + }, + ] + + inv = build_figure_inventory(structured_blocks) + + assert inv["matched_figures"] == [] + assert len(inv.get("ambiguous_figures", [])) == 1 + assert inv["ambiguous_figures"][0]["legend_block_id"] == "p4_b1" + assert len(inv["unmatched_assets"]) == 1 + assert inv["unmatched_assets"][0]["block_id"] == "p4_b2" + + +def test_no_candidate_sequential_fallback_no_longer_manufactures_match() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 6, + "block_id": "p6_b1", + "role": "figure_caption", + "text": "Figure 1. A caption with no validated candidate geometry.", + "bbox": [50, 700, 550, 740], + "page_width": 1200, + "page_height": 1600, + }, + { + "paper_id": "K001", + "page": 6, + "block_id": "p6_b2", + "role": "figure_asset", + "text": "", + "bbox": [700, 50, 1100, 300], + "page_width": 1200, + "page_height": 1600, + }, + ] + + inv = build_figure_inventory(structured_blocks) + + assert inv["matched_figures"] == [] + assert all( + figure.get("match_score", {}).get("decision") != "matched_fallback" + for figure in inv.get("matched_figures", []) + ) + assert len(inv.get("ambiguous_figures", [])) == 1 + assert inv["ambiguous_figures"][0]["legend_block_id"] == "p6_b1" + assert len(inv["unmatched_assets"]) == 1 + assert inv["unmatched_assets"][0]["block_id"] == "p6_b2" + + def test_rejected_legend_caption_score_evidence() -> None: from paperforge.worker.ocr_figures import build_figure_inventory diff --git a/tests/test_ocr_objects.py b/tests/test_ocr_objects.py index 3e8146f5..5b6b65ea 100644 --- a/tests/test_ocr_objects.py +++ b/tests/test_ocr_objects.py @@ -197,6 +197,44 @@ def test_unresolved_cluster_object_emission(tmp_path: Path) -> None: assert not (render_root / "figures" / "cluster_001.md").exists() +def test_held_figures_do_not_emit_object_notes(tmp_path: Path) -> None: + from paperforge.worker.ocr_objects import extract_and_write_objects + + render_root = tmp_path / "render" + asset_root = tmp_path / "assets" + + figure_inventory: dict[str, Any] = { + "matched_figures": [], + "held_figures": [ + { + "figure_id": "held_figure_001", + "legend_block_id": "p10_b1", + "page": 10, + "text": "Figure 1", + "figure_number": 1, + "hold_reason": "insufficient_legend_evidence", + } + ], + "unmatched_assets": [], + "rejected_legends": [], + "figure_legends": [], + "figure_assets": [], + "official_figure_count": 0, + "unresolved_clusters": [], + } + + extract_and_write_objects( + pdf_path=None, + figure_inventory=figure_inventory, + table_inventory={"tables": [], "unmatched_assets": []}, + asset_root=asset_root, + render_root=render_root, + ) + + render_files = sorted((render_root / "figures").glob("*.md")) + assert render_files == [] + + def test_crop_asset_uses_ocr_page_coordinates_when_dimensions_provided(tmp_path: Path) -> None: import fitz from PIL import Image diff --git a/tests/test_ocr_scores.py b/tests/test_ocr_scores.py index 483d7036..293b2f9b 100644 --- a/tests/test_ocr_scores.py +++ b/tests/test_ocr_scores.py @@ -62,6 +62,43 @@ def test_figure_match_score_prefers_same_page_overlap() -> None: assert "x_overlap" in result["evidence"] +def test_figure_match_score_allows_strong_same_page_geometry_without_x_overlap() -> None: + from paperforge.worker.ocr_scores import score_figure_match + + legend = {"block_id": "cap1", "page": 2, "bbox": [100, 500, 700, 540]} + asset = {"block_id": "fig1", "page": 2, "bbox": [710, 120, 1110, 480]} + + result = score_figure_match( + legend, + asset, + caption_score={"score": 0.8}, + caption_text_supported=True, + ) + + assert result["decision"] == "matched" + assert result["matched_asset_id"] == "fig1" + assert result["score"] >= 0.6 + assert "same_page" in result["evidence"] + assert "nearby_y" in result["evidence"] + assert "caption_above_or_below" in result["evidence"] + assert "caption_text_supported" in result["evidence"] + + +def test_figure_match_score_keeps_short_caption_without_x_overlap_ambiguous() -> None: + from paperforge.worker.ocr_scores import score_figure_match + + legend = {"block_id": "cap1", "page": 2, "bbox": [50, 500, 250, 530]} + asset = {"block_id": "fig1", "page": 2, "bbox": [700, 60, 1100, 420]} + + result = score_figure_match( + legend, + asset, + caption_score={"score": 0.7}, + ) + + assert result["decision"] == "ambiguous" + + def test_figure_match_score_rejects_low_caption_score() -> None: from paperforge.worker.ocr_scores import score_figure_match