diff --git a/paperforge/worker/ocr.py b/paperforge/worker/ocr.py
index 7b39072a..169d753e 100644
--- a/paperforge/worker/ocr.py
+++ b/paperforge/worker/ocr.py
@@ -1546,7 +1546,10 @@ def render_page_blocks(
append_reference_section(rendered, reference_blocks, reference_continuations)
references_emitted = True
references_section_active = False
- rendered.append(f"### {title}")
+ if title.lower() == "abstract" or re.match(r"^\d+\s", title):
+ rendered.append(f"## {title}")
+ else:
+ rendered.append(f"### {title}")
if title.lower() == "references":
references_heading_seen = True
references_section_active = True
diff --git a/paperforge/worker/ocr_blocks.py b/paperforge/worker/ocr_blocks.py
index 1ea8d78a..d452b7e6 100644
--- a/paperforge/worker/ocr_blocks.py
+++ b/paperforge/worker/ocr_blocks.py
@@ -72,6 +72,9 @@ def build_structured_blocks(
"role_confidence": role.confidence,
"evidence": role.evidence,
"span_metadata": block.get("span_metadata"),
+ "_in_visual_container": block.get("_in_visual_container", None),
+ "_container_bbox": block.get("_container_bbox", None),
+ "_container_text": block.get("_container_text", None),
"render_default": render_default,
"index_default": index_default,
}
diff --git a/paperforge/worker/ocr_document.py b/paperforge/worker/ocr_document.py
index 425a66d4..c827536a 100644
--- a/paperforge/worker/ocr_document.py
+++ b/paperforge/worker/ocr_document.py
@@ -84,6 +84,113 @@ class DocumentStructure:
layout_audit: dict | None = None
+@dataclass
+class RegionPrepass:
+ block_regions: dict[int, str]
+ frontmatter_indices: set[int]
+ structured_insert_indices: set[int]
+ body_candidate_indices: set[int]
+ confidence_by_index: dict[int, float]
+
+
+def _block_text(block: dict) -> str:
+ return str(block.get("text") or block.get("block_content") or "")
+
+
+def _looks_like_box_anchor(text: str) -> bool:
+ return bool(re.match(r"^box\s*\.?\s*\d+\b", text.strip().lower()))
+
+
+def _build_region_prepass(blocks: list[dict]) -> RegionPrepass:
+ block_regions: dict[int, str] = {}
+ confidence_by_index: dict[int, float] = {}
+ frontmatter_indices: set[int] = set()
+ structured_insert_indices: set[int] = set()
+ body_candidate_indices: set[int] = set()
+
+ by_page: dict[int, list[tuple[int, dict]]] = {}
+ for idx, block in enumerate(blocks):
+ by_page.setdefault(int(block.get("page", 1) or 1), []).append((idx, block))
+
+ for page, page_items in by_page.items():
+ page_items.sort(key=lambda item: (item[1].get("bbox") or item[1].get("block_bbox") or [0, 0, 0, 0])[1])
+ last_insert_on_page = False
+ last_insert_anchor_kind = ""
+ for idx, block in page_items:
+ role = block.get("role", "")
+ text = _block_text(block).strip().lower()
+ bbox = block.get("bbox") or block.get("block_bbox") or [0, 0, 0, 0]
+ page_height = block.get("page_height") or 1700
+ page_width = block.get("page_width") or 1200
+ y_top = bbox[1] if len(bbox) >= 4 else 0
+
+ region = "body"
+ confidence = 0.5
+
+ in_container = block.get("_in_visual_container", False)
+ if in_container and len(bbox) >= 4:
+ cbox = block.get("_container_bbox", [])
+ within_container = False
+ if len(cbox) >= 4:
+ margin_w = (bbox[2] - bbox[0]) * 0.1
+ margin_h = (bbox[3] - bbox[1]) * 0.1
+ within_container = (
+ bbox[0] >= cbox[0] - margin_w
+ and bbox[2] <= cbox[2] + margin_w
+ and bbox[1] >= cbox[1] - margin_h
+ and bbox[3] <= cbox[3] + margin_h
+ )
+ if within_container:
+ region = "structured_insert"
+ confidence = 0.85
+ last_insert_anchor_kind = "container"
+ elif (page <= 3 and ("key point" in text or text in {"sections", "highlights"})) or _looks_like_box_anchor(text):
+ region = "structured_insert"
+ confidence = 0.8
+ last_insert_anchor_kind = "box" if _looks_like_box_anchor(text) else "summary"
+ elif (
+ last_insert_on_page
+ and page <= 3
+ and len(text.split()) <= 18
+ and role not in {"section_heading", "subsection_heading", "sub_subsection_heading", "reference_heading", "backmatter_heading"}
+ ):
+ region = "structured_insert"
+ confidence = max(confidence, 0.7)
+ elif (
+ last_insert_on_page
+ and last_insert_anchor_kind == "box"
+ and len(bbox) >= 4
+ and bbox[0] < page_width * 0.25
+ and role in {"section_heading", "subsection_heading", "sub_subsection_heading", "body_paragraph"}
+ ):
+ region = "structured_insert"
+ confidence = max(confidence, 0.75)
+
+ elif page == 1 and (role in {"paper_title", "authors", "affiliation", "frontmatter_noise"} or y_top < page_height * 0.22):
+ region = "frontmatter"
+ confidence = 0.8
+ last_insert_on_page = region == "structured_insert"
+ if region != "structured_insert":
+ last_insert_anchor_kind = ""
+
+ block_regions[idx] = region
+ confidence_by_index[idx] = confidence
+ if region == "frontmatter":
+ frontmatter_indices.add(idx)
+ elif region == "structured_insert":
+ structured_insert_indices.add(idx)
+ else:
+ body_candidate_indices.add(idx)
+
+ return RegionPrepass(
+ block_regions=block_regions,
+ frontmatter_indices=frontmatter_indices,
+ structured_insert_indices=structured_insert_indices,
+ body_candidate_indices=body_candidate_indices,
+ confidence_by_index=confidence_by_index,
+ )
+
+
def _compute_span_coverage(blocks: list[dict]) -> dict:
"""Compute span metadata coverage across the document.
@@ -1459,7 +1566,19 @@ def _select_body_anchor_pages(blocks: list[dict], doc: DocumentStructure | None
return fallback if fallback else []
-def _detect_body_spine(blocks: list[dict], doc: DocumentStructure | None = None) -> dict[int, dict]:
+def _is_body_spine_training_block(block: dict, idx: int, region_prepass: RegionPrepass | None) -> bool:
+ if block.get("role") != "body_paragraph":
+ return False
+ if region_prepass is None:
+ return True
+ return region_prepass.block_regions.get(idx, "body") == "body"
+
+
+def _detect_body_spine(
+ blocks: list[dict],
+ doc: DocumentStructure | None = None,
+ region_prepass: RegionPrepass | None = None,
+) -> dict[int, dict]:
"""Detect the main body column characteristics per page.
Uses a two-pass approach:
@@ -1485,9 +1604,11 @@ def _detect_body_spine(blocks: list[dict], doc: DocumentStructure | None = None)
import statistics
by_page: dict[int, list[dict]] = {}
- for block in blocks:
+ block_idx: dict[int, int] = {}
+ for idx, block in enumerate(blocks):
page = block.get("page", 1)
by_page.setdefault(page, []).append(block)
+ block_idx[id(block)] = idx
all_pages = sorted(by_page.keys())
if not all_pages:
@@ -1516,7 +1637,7 @@ def _detect_body_spine(blocks: list[dict], doc: DocumentStructure | None = None)
for page in anchor_pages:
for b in by_page.get(page, []):
- if b.get("role") == "body_paragraph":
+ if _is_body_spine_training_block(b, block_idx.get(id(b), -1), region_prepass):
bbox = b.get("bbox", [0, 0, 0, 0])
w = bbox[2] - bbox[0]
if w >= 400:
@@ -1555,7 +1676,10 @@ def _detect_body_spine(blocks: list[dict], doc: DocumentStructure | None = None)
per_page_spine: dict[int, dict | None] = {}
for page in all_pages:
page_blocks = by_page[page]
- body_blocks = [b for b in page_blocks if b.get("role") == "body_paragraph"]
+ body_blocks = [
+ b for b in page_blocks
+ if _is_body_spine_training_block(b, block_idx.get(id(b), -1), region_prepass)
+ ]
if body_blocks:
widths = []
@@ -1837,7 +1961,7 @@ def _detect_non_body_insert_clusters(
font_mismatch = bool(block_font and spine_fonts and block_font not in spine_fonts)
if quality == "strong":
- passes = is_narrow or font_mismatch
+ passes = is_narrow or (font_mismatch and block_width < 0.9 * median_width)
elif quality == "moderate":
passes = is_narrow and (font_mismatch or len(candidates_by_page.get(page, [])) >= 1)
else:
@@ -2255,6 +2379,86 @@ def _detect_structured_insert_clusters(blocks: list[dict]) -> set[int]:
return result
+def _expand_structured_insert_cluster_with_mixed_sidebar_blocks(
+ blocks: list[dict],
+ indices: set[int],
+) -> set[int]:
+ if not indices:
+ return indices
+
+ expanded = set(indices)
+ by_page: dict[int, list[int]] = {}
+ for idx in indices:
+ page = int(blocks[idx].get("page", 1) or 1)
+ by_page.setdefault(page, []).append(idx)
+
+ heading_roles = {"section_heading", "subsection_heading", "sub_subsection_heading", "reference_heading", "backmatter_heading"}
+ mixed_roles = {"media_asset", "table_html", "unknown_structural", "structured_insert_candidate"}
+
+ for page, page_indices in by_page.items():
+ changed = True
+ while changed:
+ changed = False
+ current_page_indices = [idx for idx in expanded if int(blocks[idx].get("page", 1) or 1) == page]
+ current_texts = [str(blocks[idx].get("text") or "") for idx in current_page_indices]
+ has_box_anchor = any(_looks_like_box_anchor(text) for text in current_texts)
+ bboxes = [blocks[idx].get("bbox") or blocks[idx].get("block_bbox") for idx in current_page_indices]
+ bboxes = [bb for bb in bboxes if bb and len(bb) >= 4]
+ if not bboxes:
+ break
+
+ cluster_min_x = min(bb[0] for bb in bboxes)
+ cluster_max_x = max(bb[2] for bb in bboxes)
+ cluster_min_y = min(bb[1] for bb in bboxes)
+ cluster_max_y = max(bb[3] for bb in bboxes)
+
+ next_heading_top = None
+ for i, block in enumerate(blocks):
+ if int(block.get("page", 1) or 1) != page or i in expanded:
+ continue
+ role = block.get("role", "")
+ if role not in heading_roles:
+ continue
+ bb = block.get("bbox") or block.get("block_bbox")
+ if not bb or len(bb) < 4:
+ continue
+ block_x_center = (bb[0] + bb[2]) / 2
+ same_column = block_x_center < (cluster_min_x + cluster_max_x) / 2 + 200
+ if has_box_anchor and same_column and bb[0] <= cluster_max_x + 40 and bb[2] >= cluster_min_x - 40 and (bb[1] - cluster_max_y) <= 140:
+ continue
+ if bb[1] > cluster_max_y and (next_heading_top is None or bb[1] < next_heading_top):
+ next_heading_top = bb[1]
+
+ for i, block in enumerate(blocks):
+ if int(block.get("page", 1) or 1) != page or i in expanded:
+ continue
+ role = block.get("role", "")
+ text = str(block.get("text") or "").strip()
+ bb = block.get("bbox") or block.get("block_bbox")
+ if not bb or len(bb) < 4:
+ continue
+ if next_heading_top is not None and bb[1] >= next_heading_top:
+ continue
+
+ page_width = max(block.get("page_width", 0) or 0, 1200)
+ cluster_x_center = (cluster_min_x + cluster_max_x) / 2
+ block_x_center = (bb[0] + bb[2]) / 2
+ same_column = (
+ (cluster_x_center < page_width * 0.5 and block_x_center < page_width * 0.55)
+ or (cluster_x_center >= page_width * 0.5 and block_x_center >= page_width * 0.45)
+ )
+ overlaps_x = bb[0] <= cluster_max_x + 40 and bb[2] >= cluster_min_x - 40
+ near_y = bb[1] <= cluster_max_y + 80 and bb[3] >= cluster_min_y - 20
+ bullet_like = text.startswith(("•", "-", "*"))
+ table_like = text.lower().startswith("
dict:
"""Check resolved structure for obvious geometric contradictions.
@@ -2371,8 +2575,40 @@ def normalize_document_structure(blocks: list[dict]) -> tuple[DocumentStructure,
# Compute span coverage for degraded mode detection
doc_structure.span_coverage = _compute_span_coverage(blocks)
+ # Build region prepass: classify frontmatter/structured_insert/body by geometry
+ region_prepass = _build_region_prepass(blocks)
+ doc_structure.region_prepass = region_prepass # type: ignore[attr-defined]
+
+ # Mark prepass-identified structured insert blocks as candidates before spine
+ for idx in region_prepass.structured_insert_indices:
+ if idx < len(blocks):
+ role = blocks[idx].get("role", "")
+ if role in {"body_paragraph", "section_heading", "subsection_heading", "sub_subsection_heading"}:
+ blocks[idx]["role"] = "structured_insert_candidate"
+
+ # Detect structured insert clusters BEFORE body spine and non-body gates
+ structured_insert_indices = _detect_structured_insert_clusters(blocks)
+ for idx in structured_insert_indices:
+ if idx < len(blocks):
+ blocks[idx]["role"] = "structured_insert"
+
+ # Fallback: promote single structured_insert_candidate blocks that the
+ # region prepass independently identified as structured inserts, even
+ # when clustering could not form (only one candidate on the page).
+ for idx in region_prepass.structured_insert_indices:
+ if idx < len(blocks) and blocks[idx].get("role") == "structured_insert_candidate":
+ blocks[idx]["role"] = "structured_insert"
+
+ structured_insert_indices = {i for i, b in enumerate(blocks) if b.get("role") == "structured_insert"}
+ structured_insert_indices = _expand_structured_insert_cluster_with_mixed_sidebar_blocks(blocks, structured_insert_indices)
+ for idx in structured_insert_indices:
+ if idx < len(blocks):
+ blocks[idx]["role"] = "structured_insert"
+
+ # Detect body spine from region-approved body blocks only
+ body_spine = _detect_body_spine(blocks, doc_structure, region_prepass=region_prepass)
+
# Detect non-body insert clusters on early pages (relative to body length)
- body_spine = _detect_body_spine(blocks, doc_structure)
pw = max((b.get("page_width", 0) or 0) for b in blocks) or 1200
insert_indices = _detect_non_body_insert_clusters(
blocks,
@@ -2392,12 +2628,6 @@ def normalize_document_structure(blocks: list[dict]) -> tuple[DocumentStructure,
_mark_non_body_media(blocks)
_resolve_ambiguous_candidates(blocks, doc_structure, page_layouts)
- # Detect structured insert clusters (detached summary boxes, key points)
- structured_insert_indices = _detect_structured_insert_clusters(blocks)
- for idx in structured_insert_indices:
- if idx < len(blocks):
- blocks[idx]["role"] = "structured_insert"
-
# Run layout audit after all resolution is done
layout_audit = _run_layout_audit(blocks)
doc_structure.layout_audit = layout_audit
diff --git a/paperforge/worker/ocr_health.py b/paperforge/worker/ocr_health.py
index ca02d891..8f2421dc 100644
--- a/paperforge/worker/ocr_health.py
+++ b/paperforge/worker/ocr_health.py
@@ -60,6 +60,13 @@ def build_ocr_health(
else:
overall = "red"
+ # Compute structural health signals
+ from paperforge.worker.ocr_document import _compute_span_coverage, _detect_body_spine, _run_layout_audit
+
+ span = _compute_span_coverage(structured_blocks)
+ spine = _detect_body_spine(structured_blocks)
+ layout = _run_layout_audit(structured_blocks)
+
return {
"page_count": page_count,
"blocks_count": raw_blocks_count,
@@ -77,6 +84,15 @@ def build_ocr_health(
"empty_table_count": empty_tables,
"frontmatter_quality": frontmatter_quality,
"overall": overall,
+ "span_coverage": span,
+ "span_coverage_quality": span.get("coverage_quality", "weak"),
+ "degraded_mode_active": span.get("degraded_mode_active", True),
+ "body_spine_quality": spine.get("_meta", {}).get("quality", "weak"),
+ "body_anchor_pages": spine.get("_meta", {}).get("anchor_pages", []),
+ "body_spine_sample_count": spine.get("_meta", {}).get("sample_count", 0),
+ "layout_audit_status": layout.get("status", "unknown"),
+ "layout_anomaly_pages": layout.get("anomaly_pages", []),
+ "layout_anomaly_count": layout.get("anomaly_count", 0),
}
diff --git a/paperforge/worker/ocr_pdf_spans.py b/paperforge/worker/ocr_pdf_spans.py
index 56e93ef0..5553dcad 100644
--- a/paperforge/worker/ocr_pdf_spans.py
+++ b/paperforge/worker/ocr_pdf_spans.py
@@ -109,6 +109,48 @@ def extract_pdf_spans_for_block(
return spans if spans else None
+def _extract_visual_container_rects(page: Any) -> list:
+ """Extract visible rectangle regions (filled or bordered) from a PDF page.
+
+ Uses PyMuPDF's ``get_drawings()`` to find rectangles with:
+ - noticeable fill color (not white/transparent), OR
+ - visible border/outline
+
+ Returns list of fitz.Rect objects for each detected container.
+ """
+ import fitz
+
+ rects: list = []
+ try:
+ drawings = page.get_drawings()
+ except Exception:
+ return rects
+
+ for drawing in drawings:
+ fill = drawing.get("fill")
+ color = drawing.get("color")
+ stroke_width = drawing.get("width", 0) or 0
+ rect = drawing.get("rect")
+ if not rect:
+ continue
+
+ if rect.width < 100 or rect.height < 50:
+ continue
+
+ is_filled = False
+ if fill and len(fill) >= 3:
+ r, g, b = fill[0], fill[1], fill[2]
+ brightness = (r + g + b) / 3
+ is_filled = brightness < 0.95
+
+ has_border = bool(color) and (isinstance(color, (list, tuple))) and stroke_width > 0
+
+ if is_filled or has_border:
+ rects.append(rect)
+
+ return rects
+
+
def backfill_span_metadata_from_pdf(
raw_blocks: list[dict],
pdf_path: str | Path | None,
@@ -136,6 +178,7 @@ def backfill_span_metadata_from_pdf(
return raw_blocks
try:
+ by_page_containers: dict[int, list] = {}
for block in raw_blocks:
page_num = block.get("page", 1) - 1
bbox = block.get("bbox", [])
@@ -150,6 +193,94 @@ def backfill_span_metadata_from_pdf(
)
if spans:
block["span_metadata"] = spans
+
+ import fitz
+ for block in raw_blocks:
+ page_num = block.get("page", 1) - 1
+ bbox = block.get("bbox", [])
+ if not bbox or len(bbox) < 4:
+ continue
+ if page_num not in by_page_containers:
+ try:
+ pdf_page = doc[page_num]
+ by_page_containers[page_num] = _extract_visual_container_rects(pdf_page)
+ except Exception:
+ by_page_containers[page_num] = []
+ containers = by_page_containers[page_num]
+ if not containers:
+ continue
+ pw = block.get("page_width") or block.get("ocr_width") or 0
+ ph = block.get("page_height") or block.get("ocr_height") or 0
+ if pw <= 0 or ph <= 0:
+ try:
+ pdf_rect = doc[page_num].rect
+ pw = int(pdf_rect.width * 2) if pdf_rect.width > 0 else 1200
+ ph = int(pdf_rect.height * 2) if pdf_rect.height > 0 else 1700
+ except Exception:
+ pw, ph = 1200, 1700
+ if pw > 0 and ph > 0:
+ try:
+ pdf_page = doc[page_num]
+ block_rect = _map_ocr_bbox_to_pdf_rect(bbox, pw, ph, pdf_page)
+ except Exception:
+ continue
+ else:
+ block_rect = fitz.Rect(*bbox)
+ for container_rect in containers:
+ if block_rect.x0 > container_rect.x1 or block_rect.x1 < container_rect.x0:
+ continue
+ if block_rect.y0 > container_rect.y1 or block_rect.y1 < container_rect.y0:
+ continue
+ overlap_w = min(block_rect.x1, container_rect.x1) - max(block_rect.x0, container_rect.x0)
+ overlap_h = min(block_rect.y1, container_rect.y1) - max(block_rect.y0, container_rect.y0)
+ if overlap_w > 0 and overlap_h > 0:
+ block_w = block_rect.x1 - block_rect.x0
+ block_h = block_rect.y1 - block_rect.y0
+ overlap_area = overlap_w * overlap_h
+ block_area = block_w * block_h if block_w > 0 and block_h > 0 else 0
+ if block_area > 0 and overlap_area / block_area >= 0.3:
+ block["_in_visual_container"] = True
+ container_overlap_ratio = overlap_area / block_area
+ pw = block.get("page_width") or 0
+ ph = block.get("page_height") or 0
+ pdf_page_rect = doc[page_num].rect
+ scale_x = pw / pdf_page_rect.width if (pw and pdf_page_rect.width) else 1
+ scale_y = ph / pdf_page_rect.height if (ph and pdf_page_rect.height) else 1
+ block["_container_bbox"] = [
+ container_rect.x0 * scale_x,
+ container_rect.y0 * scale_y,
+ container_rect.x1 * scale_x,
+ container_rect.y1 * scale_y,
+ ]
+
+ in_container_parts: list[str] = []
+ pdf_page_for_text = doc[page_num]
+ block_pdf_rect = _map_ocr_bbox_to_pdf_rect(bbox, pw, ph, pdf_page_for_text) if pw and ph else None
+ for tblock in pdf_page_for_text.get_text("blocks"):
+ tx0, ty0, tx1, ty1, ttext, *_ = tblock
+ ttext = (ttext or "").strip()
+ if not ttext:
+ continue
+ in_container = (
+ max(tx0, container_rect.x0) < min(tx1, container_rect.x1)
+ and max(ty0, container_rect.y0) < min(ty1, container_rect.y1)
+ )
+ if not in_container:
+ continue
+ if block_pdf_rect:
+ in_block = (
+ max(tx0, block_pdf_rect.x0) < min(tx1, block_pdf_rect.x1)
+ and max(ty0, block_pdf_rect.y0) < min(ty1, block_pdf_rect.y1)
+ )
+ if not in_block:
+ continue
+ in_container_parts.append(ttext)
+ if in_container_parts and container_overlap_ratio > 0.7:
+ container_text = "\n".join(in_container_parts)
+ block_text = str(block.get("text") or "")
+ if container_text and len(container_text) < len(block_text) * 0.8:
+ block["_container_text"] = container_text
+ break
finally:
doc.close()
diff --git a/paperforge/worker/ocr_rebuild.py b/paperforge/worker/ocr_rebuild.py
index 46bf90d1..a91aa361 100644
--- a/paperforge/worker/ocr_rebuild.py
+++ b/paperforge/worker/ocr_rebuild.py
@@ -101,7 +101,10 @@ def run_derived_rebuild_for_keys(vault: Path, keys: list[str]) -> dict:
from paperforge.worker.ocr_profiles import write_role_span_profiles
write_role_span_profiles(structured, artifacts.blocks_structured.parent)
- # Read source metadata
+ # Read source metadata. If legacy/old OCR papers are missing canonical
+ # bibliographic metadata, enrich source_metadata.json from the formal
+ # Literature-hub note frontmatter before resolving OCR metadata.
+ _enrich_meta_from_paper_note(vault, key, artifacts.source_metadata)
source_meta = read_json(artifacts.source_metadata) if artifacts.source_metadata.exists() else {}
# Rebuild resolved metadata
@@ -114,7 +117,8 @@ def run_derived_rebuild_for_keys(vault: Path, keys: list[str]) -> dict:
metadata_dir = paper_root / "metadata"
metadata_dir.mkdir(parents=True, exist_ok=True)
frontmatter_candidates = extract_frontmatter_candidates(artifacts.blocks_structured)
- resolved = resolve_metadata(source_meta, frontmatter_candidates)
+ page1_blocks = [block for block in structured if int(block.get("page", 0) or 0) == 1]
+ resolved = resolve_metadata(source_meta, frontmatter_candidates, page1_blocks=page1_blocks)
write_resolved_metadata(metadata_dir / "resolved_metadata.json", resolved)
# Rebuild figure inventory
diff --git a/paperforge/worker/ocr_render.py b/paperforge/worker/ocr_render.py
index e057ed52..6dbc7352 100644
--- a/paperforge/worker/ocr_render.py
+++ b/paperforge/worker/ocr_render.py
@@ -1,6 +1,7 @@
from __future__ import annotations
import re
+from html import unescape
from pathlib import Path
from paperforge.worker.ocr_document import (
@@ -27,6 +28,28 @@ def _has_tail_role(block: dict) -> bool:
return block.get("role") in _TAIL_ROLES
+def _heading_number_depth_text(text: str) -> int:
+ match = re.match(r"^(\d+(?:\.\d+)*)", text.strip())
+ if not match:
+ return 0
+ return match.group(1).count(".") + 1
+
+
+def _table_html_to_lines(text: str) -> list[str]:
+ rows = re.findall(r"]*>(.*?)
", text, flags=re.IGNORECASE | re.DOTALL)
+ lines: list[str] = []
+ for row in rows:
+ cells = [unescape(re.sub(r"<[^>]+>", "", cell)).strip() for cell in re.findall(r"]*>(.*?) | ", row, flags=re.IGNORECASE | re.DOTALL)]
+ cells = [cell for cell in cells if cell]
+ if not cells:
+ continue
+ if len(cells) == 1:
+ lines.append(cells[0])
+ else:
+ lines.append(" | ".join(cells))
+ return lines
+
+
def _find_owning_heading(
body: dict,
sections: list[dict],
@@ -676,6 +699,37 @@ def render_fulltext_markdown(
unresolved_clusters_by_page.setdefault(page, []).append(cluster_id)
emitted_pages: set[int] = set()
+ last_structured_insert_page: int | None = None
+ last_structured_insert_bbox: list[float] | None = None
+
+ # Compute heading levels from span metadata font sizes when available
+ _HEADING_ROLES = {"section_heading", "subsection_heading", "sub_subsection_heading"}
+ heading_font_sizes: dict[int, float] = {}
+ for i, block in enumerate(structured_blocks):
+ if block.get("role") not in _HEADING_ROLES:
+ continue
+ span = block.get("span_metadata") or {}
+ if isinstance(span, list):
+ for s in span:
+ sz = s.get("size")
+ if sz:
+ heading_font_sizes[id(block)] = float(sz)
+ break
+ elif isinstance(span, dict):
+ sz = span.get("size")
+ if sz:
+ heading_font_sizes[id(block)] = float(sz)
+ block_heading_prefix: dict[int, str] = {}
+ if heading_font_sizes:
+ size_groups: dict[float, list[int]] = {}
+ for bid, sz in heading_font_sizes.items():
+ bucket = round(sz * 2) / 2
+ size_groups.setdefault(bucket, []).append(bid)
+ sorted_sizes = sorted(size_groups.keys(), reverse=True)
+ for level_idx, bucket in enumerate(sorted_sizes):
+ prefix = "##" if level_idx == 0 else "###"
+ for bid in size_groups[bucket]:
+ block_heading_prefix[bid] = prefix
# --- body with anchored figures/tables ---
# Find the min and max page across ALL blocks (including suppressed)
@@ -725,6 +779,8 @@ def render_fulltext_markdown(
})
if old_role in _PROTECTED:
continue
+ if old_role == "body_paragraph" and new_role in {"structured_insert", "non_body_insert"}:
+ continue
if old_role != new_role:
block["role"] = new_role
block["role_confidence"] = new_blocks[i].get("role_confidence", block.get("role_confidence", 0.5))
@@ -784,9 +840,11 @@ def render_fulltext_markdown(
ordered_blocks = _order_tail_blocks(structured_blocks, style_profiles=style_profiles)
for block in ordered_blocks:
- if not block.get("render_default", True):
- continue
role = block.get("role", "")
+ if role == "structured_insert":
+ pass # render as callout below
+ elif not block.get("render_default", True):
+ continue
if role in CONSUMED_FRONTMATTER_ROLES and block.get("page") == 1:
continue
_SKIPPED_BODY_ROLES = {
@@ -800,7 +858,8 @@ def render_fulltext_markdown(
if role in _SKIPPED_BODY_ROLES:
continue
- text = normalize_ocr_math_text(block.get("text", ""))
+ raw_text = block.get("text", "")
+ text = normalize_ocr_math_text(raw_text)
text = re.sub(r"", "", text, flags=re.DOTALL | re.IGNORECASE)
if text.strip().lower().startswith("")
lines.append("")
emitted_pages.add(block_page)
+ last_structured_insert_page = None
+ last_structured_insert_bbox = None
- if not block.get("render_default", True):
+ if role == "structured_insert":
+ pass # render as callout below
+ elif not block.get("render_default", True):
continue
if role == "backmatter_boundary_heading" or role == "backmatter_heading" or role == "reference_heading":
+ last_structured_insert_page = None
+ last_structured_insert_bbox = None
lines.append(f"## {text}")
lines.append("")
elif role in ("subsection_heading", "sub_subsection_heading", "section_heading"):
+ last_structured_insert_page = None
+ last_structured_insert_bbox = None
if role == "section_heading":
if text.strip().lower() in FRONTMATTER_NOISE:
continue
@@ -853,13 +920,67 @@ def render_fulltext_markdown(
lines.append(text)
lines.append("")
continue
- lines.append(f"### {text}")
+ lines.append(f"## {text}")
+ else:
+ prefix = block_heading_prefix.get(id(block))
+ if prefix is None:
+ depth = _heading_number_depth_text(text)
+ prefix = "##" if depth <= 1 else "###"
+ lines.append(f"{prefix} {text}")
lines.append("")
+ elif role == "structured_insert":
+ container_text = block.get("_container_text")
+ if container_text:
+ source_text = " ".join(container_text.replace("\n", " ").split())
+ else:
+ source_text = raw_text
+ callout_lines = _table_html_to_lines(source_text) if source_text.strip().lower().startswith("= 4
+ ):
+ prev = last_structured_insert_bbox
+ overlaps_x = bbox[0] <= prev[2] + 40 and bbox[2] >= prev[0] - 40
+ gap_y = bbox[1] - prev[3]
+ merge_with_previous = overlaps_x and gap_y <= 80
+ if merge_with_previous and lines and lines[-1] == "":
+ lines.pop()
+ if (
+ merge_with_previous
+ and callout_lines
+ and callout_lines[0].lstrip().startswith("•")
+ and lines
+ and lines[-1].startswith("> ")
+ and "•" in lines[-1][2:]
+ and not lines[-1].startswith("> •")
+ ):
+ lines.pop()
+ if not merge_with_previous:
+ lines.append("> [!NOTE]")
+ for cl in callout_lines:
+ lines.append(f"> {cl}")
+ if not merge_with_previous:
+ lines.append("")
+ last_structured_insert_page = block_page
+ last_structured_insert_bbox = bbox if len(bbox) >= 4 else None
elif role in ("backmatter_body", "tail_candidate_body", "body_paragraph"):
+ if last_structured_insert_page is not None:
+ lines.append("")
+ last_structured_insert_page = None
+ last_structured_insert_bbox = None
if text:
lines.append(text)
lines.append("")
else:
+ if last_structured_insert_page is not None:
+ lines.append("")
+ last_structured_insert_page = None
+ last_structured_insert_bbox = None
if text:
lines.append(text)
lines.append("")
diff --git a/paperforge/worker/ocr_roles.py b/paperforge/worker/ocr_roles.py
index 354761e0..d8f29872 100644
--- a/paperforge/worker/ocr_roles.py
+++ b/paperforge/worker/ocr_roles.py
@@ -493,6 +493,12 @@ def assign_block_role(
"review article", "research article", "original article",
"case report", "brief communication", "rapid communication",
})
+ if lower in _RUNNING_HEADER_LABELS and (block.get("page") or 1) == 1:
+ return RoleAssignment(
+ role="frontmatter_noise",
+ confidence=0.8,
+ evidence=[f"page-1 article-type label: {lower}"],
+ )
if lower in _RUNNING_HEADER_LABELS and (block.get("page") or 1) > 1:
bbox = block.get("block_bbox") or [0, 0, 0, 0]
_ph = block.get("page_height") or 1700
@@ -557,9 +563,9 @@ def assign_block_role(
)
if _is_author_byline and page_num == 1:
return RoleAssignment(
- role="frontmatter_noise",
- confidence=0.5,
- evidence=[f"author byline on page 1, not heading: {text[:60]}"],
+ role="authors",
+ confidence=0.6,
+ evidence=[f"author byline on page 1, assigned as authors: {text[:60]}"],
)
if re.search(r"(?:correspondence|orcid|@)", text.lower()) and len(text.split()) <= 20 and page_num == 1:
diff --git a/tests/test_ocr_document.py b/tests/test_ocr_document.py
index aafaa1a4..ac8047be 100644
--- a/tests/test_ocr_document.py
+++ b/tests/test_ocr_document.py
@@ -2696,6 +2696,36 @@ def test_non_body_insert_strong_spine_or_gate_ok() -> None:
assert len(result) >= 2, f"Expected at least 2 non_body_insert with strong spine, got {result}"
+def test_non_body_insert_strong_spine_does_not_use_font_only_for_body_width_blocks() -> None:
+ """Strong spine should not classify body-width paragraphs by font mismatch alone."""
+ from paperforge.worker.ocr_document import _detect_non_body_insert_clusters
+
+ blocks = [{
+ "role": "body_paragraph",
+ "page": 2,
+ "bbox": [80, 600, 590, 660],
+ "text": "This is real body text with a locally different OCR font family.",
+ "span_metadata": [{"size": 10, "font": "Helvetica"}],
+ "page_width": 1200,
+ "page_height": 1700,
+ }, {
+ "role": "body_paragraph",
+ "page": 2,
+ "bbox": [80, 700, 590, 760],
+ "text": "This continuation paragraph should remain body text as well.",
+ "span_metadata": [{"size": 10, "font": "Helvetica"}],
+ "page_width": 1200,
+ "page_height": 1700,
+ }]
+ body_spine = {
+ 2: {"median_width": 510, "all_fonts": {"times"}, "quality": "strong"},
+ "_meta": {"quality": "strong"},
+ }
+
+ result = _detect_non_body_insert_clusters(blocks, body_spine, page_width=1200)
+ assert result == set(), f"Body-width font-only mismatch should not be non_body_insert, got {result}"
+
+
def test_title_not_swept_into_non_body_insert() -> None:
"""Page 1 title-like block (long text, no bullets) is NOT marked as non_body_insert."""
from paperforge.worker.ocr_document import _detect_body_spine, _detect_non_body_insert_clusters
@@ -2771,3 +2801,86 @@ def test_layout_audit_reference_zone_overlaps_body() -> None:
assert result["status"] in ("warn", "fail"), f"Expected warn/fail, got {result['status']}"
assert "1" in result["page_warnings"], f"Page 1 should have warnings: {result['page_warnings']}"
assert result["anomaly_count"] >= 1
+
+
+# ---------------------------------------------------------------------------
+# Region prepass tests (Task 3)
+# ---------------------------------------------------------------------------
+
+
+def test_region_prepass_marks_frontmatter_insert_and_body_regions() -> None:
+ from paperforge.worker.ocr_document import _build_region_prepass
+
+ blocks = [
+ {"page": 1, "role": "paper_title", "text": "A Real Paper Title", "bbox": [80, 80, 700, 140], "page_width": 1200, "page_height": 1700},
+ {"page": 1, "role": "authors", "text": "Jane Author & John Writer", "bbox": [80, 160, 600, 200], "page_width": 1200, "page_height": 1700},
+ {"page": 1, "role": "body_paragraph", "text": "Key points", "bbox": [760, 280, 1050, 315], "page_width": 1200, "page_height": 1700},
+ {"page": 1, "role": "body_paragraph", "text": "Important short list item", "bbox": [760, 330, 1050, 370], "page_width": 1200, "page_height": 1700},
+ {"page": 2, "role": "body_paragraph", "text": "This is a real body paragraph with enough words to train the body spine.", "bbox": [80, 220, 640, 270], "page_width": 1200, "page_height": 1700},
+ ]
+
+ prepass = _build_region_prepass(blocks)
+
+ assert prepass.block_regions[0] == "frontmatter"
+ assert prepass.block_regions[1] == "frontmatter"
+ assert prepass.block_regions[2] == "structured_insert"
+ assert prepass.block_regions[3] == "structured_insert"
+ assert prepass.block_regions[4] == "body"
+
+
+def test_body_spine_ignores_region_frontmatter_and_structured_insert_blocks() -> None:
+ from paperforge.worker.ocr_document import _build_region_prepass, _detect_body_spine
+
+ blocks = [
+ {"page": 1, "role": "body_paragraph", "text": "Review article", "bbox": [80, 80, 240, 110], "page_width": 1200, "page_height": 1700, "span_metadata": [{"font": "Heading"}]},
+ {"page": 1, "role": "body_paragraph", "text": "Key points", "bbox": [760, 280, 1050, 315], "page_width": 1200, "page_height": 1700, "span_metadata": [{"font": "Box"}]},
+ ]
+ for page in range(2, 6):
+ for line in range(3):
+ blocks.append({"page": page, "role": "body_paragraph", "text": "Real body paragraph text for spine training.", "bbox": [80, 200 + line * 80, 620, 250 + line * 80], "page_width": 1200, "page_height": 1700, "span_metadata": [{"font": "Body"}]})
+
+ prepass = _build_region_prepass(blocks)
+ spine = _detect_body_spine(blocks, region_prepass=prepass)
+
+ assert "heading" not in spine[1].get("all_fonts", set()), str(spine[1].get("all_fonts"))
+ assert "box" not in spine[1].get("all_fonts", set()), str(spine[1].get("all_fonts"))
+
+
+def test_normalize_marks_structured_insert_before_non_body_insert_suppression() -> None:
+ from paperforge.worker.ocr_document import normalize_document_structure
+
+ blocks = [
+ {"page": 1, "role": "body_paragraph", "text": "Key points", "bbox": [760, 280, 1050, 315], "page_width": 1200, "page_height": 1700},
+ {"page": 1, "role": "body_paragraph", "text": "Short summary point", "bbox": [760, 330, 1050, 370], "page_width": 1200, "page_height": 1700},
+ ]
+ for page in range(2, 6):
+ for line in range(3):
+ blocks.append({"page": page, "role": "body_paragraph", "text": "Real body paragraph text for training.", "bbox": [80, 200 + line * 80, 620, 250 + line * 80], "page_width": 1200, "page_height": 1700, "span_metadata": [{"font": "Body"}]})
+
+ _, normalized = normalize_document_structure(blocks)
+
+ assert normalized[0]["role"] == "structured_insert"
+ assert normalized[1]["role"] == "structured_insert"
+
+
+def test_normalize_promotes_mixed_sidebar_blocks_into_single_structured_insert_cluster() -> None:
+ """Sidebar heading, textual table container, and continuation block should all join one structured insert."""
+ from paperforge.worker.ocr_document import normalize_document_structure
+
+ blocks = [
+ {"page": 2, "role": "body_paragraph", "text": "Key points", "bbox": [80, 220, 180, 247], "page_width": 1200, "page_height": 1700},
+ {"page": 2, "role": "media_asset", "raw_label": "table", "text": "", "bbox": [73, 270, 591, 738], "page_width": 1200, "page_height": 1700},
+ {"page": 2, "role": "unknown_structural", "text": "• Point three continues below the detected table box.", "bbox": [74, 675, 566, 743], "page_width": 1200, "page_height": 1700},
+ {"page": 2, "role": "sub_subsection_heading", "text": "Introduction", "bbox": [76, 780, 210, 803], "page_width": 1200, "page_height": 1700},
+ {"page": 2, "role": "body_paragraph", "text": "The skeleton has been considered to be a metabolically active organ for decades.", "bbox": [73, 805, 593, 1192], "page_width": 1200, "page_height": 1700},
+ ]
+ for i in range(3):
+ blocks.append({"page": 3 + i, "role": "body_paragraph", "text": "Real body paragraph text for training and normalization.", "bbox": [80, 200, 620, 260], "page_width": 1200, "page_height": 1700, "span_metadata": [{"font": "Body"}]})
+
+ _, normalized = normalize_document_structure(blocks)
+
+ assert normalized[0]["role"] == "structured_insert"
+ assert normalized[1]["role"] == "structured_insert"
+ assert normalized[2]["role"] == "structured_insert"
+ assert normalized[3]["role"] != "structured_insert"
+ assert normalized[4]["role"] == "body_paragraph"
diff --git a/tests/test_ocr_health.py b/tests/test_ocr_health.py
index a4e7298f..5fe80ead 100644
--- a/tests/test_ocr_health.py
+++ b/tests/test_ocr_health.py
@@ -124,3 +124,20 @@ def test_layout_audit_health_surface() -> None:
assert empty["layout_audit_status"] == "unknown"
assert empty["layout_anomaly_pages"] == []
assert empty["layout_anomaly_count"] == 0
+
+
+def test_ocr_health_includes_span_spine_and_layout_signals() -> None:
+ from paperforge.worker.ocr_health import build_ocr_health
+
+ blocks = [
+ {"role": "section_heading", "span_metadata": [{"font": "Body"}]},
+ {"role": "section_heading", "span_metadata": [{"font": "Body"}]},
+ {"role": "abstract_body", "span_metadata": [{"font": "Body"}]},
+ {"role": "reference_item", "span_metadata": [{"font": "Body"}]},
+ ]
+
+ health = build_ocr_health(page_count=2, raw_blocks_count=4, structured_blocks=blocks, figure_inventory={}, table_inventory={})
+
+ assert "span_coverage_quality" in health
+ assert "body_spine_quality" in health
+ assert "layout_audit_status" in health
diff --git a/tests/test_ocr_real_paper_regressions.py b/tests/test_ocr_real_paper_regressions.py
new file mode 100644
index 00000000..40fe17ff
--- /dev/null
+++ b/tests/test_ocr_real_paper_regressions.py
@@ -0,0 +1,146 @@
+from __future__ import annotations
+
+import json
+from pathlib import Path
+
+import pytest
+
+
+VAULT = Path(r"D:\L\OB\Literature-hub")
+OCR_ROOT = VAULT / "System" / "PaperForge" / "ocr"
+
+PROBLEM_KEYS = ["TSCKAVIS", "CAQNW9Q2", "A8E7SRVS", "K7R8PEKW"]
+CONTROL_KEYS = ["SAN9AYVR", "2GN9LMCW", "7C8829BD"]
+ALL_KEYS = PROBLEM_KEYS + CONTROL_KEYS
+
+
+def _paper_root(key: str) -> Path:
+ return OCR_ROOT / key
+
+
+def _structured_path(key: str) -> Path:
+ return _paper_root(key) / "structure" / "blocks.structured.jsonl"
+
+
+def _fulltext_path(key: str) -> Path:
+ return _paper_root(key) / "fulltext.md"
+
+
+def _health_path(key: str) -> Path:
+ return _paper_root(key) / "health" / "ocr_health.json"
+
+
+def _metadata_path(key: str) -> Path:
+ return _paper_root(key) / "metadata" / "resolved_metadata.json"
+
+
+def _read_json(path: Path) -> dict:
+ return json.loads(path.read_text(encoding="utf-8"))
+
+
+def _read_jsonl(path: Path) -> list[dict]:
+ return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line.strip()]
+
+
+def _require_artifacts(key: str) -> None:
+ if not VAULT.exists():
+ pytest.skip(f"Vault path not available: {VAULT}")
+ missing = [
+ str(path)
+ for path in [_structured_path(key), _fulltext_path(key), _health_path(key), _metadata_path(key)]
+ if not path.exists()
+ ]
+ if missing:
+ pytest.skip(f"OCR artifacts not present for {key}: {missing}")
+
+
+@pytest.fixture(scope="session")
+def rebuilt_real_papers() -> dict:
+ if not VAULT.exists():
+ pytest.skip(f"Vault path not available: {VAULT}")
+ from paperforge.worker.ocr_rebuild import run_derived_rebuild_for_keys
+
+ result = run_derived_rebuild_for_keys(VAULT, ALL_KEYS)
+ assert result.get("rebuild_count", 0) >= 1, result
+ return result
+
+
+@pytest.mark.parametrize("key", ALL_KEYS)
+def test_real_paper_artifacts_exist(key: str) -> None:
+ _require_artifacts(key)
+
+
+def test_real_paper_rebuild_runs(rebuilt_real_papers: dict) -> None:
+ assert rebuilt_real_papers.get("rebuild_count", 0) >= 1
+
+
+BODY_RETENTION = {
+ "CAQNW9Q2": {"min_body": 35, "max_non_body_insert": 8},
+ "A8E7SRVS": {"min_body": 45, "max_non_body_insert": 8},
+ "K7R8PEKW": {"min_body": 70, "max_non_body_insert": 8},
+ "TSCKAVIS": {"min_body": 55, "max_non_body_insert": 12},
+}
+
+
+def _role_texts(blocks: list[dict], role: str) -> list[str]:
+ return [str(block.get("text") or block.get("block_content") or "") for block in blocks if block.get("role") == role]
+
+
+@pytest.mark.parametrize("key", sorted(BODY_RETENTION))
+def test_problem_papers_retain_body_and_avoid_mass_insert_suppression(rebuilt_real_papers: dict, key: str) -> None:
+ _require_artifacts(key)
+ blocks = _read_jsonl(_structured_path(key))
+ roles = [block.get("role") for block in blocks]
+ thresholds = BODY_RETENTION[key]
+
+ assert roles.count("body_paragraph") >= thresholds["min_body"], roles.count("body_paragraph")
+ assert roles.count("non_body_insert") <= thresholds["max_non_body_insert"], roles.count("non_body_insert")
+
+
+def test_tsckavis_frontmatter_and_key_points_are_callout(rebuilt_real_papers: dict) -> None:
+ key = "TSCKAVIS"
+ _require_artifacts(key)
+ blocks = _read_jsonl(_structured_path(key))
+ fulltext = _fulltext_path(key).read_text(encoding="utf-8", errors="replace")
+ heading_text = "\n".join(
+ _role_texts(blocks, "section_heading")
+ + _role_texts(blocks, "subsection_heading")
+ + _role_texts(blocks, "sub_subsection_heading")
+ ).lower()
+
+ # Frontmatter must not leak into content headings
+ assert "review article" not in heading_text
+ assert "steve stegen" not in heading_text
+ assert "geert carmeliet" not in heading_text
+
+ # Key points must render as a callout block, not suppressed
+ assert "key points" in fulltext.lower()
+ assert "[!NOTE]" in fulltext
+ assert "skeletal stem and progenitor cells display a high metabolic flexibility" in fulltext.lower()
+
+ # Metadata must capture title and authors from OCR blocks
+ meta = _read_json(_metadata_path(key))
+ assert len(meta.get("title", {}).get("value", "")) > 10, meta.get("title")
+ assert len(meta.get("authors", {}).get("value", [])) > 0, meta.get("authors")
+ assert meta.get("authors_display", "") != "", f"authors_display is empty: {meta}"
+ assert meta.get("year", {}).get("value", 0), meta.get("year")
+ assert meta.get("journal", {}).get("value", ""), meta.get("journal")
+ assert meta.get("doi", {}).get("value", ""), meta.get("doi")
+
+
+CONTROL_MIN_BODY = {
+ "SAN9AYVR": 250,
+ "2GN9LMCW": 25,
+ "7C8829BD": 70,
+}
+
+
+@pytest.mark.parametrize("key", sorted(CONTROL_MIN_BODY))
+def test_control_papers_keep_body_and_tail_stability(rebuilt_real_papers: dict, key: str) -> None:
+ _require_artifacts(key)
+ blocks = _read_jsonl(_structured_path(key))
+ roles = [block.get("role") for block in blocks]
+ fulltext = _fulltext_path(key).read_text(encoding="utf-8", errors="replace")
+
+ assert roles.count("body_paragraph") >= CONTROL_MIN_BODY[key]
+ assert any(w in fulltext for w in ["References", "references", "REFERENCE", "Bibliography"]), "Tail reference marker not found"
diff --git a/tests/test_ocr_rebuild.py b/tests/test_ocr_rebuild.py
index 4955c32c..1d5e8143 100644
--- a/tests/test_ocr_rebuild.py
+++ b/tests/test_ocr_rebuild.py
@@ -50,3 +50,44 @@ def test_resolve_source_pdf_stale_missing() -> None:
meta = {"source_pdf": r"D:\nonexistent\path.pdf"}
result = _resolve_source_pdf_for_rebuild(vault, "SAN9AYVR", meta)
assert result is not None and result.exists(), f"Fallback failed for stale key: {result}"
+
+
+def test_enrich_source_metadata_from_paper_note(monkeypatch, tmp_path) -> None:
+ from paperforge.worker import _utils
+ from paperforge.worker.ocr_rebuild import _enrich_meta_from_paper_note
+ from paperforge.core.io import read_json
+
+ vault = tmp_path / "vault"
+ lit_dir = vault / "literature"
+ note_dir = lit_dir / "Biology"
+ note_dir.mkdir(parents=True)
+ note_path = note_dir / "TSCKAVIS.md"
+ note_path.write_text(
+ """---
+zotero_key: TSCKAVIS
+title: Metabolic regulation of skeletal cell fate and function in development and disease
+authors:
+ - Steve Stegen
+ - Geert Carmeliet
+year: 2022
+journal: Nature Reviews Endocrinology
+doi: 10.1038/s41574-021-00588-4
+---
+""",
+ encoding="utf-8",
+ )
+
+ meta_path = vault / "ocr" / "TSCKAVIS" / "source_metadata.json"
+ meta_path.parent.mkdir(parents=True)
+ meta_path.write_text("{}", encoding="utf-8")
+
+ monkeypatch.setattr(_utils, "pipeline_paths", lambda _vault: {"literature": lit_dir, "ocr": vault / "ocr"})
+
+ _enrich_meta_from_paper_note(vault, "TSCKAVIS", meta_path)
+
+ meta = read_json(meta_path)
+ assert meta["title"].startswith("Metabolic regulation")
+ assert meta["authors"] == ["Steve Stegen", "Geert Carmeliet"]
+ assert meta["year"] == 2022
+ assert meta["journal"] == "Nature Reviews Endocrinology"
+ assert meta["doi"] == "10.1038/s41574-021-00588-4"
diff --git a/tests/test_ocr_render_stabilization.py b/tests/test_ocr_render_stabilization.py
index b85c9343..e1b1b813 100644
--- a/tests/test_ocr_render_stabilization.py
+++ b/tests/test_ocr_render_stabilization.py
@@ -315,7 +315,7 @@ def test_stabilize_heading_sanity_allows_valid_short_heading() -> None:
table_inventory={},
)
- assert "### 1 Introduction" in md
+ assert "## 1 Introduction" in md
def test_stabilize_heading_sanity_downgrades_verb_heavy_heading() -> None:
@@ -374,7 +374,7 @@ def test_stabilize_heading_sanity_allows_verb_short_heading() -> None:
table_inventory={},
)
- assert "### Results are shown" in md
+ assert "## Results are shown" in md
def test_stabilize_reference_content_mapped() -> None:
@@ -916,8 +916,8 @@ def test_normalize_ocr_math_text_display_math() -> None:
assert normalize_ocr_math_text("$$ ... $$") == "$$...$$"
-def test_structured_insert_block_not_rendered() -> None:
- """structured_insert blocks should not appear in rendered markdown."""
+def test_structured_insert_renders_as_single_callout_without_swallowing_body() -> None:
+ """structured_insert should render as one callout and must not absorb body text."""
from paperforge.worker.ocr_render import render_fulltext_markdown
blocks = [
@@ -959,18 +959,127 @@ def test_structured_insert_block_not_rendered() -> None:
md = render_fulltext_markdown(structured_blocks=blocks, resolved_metadata={}, figure_inventory={}, table_inventory={})
assert "2 Results" in md
- assert "Key points" not in md, "structured_insert should not render"
- assert "Point one" not in md, "structured_insert content should not render"
+ assert "Key points" in md, "structured_insert heading should render"
+ assert "Point one" in md, "structured_insert content should render"
+ assert "Point two" in md, "structured_insert content should render"
+ assert md.count("[!NOTE]") == 1, f"Expected exactly one callout, got markdown:\n{md}"
assert "Body text continues" in md
+def test_adjacent_structured_insert_blocks_merge_into_one_callout() -> None:
+ from paperforge.worker.ocr_render import render_fulltext_markdown
+
+ blocks = [
+ {
+ "paper_id": "KEY",
+ "page": 2,
+ "block_id": "s1",
+ "role": "structured_insert",
+ "text": "Key points",
+ "render_default": False,
+ "index_default": False,
+ "bbox": [80, 100, 200, 130],
+ "page_width": 1200,
+ "page_height": 1700,
+ },
+ {
+ "paper_id": "KEY",
+ "page": 2,
+ "block_id": "s2",
+ "role": "structured_insert",
+ "text": "• Point one\n• Point two",
+ "render_default": False,
+ "index_default": False,
+ "bbox": [80, 140, 500, 260],
+ "page_width": 1200,
+ "page_height": 1700,
+ },
+ {
+ "paper_id": "KEY",
+ "page": 2,
+ "block_id": "b1",
+ "role": "body_paragraph",
+ "text": "Main body resumes here.",
+ "render_default": True,
+ "bbox": [80, 320, 500, 360],
+ "page_width": 1200,
+ "page_height": 1700,
+ },
+ ]
+
+ md = render_fulltext_markdown(structured_blocks=blocks, resolved_metadata={}, figure_inventory={}, table_inventory={})
+
+ assert md.count("[!NOTE]") == 1, f"Expected merged callout, got markdown:\n{md}"
+ assert "Key points" in md
+ assert "Point one" in md
+ assert "Point two" in md
+ assert "Key points\n\n> • Point one" not in md, f"Merged callout should not contain blank separator:\n{md}"
+ assert "Main body resumes here." in md
+
+
+def test_malformed_sidebar_tail_fragment_is_dropped_when_continuation_block_follows() -> None:
+ from paperforge.worker.ocr_render import render_fulltext_markdown
+
+ blocks = [
+ {
+ "paper_id": "KEY",
+ "page": 2,
+ "block_id": "s1",
+ "role": "structured_insert",
+ "text": "| • Point one |
| cell dysfunctionatic and leukaemic cells• their tumorigenic spread. |
",
+ "render_default": False,
+ "index_default": False,
+ "bbox": [80, 140, 500, 260],
+ "page_width": 1200,
+ "page_height": 1700,
+ },
+ {
+ "paper_id": "KEY",
+ "page": 2,
+ "block_id": "s2",
+ "role": "structured_insert",
+ "text": "• Metabolic disturbance is linked to skeletal cell dysfunction during bone pathology, and bone-metastatic and leukaemic cells hijack skeletal cell metabolism to support their tumorigenic spread.",
+ "render_default": False,
+ "index_default": False,
+ "bbox": [80, 265, 500, 320],
+ "page_width": 1200,
+ "page_height": 1700,
+ },
+ ]
+
+ md = render_fulltext_markdown(structured_blocks=blocks, resolved_metadata={}, figure_inventory={}, table_inventory={})
+
+ assert "cell dysfunctionatic and leukaemic cells• their tumorigenic spread." not in md
+ assert "Metabolic disturbance is linked to skeletal cell dysfunction" in md
+
+
def test_section_heading_renders_with_prefix() -> None:
from paperforge.worker.ocr_render import render_fulltext_markdown
blocks = [
{"paper_id": "KEY", "page": 3, "block_id": "b1", "role": "section_heading", "text": "1 Introduction", "render_default": True, "bbox": [80, 200, 500, 230], "page_width": 1200, "page_height": 1700},
]
md = render_fulltext_markdown(structured_blocks=blocks, resolved_metadata={}, figure_inventory={}, table_inventory={})
- assert "### 1 Introduction" in md, f"Expected ### prefix, got: {md[:200]}"
+ assert "## 1 Introduction" in md, f"Expected ## prefix, got: {md[:200]}"
+
+
+def test_unumbered_subsection_headings_use_double_hash_by_default() -> None:
+ from paperforge.worker.ocr_render import render_fulltext_markdown
+
+ blocks = [
+ {"paper_id": "KEY", "page": 2, "block_id": "h1", "role": "sub_subsection_heading", "text": "Introduction", "render_default": True, "bbox": [80, 200, 300, 230], "page_width": 1200, "page_height": 1700},
+ {"paper_id": "KEY", "page": 2, "block_id": "h2", "role": "subsection_heading", "text": "2.5 Electrical stimulation", "render_default": True, "bbox": [80, 260, 700, 300], "page_width": 1200, "page_height": 1700},
+ ]
+
+ md = render_fulltext_markdown(
+ structured_blocks=blocks,
+ resolved_metadata={},
+ figure_inventory={},
+ table_inventory={},
+ )
+
+ lines = md.splitlines()
+ assert "## Introduction" in lines
+ assert "### 2.5 Electrical stimulation" in lines
# === 2GN9LMCW / 7C8829BD guard tests (Task 7 -- preserve tail mainline) ===
diff --git a/tests/test_ocr_render_v2.py b/tests/test_ocr_render_v2.py
index 3b6340b3..8a65e53b 100644
--- a/tests/test_ocr_render_v2.py
+++ b/tests/test_ocr_render_v2.py
@@ -33,5 +33,5 @@ def test_render_fulltext_uses_resolved_metadata_and_object_links(tmp_path) -> No
assert "# Paper Title" in md
assert "Authors:" in md
- assert "## 1 Introduction" in md or "### 1 Introduction" in md
+ assert "## 1 Introduction" in md
assert "![[figures/figure_001.md]]" in md or "![[render/figures/figure_001.md]]" in md
diff --git a/tests/test_ocr_rendering.py b/tests/test_ocr_rendering.py
index a84fa9fa..61f80c36 100644
--- a/tests/test_ocr_rendering.py
+++ b/tests/test_ocr_rendering.py
@@ -708,7 +708,7 @@ def test_render_page_blocks_keeps_top_figure_before_body_and_keeps_author_et_al_
assert body_lines[1].startswith("![[")
assert body_lines[2] == "### 2.5 Electrical stimulation for tissue-engineered articular cartilage"
assert body_lines[6].startswith("Kwon et al. demonstrated")
- assert body_lines[7] == "### 3 Discussion"
+ assert body_lines[7] == "## 3 Discussion"
def test_validate_block_order_preserves_full_width_blocks_between_columns() -> None:
@@ -869,12 +869,12 @@ def test_render_page_blocks_keeps_abstract_heading_before_abstract_body_on_first
rendered = render_page_blocks(vault, 1, result, images_dir, page_cache_dir, pdf_doc=None)
body_lines = [line for line in rendered if line and not line.startswith("