diff --git a/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p1.png b/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p1.png new file mode 100644 index 00000000..cb40a5f3 Binary files /dev/null and b/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p1.png differ diff --git a/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p2.png b/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p2.png new file mode 100644 index 00000000..db85bb85 Binary files /dev/null and b/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p2.png differ diff --git a/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p5.png b/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p5.png new file mode 100644 index 00000000..4bccf14a Binary files /dev/null and b/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p5.png differ diff --git a/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p6.png b/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p6.png new file mode 100644 index 00000000..0d2c81a7 Binary files /dev/null and b/.tmp/tail_pdf_samples/Fitzsimmons_等_-_2008_-_A_pulsing_electri_p6.png differ diff --git a/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p1.png b/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p1.png new file mode 100644 index 00000000..5e1d3a54 Binary files /dev/null and b/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p1.png differ diff --git a/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p2.png b/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p2.png new file mode 100644 index 00000000..63abbccb Binary files /dev/null and b/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p2.png differ diff --git a/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p25.png b/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p25.png new file mode 100644 index 00000000..aa6bf95f Binary files /dev/null and b/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p25.png differ diff --git a/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p26.png b/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p26.png new file mode 100644 index 00000000..423882d5 Binary files /dev/null and b/.tmp/tail_pdf_samples/Masante_等_-_2025_-_Insights_into_bone_an_p26.png differ diff --git a/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p1.png b/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p1.png new file mode 100644 index 00000000..8716d34e Binary files /dev/null and b/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p1.png differ diff --git a/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p14.png b/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p14.png new file mode 100644 index 00000000..401e50ea Binary files /dev/null and b/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p14.png differ diff --git a/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p15.png b/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p15.png new file mode 100644 index 00000000..3c81a590 Binary files /dev/null and b/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p15.png differ diff --git a/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p2.png b/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p2.png new file mode 100644 index 00000000..8f8aa5d5 Binary files /dev/null and b/.tmp/tail_pdf_samples/Mobini_等_-_2017_-_In_vitro_effect_of_dir_p2.png differ diff --git a/docs/superpowers/checklists/2026-06-05-ocr-formal-object-checklist.md b/docs/superpowers/checklists/2026-06-05-ocr-formal-object-checklist.md new file mode 100644 index 00000000..cbba6c6a --- /dev/null +++ b/docs/superpowers/checklists/2026-06-05-ocr-formal-object-checklist.md @@ -0,0 +1,60 @@ +# OCR Formal Object Checklist + +Use this checklist before calling the OCR remediation complete. + +## A. Frontmatter + +- [ ] `paper_title` is unique and page-1-only +- [ ] authors are either recovered or correctly isolated +- [ ] affiliations are isolated from body +- [ ] DOI is localized correctly +- [ ] abstract is rendered +- [ ] frontmatter noise does not enter body or heading buckets + +## B. Headings + +- [ ] no long body paragraph is rendered as a heading +- [ ] `Figure N shows ...` body references are not headings +- [ ] heading hierarchy is consistent across the paper +- [ ] references heading is recognized + +## C. Figures + +- [ ] formal legends are distinct from body mentions +- [ ] candidate legends are not silently promoted without evidence +- [ ] `legend_only` figures remain assetless +- [ ] orphan assets stay orphaned +- [ ] figure object titles use formal figure numbers +- [ ] figure crops match the intended asset bbox + +## D. Tables + +- [ ] formal table numbers are preserved +- [ ] continuation pages merge into the same formal table +- [ ] table object titles use formal table numbers +- [ ] no raw table HTML remains in `fulltext.md` +- [ ] table crops match the intended asset bbox + +## E. Cropping / Assets + +- [ ] OCR page-image coordinate cropping is used +- [ ] cached `pages/page_XXX` images are preferred when available +- [ ] `assets/` is structured truth +- [ ] `images/` remains compatibility only +- [ ] path mapping between `assets/` and `images/` is explicit + +## F. Index / Health + +- [ ] `body` index bucket is free of frontmatter furniture +- [ ] `references` bucket contains actual reference items +- [ ] `abstract_found` is correct +- [ ] `references_found` is correct + +## G. Real Paper: `7C8829BD` + +- [ ] Figure 3 uses the real chart, not the body mention +- [ ] Figure 4 uses the real page-14 chart +- [ ] no `Check for updates` figure object remains +- [ ] Table 6 and `Table 6 (Continued)` are one formal table +- [ ] Table 7 remains Table 7 +- [ ] `fulltext.md` has no inline raw table HTML diff --git a/docs/superpowers/plans/2026-06-05-ocr-backmatter-and-closure-plan.md b/docs/superpowers/plans/2026-06-05-ocr-backmatter-and-closure-plan.md new file mode 100644 index 00000000..2b041a0f --- /dev/null +++ b/docs/superpowers/plans/2026-06-05-ocr-backmatter-and-closure-plan.md @@ -0,0 +1,274 @@ +# OCR Backmatter Tail-Ordering Closure Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Finish the last unresolved OCR issues for `7C8829BD` by fixing tail-page ordering, section-body attachment, and references-zone attachment, without reopening the already-stable figure/table/frontmatter/cropping work. + +**Architecture:** This is a narrow closure plan. Backmatter heading detection is already working well enough; the remaining bug is that tail-page blocks are rendered in the wrong structural order. The fix should happen as a dedicated `tail-page ordering` pass after role assignment but before final markdown emission. That pass should attach each `backmatter_body` to the correct `backmatter_heading`, keep `reference_item` blocks under `reference_heading`, and prevent tail sections like `Publisher's note` and `Supplementary material` from being interleaved incorrectly with references. + +**Tech Stack:** Python, pytest, structured OCR artifacts, real-paper validation on `D:\L\OB\Literature-hub\System\PaperForge\ocr\7C8829BD` + +--- + +## Already Complete + +These are done and should not be reopened in this plan: + +- frontmatter title/metadata block render +- abstract render +- figure 3/4 matching and crop recovery +- `assets/` truth + `images/` compatibility mapping +- inline raw `` removal from `fulltext.md` +- formal table count vs segment count in health +- page-marker mismatch in `meta.error` +- backmatter headings are mostly recognized as headings + +## Remaining Problems + +For `7C8829BD`, the unresolved issues are now: + +1. `Generative AI statement`, `References`, `Publisher's note`, and `Supplementary material` are still rendered in the wrong order. +2. Tail section bodies are not reliably attached to the heading that owns them. +3. Reference items start under the wrong heading region because the `References` zone boundary is not enforced strongly enough. + +Current bad pattern in `fulltext.md`: + +- `## Generative AI statement` +- `## References` +- `## Publisher's note` +- body for Generative AI statement +- body for Publisher's note +- `## Supplementary material` +- then references begin + +This shows the problem is no longer role detection; it is tail-page structural ordering. + +## Root Cause Summary + +### 1. Tail-page reading order is missing + +The current pipeline can recognize tail headings, but it still renders page 22 mostly in linear block order. + +That is insufficient because the tail page mixes: + +- left-column `References` +- right-column `Generative AI statement` +- right-column `Publisher's note` +- lower blocks for `Supplementary material` + +These need explicit structural ordering, not naive per-block rendering. + +### 2. Section-body attachment is missing + +The renderer does not yet explicitly attach: + +- `backmatter_body` -> nearest owning `backmatter_heading` +- `reference_item` -> active `reference_heading` + +So the right bodies can drift beneath the wrong heading. + +### 3. `References` needs a stronger section boundary + +Once `reference_heading` appears, the renderer should open a `references zone` and keep subsequent `reference_item` blocks there unless a later tail heading clearly owns intervening non-reference text. + +Right now, the boundary is too weak, so other tail headings can be emitted before references are laid down correctly. + +## Target Tail Contract + +For the tail portion of the paper, the rendered structure should become: + +```md +## Author contributions +... + +## Funding +... + +## Acknowledgments +... + +## Conflict of interest +... + +## Generative AI statement +... + +## Publisher's note +... + +## Supplementary material +... + +## References +... +``` + +Or, if the paper’s page-local structure clearly places `References` before certain later sections, the renderer should still maintain **section ownership consistency**: + +- each heading immediately followed by its own body +- references grouped under `## References` +- no heading/body cross-wire + +The strict requirement is not a fixed global tail order template. +The strict requirement is **correct heading-body ownership plus a coherent references zone**. + +## Task 1: Lock Tail-Ordering Failures In Tests + +**Files:** +- Modify: `tests/test_ocr_rendering.py` +- Modify: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Add a failing heading-body attachment test** + +Build a fixture with: + +- `backmatter_heading("Generative AI statement")` +- its body block +- `reference_heading("References")` +- several `reference_item` +- `backmatter_heading("Publisher's note")` +- its body block + +Expected: + +- each heading is followed by its own body +- reference items stay grouped under `References` + +- [ ] **Step 2: Add a failing tail mixed-column ordering test** + +Model a page with: + +- left-column `References` +- right-column `Generative AI statement` +- right-column `Publisher's note` +- lower `Supplementary material` + +Expected: + +- renderer does not interleave heading/body ownership incorrectly + +- [ ] **Step 3: Run tests to confirm failure** + +Run: `python -m pytest tests/test_ocr_rendering.py tests/test_ocr_render_stabilization.py -q` + +Expected: FAIL + +## Task 2: Add A Tail-Page Ordering Pass + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Possibly create helper logic inside `paperforge/worker/ocr_render.py` +- Test: `tests/test_ocr_rendering.py` + +- [ ] **Step 1: Partition tail blocks before rendering** + +For late pages containing any of: + +- `backmatter_heading` +- `reference_heading` +- `reference_item` +- `backmatter_body` + +collect them into a dedicated tail-page structure instead of rendering them one block at a time. + +- [ ] **Step 2: Attach body blocks to the nearest valid owning heading** + +Use page-local signals: + +- same column or strong horizontal overlap +- closest heading above +- no stronger competing heading between them +- short vertical gap + +Result: + +- `Generative AI statement` body attaches to `Generative AI statement` +- `Publisher's note` body attaches to `Publisher's note` +- `Supplementary material` body attaches to `Supplementary material` + +- [ ] **Step 3: Keep reference items in a dedicated references bucket** + +If a block is `reference_item`, it should not be claimed by backmatter headings. + +Reference items should be collected under the active `reference_heading`, even if nearby headings exist in another column. + +- [ ] **Step 4: Render the ordered tail sections back into markdown** + +The renderer should emit: + +- heading +- attached body blocks + +section by section, instead of by raw block sequence. + +- [ ] **Step 5: Verify** + +Run: `python -m pytest tests/test_ocr_rendering.py -q` + +Expected: PASS + +## Task 3: Strengthen The References Zone Boundary + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Possibly modify: `paperforge/worker/ocr_roles.py` only if a role distinction is missing +- Test: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Open a references zone on `reference_heading`** + +When `reference_heading` appears: + +- start a `references zone` +- collect `reference_item` blocks into it +- do not let generic body/backmatter ordering pull them elsewhere + +- [ ] **Step 2: Allow later tail sections without stealing references** + +If later headings like `Publisher's note` or `Supplementary material` exist: + +- their own body blocks may render after their headings +- but `reference_item` blocks remain under `References` + +- [ ] **Step 3: Verify** + +Run: `python -m pytest tests/test_ocr_render_stabilization.py -q` + +Expected: PASS + +## Task 4: Real-Paper Closure Audit On `7C8829BD` + +**Files:** +- Verify only + +- [ ] **Step 1: Rebuild the real paper** + +Run the current derived rebuild/backfill path for `7C8829BD`. + +- [ ] **Step 2: Verify tail structure directly in `fulltext.md`** + +Check: + +- `## Author contributions` +- `## Funding` +- `## Acknowledgments` +- `## Conflict of interest` +- `## Generative AI statement` +- `## References` +- `## Publisher's note` +- `## Supplementary material` + +all render as headings, and each body text sits under the correct heading. + +- [ ] **Step 3: Verify reference grouping** + +Check: + +- first reference item appears under `## References` +- `Publisher's note` body is not attached to `Generative AI statement` +- `Supplementary material` body is not mixed into references + +## Risks + +1. A tail-page ordering pass can accidentally overfit this paper if implemented as a hardcoded heading order template instead of ownership-based grouping. +2. Column-aware attachment rules can misfire on unusual single-column tails unless horizontal/vertical evidence is balanced carefully. +3. References-zone grouping can become too strong if it swallows genuine later tail sections; keep grouping tied to `reference_item` roles only. diff --git a/docs/superpowers/plans/2026-06-05-ocr-formal-object-remediation-plan.md b/docs/superpowers/plans/2026-06-05-ocr-formal-object-remediation-plan.md new file mode 100644 index 00000000..be732a46 --- /dev/null +++ b/docs/superpowers/plans/2026-06-05-ocr-formal-object-remediation-plan.md @@ -0,0 +1,312 @@ +# OCR Formal Object Remediation Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Repair the OCR formal-object pipeline so frontmatter, headings, figures, tables, continuation handling, and cropping all obey the new layer contracts, with `7C8829BD` as the real-paper acceptance target. + +**Architecture:** Implement this as a layered remediation, not piecemeal patching. First stabilize role and frontmatter detection, then repair figure/table formal object detection, then continuation-aware matching, then keep cropping strictly subordinate to those matching decisions, and finally update object notes/render/health/index to consume the corrected outputs. The key rule is that identity and matching decisions belong upstream, while cropping and rendering are pure consumers. + +**Tech Stack:** Python, pytest, Pillow, PyMuPDF/fitz, current OCR artifact pipeline, legacy OCR corpus at `D:\L\OB\Literature-hub` + +--- + +## File Structure + +- `paperforge/worker/ocr_roles.py` + - repair title/frontmatter/reference/body-mention role logic +- `paperforge/worker/ocr_metadata.py` + - lock metadata recovery to frontmatter analysis +- `paperforge/worker/ocr_figures.py` + - split formal legends from body mentions and candidate legends +- `paperforge/worker/ocr_tables.py` + - continuation-aware formal table model +- `paperforge/worker/ocr_objects.py` + - keep cropping subordinate to matching; fix object note numbering/sections +- `paperforge/worker/ocr_render.py` + - stop inline table HTML; consume corrected object notes +- `paperforge/worker/ocr_index.py` + - keep body bucket free of figure/table prose confusion +- `paperforge/worker/ocr_health.py` + - reflect stabilized references/tables/figures +- `paperforge/worker/ocr.py` + - pass through required page-dimension/path-map context +- `paperforge/worker/ocr_rebuild.py` + - make rebuild regenerate corrected inventories/assets/notes +- `tests/test_ocr_roles.py` +- `tests/test_ocr_metadata.py` +- `tests/test_ocr_figures.py` +- `tests/test_ocr_tables.py` +- `tests/test_ocr_objects.py` +- `tests/test_ocr_rendering.py` +- `tests/test_ocr_health.py` +- `tests/test_ocr_index.py` +- `tests/test_ocr_render_stabilization.py` + +## Task 1: Lock The Real Failure Modes In Tests + +**Files:** +- Modify: `tests/test_ocr_roles.py` +- Modify: `tests/test_ocr_figures.py` +- Modify: `tests/test_ocr_tables.py` +- Modify: `tests/test_ocr_rendering.py` +- Modify: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Add a failing body-mention vs formal-legend test** + +Fixture: + +- one prose block `Figure 3 shows ...` +- one true figure legend block on next page via `figure_title` +- one matching chart asset + +Expected: + +- prose block becomes `body_figure_mention` or body paragraph +- true legend becomes the formal legend + +- [ ] **Step 2: Add a failing legend-only no-orphan-substitution test** + +Expected: + +- a `legend_only` figure object keeps no asset +- object-writing layer does not pull in arbitrary `unmatched_assets` + +- [ ] **Step 3: Add a failing table-continuation merge test** + +Fixture: + +- `Table 6` +- `Table 6 (Continued)` +- two page-local assets + +Expected: + +- one formal table object +- two physical segments + +- [ ] **Step 4: Add a failing render test for inline table HTML** + +Expected: + +- `fulltext.md` does not contain raw `
` HTML +- table object references appear instead + +- [ ] **Step 5: Run the focused failures** + +Run: `python -m pytest tests/test_ocr_roles.py tests/test_ocr_figures.py tests/test_ocr_tables.py tests/test_ocr_rendering.py tests/test_ocr_render_stabilization.py -q` + +Expected: FAIL + +## Task 2: Repair Frontmatter And Heading Semantics + +**Files:** +- Modify: `paperforge/worker/ocr_roles.py` +- Modify: `paperforge/worker/ocr_metadata.py` +- Test: `tests/test_ocr_roles.py` +- Test: `tests/test_ocr_metadata.py` + +- [ ] **Step 1: Remove unsafe `paragraph_title -> paper_title` fallback** + +Replace with page-1/title-zone-only admission. + +- [ ] **Step 2: Add explicit `reference_content` handling** + +Map OCR `reference_content` to `reference_item`. + +- [ ] **Step 3: Split body references to figures from formal legends** + +Add explicit exclusion for patterns like: + +- `Figure 3 shows` +- `Figure 2 illustrates` +- `Figure 4 depicts` + +- [ ] **Step 4: Tighten `text -> heading` promotion** + +Only allow under strong geometric/profile conditions. + +- [ ] **Step 5: Verify** + +Run: `python -m pytest tests/test_ocr_roles.py tests/test_ocr_metadata.py -q` + +Expected: PASS + +## Task 3: Rebuild Formal Figure Detection And Matching + +**Files:** +- Modify: `paperforge/worker/ocr_figures.py` +- Possibly modify: `paperforge/worker/ocr_roles.py` +- Test: `tests/test_ocr_figures.py` + +- [ ] **Step 1: Introduce explicit figure categories** + +Represent: + +- formal legend +- candidate legend +- body mention +- matched figure +- legend-only figure +- orphan asset + +- [ ] **Step 2: Restore the old prose exclusion guard** + +The new pipeline should incorporate `master`’s anti-body-reference logic into the formal legend path. + +- [ ] **Step 3: Remove orphan substitution for legend-only figures** + +If no asset is matched, keep the figure assetless. + +- [ ] **Step 4: Keep unmatched legends/assets honest** + +Populate `unmatched_legends` and `unmatched_assets` truthfully. + +- [ ] **Step 5: Verify** + +Run: `python -m pytest tests/test_ocr_figures.py -q` + +Expected: PASS + +## Task 4: Rebuild Formal Table Detection With Continuation + +**Files:** +- Modify: `paperforge/worker/ocr_tables.py` +- Modify: `paperforge/worker/ocr_objects.py` +- Test: `tests/test_ocr_tables.py` +- Test: `tests/test_ocr_objects.py` + +- [ ] **Step 1: Add formal number + continuation model** + +Each table object should carry: + +- `formal_table_number` +- `segments` +- `is_continuation` +- `continuation_of` + +- [ ] **Step 2: Merge `Table N (Continued)` into the prior formal table** + +Do not create a new displayed formal number. + +- [ ] **Step 3: Make object note title use formal table number** + +Avoid `table_007 -> # Table 7` if the formal caption is `Table 6`. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_tables.py tests/test_ocr_objects.py -q` + +Expected: PASS + +## Task 5: Keep Cropping Strictly Subordinate To Matching + +**Files:** +- Modify: `paperforge/worker/ocr_objects.py` +- Modify: `paperforge/worker/ocr_blocks.py` +- Modify: `paperforge/worker/ocr.py` +- Modify: `paperforge/worker/ocr_rebuild.py` +- Test: `tests/test_ocr_objects.py` + +- [ ] **Step 1: Preserve OCR page dimensions into structured blocks** + +Do not drop `page_width` / `page_height`. + +- [ ] **Step 2: Prefer cached OCR page images for crop** + +Use `pages/page_XXX.(jpg|png)` first, then render-to-OCR-size fallback. + +- [ ] **Step 3: Ensure cropping does not modify matching outcome** + +No reassignment of bbox ownership inside cropping/object-writing. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_objects.py -q` + +Expected: PASS + +## Task 6: Render And Index Consumption Cleanup + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Modify: `paperforge/worker/ocr_index.py` +- Modify: `paperforge/worker/ocr_health.py` +- Test: `tests/test_ocr_rendering.py` +- Test: `tests/test_ocr_index.py` +- Test: `tests/test_ocr_health.py` + +- [ ] **Step 1: Remove inline table HTML from `fulltext.md`** + +Use table object references instead. + +- [ ] **Step 2: Make body bucket ignore formal figure/table prose confusion** + +Do not let formal captions or body mentions pollute the wrong index buckets. + +- [ ] **Step 3: Make health reflect reference/content correctness** + +Verify `references_found` once `reference_content` is mapped. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_rendering.py tests/test_ocr_index.py tests/test_ocr_health.py -q` + +Expected: PASS + +## Task 7: Real-Paper Verification On `7C8829BD` + +**Files:** +- Verify only + +- [ ] **Step 1: Rebuild the real paper** + +Run the corrected backfill / derived rebuild path on `7C8829BD`. + +- [ ] **Step 2: Verify figure behavior** + +Expected: + +- Figure 3 asset is the page-11 chart, not page-10 prose +- Figure 4 asset is the page-14 chart +- no `Check for updates` figure fallback + +- [ ] **Step 3: Verify table continuation behavior** + +Expected: + +- page 15 and 16 remain formal Table 6 +- page 19 remains formal Table 7 +- object-note display numbering follows formal table number + +- [ ] **Step 4: Verify render behavior** + +Expected: + +- no inline table HTML in `fulltext.md` +- table object links instead + +## Task 8: Final Verification + +**Files:** +- Verify only + +- [ ] **Step 1: Run focused suite** + +Run: `python -m pytest tests/test_ocr_roles.py tests/test_ocr_metadata.py tests/test_ocr_figures.py tests/test_ocr_tables.py tests/test_ocr_objects.py tests/test_ocr_rendering.py tests/test_ocr_health.py tests/test_ocr_index.py tests/test_ocr_render_stabilization.py -q` + +Expected: PASS + +- [ ] **Step 2: Run real-paper smoke checks** + +Expected: + +- assets are correct +- numbering is correct +- no orphan substitution regression + +## Risks + +1. Removing body-mention legends may reduce figure recall if candidate-legend logic is too weak. +2. Table continuation merging may accidentally over-merge distinct same-number supplements if numbering context is weak. +3. Render cleanup may break existing downstream consumers expecting inline table HTML. +4. Real-paper validation must confirm that formal-number display stays correct after continuation merging. diff --git a/docs/superpowers/plans/2026-06-05-ocr-phase-closure-cleanup.md b/docs/superpowers/plans/2026-06-05-ocr-phase-closure-cleanup.md new file mode 100644 index 00000000..d7dcb97b --- /dev/null +++ b/docs/superpowers/plans/2026-06-05-ocr-phase-closure-cleanup.md @@ -0,0 +1,332 @@ +# OCR Phase Closure Cleanup Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Close the remaining OCR stabilization gaps after the major figure/table remediation work, so the pipeline matches the intended contracts for title/frontmatter isolation, metadata cleanliness, formal-table downstream consumption, backmatter heading stability, and page-marker compatibility. + +**Architecture:** This is a closure pass, not a redesign. The major object-detection and cropping fixes are already in place and working for `7C8829BD`. The remaining issues are downstream leaks and stale fallback logic: late-paper `paper_title` pollution, frontmatter blocks leaking into body render, physical table segments still being counted and displayed too literally downstream, unclean metadata display values, unstable backmatter heading handling on the tail pages, and page-marker compatibility still reporting a mismatch. This pass should only tighten those seams. + +**Tech Stack:** Python, pytest, structured OCR artifacts, real-paper verification against `D:\L\OB\Literature-hub\System\PaperForge\ocr\7C8829BD` + +--- + +## Root Cause Summary + +The remaining problems come from six concrete causes: + +1. [`ocr_roles.py`](D:/L/Med/Research/99_System/LiteraturePipeline/github-release/.worktrees/feat-ocr-structured-pipeline/paperforge/worker/ocr_roles.py:157) + still allows page-1 zone heuristics plus generic unnumbered `paragraph_title` fallback to produce `paper_title` outside the real title object lifecycle. + +2. [`ocr_render.py`](D:/L/Med/Research/99_System/LiteraturePipeline/github-release/.worktrees/feat-ocr-structured-pipeline/paperforge/worker/ocr_render.py:103) + still renders blocks whose semantic content has already been consumed into the metadata block, so author/frontmatter material can appear twice. + +3. [`ocr_tables.py`](D:/L/Med/Research/99_System/LiteraturePipeline/github-release/.worktrees/feat-ocr-structured-pipeline/paperforge/worker/ocr_tables.py:127) + now models `formal_table_number` and `segments`, but downstream consumers such as health/render still mostly count or treat physical segment rows literally. + +4. [`ocr_metadata.py`](D:/L/Med/Research/99_System/LiteraturePipeline/github-release/.worktrees/feat-ocr-structured-pipeline/paperforge/worker/ocr_metadata.py:80) + accepts OCR author strings as-is, so metadata display still carries raw math-style affiliation markers and unclean join patterns. + +5. [`ocr_render.py`](D:/L/Med/Research/99_System/LiteraturePipeline/github-release/.worktrees/feat-ocr-structured-pipeline/paperforge/worker/ocr_render.py:115) + emits page markers only for rendered-page transitions, so the compatibility contract still drifts from `meta.page_count`. + +6. [`ocr_roles.py`](D:/L/Med/Research/99_System/LiteraturePipeline/github-release/.worktrees/feat-ocr-structured-pipeline/paperforge/worker/ocr_roles.py:130) + and [`ocr_render.py`](D:/L/Med/Research/99_System/LiteraturePipeline/github-release/.worktrees/feat-ocr-structured-pipeline/paperforge/worker/ocr_render.py:142) + still lack a distinct backmatter-heading regime. Tail-page small headings are currently split between: + - `frontmatter_noise` and therefore suppressed + - generic `section_heading` + - plain body paragraphs + This makes the tail structure unstable for: + - `Author contributions` + - `Funding` + - `Acknowledgments` + - `Conflict of interest` + - `Generative AI statement` + - `References` + - `Publisher's note` + - `Supplementary material` + +## Real-Paper Residual Issues To Eliminate + +For `7C8829BD`, the following must be fixed: + +1. `Generative AI statement` must not appear as a `paper_title` or title alternative. +2. author line must not reappear in body after already rendering in metadata. +3. `fulltext.md` must not contain raw inline `
...
` HTML. +4. health and downstream stats should distinguish formal table count from physical segment count. +5. page marker mismatch must disappear from `meta.error`. +6. metadata author display should be cleaner and less OCR-raw. +7. tail-page section structure should be explicit and stable. +8. `References` should render as a true heading, not as plain text or an accidental fallback. + +## Task 1: Lock The Remaining Real-Paper Failures In Tests + +**Files:** +- Modify: `tests/test_ocr_roles.py` +- Modify: `tests/test_ocr_rendering.py` +- Modify: `tests/test_ocr_metadata.py` +- Modify: `tests/test_ocr_health.py` +- Modify: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Add a failing late-paper-title pollution test** + +Assert that: + +- `Generative AI statement` +- `Acknowledgments` +- `Funding` +- `Conflict of interest` + +cannot become `paper_title`. + +- [ ] **Step 2: Add a failing frontmatter-duplication render test** + +Assert that when authors are already emitted in the metadata block, the same frontmatter author block is not rendered again in body flow. + +- [ ] **Step 3: Add a failing no-inline-table-html render test** + +Assert that rendered `fulltext.md` contains object references for tables, not literal `` HTML. + +- [ ] **Step 4: Add a failing formal-table-vs-segment health test** + +Assert that health can report formal table count distinctly from physical asset segments. + +- [ ] **Step 5: Add a failing backmatter-heading render test** + +Assert that end-of-article sections such as: + +- `Author contributions` +- `Funding` +- `Acknowledgments` +- `Conflict of interest` +- `Generative AI statement` +- `References` + +render with explicit section semantics instead of being suppressed or flattened. + +- [ ] **Step 6: Add a failing page-marker compatibility test** + +Assert that the rendered compatibility markdown has a marker count matching `page_count`. + +- [ ] **Step 7: Run tests to confirm failure** + +Run: `python -m pytest tests/test_ocr_roles.py tests/test_ocr_rendering.py tests/test_ocr_metadata.py tests/test_ocr_health.py tests/test_ocr_render_stabilization.py -q` + +Expected: FAIL + +## Task 2: Remove Residual `paper_title` Pollution + +**Files:** +- Modify: `paperforge/worker/ocr_roles.py` +- Modify: `paperforge/worker/ocr_metadata.py` +- Test: `tests/test_ocr_roles.py` +- Test: `tests/test_ocr_metadata.py` + +- [ ] **Step 1: Narrow `paper_title` admission** + +Only allow `paper_title` from: + +- trusted frontmatter title zone +- raw `doc_title` +- validated page-1 heading-like candidates that agree with source metadata + +Do not allow later `paragraph_title` blocks to become `paper_title`. + +- [x] **Step 2: Guard title alternatives against backmatter headings** + +`resolve_metadata()` must not preserve OCR title alternatives that originate from late-paper backmatter sections. + +- [ ] **Step 3: Verify** + +Run: `python -m pytest tests/test_ocr_roles.py tests/test_ocr_metadata.py -q` + +Expected: PASS + +## Task 3: Fully Isolate Frontmatter From Body Render + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Possibly modify: `paperforge/worker/ocr_blocks.py` +- Test: `tests/test_ocr_rendering.py` +- Test: `tests/test_ocr_render_stabilization.py` + +- [x] **Step 1: Identify already-consumed frontmatter blocks** + +Once title/authors/doi/affiliations have been emitted into metadata or abstract sections, their corresponding raw blocks should not be rendered again in body flow. + +- [x] **Step 2: Keep real body content untouched** + +Do not over-suppress neighboring introduction/body blocks while removing duplicated frontmatter material. + +- [ ] **Step 3: Verify** + +Run: `python -m pytest tests/test_ocr_rendering.py tests/test_ocr_render_stabilization.py -q` + +Expected: PASS + +## Task 4: Add A Distinct Backmatter-Heading Regime + +**Files:** +- Modify: `paperforge/worker/ocr_roles.py` +- Modify: `paperforge/worker/ocr_render.py` +- Test: `tests/test_ocr_roles.py` +- Test: `tests/test_ocr_rendering.py` +- Test: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Introduce an explicit `backmatter_heading` role** + +This should cover end-of-article structural headings that are not frontmatter noise and not numbered body headings, including: + +- `Author contributions` +- `Funding` +- `Acknowledgments` +- `Conflict of interest` +- `Generative AI statement` +- `Publisher's note` +- `Supplementary material` + +- [ ] **Step 2: Use page-position and heading-profile evidence** + +Do not classify these only by keyword membership. +Use: + +- late-page position +- small-heading geometry +- spacing above/below +- consistency with nearby backmatter layout + +- [ ] **Step 3: Render `backmatter_heading` explicitly** + +Render as a stable heading level, not as body paragraph and not as suppressed noise. + +- [ ] **Step 4: Render `reference_heading` as a true heading** + +`References` should be emitted as a heading section, not plain text. + +- [ ] **Step 5: Verify** + +Run: `python -m pytest tests/test_ocr_roles.py tests/test_ocr_rendering.py tests/test_ocr_render_stabilization.py -q` + +Expected: PASS + +## Task 5: Finish Formal-Table Downstream Consumption + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Modify: `paperforge/worker/ocr_health.py` +- Possibly modify: `paperforge/worker/ocr_objects.py` +- Test: `tests/test_ocr_rendering.py` +- Test: `tests/test_ocr_health.py` + +- [x] **Step 1: Make render consume formal table objects, not raw table HTML** + +If a table has a formal object with image truth, `fulltext.md` should link/embed the table object instead of rendering assistive HTML. + +- [x] **Step 2: Distinguish formal table count from segment count** + +Health should be able to report: + +- formal table count +- physical table segment count + +without pretending they are the same thing. + +- [x] **Step 3: Keep continuation display coherent** + +Continuation segments should still map to the same formal table number downstream. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_rendering.py tests/test_ocr_health.py -q` + +Expected: PASS + +## Task 6: Clean Metadata Display Values + +**Files:** +- Modify: `paperforge/worker/ocr_metadata.py` +- Possibly modify: `paperforge/worker/ocr_render.py` +- Test: `tests/test_ocr_metadata.py` + +- [x] **Step 1: Add display-safe author normalization** + +Normalize obvious OCR spacing artifacts in metadata-only contexts, especially: + +- `$ ^{...} $ -> $^{...}$` +- missing spaces around `and` +- repeated raw OCR separator artifacts + +This should improve metadata display without rewriting original body text. + +- [x] **Step 2: Preserve raw author block separately** + +Keep raw OCR frontmatter preserved for auditability even if display values are cleaned. + +- [ ] **Step 3: Verify** + +Run: `python -m pytest tests/test_ocr_metadata.py -q` + +Expected: PASS + +## Task 7: Restore Page-Marker Compatibility + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Possibly modify: `paperforge/worker/ocr.py` +- Test: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Emit one compatibility marker per page** + +Even if a page only contributes figure/table objects or suppressed frontmatter, the compatibility output should still preserve marker coverage. + +- [ ] **Step 2: Clear stale mismatch error when contract is satisfied** + +Once the compatibility output is correct, `meta.error` should not keep reporting `page marker mismatch`. + +- [ ] **Step 3: Verify** + +Run: `python -m pytest tests/test_ocr_render_stabilization.py -q` + +Expected: PASS + +## Task 8: Real-Paper Closure Audit + +**Files:** +- Verify only + +- [ ] **Step 1: Rebuild `7C8829BD`** + +Run the current derived rebuild/backfill path again. + +- [x] **Step 2: Confirm metadata cleanup** + +Check: + +- no `Generative AI statement` title alternative +- authors present and display-clean enough + +- [ ] **Step 3: Confirm render cleanup** + +Check: + +- no duplicated frontmatter author block in body +- no inline `
` HTML +- explicit tail-page headings render correctly +- `References` renders as a heading +- page markers cover all pages + +- [ ] **Step 4: Confirm health/state cleanup** + +Check: + +- `references_found = true` +- `abstract_found = true` +- formal table handling reflected downstream +- `meta.error` no longer reports page marker mismatch + +## Risks + +1. Over-tightening title logic may accidentally drop valid page-1 titles in edge PDFs. +2. Frontmatter de-duplication may remove content that should remain visible if block provenance is not tracked carefully. +3. Removing inline table HTML may reduce text-only usefulness if object-note assistive content is not adequate. +4. Page-marker restoration may reintroduce noisy empty pages if done mechanically instead of compatibility-aware. +5. A new `backmatter_heading` regime may overfit this paper unless it is tied to geometry and position, not just keyword lists. diff --git a/docs/superpowers/plans/2026-06-05-ocr-render-stabilization.md b/docs/superpowers/plans/2026-06-05-ocr-render-stabilization.md new file mode 100644 index 00000000..6bfcc6d0 --- /dev/null +++ b/docs/superpowers/plans/2026-06-05-ocr-render-stabilization.md @@ -0,0 +1,801 @@ +# OCR Render And Structure Stabilization Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Stabilize OCR structured roles, metadata recovery, figure/table object contracts, and `fulltext.md` rendering so legacy backfilled papers such as `D:\L\OB\Literature-hub\System\PaperForge\ocr\7C8829BD` become structurally correct, Obsidian-compatible, and safe to use as the future basis for search/evidence integration. + +**Architecture:** This phase closes the remaining gap between the OCR structured pipeline spec and the current implementation. The fix is not “render polish”; it is a coordinated repair of Layer 3 through Layer 6: first-page/frontmatter analysis, global heading profile inference, raw-label-aware role assignment, metadata recovery with multi-source locking, more conservative figure/table detection and matching, render assembly that follows the paper-note contract, and compatibility containment of the old `images/` directory while `assets/` becomes the structured truth. This phase also hardens the producer side of `role-index.json` so later search integration is not built on polluted roles. + +**Tech Stack:** Python, pytest, current OCR artifact pipeline, Obsidian-flavored Markdown, legacy OCR corpus at `D:\L\OB\Literature-hub` + +--- + +## Scope + +This phase must fix all issues that currently block OCR from being considered stable: + +1. first-page/title/authors/abstract/doi/frontmatter role stability +2. heading hierarchy stability across the whole paper +3. reference zone stability +4. figure legend detection and figure-asset matching robustness +5. table detection/matching robustness with image-as-truth rendering +6. object note contracts and asset path compatibility +7. `fulltext.md` assembly and Obsidian math syntax +8. compatibility state drift such as `images/` versus `assets/` +9. health/index outputs that currently inherit polluted roles + +This phase does **not** yet implement command-layer OCR evidence search. It prepares the OCR artifacts so that later search integration is worth doing. + +## File Structure + +This phase should focus on these files and modules: + +- `paperforge/worker/ocr_roles.py` + - Replace permissive fallback rules with raw-label-aware role mapping plus conservative heuristics. +- `paperforge/worker/ocr_metadata.py` + - Recover metadata from frontmatter and OCR raw blocks without polluting non-metadata roles. +- `paperforge/worker/ocr_figures.py` + - Harden legend detection and matching with profile-based and multi-signal scoring. +- `paperforge/worker/ocr_tables.py` + - Harden table caption/asset matching and preserve image-first truth semantics. +- `paperforge/worker/ocr_render.py` + - Render the intended paper note structure and normalize Obsidian-safe markdown/math. +- `paperforge/worker/ocr_objects.py` + - Normalize object note contracts, figure/table headings, and asset references. +- `paperforge/worker/ocr_index.py` + - Ensure role-index buckets reflect stabilized roles and are no longer polluted by frontmatter/body confusion. +- `paperforge/worker/ocr_health.py` + - Make health reflect corrected abstract/reference/figure/table state. +- `paperforge/worker/ocr.py` + - Keep orchestration stable while preserving `images/` compatibility and moving structured truth to `assets/`. +- `tests/test_ocr_roles.py` + - Add global heading/frontmatter/reference regressions. +- `tests/test_ocr_metadata.py` + - Add frontmatter-locked metadata recovery regressions. +- `tests/test_ocr_figures.py` + - Add legend detection and matching regressions. +- `tests/test_ocr_tables.py` + - Add table caption/asset matching regressions. +- `tests/test_ocr_objects.py` + - Add object note contract and image-path compatibility regressions. +- `tests/test_ocr_rendering.py` + - Add render assembly and heading-sanity regressions. +- `tests/test_ocr_health.py` + - Add abstract/reference/figure/table health regressions. +- `tests/test_ocr_index.py` + - Add role-index pollution regressions. +- `tests/test_ocr_render_stabilization.py` + - New focused end-to-end structured render fixture tests. + +## Real Paper Acceptance Target + +The real validation paper for this phase is: + +- `D:\L\OB\Literature-hub\System\PaperForge\ocr\7C8829BD` + +The phase is not complete until this paper satisfies all of: + +1. `paper_title` is assigned only to the real paper title block +2. authors and affiliations are recovered or at least correctly isolated +3. abstract is present in `fulltext.md` +4. `References` and reference items are recognized +5. no bogus heading such as `## 2 mT, f = 15 Hz...` +6. figure/table object notes are correctly formed +7. figure/table links are not dumped blindly at the tail +8. `role-index.json` body bucket no longer begins with frontmatter furniture +9. `meta.json` compatibility state remains coherent + +## Failure Modes To Eliminate + +1. `doc_title` not recognized +2. `abstract` raw blocks ignored +3. `reference_content` ignored +4. any unnumbered `paragraph_title` promoted to `paper_title` +5. generic `text` blocks promoted to headings too easily +6. metadata resolver consuming polluted role output +7. formal figure legends missed because text rules are too narrow +8. asset/legend mispairing because matching is too greedy +9. table OCR text leaking into `fulltext.md` +10. object note image links breaking in Obsidian +11. `images/` and `assets/` semantics drifting apart invisibly + +## Design Contracts For This Phase + +### 1. Frontmatter Analyzer Contract + +Frontmatter analysis is a dedicated regime, not just generic block classification. + +Rules: + +- operate on page 1 first, optionally page 2 frontmatter spillover if needed +- detect and lock: + - title zone + - author zone + - affiliation zone + - journal furniture zone + - abstract zone + - doi/journal metadata zone +- once a block is locked as frontmatter object material, it should not re-enter body/heading competition + +Signals to combine: + +- OCR raw label (`doc_title`, `abstract`, `header`, `paragraph_title`, `text`) +- block geometry +- block order on page +- frontmatter keywords +- source metadata from `raw/source_metadata.json` + +### 2. Heading Contract + +Heading detection must be globally consistent and conservative. + +Rules: + +- trust high-confidence OCR heading priors first +- do not broadly promote `text` blocks into headings +- infer a document-specific heading profile from high-confidence headings: + - numbering patterns + - block length + - bbox height/width + - left alignment / indentation + - page placement +- only admit low-confidence heading candidates if they fit that profile + +Outcomes: + +- `section_heading` +- `subsection_heading` +- `reference_heading` +- or not a heading at all + +### 3. Metadata Locking Contract + +Metadata recovery should use three sources together: + +1. source metadata (`raw/source_metadata.json`) +2. frontmatter-analyzer output +3. OCR raw first-page blocks / structured role output + +Rules: + +- Zotero/source metadata stays primary when present +- OCR/frontmatter provides: + - block localization + - alternates + - fallback when source metadata is sparse +- once title/authors/doi/abstract blocks are locked, they stop affecting unrelated role inference + +### 4. Figure Contract + +Figure handling remains caption-first, but should not depend on text prefix alone. + +Legend detection must split into: + +- high-confidence legends: + - explicit `Figure/Fig/...` + - OCR raw `figure_title` + - strong image adjacency +- candidate legends: + - panel-style text + - geometry and typography similar to known legends + - image adjacency + +Matching must combine: + +- page relationship +- geometry overlap / distance +- above/below caption convention +- panel clustering +- numbering consistency +- one-to-one or one-to-cluster consistency + +Outputs must support: + +- matched formal figure +- low-confidence match +- legend-only figure +- orphan asset + +### 5. Table Contract + +Tables remain image-first truth objects. + +Rules: + +- table image is truth +- OCR/parsed text is assistive only +- `fulltext.md` should not expand assistive OCR table text inline +- caption/asset matching should use multi-signal scoring similar to figures, with table-specific continuation handling + +### 6. Asset Compatibility Contract + +For now: + +- `assets/` is the structured truth +- `images/` remains as compatibility output only + +Rules: + +- do not delete `images/` +- maintain an explicit mapping between old `images/` and new `assets/` +- update state and object rendering so future migration can flip consumers safely +- do not let new structured code keep treating `images/` as the primary semantic directory + +### 7. Object Note Contract + +Figure object notes must not use the whole legend as the title. + +Correct shape: + +- `# Figure 1` +- image embed or image reference +- `## Legend` +- legend text +- page / confidence / optional warning if needed + +Table object notes should follow the same pattern: + +- `# Table 1` +- image +- `## Caption` +- optional assistive OCR section + +## Task 1: Lock All Current Structural Failures In Tests + +**Files:** +- Create: `tests/test_ocr_render_stabilization.py` +- Modify: `tests/test_ocr_roles.py` +- Modify: `tests/test_ocr_metadata.py` +- Modify: `tests/test_ocr_figures.py` +- Modify: `tests/test_ocr_tables.py` +- Modify: `tests/test_ocr_objects.py` +- Modify: `tests/test_ocr_rendering.py` +- Modify: `tests/test_ocr_health.py` +- Modify: `tests/test_ocr_index.py` + +- [ ] **Step 1: Add a failing frontmatter-locking role test** + +Use a fixture modeled on `7C8829BD` page 1 and assert: + +- `doc_title -> paper_title` +- `abstract -> abstract_body` +- frontmatter furniture -> `frontmatter_noise` +- author line is not `body_paragraph` by default if clearly inside author zone + +- [ ] **Step 2: Add a failing heading-sanity regression** + +Assert that a long `text` block such as the current `2 mT, f = 15 Hz...` paragraph cannot become a section heading. + +- [ ] **Step 3: Add a failing reference-zone regression** + +Assert that `reference_content` blocks after `References` are recognized as `reference_item`. + +- [ ] **Step 4: Add a failing metadata recovery regression** + +Assert that sparse source metadata plus usable OCR frontmatter still yields: + +- title +- authors +- doi + +without polluted title alternatives. + +- [ ] **Step 5: Add failing figure legend/matching regressions** + +Include cases for: + +- explicit high-confidence legend +- candidate legend with no `Figure N` prefix but strong geometry +- legend-only degradation +- orphan asset preservation + +- [ ] **Step 6: Add failing table object/render regressions** + +Assert that `fulltext.md` links to table object notes instead of inlining assistive OCR table text. + +- [ ] **Step 7: Add failing object-note contract regressions** + +Assert that: + +- object note title is `Figure 1` / `Table 1` +- image reference resolves through the current compatibility contract +- legend/caption body is separated from the title + +- [ ] **Step 8: Add failing role-index pollution regressions** + +Assert that frontmatter furniture does not enter the `body` bucket. + +- [ ] **Step 9: Run tests to verify they fail** + +Run: `python -m pytest tests/test_ocr_render_stabilization.py tests/test_ocr_roles.py tests/test_ocr_metadata.py tests/test_ocr_figures.py tests/test_ocr_tables.py tests/test_ocr_objects.py tests/test_ocr_rendering.py tests/test_ocr_health.py tests/test_ocr_index.py -q` + +Expected: FAIL because current implementation still exhibits the known structural failures. + +- [ ] **Step 10: Commit** + +```bash +git add tests/test_ocr_render_stabilization.py tests/test_ocr_roles.py tests/test_ocr_metadata.py tests/test_ocr_figures.py tests/test_ocr_tables.py tests/test_ocr_objects.py tests/test_ocr_rendering.py tests/test_ocr_health.py tests/test_ocr_index.py +git commit -m "test: lock OCR structure and render stabilization regressions" +``` + +## Task 2: Rebuild Role Assignment Around Raw Priors And Global Consistency + +**Files:** +- Modify: `paperforge/worker/ocr_roles.py` +- Test: `tests/test_ocr_roles.py` + +- [ ] **Step 1: Add explicit raw-label mappings** + +Add dedicated handling for at least: + +- `doc_title -> paper_title` +- `abstract -> abstract_body` +- `reference_content -> reference_item` +- `figure_title -> figure_caption` +- `header/footer/number -> noise` + +- [ ] **Step 2: Replace unsafe `paragraph_title` fallback** + +Remove the current “unnumbered `paragraph_title` becomes `paper_title`” rule. + +Replace it with: + +- frontmatter-aware title admission only on page 1 +- explicit references heading detection +- conservative heading classification +- `unknown_structural` fallback when not confident + +- [ ] **Step 3: Remove permissive `text -> heading` upgrades** + +`text` blocks should only be upgraded to headings under very strong constraints: + +- short length +- heading-like geometry +- heading-profile compatibility +- not parameter/prose-like + +Default should remain `body_paragraph` or `unknown`, not heading. + +- [ ] **Step 4: Introduce heading-profile-aware helpers** + +Use already-detected high-confidence headings to infer: + +- numbering pattern +- max length +- bbox/indent profile + +and use that profile to reject bogus heading candidates. + +- [ ] **Step 5: Introduce a simple frontmatter-zone helper** + +The role layer should know when a block is inside the frontmatter regime so it can prevent leakage into generic body roles. + +- [ ] **Step 6: Run tests to verify they pass** + +Run: `python -m pytest tests/test_ocr_roles.py -q` + +Expected: PASS + +- [ ] **Step 7: Commit** + +```bash +git add paperforge/worker/ocr_roles.py tests/test_ocr_roles.py +git commit -m "fix: stabilize OCR role assignment with raw priors and heading profiles" +``` + +## Task 3: Build A Real Frontmatter-Locked Metadata Recovery Path + +**Files:** +- Modify: `paperforge/worker/ocr_metadata.py` +- Possibly modify: `paperforge/worker/ocr.py` +- Test: `tests/test_ocr_metadata.py` + +- [ ] **Step 1: Expand frontmatter candidate extraction** + +Extract and preserve: + +- title candidates +- author block candidates +- affiliation block candidates +- doi candidates +- journal/publish metadata candidates +- abstract-zone evidence if helpful + +- [ ] **Step 2: Separate OCR block localization from resolved value choice** + +Resolved metadata should not just say “what the title is”; it should know which block established it so that render and role stabilization can stay aligned. + +- [ ] **Step 3: Add OCR-first fallback when source metadata is sparse** + +When Zotero/source metadata is missing: + +- use locked OCR/frontmatter candidates +- do not return empty authors/title if strong frontmatter evidence exists + +- [ ] **Step 4: Prevent polluted alternates** + +Do not admit late-paper headings like `Generative AI statement` as title alternatives. + +- [ ] **Step 5: Run tests to verify they pass** + +Run: `python -m pytest tests/test_ocr_metadata.py -q` + +Expected: PASS + +- [ ] **Step 6: Commit** + +```bash +git add paperforge/worker/ocr_metadata.py paperforge/worker/ocr.py tests/test_ocr_metadata.py +git commit -m "fix: recover OCR metadata from frontmatter without role pollution" +``` + +## Task 4: Harden Figure Legend Detection And Matching + +**Files:** +- Modify: `paperforge/worker/ocr_figures.py` +- Possibly modify: `paperforge/worker/ocr_roles.py` +- Test: `tests/test_ocr_figures.py` + +- [ ] **Step 1: Split legend detection into formal and candidate paths** + +Formal: + +- explicit figure-prefix legend +- OCR `figure_title` + +Candidate: + +- panel-style nearby text +- typography/geometry similar to formal legends +- strong adjacency to media asset + +- [ ] **Step 2: Build a legend profile from high-confidence legends** + +Use high-confidence legends to infer: + +- typical width +- typical relative placement to figure asset +- typical text length/style + +- [ ] **Step 3: Improve asset matching from greedy local to multi-signal matching** + +In scoring, include: + +- same-page / adjacent-page +- vertical relation +- overlap +- horizontal alignment +- candidate asset size +- competing legend proximity +- numbering continuity if present + +- [ ] **Step 4: Preserve explicit degradation states** + +Support: + +- matched figure +- low-confidence matched figure +- legend-only +- orphan asset + +Populate inventory fields honestly instead of silently forcing a match. + +- [ ] **Step 5: Keep `unmatched_legends` and `unmatched_assets` real** + +Do not leave them structurally present but empty by implementation accident. + +- [ ] **Step 6: Run tests to verify they pass** + +Run: `python -m pytest tests/test_ocr_figures.py -q` + +Expected: PASS + +- [ ] **Step 7: Commit** + +```bash +git add paperforge/worker/ocr_figures.py paperforge/worker/ocr_roles.py tests/test_ocr_figures.py +git commit -m "fix: harden OCR figure legend detection and matching" +``` + +## Task 5: Harden Table Matching While Keeping Image Truth + +**Files:** +- Modify: `paperforge/worker/ocr_tables.py` +- Test: `tests/test_ocr_tables.py` +- Test: `tests/test_ocr_rendering.py` + +- [ ] **Step 1: Add multi-signal caption/asset scoring** + +Use: + +- page relationship +- geometry +- width/height expectations +- continuation handling for `Table N (Continued)` + +- [ ] **Step 2: Preserve image-first truth** + +Make sure assistive OCR text stays in object notes only and is not expanded inline into `fulltext.md`. + +- [ ] **Step 3: Distinguish table object title from caption body** + +Do not use the full caption as the object note title. + +- [ ] **Step 4: Run tests to verify they pass** + +Run: `python -m pytest tests/test_ocr_tables.py tests/test_ocr_rendering.py -q` + +Expected: PASS + +- [ ] **Step 5: Commit** + +```bash +git add paperforge/worker/ocr_tables.py tests/test_ocr_tables.py tests/test_ocr_rendering.py +git commit -m "fix: stabilize OCR table matching and image-first rendering" +``` + +## Task 6: Normalize Object Note Contracts And Asset Compatibility + +**Files:** +- Modify: `paperforge/worker/ocr_objects.py` +- Modify: `paperforge/worker/ocr.py` +- Test: `tests/test_ocr_objects.py` + +- [ ] **Step 1: Normalize figure/table object note headings** + +Use: + +- `# Figure 1` +- `# Table 1` + +not the entire legend/caption text as the note heading. + +- [ ] **Step 2: Add explicit section structure inside object notes** + +For figures: + +- title +- image +- `## Legend` +- optional warning or page/confidence note + +For tables: + +- title +- image +- `## Caption` +- optional assistive OCR section + +- [ ] **Step 3: Preserve `images/` as compatibility only** + +Implement or document a clear mapping so that: + +- new structured logic prefers `assets/` +- old consumers can still find `images/` + +At minimum: + +- do not let `meta.assets_path` misrepresent the structured truth silently +- preserve a deterministic path mapping + +- [ ] **Step 4: Run tests to verify they pass** + +Run: `python -m pytest tests/test_ocr_objects.py -q` + +Expected: PASS + +- [ ] **Step 5: Commit** + +```bash +git add paperforge/worker/ocr_objects.py paperforge/worker/ocr.py tests/test_ocr_objects.py +git commit -m "fix: normalize OCR object notes and asset compatibility mapping" +``` + +## Task 7: Reassemble `fulltext.md` Around The Intended Note Contract + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Test: `tests/test_ocr_render_stabilization.py` +- Test: `tests/test_ocr_rendering.py` + +- [ ] **Step 1: Define the render order explicitly** + +Render as: + +1. `# Title` +2. metadata block +3. `## Abstract` +4. structured body sections +5. anchored figure/table object references +6. references +7. tail matter as appropriate + +- [ ] **Step 2: Add heading sanity checks at render time** + +Renderer should not blindly emit `##` for absurdly long or obviously paragraph-like headings even if a low-confidence upstream role slipped through. + +- [ ] **Step 3: Anchor figure/table links near relevant content** + +Use page/caption/section proximity rather than blind tail dumping. + +- [ ] **Step 4: Keep page markers compatible** + +Do not regress page-marker coverage while filtering frontmatter/body noise. + +- [ ] **Step 5: Run tests to verify they pass** + +Run: `python -m pytest tests/test_ocr_render_stabilization.py tests/test_ocr_rendering.py -q` + +Expected: PASS + +- [ ] **Step 6: Commit** + +```bash +git add paperforge/worker/ocr_render.py tests/test_ocr_render_stabilization.py tests/test_ocr_rendering.py +git commit -m "fix: reassemble OCR fulltext around the structured note contract" +``` + +## Task 8: Normalize Obsidian Math And Protect Health/Index Outputs + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Modify: `paperforge/worker/ocr_health.py` +- Modify: `paperforge/worker/ocr_index.py` +- Test: `tests/test_ocr_health.py` +- Test: `tests/test_ocr_index.py` +- Test: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Expand inline math normalization conservatively** + +Fix: + +- `$ ^{...} $ -> $^{...}$` +- `$ B_{rms} $ -> $B_{rms}$` +- remove padding around inline delimiters + +without over-aggressively converting prose to math. + +- [ ] **Step 2: Make health consume stabilized roles** + +After role fixes: + +- `abstract_found` should become true when real abstract exists +- `references_found` should become true when references zone exists + +- [ ] **Step 3: Make role-index consume stabilized roles** + +Ensure: + +- frontmatter furniture does not enter `body` +- references enter `references` +- abstract enters the intended bucket + +- [ ] **Step 4: Run tests to verify they pass** + +Run: `python -m pytest tests/test_ocr_health.py tests/test_ocr_index.py tests/test_ocr_render_stabilization.py -q` + +Expected: PASS + +- [ ] **Step 5: Commit** + +```bash +git add paperforge/worker/ocr_render.py paperforge/worker/ocr_health.py paperforge/worker/ocr_index.py tests/test_ocr_health.py tests/test_ocr_index.py tests/test_ocr_render_stabilization.py +git commit -m "fix: stabilize OCR math, health, and role-index outputs" +``` + +## Task 9: Real-Corpus Validation On `7C8829BD` + +**Files:** +- Verify only + +- [ ] **Step 1: Rebuild the real paper with the updated derived pipeline** + +Use the current branch code against: + +`D:\L\OB\Literature-hub\System\PaperForge\ocr\7C8829BD` + +- [ ] **Step 2: Verify structured artifacts directly** + +Check: + +- `blocks.structured.jsonl` +- `resolved_metadata.json` +- `figure_inventory.json` +- `table_inventory.json` +- `role-index.json` +- `ocr_health.json` + +- [ ] **Step 3: Verify rendered artifacts directly** + +Check: + +- `fulltext.md` +- `render/fulltext.md` +- `render/figures/*.md` +- `render/tables/*.md` + +- [ ] **Step 4: Confirm all real-paper acceptance targets** + +Especially: + +- no bogus long heading promotion +- abstract present +- references recognized +- object note headings are clean +- figure/table placement is reasonable +- `images/` compatibility is preserved while `assets/` remains the structured truth + +- [ ] **Step 5: If needed, capture the paper as a durable regression fixture** + +This paper should remain the canonical stabilization target before any wider search rollout. + +## Task 10: Final Verification + +**Files:** +- Verify only + +- [ ] **Step 1: Run focused stabilization suite** + +Run: `python -m pytest tests/test_ocr_render_stabilization.py tests/test_ocr_roles.py tests/test_ocr_metadata.py tests/test_ocr_figures.py tests/test_ocr_tables.py tests/test_ocr_objects.py tests/test_ocr_rendering.py tests/test_ocr_health.py tests/test_ocr_index.py -q` + +Expected: PASS + +- [ ] **Step 2: Run broader OCR regressions** + +Run: `python -m pytest tests/test_ocr_versions.py tests/test_ocr_render_v2.py tests/test_ocr_state_machine.py tests/test_sync.py tests/test_context.py tests/test_selection_sync_pdf.py tests/test_status.py tests/test_ocr_doctor.py tests/e2e/test_ocr_e2e.py tests/test_ocr_redo.py -q` + +Expected: PASS + +- [ ] **Step 3: Run one real-corpus smoke rebuild** + +Expected: + +- real paper becomes structurally coherent +- no obvious frontmatter/body/heading corruption remains + +- [ ] **Step 4: Commit verification-only fixes if needed** + +```bash +git add -A +git commit -m "test: finalize OCR structure and render stabilization" +``` + +## Risks And Mitigations + +1. **Risk: fixing title detection pollutes other headings** + - Mitigation: restrict `paper_title` to frontmatter/title-zone logic only. + +2. **Risk: abstract fallback accidentally captures introduction text** + - Mitigation: only trust raw `abstract`, frontmatter zone, or strong proximity to title/authors on page 1. + +3. **Risk: heading profile becomes too strict and misses real headings** + - Mitigation: build profile from high-confidence headings and allow low-confidence candidates only when strongly consistent. + +4. **Risk: figure candidate legend logic starts swallowing ordinary body text** + - Mitigation: require agreement among geometry, profile, and media adjacency before promotion. + +5. **Risk: object-note cleanup breaks old consumers using `images/`** + - Mitigation: preserve `images/` as explicit compatibility output for this phase and add deterministic mapping. + +6. **Risk: render looks nicer but role-index is still polluted** + - Mitigation: test `ocr_index.py` directly and gate the phase on clean producer outputs. + +7. **Risk: aggressive math normalization corrupts prose** + - Mitigation: keep normalization conservative and test against real OCR snippets. + +8. **Risk: local fixes only work for `7C8829BD`** + - Mitigation: include synthetic tests plus at least one additional smoke rebuild if time permits. + +## Execution Notes + +Implementation order matters: + +1. roles +2. metadata +3. figures/tables +4. object contracts +5. render +6. health/index +7. real-corpus validation + +Do not start search/evidence command integration again until this phase is complete. diff --git a/docs/superpowers/plans/2026-06-05-ocr-tail-regime-remediation-plan.md b/docs/superpowers/plans/2026-06-05-ocr-tail-regime-remediation-plan.md new file mode 100644 index 00000000..ae4a5ab3 --- /dev/null +++ b/docs/superpowers/plans/2026-06-05-ocr-tail-regime-remediation-plan.md @@ -0,0 +1,413 @@ +# OCR Tail Regime Remediation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Eliminate the remaining OCR tail-page layout corruption by introducing a bidirectional body/backmatter boundary detector, a multi-page tail-spread regroup pass, a first-class references zone, and PDF-style-assisted unnumbered heading discrimination, while cleaning out obsolete text-first noise checks. + +**Architecture:** The current failures are no longer in the main OCR pipeline. They are concentrated in the late-body / backmatter transition. The fix should not add paper-specific text rules. Instead, it should change the decision hierarchy: + +1. estimate paper-level edge noise bands +2. parse the main body from front to back +3. parse the backmatter from back to front +4. reconcile those into a tail spread +5. carve out a references zone inside that spread +6. attach tail bodies by ownership +7. use PDF style signals only to disambiguate ambiguous unnumbered headings +8. render the resolved tail structure + +This keeps the solution structural rather than lexical. + +**Tech Stack:** Python, pytest, structured OCR artifacts, PyMuPDF span metadata, real-paper validation on `D:\L\OB\Literature-hub\System\PaperForge\ocr\7C8829BD` + +--- + +## Current Failure + +The current output for `7C8829BD` still shows a layout corruption around `Funding`: + +- ordinary late-body conclusion text is rendered under `## Funding` +- the real funding continuation from page 22 is separated from `Funding` + +This means two things are still wrong: + +1. late-body text is being allowed into the tail candidate pool too easily +2. cross-page backmatter continuation is not modeled strongly enough + +At the same time, some prior noise suppression has already been reduced enough that `Supplementary material` content can appear again. So the next fix should not be “add more text rules”. It should tighten the regime. + +## Root Cause + +### 1. Tail ownership is decided too early + +In the current flow, late `text` blocks can still become `tail_candidate_body` just because they appear on a page that also contains backmatter headings. That is too broad. + +### 2. There is no explicit body-end / backmatter-start reconciliation + +The current system still lacks a true shared boundary between: + +- forward main-body parsing +- backward backmatter parsing + +So the last pages are not being split cleanly into: + +- still-body +- tail spread + +### 3. Tail handling is still too page-local + +This paper needs a spread-level interpretation of pages 21–22: + +- `Funding` starts on page 21 +- funding continuation is on page 22 +- `References` begins on page 22 as an independent region + +A page-local grouping pass cannot resolve that reliably. + +### 4. Residual text-based noise checks are still too influential + +Some old text-based noise checks still exist in `ocr_roles.py`. Even when they no longer dominate every case, they still complicate the regime and should be demoted or removed where they overlap with tail ownership. + +### 5. Unnumbered heading hierarchy still needs style support + +For papers like: + +- `Mobini 2017` +- `Fitzsimmons 2008` + +unnumbered headings differ by: + +- size +- font family +- bold/italic flags +- color + +If the system only uses text and geometry, these headings will collapse too easily into one level. + +## Design Changes + +## A. Body And Backmatter Must Be Parsed From Opposite Directions + +### Forward body spine + +Parse from the front: + +- stable numbered headings +- stable subsection headings +- body paragraph continuity +- body-local spacing + +Output: + +- conservative `body_end_candidate` + +### Backward backmatter spine + +Parse from the end: + +- `reference_heading` +- dense `reference_item` +- `backmatter_heading` +- compact tail section bodies + +Output: + +- conservative `backmatter_start_candidate` + +### Boundary reconciliation + +If these two boundaries overlap or nearly touch, define a `tail spread` rather than forcing a hard single-page cutoff. + +Only blocks inside or after that reconciled spread should enter tail ownership logic. + +## B. Tail Candidate Generation Must Be Narrower + +`tail_candidate_body` should not mean: + +- “any long text on a page that also contains backmatter headings” + +It should mean: + +- a late-page text block in the reconciled tail spread +- that is not already part of the stable forward body spine +- and is geometrically plausible as a tail-owned body block + +This avoids swallowing ordinary late conclusion text. + +## C. References Must Be A Region, Not Just A Heading + +`reference_heading` must create a `references_zone`: + +- anchored at the heading +- extending downward +- containing `reference_item` +- spanning both columns if necessary + +Blocks in this zone must be protected from backmatter ownership. + +## D. Backmatter Ownership Must Be Spread-Level + +Backmatter bodies should be assigned using: + +- below-heading geometry +- same-column preference +- horizontal overlap +- cross-page continuation allowance +- exclusion from references zone + +This is especially needed for `Funding` continuation from page 21 to page 22. + +## E. Noise Must Become Geometry-First + +Strong noise should be limited to: + +- raw `header` +- raw `footer` +- raw `number` +- edge-band artifacts + +Text-triggered noise checks should be: + +- removed if redundant +- or demoted to weak fallback only after regime/ownership decisions + +This prevents noise heuristics from competing with tail structure. + +## F. Style Must Assist Unnumbered Heading Levels + +Use PDF span metadata where available: + +- `size` +- `font` +- `flags` +- `color` +- line bbox height + +Build paper-local style profiles for: + +- primary headings +- subsection headings +- backmatter headings +- sidebar/frontmatter furniture + +Style is a supporting signal only. Geometry and regime still come first. + +## Task 1: Lock The Mechanism Failure In Tests + +**Files:** +- Modify: `tests/test_ocr_roles.py` +- Modify: `tests/test_ocr_rendering.py` +- Modify: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Add a failing tail-candidate-overreach test** + +Fixture: + +- late body paragraphs on a page that also contains backmatter headings +- true backmatter heading/body on the same page + +Expected: + +- ordinary late body paragraphs do not become tail-owned blocks + +- [ ] **Step 2: Add a failing cross-page funding continuation test** + +Fixture: + +- `Funding` heading on page N +- funding body continues on page N+1 +- references also begin on page N+1 + +Expected: + +- continuation remains under `Funding` +- references stay in `References` + +- [ ] **Step 3: Add a failing style-aware unnumbered heading test** + +Fixture: + +- two unnumbered headings with distinct visual style +- one body block with ordinary body style + +Expected: + +- headings do not flatten into one generic level by text alone +- body block is not promoted to heading + +- [ ] **Step 4: Run tests to confirm failure** + +Run: `python -m pytest tests/test_ocr_roles.py tests/test_ocr_rendering.py tests/test_ocr_render_stabilization.py -q` + +Expected: FAIL + +## Task 2: Add Bidirectional Body/Backmatter Boundary Detection + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Possibly add helper functions there +- Test: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Implement forward body spine detection** + +Detect a conservative `body_end_candidate` using: + +- stable body headings +- subsection continuity +- paragraph flow + +- [ ] **Step 2: Implement backward backmatter spine detection** + +Detect a conservative `backmatter_start_candidate` using: + +- `reference_heading` +- dense references +- backmatter headings +- short tail bodies + +- [ ] **Step 3: Reconcile into a tail spread** + +Only blocks inside or after the reconciled spread should enter tail regroup logic. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_render_stabilization.py -q` + +Expected: PASS + +## Task 3: Move Tail Ownership Out Of The Base Role Layer + +**Files:** +- Modify: `paperforge/worker/ocr_roles.py` +- Test: `tests/test_ocr_roles.py` + +- [ ] **Step 1: Narrow `tail_candidate_body` generation** + +Require: + +- tail-spread context +- not already stable forward-body content +- plausible geometry for owned tail text + +- [ ] **Step 2: Remove or demote obsolete text-first noise checks** + +Keep strong noise only for: + +- header/footer/page number +- obvious edge-band artifacts + +Demote tail-overlapping text phrases to weak fallbacks. + +- [ ] **Step 3: Preserve first-page frontmatter behavior** + +Do not regress existing frontmatter filtering. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_roles.py -q` + +Expected: PASS + +## Task 4: Add Spread-Level Tail Ownership And References Zone + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Test: `tests/test_ocr_rendering.py` +- Test: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Build a first-class `references_zone`** + +Rules: + +- anchor at `reference_heading` +- include `reference_item` +- may span both columns +- protect from backmatter ownership + +- [ ] **Step 2: Attach backmatter bodies by spread-level ownership** + +Use: + +- nearest valid heading above +- same-column preference +- cross-page continuation allowance +- references-zone exclusion + +- [ ] **Step 3: Emit groups instead of raw block order** + +Render: + +- heading +- owned bodies +- references zone + +not raw block sequence. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_rendering.py tests/test_ocr_render_stabilization.py -q` + +Expected: PASS + +## Task 5: Add Style-Aware Heading Profiles + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Possibly add helper extraction logic +- Test: `tests/test_ocr_roles.py` +- Test: `tests/test_ocr_rendering.py` + +- [ ] **Step 1: Extract PDF style metadata where available** + +Support: + +- size +- font family/name +- bold/italic flags +- color + +- [ ] **Step 2: Build local heading style profiles** + +For: + +- body headings +- subsection headings +- backmatter headings + +- [ ] **Step 3: Use style only for disambiguation** + +Style should help separate ambiguous unnumbered headings, not replace geometry/regime. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_roles.py tests/test_ocr_rendering.py -q` + +Expected: PASS + +## Task 6: Real-Paper Verification + +**Files:** +- Verify only + +- [ ] **Step 1: Rebuild `7C8829BD`** + +Run the derived rebuild/backfill path again. + +- [ ] **Step 2: Verify the page 21–22 spread** + +Expected: + +- ordinary late body text stays in the conclusion +- `Author contributions` only owns contribution text +- `Funding` owns its own body and continuation +- `References` owns the reference list + +- [ ] **Step 3: Sanity-check a style-heavy PDF** + +Use `Mobini 2017` to verify that visually distinct unnumbered headings remain separable. + +## Risks + +1. A body/backmatter boundary that is too eager can still eat conclusion text. +2. A spread that is too wide can overfit later pages. +3. Style metadata quality varies; style must remain secondary. +4. Removing too many text-noise checks globally can regress first-page suppression if not limited to tail logic. diff --git a/docs/superpowers/plans/2026-06-05-ocr-tail-zone-closure-plan.md b/docs/superpowers/plans/2026-06-05-ocr-tail-zone-closure-plan.md new file mode 100644 index 00000000..b9a89dc4 --- /dev/null +++ b/docs/superpowers/plans/2026-06-05-ocr-tail-zone-closure-plan.md @@ -0,0 +1,339 @@ +# OCR Tail Zone Closure Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Finish the OCR stabilization work by closing the last tail-page issues: page-level noise band estimation, robust backmatter body attachment, and a true references-zone layout strategy for mixed tail pages such as `7C8829BD`. + +**Architecture:** This is a narrow closure plan for the article tail only. Figure/table matching, cropping, frontmatter extraction, and formal object numbering are already stable enough and should not be reopened here. The remaining bug is that the renderer still treats the last pages too much like ordinary linear block streams. The fix is to add a page-regime-aware tail pass that first estimates `header/footer noise bands` from the whole paper, then identifies `usable content area`, then detects `references_zone` anchored by `reference_heading`, and only then attaches `backmatter_body` blocks to the correct `backmatter_heading`. In short: infer page structure first, then order/render the tail. + +**Tech Stack:** Python, pytest, structured OCR artifacts, real-paper validation on `D:\L\OB\Literature-hub\System\PaperForge\ocr\7C8829BD` + +--- + +## Scope + +This plan only covers the remaining unresolved tail-page issues: + +1. page-level noise band estimation +2. tail-page page regime classification +3. references-zone detection +4. backmatter heading/body ownership +5. tail render ordering + +This plan must not reopen: + +- figure/table matching +- crop coordinate handling +- frontmatter metadata block +- object note numbering +- formal table continuation model + +## Current Residual Problems + +For `7C8829BD`, the remaining problems are: + +1. tail-page section bodies can still attach incorrectly if block order is unusual +2. `Supplementary material` body can disappear if a generic noise rule wins over tail-page ownership +3. mixed tail pages do not always behave like ordinary left-column-then-right-column pages +4. `References` on a mixed tail page should create its own independent zone instead of competing with nearby right-column backmatter sections + +The key observation is that page 22 is not a normal two-column body page: + +- left column lower half is `References` +- right column upper and middle contain `Generative AI statement`, `Publisher's note`, and `Supplementary material` +- page-local order must be inferred by region, not by one global left-to-right sort + +## Desired Tail Strategy + +The tail pass should behave like a human reading the page: + +1. identify page-edge noise first +2. identify the usable content area +3. detect if this page is a mixed tail page +4. if `References` appears, build a `references_zone` below the `reference_heading` +5. keep reference items inside that zone, even if backmatter headings exist in another column +6. attach each remaining body block to the correct backmatter heading using geometry and ownership, not raw block sequence + +## Page Geometry Strategy + +### 1. Paper-Level Noise Bands + +Use high-confidence raw/header/footer/page-number blocks across the whole paper to estimate: + +- `header_band` +- `footer_band` + +These are not deletion rules by themselves. They are geometry priors. + +Use them to define: + +- `usable_y_min` +- `usable_y_max` + +Any tail-section body candidate should normally live within the usable content band, not in the header/footer bands. + +### 2. Tail Page Regimes + +A page near the document tail can be: + +- `tail_sections_only` +- `reference_dominant` +- `tail_mixed_sections` + +`tail_mixed_sections` applies when: + +- there are one or more backmatter headings +- and there is also a `reference_heading` or dense `reference_item` block region + +### 3. References Zone + +`References zone` should be defined from the `reference_heading`, not from generic page order. + +Rules: + +- anchor at `reference_heading` +- include only blocks whose `y` is below the heading bottom +- include `reference_item` blocks across both columns +- exclude unrelated right-column tail bodies above or outside the zone + +This means `References` becomes a structural region on the page, not just another heading in the generic sort order. + +### 4. Backmatter Section Ownership + +For backmatter headings such as: + +- `Author contributions` +- `Funding` +- `Acknowledgments` +- `Conflict of interest` +- `Generative AI statement` +- `Publisher's note` +- `Supplementary material` + +attach body blocks using: + +- same-column or strong horizontal overlap +- body `y` below heading `y` +- body within usable content band +- no intervening stronger owning heading +- not inside `references_zone` + +This is more robust than “nearest following block”. + +## Render Contract + +For a mixed tail page, render in section ownership order: + +- `## ` +- its body +- next `## ` +- its body +- ... +- `## References` +- all reference items in the references zone + +The exact global order across columns should follow the page’s actual structural ownership, not a rigid left-column-first template. + +## Task 1: Lock The Tail-Zone Failures In Tests + +**Files:** +- Modify: `tests/test_ocr_rendering.py` +- Modify: `tests/test_ocr_render_stabilization.py` +- Possibly modify: `tests/test_ocr_roles.py` + +- [ ] **Step 1: Add a failing mixed-tail-page fixture** + +Create a fixture with: + +- one left-column `reference_heading` +- multiple left/right `reference_item` below it +- one right-column `Generative AI statement` heading + body above the reference zone +- one right-column `Publisher's note` heading + body +- one right-column `Supplementary material` heading + body + +Assert that references do not steal the right-column section bodies and vice versa. + +- [ ] **Step 2: Add a failing supplementary-material body ownership test** + +Assert that a `Supplementary material` body block in the usable middle content band is rendered beneath that heading and is not suppressed as noise. + +- [ ] **Step 3: Add a failing noise-band guard test** + +Assert that blocks in the inferred footer/header band are treated as non-body candidates, but a middle-page section body is not suppressed just because it contains a weak noise phrase. + +- [ ] **Step 4: Run tests to confirm failure** + +Run: `python -m pytest tests/test_ocr_rendering.py tests/test_ocr_render_stabilization.py tests/test_ocr_roles.py -q` + +Expected: FAIL + +## Task 2: Add Paper-Level Noise Band Estimation + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Possibly add helper logic inside `paperforge/worker/ocr_render.py` +- Test: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Infer header/footer bands from the whole paper** + +Use high-confidence blocks with roles such as: + +- `noise` +- raw `header` +- raw `footer` +- raw `number` + +Estimate a conservative: + +- `header_band_max_y` +- `footer_band_min_y` + +- [ ] **Step 2: Expose usable content band helpers** + +Define helpers for: + +- `is_in_header_band(block)` +- `is_in_footer_band(block)` +- `is_in_usable_content_band(block)` + +- [ ] **Step 3: Use these only as geometric priors** + +Do not delete blocks just because they are near a band. Use the bands to guide tail ownership and zone detection. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_render_stabilization.py -q` + +Expected: PASS + +## Task 3: Add `references_zone` Detection + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Test: `tests/test_ocr_rendering.py` + +- [ ] **Step 1: Detect mixed tail pages** + +A page is `tail_mixed_sections` if it contains: + +- one or more `backmatter_heading` +- and a `reference_heading` or a dense reference area + +- [ ] **Step 2: Build a references zone anchored at `reference_heading`** + +Zone rules: + +- `block.y1 >= reference_heading.y2` +- block role is `reference_item` +- allow both columns +- stop using these blocks for ordinary backmatter attachment + +- [ ] **Step 3: Keep `reference_heading` structurally distinct** + +`reference_heading` should start the references section but should not absorb unrelated right-column backmatter text above or outside the zone. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_rendering.py -q` + +Expected: PASS + +## Task 4: Add Robust Backmatter Body Ownership + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Possibly modify: `paperforge/worker/ocr_roles.py` +- Test: `tests/test_ocr_rendering.py` +- Test: `tests/test_ocr_roles.py` + +- [ ] **Step 1: Attach tail bodies by geometry, not sequence** + +For each `backmatter_heading`, candidate body blocks must satisfy: + +- below the heading +- same column or strong horizontal overlap +- inside usable content band +- not inside references zone +- not intercepted by another nearer owning heading + +- [ ] **Step 2: Demote generic noise phrase overrides inside tail ownership** + +If a block clearly belongs to a backmatter section by geometry, do not let a weak noise phrase rule suppress it. + +- [ ] **Step 3: Keep heading-body ownership local to the page** + +Do not let one section steal bodies from a different column or from the references zone. + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_rendering.py tests/test_ocr_roles.py -q` + +Expected: PASS + +## Task 5: Tail Render Reassembly + +**Files:** +- Modify: `paperforge/worker/ocr_render.py` +- Test: `tests/test_ocr_render_stabilization.py` + +- [ ] **Step 1: Emit backmatter sections by ownership groups** + +Each heading should be followed by its attached bodies. + +- [ ] **Step 2: Emit references section by zone** + +After section grouping, emit: + +- `## References` +- then all `reference_item` in zone order + +- [ ] **Step 3: Preserve earlier stable behavior** + +Do not disturb: + +- frontmatter +- main body pages +- figure/table placement +- page-marker compatibility + +- [ ] **Step 4: Verify** + +Run: `python -m pytest tests/test_ocr_render_stabilization.py -q` + +Expected: PASS + +## Task 6: Real-Paper Closure Audit On `7C8829BD` + +**Files:** +- Verify only + +- [ ] **Step 1: Rebuild `7C8829BD`** + +Run the current derived rebuild/backfill path again. + +- [ ] **Step 2: Check page 22 specifically** + +Expected: + +- `## Acknowledgments` -> acknowledgment paragraph +- `## Conflict of interest` -> conflict paragraph +- `## Generative AI statement` -> AI statement paragraph +- `## Publisher's note` -> publisher note paragraph +- `## Supplementary material` -> supplementary material paragraph/link +- `## References` -> reference list + +- [ ] **Step 3: Confirm no heading/body cross-wire** + +Specifically: + +- `This manuscript reflects only the authors' views...` must stay under `Funding` +- `The author(s) declare that no Generative AI...` must stay under `Generative AI statement` +- `All claims expressed...` must stay under `Publisher's note` +- supplementary material line must not disappear +- references must not start under `Supplementary material` + +## Risks + +1. A references zone that is too broad may swallow nearby right-column tail content. +2. A backmatter ownership rule that is too local may fail when the body text is offset but still clearly owned by a heading. +3. Noise-band estimation can become too aggressive if based on too few pages; keep it conservative and use it only as a prior. diff --git a/docs/superpowers/plans/2026-06-05-peerj-style-remediation-plan.md b/docs/superpowers/plans/2026-06-05-peerj-style-remediation-plan.md new file mode 100644 index 00000000..f4abfaba --- /dev/null +++ b/docs/superpowers/plans/2026-06-05-peerj-style-remediation-plan.md @@ -0,0 +1,364 @@ +# OCR PeerJ-Style Remediation Plan + +Date: 2026-06-05 +Scope paper: `2GN9LMCW` +Reference PDF: `Mobini et al. 2017 - In vitro effect of direct current electrical stimulation on rat mesenchymal stem cells` + +## Goal + +Stabilize OCR parsing for journal layouts that rely on visual hierarchy rather than numbered headings, and that introduce a backmatter container section such as `ADDITIONAL INFORMATION AND DECLARATIONS`. + +This plan is intentionally narrower than the earlier tail-regime work: + +- It does **not** redesign the full OCR architecture again. +- It does **not** special-case one paper by string matching bodies. +- It **does** extend the current structured pipeline so that: + - first-page frontmatter is split into author vs affiliation vs furniture, + - unnumbered headings are assigned hierarchical levels using PDF span style, + - backmatter can begin with a container heading, + - figure inner labels are not promoted into formal legends. + +## Current Failures In `2GN9LMCW` + +### 1. Author vs affiliation confusion on page 1 + +Observed: + +- `resolved_metadata.authors.value` contains the affiliation line, not the author list. +- `authors_display` also renders the institution instead of the authors. + +Root cause: + +- Page 1 block role assignment currently labels both the true author line and at least one affiliation line as `authors`. +- Resolver then picks the wrong `authors` block. + +This is a **frontmatter zoning failure**, not only a metadata resolver failure. + +### 2. Frontmatter furniture leaks into main body + +Observed in `fulltext.md`: + +- `These authors contributed equally to this work.` +- `Submitted / Accepted / Published` +- `Distributed under Creative Commons` +- `Academic editor` +- `DOI ...` +- `Additional Information and Declarations can be found on page 10` + +These are currently rendered as normal body text. + +Root cause: + +- First-page handling still treats too many furniture blocks as generic `body_paragraph`. +- Suppression is not using the strong layout cues available in this journal. + +### 3. Unnumbered headings collapse into a single level + +Observed: + +- `MATERIALS AND METHODS` +- `Groups` +- `Cell preparation and culture` +- `Electrical stimulation of cells` +- `Cell viability and activity` +- `Osteogenic differentiation` +- `Data analysis` + +are mostly rendered at the same heading level. + +Root cause: + +- Current heading logic can detect "heading-ness", but does not robustly assign heading **level** for unnumbered layouts. +- This journal depends heavily on visual style, not numbering. + +### 4. Backmatter boundary is not modeled + +Observed: + +- `ADDITIONAL INFORMATION AND DECLARATIONS` is rendered as a normal section heading. +- Page 10/11 content under that container is not grouped as one backmatter region. +- `Funding`, `Grant Disclosures`, `Competing Interests`, `Author Contributions`, `Data Availability`, `Supplemental Information` are inconsistently treated as main-section or backmatter headings. + +Root cause: + +- Current tail logic assumes backmatter starts directly from ordinary backmatter headings or references. +- This paper has a **container heading** that marks the backmatter block before individual sub-sections appear. + +### 5. Author-contribution bullets and declarations are not attached cleanly + +Observed: + +- Some author contribution lines are downgraded incorrectly. +- The page 11 declarations area is only partially grouped. + +Root cause: + +- Once the backmatter container is missed, section ownership on page 11 becomes unstable. + +### 6. Figure 4 legend is wrong + +Observed: + +- Figure 4 is currently built from `Days post culture in osteogenic differentiation supplemented medium`, which is not a formal caption. + +Root cause: + +- `raw_label == figure_title` is still trusted too much. +- On multi-panel chart pages, OCR may tag axis labels or internal chart text as `figure_title`. + +This is a **formal legend validation** failure. + +## Design Decisions + +### A. `ADDITIONAL INFORMATION AND DECLARATIONS` becomes a backmatter boundary heading + +This is not a new universal requirement for every journal. + +It is a bounded rule: + +- treat such headings as a `backmatter_boundary_heading` when they appear in the late-paper region, +- and only when followed by a cluster of declaration-like sub-sections and/or references, +- not as a global text special case for all contexts. + +This gives us a container for journals like PeerJ without forcing the same pattern onto unrelated papers. + +### B. PDF style is a primary signal for unnumbered heading hierarchy + +For unnumbered layouts: + +- `size` +- `font` +- `flags` +- `color` +- bbox height / width +- spacing before/after + +become primary heading-level signals. + +For numbered layouts: + +- numbering remains the first structural prior, +- style acts as validation / disambiguation, +- not as a replacement. + +## Implementation Plan + +### Task 1. Strengthen first-page frontmatter zoning + +Files: + +- `paperforge/worker/ocr_roles.py` +- `paperforge/worker/ocr_metadata.py` + +Implementation: + +1. Introduce explicit page-1 frontmatter zone analysis: + - `title_zone` + - `author_zone` + - `affiliation_zone` + - `journal_furniture_zone` + - `abstract_zone` + +2. Use text + geometry + style together: + - authors: + - multiple human names, + - `and`, + - author markers like `*`, `1`, `2`, + - near title, above affiliations + - affiliations: + - institution keywords, + - city/country patterns, + - numbered superscripts, + - directly below author block + - furniture: + - left/side column small-font metadata, + - editorial / DOI / submission / copyright content + +3. Stop allowing affiliation lines to keep the `authors` role. + +4. Resolver change: + - prefer `authors` from the true author zone, + - treat affiliation blocks as raw frontmatter, not author alternatives. + +Acceptance: + +- `resolved_metadata.authors` for `2GN9LMCW` contains the author names, not the institution line. +- page-1 furniture lines no longer appear in body render. + +### Task 2. Build style-aware heading hierarchy for unnumbered papers + +Files: + +- `paperforge/worker/ocr_render.py` +- `paperforge/worker/ocr_roles.py` + +Implementation: + +1. Extend style profile extraction to use: + - font size, + - font family, + - flags/boldness, + - color, + - bbox height, + - spacing around the block. + +2. Build document-level clusters for heading families: + - title style, + - section-heading style, + - subsection-heading style, + - backmatter-heading style, + - body style. + +3. For unnumbered heading candidates: + - classify by nearest style cluster, + - then validate with layout context. + +4. For numbered heading candidates: + - numbering stays primary, + - style must still be compatible before final role assignment. + +Acceptance: + +- `MATERIALS AND METHODS` remains top-level. +- `Groups`, `Cell preparation and culture`, `Electrical stimulation of cells`, `Cell viability and activity`, `Osteogenic differentiation`, `Data analysis` are assigned lower-level headings consistently. +- Similar logic remains valid on numbered-layout papers. + +### Task 3. Add backmatter boundary container handling + +Files: + +- `paperforge/worker/ocr_roles.py` +- `paperforge/worker/ocr_render.py` + +Implementation: + +1. Add a new role: + - `backmatter_boundary_heading` + +2. Detect it only when all are true: + - it appears in late pages, + - visually matches heading style, + - followed by multiple declaration-like section headings and/or reference heading, + - not part of the main body section flow. + +3. Tail logic changes: + - once a `backmatter_boundary_heading` is detected, + - all subsequent eligible blocks until `reference_heading` belong to a backmatter container regime, + - ordinary body section rules are no longer used there. + +4. Render policy: + - either render the boundary heading explicitly, + - or treat it as a non-emitted structural container if that reads better, + - but its ownership effect must still apply. + +Acceptance: + +- Page 10/11 of `2GN9LMCW` is treated as a backmatter region beginning at `ADDITIONAL INFORMATION AND DECLARATIONS`. + +### Task 4. Group declaration sub-sections within the backmatter container + +Files: + +- `paperforge/worker/ocr_render.py` +- potentially `paperforge/worker/ocr_roles.py` + +Implementation: + +1. Inside the backmatter container: + - identify sibling section headings: + - `Funding` + - `Grant Disclosures` + - `Competing Interests` + - `Author Contributions` + - `Data Availability` + - `Supplemental Information` + - these are not main body headings anymore + +2. Attach bodies by: + - same-page geometry, + - same column preference, + - nearest valid heading above, + - until interrupted by a stronger sibling heading + +3. Preserve `references_zone` as a terminal zone: + - once `reference_heading` starts, + - later reference items do not re-enter declaration sections. + +Acceptance: + +- `Funding` contains its declaration body. +- `Grant Disclosures` contains grant list. +- `Competing Interests` contains the competing-interest sentence. +- `Author Contributions` contains all contribution lines. +- `Data Availability` and `Supplemental Information` each retain their own body text. + +### Task 5. Tighten figure-title promotion into formal legend + +Files: + +- `paperforge/worker/ocr_roles.py` +- `paperforge/worker/ocr_figures.py` + +Implementation: + +1. `figure_title` becomes only a strong prior, not automatic formal legend. + +2. Add formal-legend validation: + - contains `Figure N` / `Fig. N`, or + - sits in a caption-like region spanning the figure cluster, + - not obviously panel-internal text, + - not obviously axis/title-only text, + - width and placement compatible with a caption line, not just an internal chart label + +3. For multi-panel chart pages: + - prefer captions that describe the whole panel group + - reject isolated inner labels into `figure_inner_text` / rejected candidate + +Acceptance: + +- Figure 4 is no longer built from the axis-title text. +- If no formal caption can be validated, it should degrade safely rather than fabricate a bad figure note. + +## Risks + +### Risk 1. Overfitting PeerJ left-column furniture + +Mitigation: + +- base rules on zone + style + role interaction, +- not raw text alone, +- and keep the container-heading behavior gated to late-paper declaration clusters. + +### Risk 2. Style clustering destabilizes numbered papers + +Mitigation: + +- numbering remains primary when present, +- style only validates or refines level. + +### Risk 3. Backmatter boundary absorbs real discussion text + +Mitigation: + +- only activate after the main body end, +- require a declaration-cluster pattern, +- and stop at `reference_heading`. + +### Risk 4. Figure 4 fix reduces recall on normal legends + +Mitigation: + +- preserve current high-confidence numbered legend path, +- only tighten the ambiguous `figure_title` fallback path. + +## Verification Checklist + +- [ ] `2GN9LMCW` authors render as actual authors, not affiliation lines +- [ ] page-1 submission/editor/DOI furniture is suppressed from main body +- [ ] `MATERIALS AND METHODS` and its child headings show a consistent hierarchy +- [ ] `ADDITIONAL INFORMATION AND DECLARATIONS` triggers backmatter container handling +- [ ] `Funding`, `Grant Disclosures`, `Competing Interests`, `Author Contributions`, `Data Availability`, `Supplemental Information` each keep their own body text +- [ ] `REFERENCES` starts a separate references zone +- [ ] Figure 4 no longer uses the axis-title text as its formal legend +- [ ] existing numbered-heading papers still pass regression tests +- [ ] existing tail-spread regression paper `7C8829BD` still renders correctly diff --git a/docs/superpowers/specs/2026-06-05-ocr-formal-object-detection-and-cropping-design.md b/docs/superpowers/specs/2026-06-05-ocr-formal-object-detection-and-cropping-design.md new file mode 100644 index 00000000..ac314c8a --- /dev/null +++ b/docs/superpowers/specs/2026-06-05-ocr-formal-object-detection-and-cropping-design.md @@ -0,0 +1,352 @@ +# OCR Formal Object Detection And Cropping Design + +> **Status:** Proposed +> **Date:** 2026-06-05 +> **Audience:** Maintainers, contributors, agentic implementers + +## 1. Goal + +Stabilize PaperForge OCR by separating four concerns that are currently entangled: + +1. frontmatter understanding +2. formal object detection +3. object matching and continuation handling +4. asset cropping and render consumption + +The immediate trigger is that the current structured pipeline still misclassifies: + +- paper title vs ordinary headings +- abstract vs body +- body mentions of `Figure N` vs formal figure legends +- table continuation pages vs distinct formal tables + +and recently regressed figure/table asset cropping by mixing OCR image coordinates with PDF coordinates. + +This design defines the contracts that should govern these layers so that: + +- formal figures and tables are detected conservatively +- body mentions do not become legends +- continuation tables remain one formal table +- legend-only figures remain legend-only +- asset cropping is stable and independent from object identity logic + +## 2. Problem Statement + +The current OCR pipeline has three architecture problems: + +### 2.1 Role leakage + +Generic role heuristics are still too permissive: + +- unnumbered `paragraph_title` can become `paper_title` +- generic `text` can become a heading +- any `Figure/Fig`-prefixed text can become `figure_caption` + +These mistakes then pollute metadata resolution, rendering, health, and indexing. + +### 2.2 Formal object confusion + +The current figure/table pipeline does not strongly separate: + +- formal legends +- body mentions of figures/tables +- candidate legends +- continuation segments +- orphan assets + +As a result: + +- `Figure 3 shows ...` in body prose can become the formal legend for Figure 3 +- a legend-only figure can later be assigned an unrelated orphan asset +- `Table 6 (Continued)` can become a separate formal table object + +### 2.3 Cropping-layer regression + +The current `assets/` object extraction path introduced a coordinate-system regression: + +- OCR block bboxes are in OCR rendered page-image coordinates +- new object extraction treated them as PDF coordinates + +The old `images/blocks/` path was stable because it: + +1. rendered a page image into OCR coordinates +2. cropped from that page image with OCR bbox coordinates + +The new object layer must preserve that stability. + +## 3. Design Principles + +1. Frontmatter is a dedicated regime, not generic body parsing. +2. Formal object identity must be decided before rendering. +3. Matching decides identity and bbox ownership; cropping only materializes assets. +4. `legend_only` means no asset is assigned. +5. `orphan_asset` means no formal legend claims that asset. +6. Table continuation is one formal table with multiple physical segments. +7. `assets/` is structured truth; `images/` remains compatibility only. +8. Search/index should consume only stabilized formal roles, not raw ambiguous roles. + +## 4. Layer Contracts + +## 4.1 Frontmatter Analyzer Layer + +Responsibility: + +- detect title, authors, affiliations, doi, abstract, and journal furniture on the first page +- lock those blocks so they do not later compete for body or heading roles + +Inputs: + +- OCR raw labels such as `doc_title`, `abstract`, `paragraph_title`, `text`, `header` +- block order and geometry +- source metadata from `raw/source_metadata.json` + +Outputs: + +- frontmatter object assignments +- confidence and evidence traces +- structured roles for: + - `paper_title` + - `authors` + - `affiliation` + - `doi` + - `journal_meta` + - `abstract_heading` + - `abstract_body` + - `frontmatter_noise` + +Rules: + +- `paper_title` should normally be unique and page-1-only +- OCR `doc_title` should have very high priority +- source metadata should validate and localize, not blindly overwrite block identity +- locked frontmatter blocks should not re-enter generic role inference + +## 4.2 Heading Analysis Layer + +Responsibility: + +- determine the paper’s heading hierarchy globally + +Inputs: + +- high-confidence OCR heading priors +- numbered heading patterns +- block geometry and alignment +- page-local layout + +Outputs: + +- `section_heading` +- `subsection_heading` +- `reference_heading` +- or non-heading fallback + +Rules: + +- generic `text -> heading` promotion must be extremely strict +- long prose-like blocks must never become headings +- heading decisions should use a document-level profile: + - typical numbering + - typical width/height + - left alignment / indentation + - typical position relative to columns + +## 4.3 Formal Figure Detection Layer + +Responsibility: + +- detect formal figure legends separately from body references + +Required categories: + +- `formal_figure_legend` +- `candidate_figure_legend` +- `body_figure_mention` +- `figure_asset` +- `orphan_media` + +Rules: + +### Formal legend + +High-confidence formal figure legends include: + +- OCR raw `figure_title` +- caption blocks that match formal legend patterns such as `Figure 1`, `Fig. 1`, `FIGURE 4` + +But explicit exclusions are required: + +- body sentences such as `Figure 3 shows ...`, `Figure 2 illustrates ...` +- inline references such as `as shown in Figure 2` + +### Candidate legend + +Candidate legends may be admitted when all of these align: + +- typography and geometry match known figure legend profile +- strong adjacency to a figure asset or asset cluster +- not shaped like running body prose +- no stronger competing formal legend nearby + +Candidate legends must not silently promote to formal legends without sufficient evidence. + +## 4.4 Figure Matching Layer + +Responsibility: + +- map each formal figure legend to zero or more asset blocks + +Inputs: + +- formal legends +- candidate legends +- clustered figure assets + +Outputs: + +- `matched_figure` +- `low_confidence_figure` +- `legend_only_figure` +- `orphan_asset` + +Rules: + +- matching should use page relationship, geometry, clustering, and numbering consistency +- body mentions must never claim assets +- a `legend_only_figure` must remain assetless +- unmatched assets remain orphan assets + +Disallowed behavior: + +- object-writing layer assigning a random orphan asset to a `legend_only_figure` + +## 4.5 Table Detection And Continuation Layer + +Responsibility: + +- detect formal tables +- merge continuation pages +- preserve image-first truth semantics + +Required concepts: + +- `formal_table_caption` +- `candidate_table_caption` +- `table_asset_segment` +- `table_continuation` +- `orphan_table_asset` + +Rules: + +- `Table 6` and `Table 6 (Continued)` are one formal table object +- one formal table may have multiple asset segments +- `fulltext.md` should reference the table object, not inline OCR table HTML +- assistive OCR table text belongs in object notes only + +## 4.6 Asset Cropping Layer + +Responsibility: + +- materialize selected bbox sets into image assets + +Inputs: + +- page number +- bbox or bbox cluster chosen by the matching layer +- OCR page coordinate system + +Rules: + +- cropping must use OCR page-image coordinates, not raw PDF coordinates +- preferred order: + 1. existing cached OCR page image + 2. render PDF page into OCR page dimensions, then crop + 3. direct PDF clipping only as a last-resort fallback + +Important boundary: + +- cropping does not decide which bbox is correct +- cropping does not assign or change formal object identity + +## 4.7 Object Note Layer + +Responsibility: + +- render figure and table object notes from already-decided formal objects + +Figure note contract: + +- `# Figure ` +- image +- `## Legend` +- legend text +- optional page / confidence / warning metadata + +Table note contract: + +- `# Table ` +- one or more table images +- `## Caption` +- optional continuation info +- optional assistive OCR section + +Rules: + +- titles should use formal numbers, not sequential inventory indexes +- continuation segments should not create new displayed formal numbers + +## 4.8 Render Layer + +Responsibility: + +- assemble `fulltext.md` from stabilized structured roles and formal objects + +Rules: + +- `fulltext.md` should contain: + - title + - metadata + - abstract + - body + - anchored figure/table references + - references +- tables should be represented by object references or image embeds, not inline HTML +- render should not reinterpret ambiguous raw OCR blocks + +## 5. Compatibility Contract + +`images/` remains temporarily as compatibility output. + +Rules: + +- `assets/` is structured truth +- `images/` is compatibility only +- path mapping between `assets/` and `images/` should be explicit in `meta.json` +- future deletion of `images/` requires a separate compatibility migration + +## 6. Validation Requirements + +The real validation paper is: + +- `D:\L\OB\Literature-hub\System\PaperForge\ocr\7C8829BD` + +Required real-paper outcomes: + +1. abstract is present and isolated +2. title is unique and not polluted by backmatter headings +3. `Figure 3` body prose mention is not used as the formal legend +4. `Figure 4` does not receive an orphan asset fallback +5. `Table 6 (Continued)` remains part of formal Table 6 +6. `Table 7` does not drift into `table_009` display numbering +7. `fulltext.md` does not inline raw table HTML +8. asset crops match the old stable page-image crop behavior + +## 7. Out Of Scope + +This design does not yet specify: + +- command-layer search integration +- evidence retrieval API +- plugin UI updates for formal-object inspection + +Those should consume the stabilized outputs from this design rather than define them. diff --git a/paperforge/worker/ocr.py b/paperforge/worker/ocr.py index 86f3688c..7cdd22e5 100644 --- a/paperforge/worker/ocr.py +++ b/paperforge/worker/ocr.py @@ -1782,12 +1782,21 @@ def postprocess_ocr_result(vault: Path, key: str, all_results: list[dict]) -> tu ocr_asset_root = ocr_root / "assets" ocr_render_root = ocr_root / "render" + page_dimensions_by_page: dict[int, tuple[int, int]] = {} + for block in structured: + page = int(block.get("page", 0) or 0) + width = int(block.get("page_width", 0) or 0) + height = int(block.get("page_height", 0) or 0) + if page and width and height and page not in page_dimensions_by_page: + page_dimensions_by_page[page] = (width, height) + extract_and_write_objects( pdf_path=source_pdf_path, figure_inventory=figure_inventory, table_inventory=table_inventory, asset_root=ocr_asset_root, render_root=ocr_render_root, + page_dimensions_by_page=page_dimensions_by_page, ) # --- Phase 3: structured renderer --- @@ -1798,6 +1807,7 @@ def postprocess_ocr_result(vault: Path, key: str, all_results: list[dict]) -> tu resolved_metadata=resolved, figure_inventory=figure_inventory, table_inventory=table_inventory, + page_count=page_num, ) write_render_outputs( render_root=ocr_root / "render", diff --git a/paperforge/worker/ocr_blocks.py b/paperforge/worker/ocr_blocks.py index 4821d999..88e34548 100644 --- a/paperforge/worker/ocr_blocks.py +++ b/paperforge/worker/ocr_blocks.py @@ -8,40 +8,56 @@ from paperforge.worker.ocr_roles import assign_block_role def build_structured_blocks(raw_blocks: list[dict]) -> list[dict]: - rows = [] + # Group raw blocks by page so assign_block_role can see page-local context + by_page: dict[int, list[dict]] = {} for block in raw_blocks: - mapped = { - "block_label": block.get("raw_label", "unknown"), - "block_content": block.get("text", ""), - "block_bbox": block.get("bbox", [0, 0, 0, 0]), - } - role = assign_block_role( - mapped, - page_blocks=[], - page_width=block.get("page_width", 0), - page_height=block.get("page_height", 0), - ) - render_default = role.role not in {"noise", "unknown_structural"} - index_default = True - if role.role in {"noise", "page_header", "page_footer", "frontmatter_noise"}: - render_default = False - if role.role in {"noise", "frontmatter_noise", "table_html"}: - index_default = False - row = { - "paper_id": block["paper_id"], - "page": block["page"], - "block_id": block["block_id"], - "raw_label": block.get("raw_label", "unknown"), - "raw_order": block.get("raw_order", 0), - "bbox": block.get("bbox", [0, 0, 0, 0]), - "text": block.get("text", ""), - "role": role.role, - "role_confidence": role.confidence, - "evidence": role.evidence, - "render_default": render_default, - "index_default": index_default, - } - rows.append(row) + page = block.get("page", 1) + by_page.setdefault(page, []).append(block) + + rows = [] + for page in sorted(by_page.keys()): + raw_page_blocks = by_page[page] + # Build mapped versions for role-assignment consumption + # (assign_block_role expects block_label/block_content keys) + page_as_role_input: list[dict] = [] + for raw_block in raw_page_blocks: + page_as_role_input.append({ + "block_label": raw_block.get("raw_label", "unknown"), + "block_content": raw_block.get("text", ""), + "block_bbox": raw_block.get("bbox", [0, 0, 0, 0]), + "page": raw_block.get("page", 1), + }) + for i, block in enumerate(raw_page_blocks): + role_input = page_as_role_input[i] + role = assign_block_role( + role_input, + page_blocks=page_as_role_input, + page_width=block.get("page_width", 0), + page_height=block.get("page_height", 0), + ) + render_default = role.role not in {"noise", "unknown_structural"} + index_default = True + if role.role in {"noise", "page_header", "page_footer", "frontmatter_noise"}: + render_default = False + if role.role in {"noise", "frontmatter_noise", "table_html"}: + index_default = False + row = { + "paper_id": block["paper_id"], + "page": block["page"], + "block_id": block["block_id"], + "raw_label": block.get("raw_label", "unknown"), + "raw_order": block.get("raw_order", 0), + "bbox": block.get("bbox", [0, 0, 0, 0]), + "text": block.get("text", ""), + "page_width": block.get("page_width", 0), + "page_height": block.get("page_height", 0), + "role": role.role, + "role_confidence": role.confidence, + "evidence": role.evidence, + "render_default": render_default, + "index_default": index_default, + } + rows.append(row) return rows diff --git a/paperforge/worker/ocr_figures.py b/paperforge/worker/ocr_figures.py index 855725d0..879de61f 100644 --- a/paperforge/worker/ocr_figures.py +++ b/paperforge/worker/ocr_figures.py @@ -13,6 +13,30 @@ _FIGURE_NUMBER_PATTERN = re.compile( flags=re.IGNORECASE, ) +_BODY_MENTION_VERBS = ( + "shows", + "illustrates", + "depicts", + "presents", + "summarizes", + "demonstrates", + "displays", + "reveals", + "indicates", + "highlights", + "compares", + "outlines", + "reports", + "lists", +) + +_BODY_MENTION_PATTERN = re.compile( + r"\b(?:Figure|Fig\.?|Supplementary\s+Figure|Supplementary\s+Fig\.?|" + r"Extended\s+Data\s+Figure|Extended\s+Data\s+Fig\.?)\s+" + r"\d+\.?\s+(?:" + "|".join(_BODY_MENTION_VERBS) + r")\b", + flags=re.IGNORECASE, +) + def _extract_figure_number(text: str) -> int | None: m = _FIGURE_NUMBER_PATTERN.search(text) @@ -51,15 +75,28 @@ def _centroid_y(bbox: list[float]) -> float: return (bbox[1] + bbox[3]) / 2 +def _is_body_mention(block: dict) -> bool: + raw_role = block.get("raw_role", block.get("role", "")) + if raw_role == "body_paragraph": + return True + if block.get("block_label", "") == "text": + text = block.get("text", "") + return bool(_BODY_MENTION_PATTERN.search(text)) + return False + + def build_figure_inventory(structured_blocks: list[dict]) -> dict[str, Any]: legends: list[dict] = [] assets: list[dict] = [] + unmatched_legends: list[dict] = [] unmatched_assets: list[dict] = [] matched_figures: list[dict] = [] for block in structured_blocks: role = block.get("role", "") if role == "figure_caption": + if _is_body_mention(block): + continue legends.append(block) elif role == "figure_asset": assets.append(block) @@ -75,7 +112,7 @@ def build_figure_inventory(structured_blocks: list[dict]) -> dict[str, Any]: legend_text = legend.get("text", "") fig_num = _extract_figure_number(legend_text) - candidate_pages = [legend_page, legend_page + 1, legend_page - 1] + candidate_pages = [legend_page] matched_assets = [] for page in candidate_pages: @@ -93,8 +130,7 @@ def build_figure_inventory(structured_blocks: list[dict]) -> dict[str, Any]: overlap = _compute_overlap_score(legend_bbox, asset_bbox) dist_y = abs(_centroid_y(legend_bbox) - _centroid_y(asset_bbox)) direction_bonus = 1.0 if _centroid_y(asset_bbox) < _centroid_y(legend_bbox) else 0.5 - same_page_bonus = 2.0 if page == legend_page else 0.0 - score = overlap * 10 + direction_bonus + same_page_bonus - dist_y * 0.01 + score = overlap * 10 + direction_bonus + 2.0 - dist_y * 0.01 candidates_for_page.append( { "asset_index": i, @@ -102,7 +138,7 @@ def build_figure_inventory(structured_blocks: list[dict]) -> dict[str, Any]: "score": score, "overlap": overlap, "distance_y": dist_y, - "same_page": page == legend_page, + "same_page": True, } ) @@ -118,6 +154,8 @@ def build_figure_inventory(structured_blocks: list[dict]) -> dict[str, Any]: if matched_assets: break + is_legend_only = len(matched_assets) == 0 + matched_figures.append( { "legend_block_id": legend.get("block_id", ""), @@ -131,11 +169,14 @@ def build_figure_inventory(structured_blocks: list[dict]) -> dict[str, Any]: } for a in matched_assets ], - "confidence": 0.85 if matched_assets else 0.4, - "flags": [] if matched_assets else ["legend_only"], + "confidence": 0.85 if not is_legend_only else 0.4, + "flags": [] if not is_legend_only else ["legend_only"], } ) + if is_legend_only: + unmatched_legends.append(legend) + for i, asset in enumerate(assets): if i not in used_asset_indices: unmatched_assets.append(asset) @@ -144,7 +185,7 @@ def build_figure_inventory(structured_blocks: list[dict]) -> dict[str, Any]: "figure_legends": legends, "figure_assets": assets, "matched_figures": matched_figures, - "unmatched_legends": [], + "unmatched_legends": unmatched_legends, "unmatched_assets": unmatched_assets, "official_figure_count": len(matched_figures), } diff --git a/paperforge/worker/ocr_health.py b/paperforge/worker/ocr_health.py index 8c6f6aed..adf7e417 100644 --- a/paperforge/worker/ocr_health.py +++ b/paperforge/worker/ocr_health.py @@ -27,8 +27,12 @@ def build_ocr_health( figure_asset_count = len(figure_inventory.get("matched_figures", [])) unmatched_legends = len(figure_inventory.get("unmatched_legends", [])) unmatched_figure_assets = len(figure_inventory.get("unmatched_assets", [])) - table_asset_count = sum(1 for t in table_inventory.get("tables", []) if t.get("has_asset")) - empty_tables = sum(1 for t in table_inventory.get("tables", []) if not t.get("has_asset")) + tables = table_inventory.get("tables", []) + table_asset_count = sum(1 for t in tables if t.get("has_asset")) + empty_tables = sum(1 for t in tables if not t.get("has_asset")) + formal_table_count = len([t for t in tables if not t.get("is_continuation")]) + entries_with_asset = sum(1 for t in tables if t.get("has_asset")) + table_segment_count = entries_with_asset + len(table_inventory.get("unmatched_assets", [])) media_without_caption = unmatched_figure_assets caption_without_media = unmatched_legends + len(table_inventory.get("unmatched_captions", [])) @@ -66,6 +70,8 @@ def build_ocr_health( "figure_asset_count": figure_asset_count, "table_caption_count": table_caption_count, "table_asset_count": table_asset_count, + "formal_table_count": formal_table_count, + "table_segment_count": table_segment_count, "media_without_caption_count": media_without_caption, "caption_without_media_count": caption_without_media, "empty_table_count": empty_tables, diff --git a/paperforge/worker/ocr_metadata.py b/paperforge/worker/ocr_metadata.py index 31f86e9f..fce835e5 100644 --- a/paperforge/worker/ocr_metadata.py +++ b/paperforge/worker/ocr_metadata.py @@ -8,6 +8,19 @@ from typing import Any from paperforge.core.io import write_json +def _clean_author_display(author_text: str) -> str: + text = author_text + text = re.sub(r'\$\s+\^', '$^', text) + text = re.sub(r'\^\s+\{', '^{', text) + text = re.sub(r'\}\s+\$', '}$', text) + text = re.sub(r',(\s*)and(\s*)', r', and ', text) + text = re.sub(r';(\s*)and(\s*)', r'; and ', text) + text = re.sub(r'([,;:])\1+', r'\1', text) + text = re.sub(r'\s{2,}', ' ', text) + text = text.strip().strip(';, ') + return text.strip() + + def extract_frontmatter_candidates(blocks_structured_path: Path) -> dict[str, Any]: candidates: dict[str, Any] = { "title": None, @@ -61,6 +74,25 @@ def resolve_metadata( # --- title --- zotero_title = source_metadata.get("title", "") ocr_title = frontmatter_candidates.get("title", "") + _BACKMATTER_TITLE_DENY_LIST = { + "generative ai statement", + "acknowledgments", + "acknowledgements", + "funding", + "conflict of interest", + "competing interests", + "data availability", + "supplementary materials", + "supplementary material", + "author contributions", + "declaration of competing interest", + "credit authorship contribution statement", + "ethical statement", + "ethics statement", + "institutional review board", + } + if ocr_title and ocr_title.lower().strip() in _BACKMATTER_TITLE_DENY_LIST: + ocr_title = "" title_entry: dict[str, Any] = { "value": zotero_title or ocr_title, "source": "zotero" if zotero_title else ("ocr_frontmatter" if ocr_title else "unknown"), @@ -68,11 +100,13 @@ def resolve_metadata( title_entry["confidence"] = 0.99 if zotero_title else (0.7 if ocr_title else 0.3) alternatives = [] if ocr_title and ocr_title != zotero_title: - alternatives.append({ - "value": ocr_title, - "source": "ocr_frontmatter", - "confidence": 0.7, - }) + alternatives.append( + { + "value": ocr_title, + "source": "ocr_frontmatter", + "confidence": 0.7, + } + ) if alternatives: title_entry["alternatives"] = alternatives resolved["title"] = title_entry @@ -81,8 +115,7 @@ def resolve_metadata( zotero_authors = source_metadata.get("authors", []) ocr_authors_text = frontmatter_candidates.get("authors_text", "") ocr_author_list = ( - [a.strip() for a in re.split(r",\s+(?=[A-Z])", ocr_authors_text) if a.strip()] - if ocr_authors_text else [] + [a.strip() for a in re.split(r",\s+(?=[A-Z])", ocr_authors_text) if a.strip()] if ocr_authors_text else [] ) if isinstance(zotero_authors, list) and len(zotero_authors) > 0: @@ -104,6 +137,14 @@ def resolve_metadata( "confidence": 0.3, } + # --- authors_display (cleaned string for UI) --- + if isinstance(zotero_authors, list) and len(zotero_authors) > 0: + resolved["authors_display"] = ", ".join(zotero_authors) + elif ocr_authors_text: + resolved["authors_display"] = _clean_author_display(ocr_authors_text) + else: + resolved["authors_display"] = "" + # --- year --- zotero_year = source_metadata.get("year", 0) if zotero_year: diff --git a/paperforge/worker/ocr_objects.py b/paperforge/worker/ocr_objects.py index b330d4b8..6fbb72e1 100644 --- a/paperforge/worker/ocr_objects.py +++ b/paperforge/worker/ocr_objects.py @@ -1,5 +1,6 @@ from __future__ import annotations +import contextlib import re from pathlib import Path from typing import Any @@ -12,7 +13,7 @@ def render_figure_object_markdown(figure: dict[str, Any]) -> str: # Extract figure number for the title figure_id = figure.get("figure_id", "") if figure_id and not figure_id.startswith("orphan_"): - m = re.search(r'\d+', figure_id) + m = re.search(r"\d+", figure_id) num = str(int(m.group())) if m else figure_id label = f"Figure {num}" else: @@ -35,10 +36,14 @@ def render_table_object_markdown(table: dict[str, Any]) -> str: caption = table.get("caption", "") image_relpath = table.get("image_relpath", "") - table_id_raw = table.get("table_id", "unknown") - m = re.search(r'\d+', table_id_raw) - table_id = str(int(m.group())) if m else table_id_raw - label = f"Table {table_id}" + formal_num = table.get("formal_table_number") + if formal_num is not None: + label = f"Table {formal_num}" + else: + table_id_raw = table.get("table_id", "unknown") + m = re.search(r"\d+", table_id_raw) + table_id = str(int(m.group())) if m else table_id_raw + label = f"Table {table_id}" parts = [f"# {label}", "", f"![](../../{image_relpath})", ""] if caption: @@ -58,31 +63,78 @@ def _write_object_markdown(md: str, dst: Path) -> None: dst.write_text(md.strip() + "\n", encoding="utf-8") +def _find_cached_page_image(page_cache_dir: Path | None, page_num: int) -> Path | None: + if page_cache_dir is None: + return None + for suffix in (".jpg", ".png"): + candidate = page_cache_dir / f"page_{page_num:03d}{suffix}" + if candidate.exists(): + return candidate + return None + + def _crop_asset_from_pdf( pdf_path: Path, page_num: int, bbox: list[float], dst: Path, + *, + page_width: int = 0, + page_height: int = 0, + page_cache_dir: Path | None = None, ) -> bool: """Crop a region from a PDF page and save as JPEG. - Args: - pdf_path: Path to the PDF file. - page_num: 1-based page number. - bbox: [x1, y1, x2, y2] bounding box in PDF coordinates. - dst: Output path for the JPEG file. - - Returns: - True if crop succeeded, False otherwise. + The OCR bbox values are in rendered page-image coordinates, not PDF point + coordinates. To preserve the previously stable behavior, first render the + PDF page into the OCR page dimensions, then crop with the OCR bbox. """ + if dst.exists(): + with contextlib.suppress(Exception): + dst.unlink() + + cached_page_image = _find_cached_page_image(page_cache_dir, page_num) + if cached_page_image is not None: + try: + from paperforge.worker.ocr import crop_block_asset + except ImportError: + return False + return crop_block_asset(cached_page_image, [int(v) for v in bbox], dst) + + if not pdf_path.exists(): + return False + + if page_width > 0 and page_height > 0 and page_cache_dir is not None: + try: + import fitz + + from paperforge.worker.ocr import crop_block_asset, render_pdf_page_cached + except ImportError: + return False + + try: + doc = fitz.open(str(pdf_path)) + page_image_path = page_cache_dir / f"page_{page_num:03d}.jpg" + rendered = render_pdf_page_cached( + doc, + page_num, + target_width=page_width, + target_height=page_height, + destination=page_image_path, + ) + doc.close() + if not rendered: + return False + return crop_block_asset(rendered, [int(v) for v in bbox], dst) + except Exception: + return False + + # Fallback path for callers lacking OCR page dimensions. try: import fitz except ImportError: return False - if not pdf_path.exists(): - return False - try: doc = fitz.open(str(pdf_path)) page = doc[page_num - 1] @@ -102,58 +154,83 @@ def extract_and_write_objects( table_inventory: dict[str, Any], asset_root: Path, render_root: Path, + *, + page_dimensions_by_page: dict[int, tuple[int, int]] | None = None, ) -> None: - """Extract figure/table asset crops from PDF and write object markdown. - - Args: - pdf_path: Path to the source PDF (may be None). - figure_inventory: The figure inventory dict. - table_inventory: The table inventory dict. - asset_root: Root directory for assets (e.g., /assets/). - render_root: Root directory for render objects (e.g., /render/). - """ + """Extract figure/table asset crops from PDF and write object markdown.""" figures_asset_dir = asset_root / "figures" tables_asset_dir = asset_root / "tables" orphans_asset_dir = asset_root / "orphans" figures_render_dir = render_root / "figures" tables_render_dir = render_root / "tables" + page_cache_dir = asset_root.parent / "pages" - for d in (figures_asset_dir, tables_asset_dir, orphans_asset_dir, - figures_render_dir, tables_render_dir): + for d in ( + figures_asset_dir, + tables_asset_dir, + orphans_asset_dir, + figures_render_dir, + tables_render_dir, + page_cache_dir, + ): d.mkdir(parents=True, exist_ok=True) + if page_dimensions_by_page is None: + page_dimensions_by_page = {} + + def _page_dims(page_num: int) -> tuple[int, int]: + return page_dimensions_by_page.get(page_num, (0, 0)) + # Process matched figures for i, match in enumerate(figure_inventory.get("matched_figures", [])): fig_id = f"figure_{i + 1:03d}" caption_text = match.get("text", "") page = match.get("page", 0) + page_width, page_height = _page_dims(page) asset_path_rel = f"assets/figures/{fig_id}.jpg" asset_path_abs = figures_asset_dir / f"{fig_id}.jpg" was_cropped = False for asset_info in match.get("matched_assets", []): bbox = asset_info.get("bbox", [0, 0, 0, 0]) - if pdf_path and bbox and all(v > 0 for v in bbox): - if _crop_asset_from_pdf(pdf_path, page, bbox, asset_path_abs): - was_cropped = True - break + if pdf_path and bbox and all(v > 0 for v in bbox) and _crop_asset_from_pdf( + pdf_path, + page, + bbox, + asset_path_abs, + page_width=page_width, + page_height=page_height, + page_cache_dir=page_cache_dir, + ): + was_cropped = True + break if not was_cropped: for asset in figure_inventory.get("unmatched_assets", []): bbox = asset.get("bbox", [0, 0, 0, 0]) asset_page = asset.get("page", 0) - if pdf_path and bbox and all(v > 0 for v in bbox): - if _crop_asset_from_pdf(pdf_path, asset_page, bbox, asset_path_abs): - was_cropped = True - break + asset_page_width, asset_page_height = _page_dims(asset_page) + if pdf_path and bbox and all(v > 0 for v in bbox) and _crop_asset_from_pdf( + pdf_path, + asset_page, + bbox, + asset_path_abs, + page_width=asset_page_width, + page_height=asset_page_height, + page_cache_dir=page_cache_dir, + ): + was_cropped = True + break - md = render_figure_object_markdown({ - "figure_id": fig_id, - "page": page, - "caption": caption_text, - "image_relpath": asset_path_rel, - "confidence": match.get("confidence", 0.5), - }) + md = render_figure_object_markdown( + { + "figure_id": fig_id, + "page": page, + "caption": caption_text, + "image_relpath": asset_path_rel, + "confidence": match.get("confidence", 0.5), + } + ) _write_object_markdown(md, figures_render_dir / f"{fig_id}.md") # Process unmatched assets as orphans @@ -163,20 +240,30 @@ def extract_and_write_objects( orphan_id = f"orphan_{orphan_count:03d}" bbox = asset.get("bbox", [0, 0, 0, 0]) page = asset.get("page", 0) + page_width, page_height = _page_dims(page) asset_path_rel = f"assets/orphans/{orphan_id}.jpg" asset_path_abs = orphans_asset_dir / f"{orphan_id}.jpg" - was_cropped = False if pdf_path and bbox and all(v > 0 for v in bbox): - was_cropped = _crop_asset_from_pdf(pdf_path, page, bbox, asset_path_abs) + _crop_asset_from_pdf( + pdf_path, + page, + bbox, + asset_path_abs, + page_width=page_width, + page_height=page_height, + page_cache_dir=page_cache_dir, + ) - md = render_figure_object_markdown({ - "figure_id": orphan_id, - "page": page, - "caption": "", - "image_relpath": asset_path_rel, - "confidence": 0.3, - }) + md = render_figure_object_markdown( + { + "figure_id": orphan_id, + "page": page, + "caption": "", + "image_relpath": asset_path_rel, + "confidence": 0.3, + } + ) _write_object_markdown(md, figures_render_dir / f"{orphan_id}.md") # Process tables @@ -184,21 +271,33 @@ def extract_and_write_objects( tbl_id = f"table_{i + 1:03d}" caption_text = table.get("caption_text", "") page = table.get("page", 0) + page_width, page_height = _page_dims(page) asset_bbox = table.get("asset_bbox", [0, 0, 0, 0]) asset_path_rel = f"assets/tables/{tbl_id}.jpg" asset_path_abs = tables_asset_dir / f"{tbl_id}.jpg" was_cropped = False if table.get("has_asset") and pdf_path and asset_bbox and all(v > 0 for v in asset_bbox): - was_cropped = _crop_asset_from_pdf(pdf_path, page, asset_bbox, asset_path_abs) + was_cropped = _crop_asset_from_pdf( + pdf_path, + page, + asset_bbox, + asset_path_abs, + page_width=page_width, + page_height=page_height, + page_cache_dir=page_cache_dir, + ) - md = render_table_object_markdown({ - "table_id": tbl_id, - "page": page, - "caption": caption_text, - "image_relpath": asset_path_rel, - "confidence": 0.85 if was_cropped else 0.4, - }) + md = render_table_object_markdown( + { + "table_id": tbl_id, + "page": page, + "caption": caption_text, + "image_relpath": asset_path_rel, + "confidence": 0.85 if was_cropped else 0.4, + "formal_table_number": table.get("formal_table_number") or table.get("table_number"), + } + ) _write_object_markdown(md, tables_render_dir / f"{tbl_id}.md") # Process unmatched table assets as orphans @@ -207,17 +306,28 @@ def extract_and_write_objects( orphan_id = f"orphan_{orphan_count:03d}" bbox = asset.get("bbox", [0, 0, 0, 0]) page = asset.get("page", 0) + page_width, page_height = _page_dims(page) asset_path_rel = f"assets/orphans/{orphan_id}.jpg" asset_path_abs = orphans_asset_dir / f"{orphan_id}.jpg" if pdf_path and bbox and all(v > 0 for v in bbox): - _crop_asset_from_pdf(pdf_path, page, bbox, asset_path_abs) + _crop_asset_from_pdf( + pdf_path, + page, + bbox, + asset_path_abs, + page_width=page_width, + page_height=page_height, + page_cache_dir=page_cache_dir, + ) - md = render_figure_object_markdown({ - "figure_id": orphan_id, - "page": page, - "caption": "", - "image_relpath": asset_path_rel, - "confidence": 0.3, - }) + md = render_figure_object_markdown( + { + "figure_id": orphan_id, + "page": page, + "caption": "", + "image_relpath": asset_path_rel, + "confidence": 0.3, + } + ) _write_object_markdown(md, figures_render_dir / f"{orphan_id}.md") diff --git a/paperforge/worker/ocr_rebuild.py b/paperforge/worker/ocr_rebuild.py index eddca8e9..6da8e10c 100644 --- a/paperforge/worker/ocr_rebuild.py +++ b/paperforge/worker/ocr_rebuild.py @@ -38,6 +38,7 @@ def run_derived_rebuild_for_keys(vault: Path, keys: list[str]) -> dict: render outputs, and health — from stored raw blocks only. """ from paperforge.worker._utils import pipeline_paths, read_jsonl + from paperforge.worker.ocr import validate_ocr_meta from paperforge.worker.ocr_artifacts import artifact_paths_for_root ocr_root = pipeline_paths(vault)["ocr"] @@ -93,22 +94,36 @@ def run_derived_rebuild_for_keys(vault: Path, keys: list[str]) -> dict: ocr_meta = read_json(artifacts.meta_json) if artifacts.meta_json.exists() else {} source_pdf_path = Path(ocr_meta.get("source_pdf", "")) if ocr_meta.get("source_pdf") else None + page_dimensions_by_page: dict[int, tuple[int, int]] = {} + for block in structured: + page = int(block.get("page", 0) or 0) + width = int(block.get("page_width", 0) or 0) + height = int(block.get("page_height", 0) or 0) + if page and width and height and page not in page_dimensions_by_page: + page_dimensions_by_page[page] = (width, height) + extract_and_write_objects( pdf_path=source_pdf_path, figure_inventory=figure_inventory, table_inventory=table_inventory, asset_root=paper_root / "assets", render_root=paper_root / "render", + page_dimensions_by_page=page_dimensions_by_page, ) # Rebuild render output from paperforge.worker.ocr_render import render_fulltext_markdown, write_render_outputs + rebuild_page_count = ocr_meta.get("page_count", 0) or 0 + if not rebuild_page_count: + all_rebuild_pages = {int(b["page"]) for b in structured if b.get("page")} + rebuild_page_count = max(all_rebuild_pages) if all_rebuild_pages else 0 markdown = render_fulltext_markdown( structured_blocks=structured, resolved_metadata=resolved, figure_inventory=figure_inventory, table_inventory=table_inventory, + page_count=rebuild_page_count, ) write_render_outputs( render_root=paper_root / "render", @@ -140,6 +155,15 @@ def run_derived_rebuild_for_keys(vault: Path, keys: list[str]) -> dict: # Update version state in meta.json meta = read_json(artifacts.meta_json) if artifacts.meta_json.exists() else {} meta = _apply_post_rebuild_version_flags(meta) + # Rebuild regenerated the derived outputs; validate from a clean + # optimistic status instead of short-circuiting on a stale + # done_incomplete value from a previous render. + meta["ocr_status"] = "done" + # Re-validate and clear stale errors (e.g. page marker mismatch from pre-fix render) + paths_dict = {"ocr": pipeline_paths(vault)["ocr"]} + _status, _err = validate_ocr_meta(paths_dict, meta) + meta["ocr_status"] = _status + meta["error"] = _err if _err else "" write_json(artifacts.meta_json, meta) rebuilt_count += 1 @@ -155,6 +179,7 @@ def _enrich_meta_from_paper_note(vault: Path, key: str, meta_path: Path) -> None """ try: from paperforge.worker._utils import pipeline_paths + paths = pipeline_paths(vault) lit_dir = paths.get("literature") if lit_dir and lit_dir.exists(): @@ -164,6 +189,7 @@ def _enrich_meta_from_paper_note(vault: Path, key: str, meta_path: Path) -> None note = next((m for m in matches if m.stem == key), None) if note and note.exists(): from paperforge.adapters.obsidian_frontmatter import read_frontmatter_dict + content = note.read_text(encoding="utf-8", errors="replace") fm = read_frontmatter_dict(content) if not fm: @@ -233,11 +259,23 @@ def backfill_from_result(vault: Path, key: str) -> dict: meta_before = read_json(paper_dir / "meta.json") if (paper_dir / "meta.json").exists() else {} src = str(meta_before.get("source_pdf", "")) if src and "storage:" in src: - from paperforge.pdf_resolver import resolve_pdf_path - from paperforge.config import load_vault_config - cfg = load_vault_config(vault) - zotero_dir = cfg.get("zotero_dir", "") - resolved = resolve_pdf_path(src, True, vault, Path(zotero_dir) if zotero_dir else None) + from paperforge.pdf_resolver import resolve_junction, resolve_pdf_path + + pf_cfg = read_json(vault / "paperforge.json") if (vault / "paperforge.json").exists() else {} + zotero_path = None + zotero_dir = pf_cfg.get("zotero_data_dir", "") or pf_cfg.get("zotero_link", "") + if zotero_dir: + zotero_path = Path(zotero_dir) + if not zotero_path.is_absolute(): + zotero_path = resolve_junction((vault / zotero_dir).resolve()) + resolved = resolve_pdf_path(src, True, vault, zotero_path) + if not resolved and zotero_path and src.startswith("storage:"): + storage_key = src[len("storage:") :].split("/")[0].strip() + storage_dir = (zotero_path / "storage" / storage_key).resolve() + if storage_dir.exists(): + pdfs = [f for f in storage_dir.iterdir() if f.suffix.lower() == ".pdf"] + if pdfs: + resolved = str(pdfs[0]) if resolved: meta_before["source_pdf"] = resolved write_json(paper_dir / "meta.json", meta_before) diff --git a/paperforge/worker/ocr_render.py b/paperforge/worker/ocr_render.py index c629db8e..051f6c72 100644 --- a/paperforge/worker/ocr_render.py +++ b/paperforge/worker/ocr_render.py @@ -7,10 +7,10 @@ from paperforge.worker.ocr_roles import FRONTMATTER_NOISE def _normalize_latex(text: str) -> str: - text = re.sub(r'\$\s+', '$', text) - text = re.sub(r'\s+\$', '$', text) - text = re.sub(r'\$\^\{\s+', '$^{', text) - text = re.sub(r'\s+\}\$', '}$', text) + text = re.sub(r"\$\s+", "$", text) + text = re.sub(r"\s+\$", "$", text) + text = re.sub(r"\$\^\{\s+", "$^{", text) + text = re.sub(r"\s+\}\$", "}$", text) return text @@ -20,18 +20,857 @@ def _is_bogus_heading(text: str) -> bool: return True if t.count(". ") > 1: return True - if any(v in t.lower().split() for v in ["is", "are", "was", "were", "have", "has", "been"]): - if len(t) > 50: + return any(v in t.lower().split() for v in ["is", "are", "was", "were", "have", "has", "been"]) and len(t) > 50 + + +_TAIL_ROLES = frozenset( + { + "backmatter_heading", + "backmatter_body", + "tail_candidate_body", + "reference_heading", + "reference_item", + } +) + + +def _has_tail_role(block: dict) -> bool: + return block.get("role") in _TAIL_ROLES + + +def _get_column(block: dict, page_width: float = 1200) -> int: + bbox = block.get("bbox") or block.get("block_bbox") + if bbox and len(bbox) >= 4: + x_center = (bbox[0] + bbox[2]) / 2 + return 0 if x_center < page_width / 2 else 1 + return 0 + + +def _estimate_noise_bands( + structured_blocks: list[dict], +) -> tuple[float | None, float | None]: + header_candidates: list[float] = [] + footer_candidates: list[float] = [] + + for block in structured_blocks: + role = block.get("role", "") + bbox = block.get("bbox") or block.get("block_bbox") + if not bbox or len(bbox) < 4: + continue + page_height = block.get("page_height", 0) or 0 + if page_height == 0: + continue + y2, y1 = bbox[3], bbox[1] + + noise_roles = {"noise", "header", "footer", "number"} + raw_label = block.get("raw_label", "") + if role in noise_roles or raw_label in ("header", "footer", "number"): + if y2 < page_height * 0.15: + header_candidates.append(y2) + if y1 > page_height * 0.85: + footer_candidates.append(y1) + + header_band = max(header_candidates) if header_candidates else None + footer_band = min(footer_candidates) if footer_candidates else None + return header_band, footer_band + + +def _is_in_usable_content( + block: dict, + header_band: float | None, + footer_band: float | None, +) -> bool: + bbox = block.get("bbox") or block.get("block_bbox") + if not bbox or len(bbox) < 4: + return True + y1, y2 = bbox[1], bbox[3] + if header_band is not None and y2 < header_band: + return False + return not (footer_band is not None and y1 > footer_band) + + +def _find_owning_heading( + body: dict, + sections: list[dict], + page_width: float = 1200, +) -> int | None: + body_bbox = body.get("bbox") or body.get("block_bbox") + if not body_bbox or len(body_bbox) < 4: + return None + + body_y = body_bbox[1] + body_col = _get_column(body, page_width) + + candidates: list[tuple[int, float]] = [] + for i, sec in enumerate(sections): + h = sec["heading"] + h_bbox = h.get("bbox") or h.get("block_bbox") + if not h_bbox or len(h_bbox) < 4: + continue + h_bottom = h_bbox[3] + if h_bottom > body_y: + continue + h_col = _get_column(h, page_width) + dist = body_y - h_bottom + col_penalty = 0.0 if h_col == body_col else 10000.0 + candidates.append((i, dist + col_penalty)) + + if not candidates: + return None + candidates.sort(key=lambda x: x[1]) + return candidates[0][0] + + +def _has_same_column_anchor_above( + body: dict, + anchors: list[dict], + page_width: float = 1200, +) -> bool: + body_bbox = body.get("bbox") or body.get("block_bbox") + if not body_bbox or len(body_bbox) < 4: + return False + + body_y = body_bbox[1] + body_col = _get_column(body, page_width) + + for anchor in anchors: + anchor_bbox = anchor.get("bbox") or anchor.get("block_bbox") + if not anchor_bbox or len(anchor_bbox) < 4: + continue + if _get_column(anchor, page_width) != body_col: + continue + if anchor_bbox[3] <= body_y: return True return False +def _extract_style_profile(block: dict) -> dict | None: + span_meta = block.get("span_metadata") + if not span_meta: + return None + + # List of per-character spans (future format from PyMuPDF) + if isinstance(span_meta, list) and len(span_meta) > 0: + sizes = [] + fonts = set() + flags_list = [] + colors = [] + for s in span_meta: + if not isinstance(s, dict): + continue + size = s.get("size") or 0 + if size: + sizes.append(size) + font = s.get("font", "") + if font: + fonts.add(font) + flags_list.append(s.get("flags", 0) or 0) + colors.append(s.get("color", 0) or 0) + + if not sizes: + return None + + return { + "mean_size": sum(sizes) / len(sizes), + "max_size": max(sizes), + "font_families": fonts, + "is_bold": any(f & 16 for f in flags_list), + "is_italic": any(f & 4 for f in flags_list), + "is_colored": any(c != 0 for c in colors), + } + + # Flat dict format (backward compat with test data) + if isinstance(span_meta, dict): + size = span_meta.get("size", 0) or 0 + flags = span_meta.get("flags", "") + if isinstance(flags, str): + is_bold = "bold" in flags.lower() + is_italic = "italic" in flags.lower() + else: + is_bold = bool(flags & 16) if flags else False + is_italic = bool(flags & 4) if flags else False + + if not size and not is_bold: + return None + + return { + "mean_size": float(size), + "max_size": float(size), + "font_families": set(), + "is_bold": is_bold, + "is_italic": is_italic, + "is_colored": False, + } + + return None + + +def _build_heading_style_profiles(blocks: list[dict]) -> dict: + heading_roles = frozenset({"section_heading", "subsection_heading", "backmatter_heading", "reference_heading"}) + profiles_with_sizes = [] + + for block in blocks: + if block.get("role") in heading_roles: + profile = _extract_style_profile(block) + if profile is not None: + profiles_with_sizes.append(profile["mean_size"]) + + if len(profiles_with_sizes) < 3: + return {} + + unique_sizes = sorted(set(profiles_with_sizes)) + if not unique_sizes: + return {} + + clusters = [] + current = [unique_sizes[0]] + for s in unique_sizes[1:]: + if s <= current[-1] + 2.0: + current.append(s) + else: + clusters.append(current) + current = [s] + clusters.append(current) + + # Build full profiles for each cluster + cluster_profiles = [] + for cluster in clusters: + matching = [] + for block in blocks: + if block.get("role") in heading_roles: + p = _extract_style_profile(block) + if p is not None and any(abs(p["mean_size"] - s) <= 2 for s in cluster): + matching.append(p) + cluster_profiles.append(matching) + + clusters.sort(key=lambda c: sum(c) / len(c), reverse=True) + + keys = ["primary", "subsection", "backmatter", "body"] + result = {} + for i, (cluster, profiles) in enumerate(zip(clusters, cluster_profiles, strict=False)): + if i >= len(keys): + break + is_bold = any(p["is_bold"] for p in profiles) + fonts = set() + for p in profiles: + fonts.update(p["font_families"]) + result[keys[i]] = { + "size_min": min(cluster), + "size_max": max(cluster), + "bold": is_bold, + "fonts": fonts, + } + + return result + + +def _disambiguate_heading_role(block: dict, style_profiles: dict) -> str | None: + profile = _extract_style_profile(block) + if profile is None: + return None + + size = profile["mean_size"] + + for role_key in ("primary", "subsection", "backmatter", "body"): + cfg = style_profiles.get(role_key) + if cfg is None: + continue + if cfg["size_min"] <= size <= cfg["size_max"]: + if role_key == "primary": + return "section_heading" + elif role_key == "subsection": + return "subsection_heading" + elif role_key == "backmatter": + if profile["is_bold"]: + return "backmatter_heading" + return None + elif role_key == "body": + return None + + return None + + +def _reorder_tail_run( + tail_blocks: list[dict], + carried_ref: dict | None = None, + carried_backmatter: dict | None = None, + header_band: float | None = None, + footer_band: float | None = None, + page_width: float = 1200, +) -> tuple[list[dict], dict | None, dict | None]: + """Group tail blocks using geometric ownership and reference zone. + + Non-tail pass blocks (body_paragraph, tail_candidate_body, etc.) are + emitted first, preserving their natural page order. Backmatter sections + (heading + body) are emitted next in column-sorted order. References + zone (heading + all items below it) is emitted last. Body blocks are + attached to the nearest backmatter heading above them in the same column, + falling back to absolute proximity with a cross-column penalty. + + When blocks lack bbox data, falls back to FIFO matching for backward + compatibility. + """ + if not tail_blocks: + return tail_blocks, carried_ref + + # Quick check: do blocks have bbox data? If not, use FIFO fallback. + has_geo = any( + block.get("bbox") or block.get("block_bbox") + for block in tail_blocks + if block.get("role") in ("backmatter_heading", "reference_heading", "backmatter_body") + ) + + if not has_geo: + ordered, next_ref = _reorder_tail_run_fifo(tail_blocks, carried_ref) + return ordered, next_ref, carried_backmatter + + backmatter_sections: list[dict] = [] + ref_section: dict | None = carried_ref + ref_items: list[dict] = [] + non_tail_pass: list[dict] = [] + orphan_blocks: list[dict] = [] + carried_bodies: list[dict] = [] + + # Phase 1 — classify + body_pool: list[dict] = [] + for block in tail_blocks: + role = block.get("role") + if role == "backmatter_heading": + backmatter_sections.append({"heading": block, "bodies": []}) + elif role == "reference_heading": + ref_section = {"heading": block, "bodies": []} + elif role == "reference_item": + if _is_in_usable_content(block, header_band, footer_band): + ref_items.append(block) + else: + non_tail_pass.append(block) + elif role == "backmatter_body": + if _is_in_usable_content(block, header_band, footer_band): + body_pool.append(block) + else: + non_tail_pass.append(block) + else: + non_tail_pass.append(block) + + # Phase 2 — build references zone + ref_heading = ref_section.get("heading") if ref_section and ref_section is not carried_ref else None + ref_bottom: float = 0.0 + if ref_heading: + rh_bbox = ref_heading.get("bbox") or ref_heading.get("block_bbox") + if rh_bbox and len(rh_bbox) >= 4: + ref_bottom = rh_bbox[3] + + # Phase 3 — assign ref items to zone, backmatter bodies to body pool + for block in ref_items: + if ref_heading: + bbox = block.get("bbox") or block.get("block_bbox") + if bbox and len(bbox) >= 4 and bbox[1] >= ref_bottom: + ref_section["bodies"].append(block) + else: + body_pool.append(block) + else: + body_pool.append(block) + + # Phase 4 — geometric body attachment + first_local_anchor_top: float | None = None + local_heading_tops = [] + for sec in backmatter_sections: + bbox = sec["heading"].get("bbox") or sec["heading"].get("block_bbox") + if bbox and len(bbox) >= 4: + local_heading_tops.append(bbox[1]) + if ref_heading: + bbox = ref_heading.get("bbox") or ref_heading.get("block_bbox") + if bbox and len(bbox) >= 4: + local_heading_tops.append(bbox[1]) + if local_heading_tops: + first_local_anchor_top = min(local_heading_tops) + + for body in body_pool: + idx = _find_owning_heading(body, backmatter_sections, page_width) + if idx is not None: + backmatter_sections[idx]["bodies"].append(body) + elif carried_backmatter is not None: + bbox = body.get("bbox") or body.get("block_bbox") + if bbox and len(bbox) >= 4: + body_top = bbox[1] + if (first_local_anchor_top is None or body_top < first_local_anchor_top) and ( + not ref_heading or body_top < ref_bottom + ): + carried_bodies.append(body) + continue + elif ref_section is not None: + if ref_section is carried_ref: + ref_section["bodies"].append(body) + orphan_blocks.append(body) + else: + ref_section["bodies"].append(body) + else: + orphan_blocks.append(body) + + # Phase 5 — emit: non-tail pass (body/unknown), then backmatter, then references + result: list[dict] = [] + result.extend(non_tail_pass) + result.extend(carried_bodies) + for sec in backmatter_sections: + result.append(sec["heading"]) + result.extend(sec["bodies"]) + if ref_section is not None and ref_section is not carried_ref: + result.append(ref_section["heading"]) + result.extend(ref_section["bodies"]) + result.extend(orphan_blocks) + + next_backmatter = carried_backmatter + if backmatter_sections: + next_backmatter = backmatter_sections[-1]["heading"] + if ref_section is not None and ref_section is not carried_ref: + next_backmatter = None + + return result, ref_section if ref_section else carried_ref, next_backmatter + + +def _reorder_tail_run_fifo( + tail_blocks: list[dict], + carried_ref: dict | None = None, +) -> tuple[list[dict], dict | None]: + """FIFO fallback for blocks without bbox data.""" + backmatter_sections: list[dict] = [] + ref_section: dict | None = carried_ref + heading_queue: list[dict] = [] + orphan_bodies: list[dict] = [] + orphan_ref_items: list[dict] = [] + non_tail_pass: list[dict] = [] + + for block in tail_blocks: + role = block.get("role") + if role == "backmatter_heading": + sec = {"heading": block, "bodies": []} + backmatter_sections.append(sec) + heading_queue.append(sec) + elif role == "reference_heading": + ref_section = {"heading": block, "bodies": []} + elif role == "backmatter_body": + if heading_queue: + heading_queue.pop(0)["bodies"].append(block) + elif ref_section is not None: + if ref_section is carried_ref: + orphan_ref_items.append(block) + else: + ref_section["bodies"].append(block) + else: + orphan_bodies.append(block) + elif role == "body_paragraph": + non_tail_pass.append(block) + elif role == "reference_item": + if ref_section is not None: + if ref_section is carried_ref: + orphan_ref_items.append(block) + else: + ref_section["bodies"].append(block) + else: + non_tail_pass.append(block) + + result: list[dict] = [] + result.extend(non_tail_pass) + for sec in backmatter_sections: + result.append(sec["heading"]) + result.extend(sec["bodies"]) + if ref_section is not None and ref_section is not carried_ref: + result.append(ref_section["heading"]) + result.extend(ref_section["bodies"]) + result.extend(orphan_bodies) + result.extend(orphan_ref_items) + + return result, ref_section + + +def _promote_tail_body_candidates( + blocks: list[dict], + tail_spread: tuple[int, int] | None, + header_band: float | None = None, + footer_band: float | None = None, +) -> list[dict]: + """Promote plausible tail bodies from plain body_paragraph blocks. + + Base role assignment should stay conservative. This pass upgrades only + those body paragraphs that are geometrically compatible with tail section + ownership inside the reconciled tail spread. + """ + if tail_spread is None: + return blocks + + spread_start, spread_end = tail_spread + by_page: dict[int, list[int]] = {} + for idx, block in enumerate(blocks): + page = block.get("page") + if page is not None: + by_page.setdefault(page, []).append(idx) + + result = [dict(block) for block in blocks] + for page, indices in by_page.items(): + if page < spread_start or page > spread_end: + continue + + page_blocks = [result[i] for i in indices] + local_headings = [b for b in page_blocks if b.get("role") == "backmatter_heading"] + ref_heading = next((b for b in page_blocks if b.get("role") == "reference_heading"), None) + local_anchors = local_headings + local_tops = [] + for anchor in [*local_headings, *([ref_heading] if ref_heading else [])]: + bbox = anchor.get("bbox") or anchor.get("block_bbox") + if bbox and len(bbox) >= 4: + local_tops.append(bbox[1]) + first_local_anchor_top = min(local_tops) if local_tops else None + + for idx in indices: + block = result[idx] + if block.get("role") != "body_paragraph": + continue + if not _is_in_usable_content(block, header_band, footer_band): + continue + + bbox = block.get("bbox") or block.get("block_bbox") + if not bbox or len(bbox) < 4: + continue + + promote = False + page_width = block.get("page_width", 1200) or 1200 + if local_anchors and _has_same_column_anchor_above(block, local_anchors, page_width): + promote = True + elif page > spread_start: + body_top = bbox[1] + if first_local_anchor_top is None or body_top < first_local_anchor_top: + promote = True + + if promote: + block["role"] = "tail_candidate_body" + block["evidence"] = list(block.get("evidence") or []) + ["promoted in tail spread from body_paragraph"] + + return result + + +def _find_best_anchor( + body: dict, + anchors: list[dict], + ref_heading: dict | None = None, + page_width: float = 1200, +) -> int | None: + """Find the best backmatter heading anchor for a tail candidate body. + + Prefers same column, then nearest heading above in any column. + Handles cross-page continuations: an anchor on an earlier page is + always treated as above the body (regardless of raw Y coordinate). + Excludes ref_heading. Returns the anchor index or None. + """ + body_bbox = body.get("bbox") or body.get("block_bbox") + if not body_bbox or len(body_bbox) < 4: + return None + body_y = body_bbox[1] + body_page = body.get("page", 0) or 0 + body_mid = (body_bbox[0] + body_bbox[2]) / 2 + pw_mid = page_width / 2 + body_col = 0 if body_mid < pw_mid else 1 + + best_same: tuple[int, float] | None = None + best_other: tuple[int, float] | None = None + + for idx, anchor in enumerate(anchors): + if anchor is ref_heading: + continue + a_bbox = anchor.get("bbox") or anchor.get("block_bbox") + if not a_bbox or len(a_bbox) < 4: + continue + + anchor_page = anchor.get("page", 0) or 0 + if anchor_page > body_page: + continue # anchor on later page → cannot own body + + a_bottom = a_bbox[3] + if anchor_page == body_page and a_bottom > body_y: + continue # anchor below body on same page + + # Cross-page anchors (earlier page) are always valid — + # raw Y comparison across pages is meaningless. + a_mid = (a_bbox[0] + a_bbox[2]) / 2 + a_col = 0 if a_mid < pw_mid else 1 + + if anchor_page == body_page: + dist = body_y - a_bottom + else: + page_extent = body.get("page_height", 0) or page_width + dist = (body_page - anchor_page) * page_extent + max(0.0, page_extent - a_bottom) + + if body_col == a_col: + if best_same is None or dist < best_same[1]: + best_same = (idx, dist) + else: + if best_other is None or dist < best_other[1]: + best_other = (idx, dist) + + best = best_same or best_other + return best[0] if best is not None else None + + +def _detect_forward_body_end(blocks: list[dict]) -> int | None: + """Scan blocks front-to-back and return the last page of stable body. + + Tracks pages with body headings (section_heading, subsection_heading) + and body_paragraph continuity. When a page has tail roles + (backmatter_heading, reference_heading, etc.) and no body content, + the body is considered to have ended on the preceding clean body page. + Returns None if no clear body/backmatter boundary is found. + """ + if not blocks: + return None + by_page: dict[int, list[dict]] = {} + for block in blocks: + p = block.get("page") + if p is not None: + by_page.setdefault(p, []).append(block) + pages = sorted(by_page.keys()) + if not pages: + return None + + last_clean_body_page: int | None = None + + for page in pages: + roles = {b.get("role") for b in by_page[page]} + has_body = bool(roles & {"body_paragraph", "section_heading", "subsection_heading"}) + has_tail = bool(roles & _TAIL_ROLES) + + if has_body and not has_tail: + last_clean_body_page = page + elif has_tail: + if last_clean_body_page is not None: + return last_clean_body_page + if not has_body: + prev_idx = pages.index(page) - 1 + if prev_idx >= 0: + return pages[prev_idx] + return None + + return None + + +def _detect_backward_backmatter_start(blocks: list[dict]) -> int | None: + """Scan blocks backward and return the page where backmatter begins. + + Starting from the last page, looks for the first reference_heading or + backmatter_heading. Dense reference pages (>= 4 reference_item blocks) + are a strong signal. Short backmatter_body blocks near headings confirm + the backmatter zone. Returns None if no backmatter found. + """ + if not blocks: + return None + by_page: dict[int, list[dict]] = {} + for block in blocks: + p = block.get("page") + if p is not None: + by_page.setdefault(p, []).append(block) + pages = sorted(by_page.keys(), reverse=True) + if not pages: + return None + + for page in pages: + page_blocks = by_page[page] + roles = {b.get("role") for b in page_blocks} + + if "reference_heading" in roles or "backmatter_heading" in roles: + return page + + dense_refs = sum(1 for b in page_blocks if b.get("role") == "reference_item") + if dense_refs >= 4: + return page + + return None + + +def _reconcile_tail_spread(blocks: list[dict]) -> tuple[int, int] | None: + """Reconcile forward and backward scans into a tail spread page range. + + Returns (start_page, end_page) or None when no tail spread exists. + """ + forward_end = _detect_forward_body_end(blocks) + backward_start = _detect_backward_backmatter_start(blocks) + + if forward_end is None and backward_start is None: + return None + + max_page = 0 + for block in blocks: + p = block.get("page") + if p is not None and p > max_page: + max_page = p + + if forward_end is None and backward_start is not None: + start = max(1, backward_start - 2) + return (start, max_page) + + if backward_start is None and forward_end is not None: + return None + + if forward_end is not None and backward_start is not None: + if forward_end < backward_start: + return (forward_end + 1, backward_start) + else: + return (backward_start, forward_end) + + return None + + +def _assign_tail_spread_ownership( + blocks: list[dict], + tail_spread: tuple[int, int] | None = None, +) -> list[dict]: + """Assign tail_candidate_body blocks to backmatter anchors across pages. + + When a tail_spread boundary is provided, only tail_candidate_body blocks + within the spread are eligible for anchor matching. Blocks outside the + spread revert to body_paragraph. Inside the spread, tail_candidate_body + is replaced with backmatter_body when the block sits below a valid + backmatter heading (same-column preferred). Marks spread-assigned bodies + with _spread_anchor so the page-local reorder pass does not reassign them. + Unanchored candidates inside the spread stay as tail_candidate_body. + """ + anchors = [b for b in blocks if b.get("role") == "backmatter_heading"] + ref_heading = next((b for b in blocks if b.get("role") == "reference_heading"), None) + + if tail_spread is not None: + spread_start, spread_end = tail_spread + else: + spread_start, spread_end = 0, 0 + + if not anchors: + return [{**b, "role": "body_paragraph"} if b.get("role") == "tail_candidate_body" else b for b in blocks] + + result = list(blocks) + for i, block in enumerate(result): + if block.get("role") != "tail_candidate_body": + continue + + block_page = block.get("page", 0) or 0 + + # Outside the reconciled tail spread → revert to body_paragraph + if tail_spread is not None and (block_page < spread_start or block_page > spread_end): + result[i] = dict(block) + result[i]["role"] = "body_paragraph" + continue + + # Inside tail spread (or no spread set): try geometric anchor matching + pw = block.get("page_width", 0) or 1200 + idx = _find_best_anchor(block, anchors, ref_heading, pw) + result[i] = dict(block) + if idx is not None: + anchor_page = anchors[idx].get("page", 0) + result[i]["role"] = "backmatter_body" + result[i]["_spread_anchor"] = anchor_page + else: + # If no anchor match inside the spread, revert to plain body. + result[i]["role"] = "body_paragraph" + return result + + +def _sort_blocks_by_column(blocks: list[dict], page_width: int) -> list[dict]: + """Sort blocks on a page into natural reading order (left col then right + col, top to bottom within each column), using bbox x/y positions. + Blocks without bbox data are left in their original relative order. + """ + midpoint = page_width / 2 + + def _column_key(block: dict) -> tuple[int, int, int]: + bbox = block.get("bbox") or block.get("block_bbox") + if bbox and len(bbox) >= 4: + x_center = (bbox[0] + bbox[2]) / 2 + col = 0 if x_center < midpoint else 1 + return (col, bbox[1], bbox[0]) + return (0, 0, 0) + + return sorted(blocks, key=_column_key) + + +def _order_tail_blocks(blocks: list[dict], style_profiles: dict | None = None) -> list[dict]: + """Fix block reading order on tail pages with mixed-column layout. + + Two-column tail pages can have blocks in non-reading order (e.g. + left-column References at y=705 placed between right-column Gen AI + heading at y=154 and Pub note heading at y=297). This sorts all + blocks on such pages by (column, y-position), then groups them into + backmatter sections and a references zone using geometric ownership. + + Before per-page sorting, runs a multi-page tail-spread ownership pass + that resolves tail_candidate_body blocks against backmatter anchors + across page boundaries. Non-tail pages are untouched. + """ + if not blocks: + return blocks + + # Step 0: Reconcile tail spread boundary + tail_spread = _reconcile_tail_spread(blocks) + + # Step 0.5: only inside the reconciled tail spread, promote plausible + # body paragraphs into tail candidates using geometry rather than + # page-level text heuristics. + header_band, footer_band = _estimate_noise_bands(blocks) + blocks = _promote_tail_body_candidates(blocks, tail_spread, header_band=header_band, footer_band=footer_band) + + # Step 1: cross-page tail spread ownership (now boundary-aware) + blocks = _assign_tail_spread_ownership(blocks, tail_spread) + + # Find pages that contain tail blocks and their page widths + tail_pages: set[int] = set() + page_widths: dict[int, int] = {} + for block in blocks: + if _has_tail_role(block): + page = block.get("page") + if page is not None: + tail_pages.add(page) + pw = block.get("page_width") or 0 + p = block.get("page") + if p is not None and pw: + page_widths.setdefault(p, pw) + + if not tail_pages: + return blocks + + # Group blocks by page + by_page: dict[int, list[dict]] = {} + for block in blocks: + p = block.get("page") + if p is not None: + by_page.setdefault(p, []).append(block) + + # For tail pages: column-sort then group into sections using + # geometric ownership that handles cross-column body attachment + # and references zone boundaries. + carried_ref: dict | None = None + carried_backmatter: dict | None = None + result: list[dict] = [] + for page in sorted(by_page.keys()): + page_blocks = by_page[page] + if page in tail_pages: + pw = page_widths.get(page, 1200) + sorted_blocks = _sort_blocks_by_column(page_blocks, pw) + ordered, carried_ref, carried_backmatter = _reorder_tail_run( + sorted_blocks, + carried_ref, + carried_backmatter, + header_band=header_band, + footer_band=footer_band, + page_width=pw, + ) + result.extend(ordered) + else: + result.extend(page_blocks) + + return result + + def render_fulltext_markdown( *, structured_blocks: list[dict], resolved_metadata: dict, figure_inventory: dict, table_inventory: dict, + page_count: int | None = None, ) -> str: lines: list[str] = [] @@ -42,9 +881,13 @@ def render_fulltext_markdown( lines.append("") # --- authors --- - authors = resolved_metadata.get("authors", {}).get("value", []) - if authors: - lines.append(f"**Authors:** {', '.join(authors)}") + authors_display = resolved_metadata.get("authors_display", "") + if not authors_display: + authors = resolved_metadata.get("authors", {}).get("value", []) + if authors: + authors_display = ", ".join(authors) + if authors_display: + lines.append(f"**Authors:** {authors_display}") lines.append("") # --- metadata block --- @@ -100,14 +943,39 @@ def render_fulltext_markdown( max_page = max(all_pages) if all_pages else 0 current_page: int | None = None - for block in structured_blocks: + CONSUMED_FRONTMATTER_ROLES = frozenset( + { + "paper_title", + "authors", + "doi", + "affiliation", + "email", + "correspondence", + } + ) + + style_profiles = _build_heading_style_profiles(structured_blocks) + ordered_blocks = _order_tail_blocks(structured_blocks, style_profiles=style_profiles) + + for block in ordered_blocks: if not block.get("render_default", True): continue role = block.get("role", "") - if role in ("abstract_heading", "abstract_body", "figure_caption", "table_caption", "frontmatter_noise", "table_html"): + if role in CONSUMED_FRONTMATTER_ROLES and block.get("page") == 1: + continue + _SKIPPED_BODY_ROLES = { + "abstract_heading", + "abstract_body", + "figure_caption", + "table_caption", + "frontmatter_noise", + "table_html", + } + if role in _SKIPPED_BODY_ROLES: continue text = _normalize_latex(block.get("text", "")) + text = re.sub(r"]*>.*?
", "", text, flags=re.DOTALL | re.IGNORECASE) if text.strip().lower().startswith(" (current_page or 0): + lines.append(f"") + lines.append("") for fig_id in figures_by_page.get(p, []): lines.append(f"![[render/figures/{fig_id}.md]]") lines.append("") - had_objects = True for tbl_id in tables_by_page.get(p, []): lines.append(f"![[render/tables/{tbl_id}.md]]") lines.append("") - had_objects = True - if not had_objects and p > (current_page or 0): - lines.append(f"") - lines.append("") return "\n".join(lines).strip() + "\n" diff --git a/paperforge/worker/ocr_roles.py b/paperforge/worker/ocr_roles.py index a38663d4..0d29f6ba 100644 --- a/paperforge/worker/ocr_roles.py +++ b/paperforge/worker/ocr_roles.py @@ -26,6 +26,43 @@ _TABLE_PREFIX_PATTERN = re.compile( flags=re.IGNORECASE, ) +_BACKMATTER_TITLE_DENY_LIST = { + "generative ai statement", + "acknowledgments", + "acknowledgements", + "funding", + "conflict of interest", + "competing interests", + "data availability", + "supplementary materials", + "supplementary material", + "author contributions", + "declaration of competing interest", + "credit authorship contribution statement", + "ethical statement", + "ethics statement", + "institutional review board", +} + +_BACKMATTER_HEADINGS = { + "author contributions", + "funding", + "acknowledgments", + "acknowledgements", + "conflict of interest", + "competing interests", + "data availability", + "supplementary materials", + "supplementary material", + "generative ai statement", + "declaration of competing interest", + "ethical statement", + "ethics statement", + "institutional review board", + "credit authorship contribution statement", + "publisher's note", +} + FRONTMATTER_NOISE = { "open access", "copyright", @@ -38,20 +75,8 @@ FRONTMATTER_NOISE = { "accepted", "published", "present address", - "supplementary material", "these authors have contributed equally", - "publisher's note", - "competing interests", - "conflict of interest", - "data availability", - "acknowledgments", - "acknowledgements", - "author contributions", - "supplementary materials", - "funding", - "ethical statement", "informed consent", - "institutional review board", "orcid", } @@ -67,11 +92,6 @@ _CITATION_LINE_PATTERN = re.compile( # author list with superscript affiliation markers like "$^{1,2\dagger}$" _AUTHOR_AFFILIATION_MARKER = re.compile(r"\$\s*\^\{") -# keyword list: comma-separated short phrases, no sentence structure -_KEYWORD_BLOCK_PATTERN = re.compile( - r"^[a-z][a-z\s\-]+\([A-Z]+\)(?:,\s*[a-z][a-z\s\-]+(?:\([A-Z]+\))?)*$", -) - def _has_heading_numbering(text: str) -> bool: return bool(_HEADING_NUMBER_PATTERN.match(text.strip())) @@ -94,12 +114,26 @@ def assign_block_role( page_blocks: list[dict], page_width: int = 0, page_height: int = 0, + style_profiles: dict | None = None, ) -> RoleAssignment: raw_label = str(block.get("block_label", "") or "").strip() text = str(block.get("block_content", "") or "").strip() # Figure / table caption patterns override any prior if _has_figure_prefix(text): + if raw_label == "text": + verb_patterns = ["shows", "illustrates", "depicts", "demonstrates", "presents", "summarizes"] + has_verb = any(v in text.lower() for v in verb_patterns) + sentence_markers = [" is ", " are ", " was ", " were "] + has_sentence = any(m in text.lower() for m in sentence_markers) + is_long = len(text) > 80 + + if (has_verb and has_sentence) or is_long: + return RoleAssignment( + role="body_paragraph", + confidence=0.6, + evidence=[f"body reference to figure, not caption: {text[:60]}"], + ) return RoleAssignment( role="figure_caption", confidence=0.9, @@ -135,18 +169,49 @@ def assign_block_role( confidence=0.9, evidence=[f"references heading: {text[:60]}"], ) + # Backmatter heading detection (tail-zone + text evidence) + # Known backmatter phrases on tail pages (page > 1) are unambiguous - + # full-width headings are common in real papers, so geometric checks + # are not used here. Page-1 blocks with these phrases are frontmatter + # noise (already caught above) or genuine backmatter that fell through. + if lower in _BACKMATTER_HEADINGS: + page_num = block.get("page", 1) or 1 + if page_num > 1: + return RoleAssignment( + role="backmatter_heading", + confidence=0.8, + evidence=[f"backmatter heading on page {page_num}: {text[:60]}"], + ) + return RoleAssignment( + role="section_heading", + confidence=0.5, + evidence=[f"backmatter heading text on page 1, treated as section: {text[:60]}"], + ) if _has_heading_numbering(text): return RoleAssignment( role="section_heading" if re.match(r"^\d+\s", text) else "subsection_heading", confidence=0.85, evidence=[f"paragraph_title label with numbering: {text[:60]}"], ) + bbox = block.get("block_bbox", [0, 0, 0, 0]) page_num = block.get("page", 1) or 1 if page_num <= 1: + if bbox[1] < max(page_height, 1) * 0.25: + if lower in _BACKMATTER_TITLE_DENY_LIST: + return RoleAssignment( + role="section_heading", + confidence=0.5, + evidence=[f"backmatter title in title zone, treated as heading: {text[:60]}"], + ) + return RoleAssignment( + role="paper_title", + confidence=0.7, + evidence=[f"unnumbered paragraph_title in title zone on page 1: {text[:60]}"], + ) return RoleAssignment( - role="paper_title", - confidence=0.7, - evidence=[f"unnumbered paragraph_title on page 1, likely paper title: {text[:60]}"], + role="section_heading", + confidence=0.5, + evidence=[f"unnumbered paragraph_title on page 1 outside title zone: {text[:60]}"], ) return RoleAssignment( role="section_heading", @@ -221,7 +286,7 @@ def assign_block_role( return RoleAssignment( role="table_html", confidence=0.95, - evidence=[f"inline table HTML"], + evidence=["inline table HTML"], ) # Check for abstract heading (may appear as text block, not paragraph_title) @@ -248,7 +313,11 @@ def assign_block_role( evidence=[f"contact/email: {text[:60]}"], ) - # Check for frontmatter noise phrases (broader than exact startswith) + # Check for frontmatter noise phrases — demoted to weak fallback. + # Backmatter-referring phrases (supplementary material, publisher's note) + # are intentionally removed: they are valid backmatter headings and + # should never suppress body text. Tail ownership is resolved by + # the renderer's multi-page tail spread, not by the role layer. noise_phrases = [ "citation:", "to cite this article", @@ -256,8 +325,6 @@ def assign_block_role( "orcid", "these authors have contributed", "equal contribution", - "supplementary material", - "publisher's note", ] if any(phrase in lower_txt for phrase in noise_phrases): return RoleAssignment( @@ -278,11 +345,16 @@ def assign_block_role( # Distinguish from citation lines (which have year in parens) and # affiliation blocks (which have institutional keywords) has_year_parens = bool(re.search(r"\(\d{4}[a-z]?\)", text)) - has_inst_keyword = any(kw in lower_txt for kw in - ["department", "university", "institute", "college", "school of"]) - if (_AUTHOR_AFFILIATION_MARKER.search(text) and "," in text - and not has_year_parens and not has_inst_keyword - and len(text) < 500): + has_inst_keyword = any( + kw in lower_txt for kw in ["department", "university", "institute", "college", "school of"] + ) + if ( + _AUTHOR_AFFILIATION_MARKER.search(text) + and "," in text + and not has_year_parens + and not has_inst_keyword + and len(text) < 500 + ): return RoleAssignment( role="authors", confidence=0.8, @@ -319,16 +391,62 @@ def assign_block_role( confidence=0.7, evidence=[f"frontmatter noise text: {text[:60]}"], ) - if ( - _has_heading_numbering(text) - and len(text) < 80 - and ". " not in text - ): + if _has_heading_numbering(text) and len(text) < 80 and ". " not in text: + bbox = block.get("block_bbox", [0, 0, 0, 0]) + x1, y1, x2 = bbox[0], bbox[1], bbox[2] + block_width = x2 - x1 + in_top_80 = y1 < max(page_height, 1) * 0.8 + wide_enough = block_width > max(page_width, 1) * 0.3 + sentence_verbs = [" is ", " are ", " was ", " were ", " have ", " has ", " been "] + no_sentence_verbs = not (len(text) > 50 and any(v in text.lower() for v in sentence_verbs)) + if in_top_80 and wide_enough and no_sentence_verbs: + return RoleAssignment( + role="section_heading" if re.match(r"^\d+\s", text) else "subsection_heading", + confidence=0.65, + evidence=[f"numbered text block: {text[:60]}"], + ) + # References heading from text block + if lower_txt in ("references", "bibliography") and len(text) < 30: return RoleAssignment( - role="section_heading" if re.match(r"^\d+\s", text) else "subsection_heading", - confidence=0.65, - evidence=[f"numbered text block: {text[:60]}"], + role="reference_heading", + confidence=0.8, + evidence=[f"references heading from text block: {text[:40]}"], ) + + # Style-aware heading disambiguation (Task 5) + if style_profiles is not None and style_profiles: + from paperforge.worker.ocr_render import _disambiguate_heading_role + + style_suggested = _disambiguate_heading_role(block, style_profiles) + if style_suggested is not None: + bbox = block.get("block_bbox", [0, 0, 0, 0]) + in_top_80 = page_height == 0 or (len(bbox) >= 4 and bbox[1] < page_height * 0.8) + if in_top_80 and len(text) >= 5 and len(text) < 60 and ". " not in text: + return RoleAssignment( + role=style_suggested, + confidence=0.65, + evidence=[f"style-aware heading detection: role={style_suggested}, text={text[:40]}"], + ) + + # Visual heading detection: large or bold text blocks (fallback thresholds) + span_meta = block.get("span_metadata", {}) or {} + if isinstance(span_meta, dict): + font_size = span_meta.get("size", 0) or 0 + font_flags = (span_meta.get("flags", "") or "").lower() + else: + font_size = 0 + font_flags = "" + is_visually_prominent = (font_size >= 12 and "bold" in font_flags) or font_size >= 14 + if is_visually_prominent and len(text) >= 5 and len(text) < 60 and ". " not in text: + bbox = block.get("block_bbox", [0, 0, 0, 0]) + in_top_80 = page_height == 0 or (len(bbox) >= 4 and bbox[1] < page_height * 0.8) + if in_top_80: + return RoleAssignment( + role="section_heading", + confidence=0.65, + evidence=[f"heading-style text block: size={font_size}, flags={font_flags}, text={text[:40]}"], + ) + if len(text) < 20: return RoleAssignment( role="unknown_structural", diff --git a/paperforge/worker/ocr_tables.py b/paperforge/worker/ocr_tables.py index 3f44d05c..c394ba27 100644 --- a/paperforge/worker/ocr_tables.py +++ b/paperforge/worker/ocr_tables.py @@ -6,12 +6,13 @@ from typing import Any from paperforge.core.io import write_json - _TABLE_PREFIX_PATTERN = re.compile( r"^(?:Table|Supplementary\s+Table|Extended\s+Data\s+Table)\s+(\d+(?:\.\d+)?)", flags=re.IGNORECASE, ) +_CONTINUATION_PATTERN = re.compile(r"\(cont(?:inued)?\.?\)", re.IGNORECASE) + def _extract_table_number(text: str) -> int | None: m = _TABLE_PREFIX_PATTERN.search(text) @@ -23,6 +24,41 @@ def _extract_table_number(text: str) -> int | None: return None +def _is_continuation_caption(text: str) -> bool: + return bool(_CONTINUATION_PATTERN.search(text)) + + +def _extract_base_table_number(text: str) -> int | None: + cleaned = _CONTINUATION_PATTERN.sub("", text).strip() + return _extract_table_number(cleaned) + + +def _compute_asset_score( + asset: dict, + caption_bottom: float, +) -> float: + asset_bbox = asset.get("bbox", [0, 0, 0, 0]) + asset_top = asset_bbox[1] if len(asset_bbox) > 1 else 0 + distance = caption_bottom - asset_top + if distance > 0: + return distance + return abs(distance) + 100000.0 + + +def _pick_best_asset( + page_assets: list[tuple[int, dict]], + caption_bottom: float, +) -> tuple[int, dict] | None: + best = None + best_score = float("inf") + for i, asset in page_assets: + score = _compute_asset_score(asset, caption_bottom) + if score < best_score: + best_score = score + best = (i, asset) + return best + + def build_table_inventory(structured_blocks: list[dict]) -> dict[str, Any]: tables: list[dict] = [] captions: list[dict] = [] @@ -45,39 +81,70 @@ def build_table_inventory(structured_blocks: list[dict]) -> dict[str, Any]: caption_page = caption.get("page", 0) caption_text = caption.get("text", "") table_num = _extract_table_number(caption_text) + formal_table_number = _extract_base_table_number(caption_text) + is_cont = _is_continuation_caption(caption_text) + + candidate_pages = [caption_page] if is_cont else [caption_page, caption_page + 1] + + caption_bbox = caption.get("bbox", [0, 0, 0, 0]) + caption_bottom = caption_bbox[3] if len(caption_bbox) > 3 else 0 - candidate_pages = [caption_page, caption_page + 1, caption_page - 1] matched_asset = None - for page in candidate_pages: if page < 1: continue - for i, asset in enumerate(assets): - if asset.get("page", 0) == page and i not in used_asset_indices: - matched_asset = asset - used_asset_indices.add(i) - break - if matched_asset: + page_assets = [ + (i, asset) + for i, asset in enumerate(assets) + if i not in used_asset_indices and asset.get("page", 0) == page + ] + if not page_assets: + continue + best = _pick_best_asset(page_assets, caption_bottom) + if best is not None: + best_idx, best_asset = best + matched_asset = best_asset + used_asset_indices.add(best_idx) break + continuation_of = None + if is_cont and formal_table_number is not None: + for t in tables: + tt = t.get("formal_table_number") + if tt == formal_table_number and not t.get("is_continuation"): + continuation_of = formal_table_number + break + + segments: list[dict] = [] + if matched_asset: + segments.append({ + "page": matched_asset.get("page", 0), + "asset_block_id": matched_asset.get("block_id", ""), + "asset_bbox": matched_asset.get("bbox", [0, 0, 0, 0]), + "is_continuation": is_cont, + }) + tables.append({ "caption_block_id": caption.get("block_id", ""), "page": caption_page, "caption_text": caption_text, "table_number": table_num, + "formal_table_number": formal_table_number, "asset_block_id": matched_asset.get("block_id", "") if matched_asset else "", "asset_bbox": matched_asset.get("bbox", [0, 0, 0, 0]) if matched_asset else [], "assistive_text": (matched_asset.get("text", "") or "") if matched_asset else "", "truth_source": "image", "has_asset": matched_asset is not None, + "segments": segments, + "is_continuation": is_cont, + "continuation_of": continuation_of, }) + cap_block_ids_with_asset = { + t["caption_block_id"] for t in tables if t["has_asset"] + } for caption in captions: - if not any( - t["caption_block_id"] == caption.get("block_id", "") - for t in tables - if t["has_asset"] - ): + if caption.get("block_id", "") not in cap_block_ids_with_asset: unmatched_captions.append(caption) for i, asset in enumerate(assets): @@ -88,7 +155,9 @@ def build_table_inventory(structured_blocks: list[dict]) -> dict[str, Any]: "tables": tables, "unmatched_captions": unmatched_captions, "unmatched_assets": unmatched_assets, - "official_table_count": len([t for t in tables if t["has_asset"]]), + "official_table_count": len( + [t for t in tables if t["has_asset"] and not t["is_continuation"]] + ), } diff --git a/tests/test_ocr_figures.py b/tests/test_ocr_figures.py index f26990a0..417b6ed7 100644 --- a/tests/test_ocr_figures.py +++ b/tests/test_ocr_figures.py @@ -220,3 +220,74 @@ def test_unmatched_legends_populated() -> None: assert len(inventory["unmatched_legends"]) == 1 assert inventory["unmatched_legends"][0]["block_id"] == "p3_b1" + + +def test_body_mention_not_mistaken_for_formal_legend() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 1, + "block_id": "p1_b1", + "role": "figure_caption", + "block_label": "text", + "text": "Figure 3 shows quantitative results of cell migration under applied DC electric field.", + "bbox": [50, 100, 550, 130], + }, + { + "paper_id": "K001", + "page": 2, + "block_id": "p2_b1", + "role": "figure_caption", + "block_label": "figure_title", + "text": "Figure 3. Quantitative analysis of cell migration under DC electric field stimulation.", + "bbox": [50, 700, 550, 740], + }, + { + "paper_id": "K001", + "page": 2, + "block_id": "p2_b2", + "role": "figure_asset", + "text": "", + "bbox": [50, 50, 550, 680], + }, + ] + + inventory = build_figure_inventory(structured_blocks) + + assert len(inventory["matched_figures"]) == 1 + assert inventory["matched_figures"][0]["legend_block_id"] == "p2_b1" + assert len(inventory["matched_figures"][0]["matched_assets"]) == 1 + assert inventory["matched_figures"][0]["matched_assets"][0]["block_id"] == "p2_b2" + + +def test_legend_does_not_steal_offpage_asset() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + + structured_blocks = [ + { + "paper_id": "K001", + "page": 1, + "block_id": "p1_b1", + "role": "figure_caption", + "text": "Figure 1. A caption with no asset on the same page.", + "bbox": [50, 700, 550, 740], + }, + { + "paper_id": "K001", + "page": 2, + "block_id": "p2_b1", + "role": "figure_asset", + "text": "", + "bbox": [50, 50, 550, 400], + }, + ] + + inventory = build_figure_inventory(structured_blocks) + + assert len(inventory["matched_figures"]) == 1 + assert len(inventory["matched_figures"][0]["matched_assets"]) == 0 + assert "legend_only" in inventory["matched_figures"][0]["flags"] + assert len(inventory["unmatched_assets"]) == 1 + assert inventory["unmatched_assets"][0]["block_id"] == "p2_b1" diff --git a/tests/test_ocr_health.py b/tests/test_ocr_health.py index a606a10c..da2f6555 100644 --- a/tests/test_ocr_health.py +++ b/tests/test_ocr_health.py @@ -31,3 +31,31 @@ def test_health_report_is_independent_from_ocr_status() -> None: assert report["page_count"] == 3 assert report["figure_caption_count"] == 1 assert report["overall"] in {"yellow", "red"} + + +def test_health_report_distinguishes_formal_tables_from_segments() -> None: + from paperforge.worker.ocr_health import build_ocr_health + + table_inventory = { + "tables": [ + {"table_id": "tbl_001", "has_asset": True}, + {"table_id": "tbl_002", "has_asset": True}, + ], + "unmatched_captions": [], + "unmatched_assets": [ + {"asset_id": "seg_001"}, + {"asset_id": "seg_002"}, + {"asset_id": "seg_003"}, + ], + } + + report = build_ocr_health( + page_count=5, + raw_blocks_count=50, + structured_blocks=[], + figure_inventory={}, + table_inventory=table_inventory, + ) + + assert report.get("formal_table_count", 0) == 2 + assert report.get("table_segment_count", 0) == 5 diff --git a/tests/test_ocr_objects.py b/tests/test_ocr_objects.py index 887e2f89..b0b3261a 100644 --- a/tests/test_ocr_objects.py +++ b/tests/test_ocr_objects.py @@ -1,5 +1,7 @@ from __future__ import annotations +from pathlib import Path + def test_figure_object_markdown_links_image_and_legend() -> None: @@ -63,3 +65,56 @@ def test_stabilize_object_wikilink_uses_correct_relative_path() -> None: }) assert "![](../../assets/figures/figure_001.jpg)" in md + + +def test_crop_asset_uses_ocr_page_coordinates_when_dimensions_provided(tmp_path: Path) -> None: + import fitz + from PIL import Image + + from paperforge.worker.ocr_objects import _crop_asset_from_pdf + + pdf_path = tmp_path / "sample.pdf" + doc = fitz.open() + page = doc.new_page(width=300, height=400) + page.draw_rect(fitz.Rect(25, 25, 50, 50), color=(1, 0, 0), fill=(1, 0, 0)) + doc.save(pdf_path) + doc.close() + + dst = tmp_path / "crop.jpg" + ok = _crop_asset_from_pdf( + pdf_path, + 1, + [50, 50, 100, 100], + dst, + page_width=600, + page_height=800, + page_cache_dir=tmp_path / "pages", + ) + + assert ok is True + with Image.open(dst) as img: + assert img.size == (50, 50) + + +def test_crop_asset_prefers_cached_page_image_when_available(tmp_path: Path) -> None: + from PIL import Image + + from paperforge.worker.ocr_objects import _crop_asset_from_pdf + + page_cache_dir = tmp_path / "pages" + page_cache_dir.mkdir() + page_image = page_cache_dir / "page_001.jpg" + Image.new("RGB", (600, 800), "white").save(page_image) + + dst = tmp_path / "crop.jpg" + ok = _crop_asset_from_pdf( + tmp_path / "missing.pdf", + 1, + [50, 50, 100, 100], + dst, + page_cache_dir=page_cache_dir, + ) + + assert ok is True + with Image.open(dst) as img: + assert img.size == (50, 50) diff --git a/tests/test_ocr_render_stabilization.py b/tests/test_ocr_render_stabilization.py index 640371fd..810fdcac 100644 --- a/tests/test_ocr_render_stabilization.py +++ b/tests/test_ocr_render_stabilization.py @@ -5,9 +5,30 @@ def test_stabilize_render_suppresses_frontmatter_noise() -> None: from paperforge.worker.ocr_render import render_fulltext_markdown structured_blocks = [ - {"paper_id": "KEY001", "page": 1, "block_id": "b1", "role": "section_heading", "text": "OPEN ACCESS", "render_default": True}, - {"paper_id": "KEY001", "page": 1, "block_id": "b2", "role": "section_heading", "text": "CITATION", "render_default": True}, - {"paper_id": "KEY001", "page": 1, "block_id": "b3", "role": "body_paragraph", "text": "Real body text should render.", "render_default": True}, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "section_heading", + "text": "OPEN ACCESS", + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b2", + "role": "section_heading", + "text": "CITATION", + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b3", + "role": "body_paragraph", + "text": "Real body text should render.", + "render_default": True, + }, ] md = render_fulltext_markdown( @@ -33,9 +54,30 @@ def test_stabilize_render_output_starts_with_metadata_and_abstract() -> None: "doi": {"value": "10.1000/xyz"}, } structured_blocks = [ - {"paper_id": "KEY001", "page": 1, "block_id": "b1", "role": "abstract_heading", "text": "Abstract", "render_default": True}, - {"paper_id": "KEY001", "page": 1, "block_id": "b2", "role": "abstract_body", "text": "This is the abstract text.", "render_default": True}, - {"paper_id": "KEY001", "page": 2, "block_id": "b3", "role": "section_heading", "text": "1 Introduction", "render_default": True}, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "abstract_heading", + "text": "Abstract", + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b2", + "role": "abstract_body", + "text": "This is the abstract text.", + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 2, + "block_id": "b3", + "role": "section_heading", + "text": "1 Introduction", + "render_default": True, + }, ] md = render_fulltext_markdown( @@ -57,9 +99,30 @@ def test_stabilize_figure_not_appended_at_end() -> None: from paperforge.worker.ocr_render import render_fulltext_markdown structured_blocks = [ - {"paper_id": "KEY001", "page": 1, "block_id": "b1", "role": "body_paragraph", "text": "Body text.", "render_default": True}, - {"paper_id": "KEY001", "page": 2, "block_id": "b2", "role": "figure_caption", "text": "Figure 1. Results.", "render_default": True}, - {"paper_id": "KEY001", "page": 3, "block_id": "b3", "role": "section_heading", "text": "2 Discussion", "render_default": True}, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "body_paragraph", + "text": "Body text.", + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 2, + "block_id": "b2", + "role": "figure_caption", + "text": "Figure 1. Results.", + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 3, + "block_id": "b3", + "role": "section_heading", + "text": "2 Discussion", + "render_default": True, + }, ] figure_inventory = { "matched_figures": [{"figure_id": "fig_001", "page": 2}], @@ -81,8 +144,14 @@ def test_stabilize_latex_normalization() -> None: from paperforge.worker.ocr_render import render_fulltext_markdown structured_blocks = [ - {"paper_id": "KEY001", "page": 1, "block_id": "b1", "role": "body_paragraph", - "text": "Expression $ ^{1} $ and $ ^{\\u2020} $ should be compact.", "render_default": True}, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "body_paragraph", + "text": "Expression $ ^{1} $ and $ ^{\\u2020} $ should be compact.", + "render_default": True, + }, ] md = render_fulltext_markdown( @@ -101,10 +170,22 @@ def test_stabilize_no_inline_table_html() -> None: from paperforge.worker.ocr_render import render_fulltext_markdown structured_blocks = [ - {"paper_id": "KEY001", "page": 1, "block_id": "b1", "role": "table_html", - "text": "
data
", "render_default": True}, - {"paper_id": "KEY001", "page": 1, "block_id": "b2", "role": "body_paragraph", - "text": "Real body text.", "render_default": True}, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "table_html", + "text": "
data
", + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b2", + "role": "body_paragraph", + "text": "Real body text.", + "render_default": True, + }, ] md = render_fulltext_markdown( @@ -123,8 +204,10 @@ def test_stabilize_author_recovery_from_ocr() -> None: import json, tempfile, pathlib blocks_data = ( - json.dumps({"role": "paper_title", "text": "Test Paper"}) + "\n" - + json.dumps({"role": "authors", "text": "Alice Smith, Bob Jones"}) + "\n" + json.dumps({"role": "paper_title", "text": "Test Paper"}) + + "\n" + + json.dumps({"role": "authors", "text": "Alice Smith, Bob Jones"}) + + "\n" ) with tempfile.NamedTemporaryFile(mode="w", suffix=".jsonl", delete=False, encoding="utf-8") as f: f.write(blocks_data) @@ -146,10 +229,22 @@ def test_stabilize_heading_sanity_downgrades_long_heading() -> None: # A very long section_heading (>100 chars) should be downgraded to body text long_text = "This is a very long heading that exceeds one hundred characters and should definitely be downgraded to a body paragraph instead of being rendered as a heading" structured_blocks = [ - {"paper_id": "KEY001", "page": 1, "block_id": "b1", "role": "section_heading", - "text": long_text, "render_default": True}, - {"paper_id": "KEY001", "page": 1, "block_id": "b2", "role": "body_paragraph", - "text": "Normal body text.", "render_default": True}, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "section_heading", + "text": long_text, + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b2", + "role": "body_paragraph", + "text": "Normal body text.", + "render_default": True, + }, ] md = render_fulltext_markdown( @@ -170,11 +265,22 @@ def test_stabilize_heading_sanity_downgrades_multi_sentence_heading() -> None: # A heading with multiple sentence-ending periods should be downgraded structured_blocks = [ - {"paper_id": "KEY001", "page": 1, "block_id": "b1", "role": "section_heading", - "text": "This is a heading. It has multiple sentences. This is the third one.", - "render_default": True}, - {"paper_id": "KEY001", "page": 1, "block_id": "b2", "role": "body_paragraph", - "text": "Normal body text.", "render_default": True}, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "section_heading", + "text": "This is a heading. It has multiple sentences. This is the third one.", + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b2", + "role": "body_paragraph", + "text": "Normal body text.", + "render_default": True, + }, ] md = render_fulltext_markdown( @@ -192,8 +298,14 @@ def test_stabilize_heading_sanity_allows_valid_short_heading() -> None: from paperforge.worker.ocr_render import render_fulltext_markdown structured_blocks = [ - {"paper_id": "KEY001", "page": 1, "block_id": "b1", "role": "section_heading", - "text": "1 Introduction", "render_default": True}, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "section_heading", + "text": "1 Introduction", + "render_default": True, + }, ] md = render_fulltext_markdown( @@ -211,11 +323,22 @@ def test_stabilize_heading_sanity_downgrades_verb_heavy_heading() -> None: # A heading longer than 50 chars with common sentence verbs should be downgraded structured_blocks = [ - {"paper_id": "KEY001", "page": 1, "block_id": "b1", "role": "section_heading", - "text": "This is a method that was used for the experiment and has many words", - "render_default": True}, - {"paper_id": "KEY001", "page": 1, "block_id": "b2", "role": "body_paragraph", - "text": "Normal body text.", "render_default": True}, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "section_heading", + "text": "This is a method that was used for the experiment and has many words", + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b2", + "role": "body_paragraph", + "text": "Normal body text.", + "render_default": True, + }, ] md = render_fulltext_markdown( @@ -234,8 +357,14 @@ def test_stabilize_heading_sanity_allows_verb_short_heading() -> None: # A short heading with verbs is OK (under 50 chars) structured_blocks = [ - {"paper_id": "KEY001", "page": 1, "block_id": "b1", "role": "section_heading", - "text": "Results are shown", "render_default": True}, + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "section_heading", + "text": "Results are shown", + "render_default": True, + }, ] md = render_fulltext_markdown( @@ -258,3 +387,238 @@ def test_stabilize_reference_content_mapped() -> None: assert assignment.role == "reference_item" assert assignment.confidence >= 0.8 + + +def test_stabilize_page_marker_count_matches_page_count() -> None: + from paperforge.worker.ocr_render import render_fulltext_markdown + + structured_blocks = [ + { + "paper_id": "KEY001", + "page": 1, + "block_id": "b1", + "role": "body_paragraph", + "text": "Page 1 body.", + "render_default": True, + }, + { + "paper_id": "KEY001", + "page": 3, + "block_id": "b2", + "role": "body_paragraph", + "text": "Page 3 body.", + "render_default": True, + }, + ] + + md = render_fulltext_markdown( + structured_blocks=structured_blocks, + resolved_metadata={}, + figure_inventory={}, + table_inventory={}, + page_count=5, + ) + + page_markers = [line for line in md.split("\n") if line.strip().startswith("