From acd28d321ec7c0ed3acaac26123cc105f8adca59 Mon Sep 17 00:00:00 2001 From: Research Assistant Date: Mon, 8 Jun 2026 12:13:11 +0800 Subject: [PATCH] test: make OCR real-paper regressions reproducible --- docs/COMMANDS.md | 28 +++++ .../figure_table_tail_structured.jsonl | 14 +++ .../ocr_synthetic/two_column_structured.jsonl | 9 ++ tests/test_ocr_real_paper_regressions.py | 115 ++++++++++++------ tests/test_ocr_synthetic_fixtures.py | 45 +++++++ 5 files changed, 174 insertions(+), 37 deletions(-) create mode 100644 tests/fixtures/ocr_synthetic/figure_table_tail_structured.jsonl create mode 100644 tests/fixtures/ocr_synthetic/two_column_structured.jsonl create mode 100644 tests/test_ocr_synthetic_fixtures.py diff --git a/docs/COMMANDS.md b/docs/COMMANDS.md index d9760ee4..4f2e8d5e 100644 --- a/docs/COMMANDS.md +++ b/docs/COMMANDS.md @@ -102,3 +102,31 @@ paperforge runtime-health --json - [使用教程](getting-started.md) - [故障排除](troubleshooting.md) - [更新指南](update-upgrade.md) + +--- + +## 测试 + +### 合成 OCR 测试(CI 默认) + +```bash +python -m pytest tests/test_ocr_synthetic_fixtures.py -q --tb=short +``` + +### 真实文献 OCR 回归测试(可选集成测试) + +需要设置环境变量指定 vault 路径和要测试的文献 key: + +**Unix/macOS:** + +```bash +PAPERFORGE_REAL_OCR_VAULT="/path/to/Literature-hub" PAPERFORGE_REAL_OCR_KEYS="SAN9AYVR,2GN9LMCW" python -m pytest tests/test_ocr_real_paper_regressions.py -q +``` + +**PowerShell:** + +```powershell +$env:PAPERFORGE_REAL_OCR_VAULT="D:\path\to\Literature-hub"; $env:PAPERFORGE_REAL_OCR_KEYS="SAN9AYVR,2GN9LMCW"; python -m pytest tests/test_ocr_real_paper_regressions.py -q +``` + +不设置环境变量时,真实文献测试自动 SKIP。 diff --git a/tests/fixtures/ocr_synthetic/figure_table_tail_structured.jsonl b/tests/fixtures/ocr_synthetic/figure_table_tail_structured.jsonl new file mode 100644 index 00000000..584813de --- /dev/null +++ b/tests/fixtures/ocr_synthetic/figure_table_tail_structured.jsonl @@ -0,0 +1,14 @@ +{"block_id":"p1_b1","page":1,"role":"paper_title","text":"Synthetic OCR Paper","bbox":[90,60,720,100],"page_width":1200,"page_height":1600,"render_default":false} +{"block_id":"p1_b2","page":1,"role":"abstract_heading","text":"Abstract","bbox":[90,180,260,220],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p1_b3","page":1,"role":"abstract_body","text":"This fixture includes figures tables inserts and tail content.","bbox":[90,230,1000,290],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p2_img1","page":2,"role":"media_asset","raw_label":"image","text":"","bbox":[100,180,700,620],"page_width":1200,"page_height":1600,"render_default":false} +{"block_id":"p2_cap1","page":2,"role":"figure_caption","text":"Figure 1. Synthetic figure caption.","bbox":[100,640,900,690],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p2_table1","page":2,"role":"media_asset","raw_label":"table","text":"","bbox":[100,820,900,1120],"page_width":1200,"page_height":1600,"render_default":false} +{"block_id":"p2_tcap1","page":2,"role":"table_caption","text":"Table 1. Synthetic table caption.","bbox":[100,760,900,810],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p2_insert1","page":2,"role":"structured_insert","text":"Key points","bbox":[930,180,1130,220],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p2_insert2","page":2,"role":"structured_insert","text":"- Synthetic insert item","bbox":[930,230,1130,270],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p3_back1","page":3,"role":"backmatter_boundary_heading","text":"Additional Information and Declarations","bbox":[90,90,800,130],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p3_back2","page":3,"role":"backmatter_heading","text":"Funding","bbox":[90,150,260,190],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p3_back3","page":3,"role":"backmatter_body","text":"No external funding was received.","bbox":[90,200,900,250],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p3_refh","page":3,"role":"reference_heading","text":"References","bbox":[90,360,300,400],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p3_ref1","page":3,"role":"reference_item","text":"1. Author A. Synthetic reference. Journal. 2026.","bbox":[90,420,950,470],"page_width":1200,"page_height":1600,"render_default":true} diff --git a/tests/fixtures/ocr_synthetic/two_column_structured.jsonl b/tests/fixtures/ocr_synthetic/two_column_structured.jsonl new file mode 100644 index 00000000..2415f14a --- /dev/null +++ b/tests/fixtures/ocr_synthetic/two_column_structured.jsonl @@ -0,0 +1,9 @@ +{"block_id":"p1_b1","page":1,"role":"paper_title","text":"Synthetic Two Column OCR Paper","bbox":[90,60,720,100],"page_width":1200,"page_height":1600,"render_default":false} +{"block_id":"p1_b2","page":1,"role":"authors","text":"A. Author, B. Writer","bbox":[90,120,650,150],"page_width":1200,"page_height":1600,"render_default":false} +{"block_id":"p1_b3","page":1,"role":"abstract_heading","text":"Abstract","bbox":[90,210,240,240],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p1_b4","page":1,"role":"abstract_body","text":"This synthetic abstract verifies frontmatter rendering.","bbox":[90,250,980,310],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p2_b1","page":2,"role":"section_heading","text":"1. Introduction","bbox":[90,90,430,130],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p2_b2","page":2,"role":"body_paragraph","text":"Left column body paragraph one.","bbox":[90,150,530,210],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p2_b3","page":2,"role":"body_paragraph","text":"Right column body paragraph one.","bbox":[650,150,1090,210],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p3_b1","page":3,"role":"reference_heading","text":"References","bbox":[90,90,300,130],"page_width":1200,"page_height":1600,"render_default":true} +{"block_id":"p3_b2","page":3,"role":"reference_item","text":"1. Author A. Synthetic reference. Journal. 2026.","bbox":[90,150,950,210],"page_width":1200,"page_height":1600,"render_default":true} diff --git a/tests/test_ocr_real_paper_regressions.py b/tests/test_ocr_real_paper_regressions.py index 1d1ccb21..da2e2975 100644 --- a/tests/test_ocr_real_paper_regressions.py +++ b/tests/test_ocr_real_paper_regressions.py @@ -1,37 +1,67 @@ from __future__ import annotations import json +import os from pathlib import Path import pytest - -VAULT = Path(r"D:\L\OB\Literature-hub") -OCR_ROOT = VAULT / "System" / "PaperForge" / "ocr" +REAL_VAULT_ENV = "PAPERFORGE_REAL_OCR_VAULT" +REAL_KEYS_ENV = "PAPERFORGE_REAL_OCR_KEYS" PROBLEM_KEYS = ["TSCKAVIS", "CAQNW9Q2", "A8E7SRVS", "K7R8PEKW"] CONTROL_KEYS = ["SAN9AYVR", "2GN9LMCW", "7C8829BD"] -ALL_KEYS = PROBLEM_KEYS + CONTROL_KEYS -def _paper_root(key: str) -> Path: - return OCR_ROOT / key +def _real_ocr_vault() -> Path | None: + value = os.environ.get(REAL_VAULT_ENV) + return Path(value) if value else None -def _structured_path(key: str) -> Path: - return _paper_root(key) / "structure" / "blocks.structured.jsonl" +def _real_ocr_keys() -> list[str]: + value = os.environ.get(REAL_KEYS_ENV, "") + return [k.strip() for k in value.split(",") if k.strip()] -def _fulltext_path(key: str) -> Path: - return _paper_root(key) / "fulltext.md" +def _get_vault() -> Path: + vault = _real_ocr_vault() + if vault is None: + pytest.skip(f"Set {REAL_VAULT_ENV} and {REAL_KEYS_ENV} for real OCR integration tests") + return vault -def _health_path(key: str) -> Path: - return _paper_root(key) / "health" / "ocr_health.json" +def _get_ocr_root(vault: Path) -> Path: + return vault / "System" / "PaperForge" / "ocr" -def _metadata_path(key: str) -> Path: - return _paper_root(key) / "metadata" / "resolved_metadata.json" +def _get_keys() -> list[str]: + keys = _real_ocr_keys() + if not keys: + vault = _real_ocr_vault() + if vault is None: + pytest.skip(f"Set {REAL_VAULT_ENV} and {REAL_KEYS_ENV} for real OCR integration tests") + return PROBLEM_KEYS + CONTROL_KEYS + return keys + + +def _paper_root(ocr_root: Path, key: str) -> Path: + return ocr_root / key + + +def _structured_path(ocr_root: Path, key: str) -> Path: + return _paper_root(ocr_root, key) / "structure" / "blocks.structured.jsonl" + + +def _fulltext_path(ocr_root: Path, key: str) -> Path: + return _paper_root(ocr_root, key) / "fulltext.md" + + +def _health_path(ocr_root: Path, key: str) -> Path: + return _paper_root(ocr_root, key) / "health" / "ocr_health.json" + + +def _metadata_path(ocr_root: Path, key: str) -> Path: + return _paper_root(ocr_root, key) / "metadata" / "resolved_metadata.json" def _read_json(path: Path) -> dict: @@ -42,12 +72,10 @@ def _read_jsonl(path: Path) -> list[dict]: return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line.strip()] -def _require_artifacts(key: str) -> None: - if not VAULT.exists(): - pytest.skip(f"Vault path not available: {VAULT}") +def _require_artifacts(ocr_root: Path, key: str) -> None: missing = [ str(path) - for path in [_structured_path(key), _fulltext_path(key), _health_path(key), _metadata_path(key)] + for path in [_structured_path(ocr_root, key), _fulltext_path(ocr_root, key), _health_path(ocr_root, key), _metadata_path(ocr_root, key)] if not path.exists() ] if missing: @@ -55,19 +83,32 @@ def _require_artifacts(key: str) -> None: @pytest.fixture(scope="session") -def rebuilt_real_papers() -> dict: - if not VAULT.exists(): - pytest.skip(f"Vault path not available: {VAULT}") +def _vault() -> Path: + return _get_vault() + + +@pytest.fixture(scope="session") +def _ocr_root(_vault: Path) -> Path: + return _get_ocr_root(_vault) + + +@pytest.fixture(scope="session") +def _all_keys() -> list[str]: + return _get_keys() + + +@pytest.fixture(scope="session") +def rebuilt_real_papers(_vault: Path, _all_keys: list[str]) -> dict: from paperforge.worker.ocr_rebuild import run_derived_rebuild_for_keys - result = run_derived_rebuild_for_keys(VAULT, ALL_KEYS) + result = run_derived_rebuild_for_keys(_vault, _all_keys) assert result.get("rebuild_count", 0) >= 1, result return result -@pytest.mark.parametrize("key", ALL_KEYS) -def test_real_paper_artifacts_exist(key: str) -> None: - _require_artifacts(key) +@pytest.mark.parametrize("key", PROBLEM_KEYS + CONTROL_KEYS) +def test_real_paper_artifacts_exist(_ocr_root: Path, key: str) -> None: + _require_artifacts(_ocr_root, key) def test_real_paper_rebuild_runs(rebuilt_real_papers: dict) -> None: @@ -87,9 +128,9 @@ def _role_texts(blocks: list[dict], role: str) -> list[str]: @pytest.mark.parametrize("key", sorted(BODY_RETENTION)) -def test_problem_papers_retain_body_and_avoid_mass_insert_suppression(rebuilt_real_papers: dict, key: str) -> None: - _require_artifacts(key) - blocks = _read_jsonl(_structured_path(key)) +def test_problem_papers_retain_body_and_avoid_mass_insert_suppression(rebuilt_real_papers: dict, _ocr_root: Path, key: str) -> None: + _require_artifacts(_ocr_root, key) + blocks = _read_jsonl(_structured_path(_ocr_root, key)) roles = [block.get("role") for block in blocks] thresholds = BODY_RETENTION[key] @@ -97,11 +138,11 @@ def test_problem_papers_retain_body_and_avoid_mass_insert_suppression(rebuilt_re assert roles.count("non_body_insert") <= thresholds["max_non_body_insert"], roles.count("non_body_insert") -def test_tsckavis_frontmatter_and_key_points_are_callout(rebuilt_real_papers: dict) -> None: +def test_tsckavis_frontmatter_and_key_points_are_callout(rebuilt_real_papers: dict, _ocr_root: Path) -> None: key = "TSCKAVIS" - _require_artifacts(key) - blocks = _read_jsonl(_structured_path(key)) - fulltext = _fulltext_path(key).read_text(encoding="utf-8", errors="replace") + _require_artifacts(_ocr_root, key) + blocks = _read_jsonl(_structured_path(_ocr_root, key)) + fulltext = _fulltext_path(_ocr_root, key).read_text(encoding="utf-8", errors="replace") heading_text = "\n".join( _role_texts(blocks, "section_heading") + _role_texts(blocks, "subsection_heading") @@ -119,7 +160,7 @@ def test_tsckavis_frontmatter_and_key_points_are_callout(rebuilt_real_papers: di assert "skeletal stem and progenitor cells display a high metabolic flexibility" in fulltext.lower() # Metadata must capture title and authors from OCR blocks - meta = _read_json(_metadata_path(key)) + meta = _read_json(_metadata_path(_ocr_root, key)) assert len(meta.get("title", {}).get("value", "")) > 10, meta.get("title") assert len(meta.get("authors", {}).get("value", [])) > 0, meta.get("authors") assert meta.get("authors_display", "") != "", f"authors_display is empty: {meta}" @@ -136,11 +177,11 @@ CONTROL_MIN_BODY = { @pytest.mark.parametrize("key", sorted(CONTROL_MIN_BODY)) -def test_control_papers_keep_body_and_tail_stability(rebuilt_real_papers: dict, key: str) -> None: - _require_artifacts(key) - blocks = _read_jsonl(_structured_path(key)) +def test_control_papers_keep_body_and_tail_stability(rebuilt_real_papers: dict, _ocr_root: Path, key: str) -> None: + _require_artifacts(_ocr_root, key) + blocks = _read_jsonl(_structured_path(_ocr_root, key)) roles = [block.get("role") for block in blocks] - fulltext = _fulltext_path(key).read_text(encoding="utf-8", errors="replace") + fulltext = _fulltext_path(_ocr_root, key).read_text(encoding="utf-8", errors="replace") assert roles.count("body_paragraph") >= CONTROL_MIN_BODY[key] assert any(w in fulltext for w in ["References", "references", "REFERENCE", "Bibliography"]), "Tail reference marker not found" diff --git a/tests/test_ocr_synthetic_fixtures.py b/tests/test_ocr_synthetic_fixtures.py new file mode 100644 index 00000000..5cf1b252 --- /dev/null +++ b/tests/test_ocr_synthetic_fixtures.py @@ -0,0 +1,45 @@ +from __future__ import annotations + +import json +from pathlib import Path + + +FIXTURE_ROOT = Path("tests/fixtures/ocr_synthetic") + + +def _load_jsonl(path: Path) -> list[dict]: + return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line.strip()] + + +def test_synthetic_fixtures_load() -> None: + for path in FIXTURE_ROOT.glob("*.jsonl"): + blocks = _load_jsonl(path) + assert blocks + assert all("block_id" in b and "role" in b and "bbox" in b for b in blocks) + + +def test_synthetic_fixture_runs_core_downstream_modules() -> None: + from paperforge.worker.ocr_figures import build_figure_inventory + from paperforge.worker.ocr_tables import build_table_inventory + from paperforge.worker.ocr_render import render_fulltext_markdown + from paperforge.worker.ocr_health import build_ocr_health + + blocks = _load_jsonl(FIXTURE_ROOT / "figure_table_tail_structured.jsonl") + figure_inventory = build_figure_inventory(blocks) + table_inventory = build_table_inventory(blocks) + markdown = render_fulltext_markdown( + structured_blocks=blocks, + resolved_metadata={"title": {"value": "Synthetic OCR Paper"}, "authors": {"value": ["A. Author"]}}, + figure_inventory=figure_inventory, + table_inventory=table_inventory, + page_count=max(int(b.get("page", 1)) for b in blocks), + ) + health = build_ocr_health( + page_count=3, + raw_blocks_count=len(blocks), + structured_blocks=blocks, + figure_inventory=figure_inventory, + table_inventory=table_inventory, + ) + assert "Synthetic OCR Paper" in markdown + assert health["blocks_count"] == len(blocks)