mirror of
https://github.com/lllin000/PaperForge.git
synced 2026-07-22 06:50:53 +00:00
feat: add confidence distributions to OCR health report
This commit is contained in:
parent
c8b4846998
commit
dbab9c5ddf
2 changed files with 83 additions and 0 deletions
|
|
@ -106,6 +106,33 @@ def build_ocr_health(
|
|||
"tail_boundary_confidence": tail_score.get("score", 0.0),
|
||||
}
|
||||
report.update(decision_summary)
|
||||
|
||||
def _score_distribution(scores: list[float]) -> dict:
|
||||
return {
|
||||
"high": sum(1 for s in scores if s >= 0.75),
|
||||
"medium": sum(1 for s in scores if 0.4 <= s < 0.75),
|
||||
"low": sum(1 for s in scores if s < 0.4),
|
||||
}
|
||||
|
||||
fig_scores = []
|
||||
for mf in figure_inventory.get("matched_figures", []):
|
||||
cs = mf.get("caption_score", {})
|
||||
if "score" in cs:
|
||||
fig_scores.append(cs["score"])
|
||||
for rl in figure_inventory.get("rejected_legends", []):
|
||||
cs = rl.get("caption_score", {})
|
||||
if "score" in cs:
|
||||
fig_scores.append(cs["score"])
|
||||
|
||||
table_scores = []
|
||||
for t in tables:
|
||||
ms = t.get("match_score", {})
|
||||
if "score" in ms:
|
||||
table_scores.append(ms["score"])
|
||||
|
||||
report["figure_match_confidence_distribution"] = _score_distribution(fig_scores)
|
||||
report["table_match_confidence_distribution"] = _score_distribution(table_scores)
|
||||
|
||||
return report
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -182,3 +182,59 @@ def test_ocr_health_includes_tail_boundary_confidence() -> None:
|
|||
)
|
||||
assert "tail_boundary_confidence" in report
|
||||
assert isinstance(report["tail_boundary_confidence"], (int, float))
|
||||
|
||||
|
||||
def test_ocr_health_includes_confidence_distributions() -> None:
|
||||
from paperforge.worker.ocr_health import build_ocr_health
|
||||
|
||||
structured_blocks = [
|
||||
{"role": "section_heading", "text": "1 Introduction"},
|
||||
{"role": "body_paragraph", "text": "Body"},
|
||||
]
|
||||
figure_inventory = {
|
||||
"matched_figures": [
|
||||
{
|
||||
"figure_id": "figure_001",
|
||||
"caption_score": {"score": 0.9, "decision": "figure_caption", "evidence": ["figure_number"]},
|
||||
},
|
||||
{
|
||||
"figure_id": "figure_002",
|
||||
"caption_score": {"score": 0.5, "decision": "figure_caption_candidate", "evidence": []},
|
||||
},
|
||||
{
|
||||
"figure_id": "figure_003",
|
||||
"caption_score": {"score": 0.3, "decision": "rejected", "evidence": []},
|
||||
},
|
||||
],
|
||||
"unmatched_legends": [],
|
||||
"unmatched_assets": [],
|
||||
}
|
||||
table_inventory = {
|
||||
"tables": [
|
||||
{"match_score": {"score": 0.85, "decision": "matched", "evidence": ["same_page"]}},
|
||||
{"match_score": {"score": 0.40, "decision": "ambiguous", "evidence": []}},
|
||||
{"match_score": {"score": 0.20, "decision": "ambiguous", "evidence": []}},
|
||||
],
|
||||
"unmatched_captions": [],
|
||||
"unmatched_assets": [],
|
||||
}
|
||||
report = build_ocr_health(
|
||||
page_count=3,
|
||||
raw_blocks_count=20,
|
||||
structured_blocks=structured_blocks,
|
||||
figure_inventory=figure_inventory,
|
||||
table_inventory=table_inventory,
|
||||
)
|
||||
assert "figure_match_confidence_distribution" in report
|
||||
assert "table_match_confidence_distribution" in report
|
||||
assert "tail_boundary_confidence" in report
|
||||
|
||||
fig_dist = report["figure_match_confidence_distribution"]
|
||||
assert fig_dist["high"] == 1
|
||||
assert fig_dist["medium"] == 1
|
||||
assert fig_dist["low"] == 1
|
||||
|
||||
tbl_dist = report["table_match_confidence_distribution"]
|
||||
assert tbl_dist["high"] == 1
|
||||
assert tbl_dist["medium"] == 1
|
||||
assert tbl_dist["low"] == 1
|
||||
|
|
|
|||
Loading…
Reference in a new issue