mirror of
https://github.com/lllin000/PaperForge.git
synced 2026-07-22 06:50:53 +00:00
feat: bump PaddleOCR model from VL-1.5 to VL-1.6
This commit is contained in:
parent
dde9c903fe
commit
c1c5b9b8ba
3 changed files with 11 additions and 6 deletions
|
|
@ -61,7 +61,7 @@ def ocr_doctor(config: dict[str, str] | None, live: bool = False) -> dict:
|
|||
resp = requests.post(
|
||||
job_url,
|
||||
headers={"Authorization": f"bearer {token}"},
|
||||
json={"model": os.environ.get("PADDLEOCR_MODEL", "PaddleOCR-VL-1.5")},
|
||||
json={"model": os.environ.get("PADDLEOCR_MODEL", "PaddleOCR-VL-1.6")},
|
||||
timeout=30,
|
||||
)
|
||||
if resp.status_code == 401:
|
||||
|
|
@ -98,7 +98,7 @@ def ocr_doctor(config: dict[str, str] | None, live: bool = False) -> dict:
|
|||
# L3 — API response structure
|
||||
try:
|
||||
minimal_payload = {
|
||||
"model": os.environ.get("PADDLEOCR_MODEL", "PaddleOCR-VL-1.5"),
|
||||
"model": os.environ.get("PADDLEOCR_MODEL", "PaddleOCR-VL-1.6"),
|
||||
"optionalPayload": json.dumps({"useDocOrientationClassify": False}),
|
||||
}
|
||||
resp = requests.post(
|
||||
|
|
@ -159,7 +159,7 @@ def ocr_doctor(config: dict[str, str] | None, live: bool = False) -> dict:
|
|||
resp = requests.post(
|
||||
job_url,
|
||||
headers={"Authorization": f"bearer {token}"},
|
||||
data={"model": os.environ.get("PADDLEOCR_MODEL", "PaddleOCR-VL-1.5")},
|
||||
data={"model": os.environ.get("PADDLEOCR_MODEL", "PaddleOCR-VL-1.6")},
|
||||
files={"file": f},
|
||||
timeout=120,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -718,7 +718,7 @@ def headless_setup(
|
|||
{
|
||||
"PADDLEOCR_API_TOKEN": raw_key,
|
||||
"PADDLEOCR_JOB_URL": paddleocr_url,
|
||||
"PADDLEOCR_MODEL": "PaddleOCR-VL-1.5",
|
||||
"PADDLEOCR_MODEL": "PaddleOCR-VL-1.6",
|
||||
"ZOTERO_DATA_DIR": zotero_data or "",
|
||||
},
|
||||
)
|
||||
|
|
|
|||
|
|
@ -66,7 +66,7 @@ def ensure_ocr_meta(vault: Path, row: dict) -> dict:
|
|||
meta = {}
|
||||
meta.setdefault("zotero_key", key)
|
||||
meta.setdefault("source_pdf", row.get("pdf_path", ""))
|
||||
meta.setdefault("ocr_provider", "PaddleOCR-VL-1.5")
|
||||
meta.setdefault("ocr_provider", "PaddleOCR-VL-1.6")
|
||||
meta.setdefault("mode", "async")
|
||||
meta.setdefault("ocr_status", "pending")
|
||||
meta.setdefault("ocr_job_id", "")
|
||||
|
|
@ -1029,6 +1029,8 @@ def is_embedded_figure_text_block(block: dict, blocks: list[dict], page_width: i
|
|||
label = block.get("block_label", "")
|
||||
if label not in {"text", "paragraph_title"}:
|
||||
return False
|
||||
if label == "paragraph_title":
|
||||
return False
|
||||
text = clean_block_text(block.get("block_content", ""))
|
||||
if not text:
|
||||
return False
|
||||
|
|
@ -1681,6 +1683,9 @@ def run_ocr(vault: Path, verbose: bool = False, no_progress: bool = False) -> in
|
|||
meta["retry_count"] = 0
|
||||
write_json(paths["ocr"] / key / "meta.json", meta)
|
||||
ocr_queue = sync_ocr_queue(paths, target_rows)
|
||||
print(f"DEBUG: target_keys count={len(target_keys)}, target_rows count={len(target_rows)}, queue count={len(ocr_queue)}", flush=True)
|
||||
if ocr_queue:
|
||||
print(f"DEBUG: first item={ocr_queue[0].get('zotero_key')} status={ocr_queue[0].get('queue_status')}", flush=True)
|
||||
max_items_raw = os.environ.get("PADDLEOCR_MAX_ITEMS", "").strip()
|
||||
max_items = 3
|
||||
if max_items_raw:
|
||||
|
|
@ -1703,7 +1708,7 @@ def run_ocr(vault: Path, verbose: bool = False, no_progress: bool = False) -> in
|
|||
# Fallback: parse vault-root .env
|
||||
token = _read_dotenv(vault, "PADDLEOCR_API_TOKEN")
|
||||
job_url = os.environ.get("PADDLEOCR_JOB_URL", "https://paddleocr.aistudio-app.com/api/v2/ocr/jobs").strip()
|
||||
model = os.environ.get("PADDLEOCR_MODEL", "PaddleOCR-VL-1.5").strip()
|
||||
model = os.environ.get("PADDLEOCR_MODEL", "PaddleOCR-VL-1.6").strip()
|
||||
optional_payload = {"useDocOrientationClassify": False, "useDocUnwarping": False, "useChartRecognition": False}
|
||||
changed = 0
|
||||
active_submitted = 0
|
||||
|
|
|
|||
Loading…
Reference in a new issue