diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3b6c1bfa..8e18df20 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1,24 +1,232 @@ +# CI — Plasma Matrix Pipeline (L0-L5 Merge Gate) +# +# Runs on push/PR to main/master. Uses path-filtered triggers for efficient +# resource usage and a re-actors/alls-green aggregator for clean branch protection. +# +# Layer structure: +# L0: Version consistency check (version/__init__.py changes) +# L1: Unit tests — 3 OS x 3 Python plasma matrix +# L2: CLI contract tests — 2 Python x 1 OS +# L3: Plugin tests — Node 20 single config +# L4: E2E + Consistency audit — Python 3.11 single config +# L5: Journey tests — informational, not in merge gate +# Gate: re-actors/alls-green aggregator (excludes L5) + name: CI -on: +"on": push: branches: [main, master] + paths-ignore: + - "**.md" + - "docs/**" pull_request: branches: [main, master] + paths-ignore: + - "**.md" + - "docs/**" + +env: + PYTHONIOENCODING: utf-8 jobs: - plugin-tests: - name: Plugin Tests (L3) + # --------------------------------------------------------------------------- + # Path-filter detection — decides which jobs to run + # --------------------------------------------------------------------------- + changes: + name: Detect Changed Paths + runs-on: ubuntu-latest + outputs: + version: ${{ steps.filter.outputs.version }} + plugin: ${{ steps.filter.outputs.plugin }} + ocr: ${{ steps.filter.outputs.ocr }} + core: ${{ steps.filter.outputs.core }} + any_changed: ${{ steps.filter.outputs.any_changed }} + steps: + - uses: actions/checkout@v4 + - uses: dorny/paths-filter@v3 + id: filter + with: + filters: | + version: + - 'paperforge/__init__.py' + - 'paperforge/plugin/manifest.json' + - 'paperforge/plugin/versions.json' + - 'CHANGELOG.md' + - 'pyproject.toml' + plugin: + - 'paperforge/plugin/**' + - 'paperforge/plugin/package.json' + ocr: + - 'paperforge/worker/ocr.py' + - 'paperforge/ocr_diagnostics.py' + core: + - 'paperforge/**' + - 'tests/**' + - 'fixtures/**' + - 'pyproject.toml' + - 'scripts/**' + + # --------------------------------------------------------------------------- + # L0 — Version consistency check + # --------------------------------------------------------------------------- + version-check: + name: L0 — Version Sync Check + needs: changes + if: ${{ needs.changes.outputs.version == 'true' || needs.changes.outputs.core == 'true' }} + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.11" + - name: Install package + run: pip install -e . + - name: Check version consistency + run: | + if [ -f scripts/check_version_sync.py ]; then + python scripts/check_version_sync.py + else + python -c "from paperforge import __version__; import json; \ + m = json.load(open('paperforge/plugin/manifest.json')); \ + assert m['version'] == __version__, \ + f'Manifest {m[\"version\"]} != {__version__}'; \ + print(f'Version sync: OK ({__version__})')" + fi + + # --------------------------------------------------------------------------- + # L1 — Unit tests (plasma matrix: 3 OS x 3 Python) + # --------------------------------------------------------------------------- + unit-tests: + name: L1 — Unit Tests (${{ matrix.os }}, py${{ matrix.python-version }}) + needs: changes + if: ${{ needs.changes.outputs.plugin != 'true' || needs.changes.outputs.core == 'true' }} + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest, macos-latest] + python-version: ["3.10", "3.11", "3.12"] + runs-on: ${{ matrix.os }} + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "${{ matrix.python-version }}" + - name: Install package with test deps + run: | + pip install -e ".[test]" + - name: Run unit tests (root tests, not layer-marked) + run: | + python -m pytest tests/ \ + --ignore=tests/sandbox \ + --ignore=tests/cli \ + --ignore=tests/e2e \ + --ignore=tests/journey \ + --ignore=tests/chaos \ + --ignore=tests/audit \ + -v --tb=short --timeout=60 \ + -x + + # --------------------------------------------------------------------------- + # L2 — CLI contract tests (2 Python x 1 OS) + # --------------------------------------------------------------------------- + cli-tests: + name: L2 — CLI Contracts (py${{ matrix.python-version }}) + needs: changes + if: ${{ needs.changes.outputs.plugin != 'true' || needs.changes.outputs.core == 'true' }} + strategy: + matrix: + python-version: ["3.10", "3.12"] + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "${{ matrix.python-version }}" + - name: Install package with test deps + run: pip install -e ".[test]" + - name: Run CLI contract tests + run: python -m pytest tests/cli/ -m cli -v --tb=short --timeout=60 -x + + # --------------------------------------------------------------------------- + # L3 — Plugin tests (Node 20, single config) + # --------------------------------------------------------------------------- + plugin-tests: + name: L3 — Plugin Tests + needs: changes + if: ${{ needs.changes.outputs.plugin == 'true' || needs.changes.outputs.core == 'true' }} runs-on: ubuntu-latest - if: contains(join(github.event.head_commit.modified, ' '), 'paperforge/plugin/') || contains(join(github.event.head_commit.added, ' '), 'paperforge/plugin/') || github.ref == 'refs/heads/main' steps: - uses: actions/checkout@v4 - uses: actions/setup-node@v4 with: node-version: 20 - cache: 'npm' + cache: "npm" cache-dependency-path: paperforge/plugin/package-lock.json - run: npm ci working-directory: paperforge/plugin - run: npx vitest run --reporter=verbose working-directory: paperforge/plugin + + # --------------------------------------------------------------------------- + # L4 — E2E + consistency audit tests (single config) + # --------------------------------------------------------------------------- + e2e-tests: + name: L4 — E2E + Audit + needs: changes + if: ${{ needs.changes.outputs.plugin != 'true' || needs.changes.outputs.core == 'true' }} + runs-on: ubuntu-latest + strategy: + matrix: + python-version: ["3.11"] + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "${{ matrix.python-version }}" + - name: Install package with test deps + run: pip install -e ".[test]" + - name: Run E2E tests + run: python -m pytest tests/e2e/ -m e2e -v --tb=short --timeout=120 -x + - name: Run consistency audit tests + run: python -m pytest tests/audit/ -m audit -v --tb=short --timeout=120 -x + + # --------------------------------------------------------------------------- + # L5 — Journey tests (informational, NOT in merge gate) + # --------------------------------------------------------------------------- + journey-tests: + name: L5 — Journey Tests + needs: changes + if: ${{ needs.changes.outputs.any_changed == 'true' }} + runs-on: ubuntu-latest + strategy: + matrix: + python-version: ["3.11"] + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "${{ matrix.python-version }}" + - name: Install package with test deps + run: pip install -e ".[test]" + - name: Run journey tests + run: python -m pytest tests/journey/ -m journey -v --tb=short --timeout=120 + + # --------------------------------------------------------------------------- + # Merge gate — single-status aggregator for branch protection + # --------------------------------------------------------------------------- + alls-green: + name: All Checks Passed + if: always() + needs: + - version-check + - unit-tests + - cli-tests + - plugin-tests + - e2e-tests + runs-on: ubuntu-latest + steps: + - uses: re-actors/alls-green@v1 + with: + allowed-skips: version-check, plugin-tests + jobs: ${{ toJSON(needs) }} diff --git a/INSTALLATION.md b/INSTALLATION.md new file mode 100644 index 00000000..6a771b77 --- /dev/null +++ b/INSTALLATION.md @@ -0,0 +1,61 @@ +# PaperForge Installation Guide + +> Canonical install reference. For other docs, see [README.md](README.md). + +--- + +## Method 1: Obsidian Plugin (Recommended) + +1. Download the plugin files from the [latest release](https://github.com/LLLin000/PaperForge/releases/latest). +2. Extract and copy the files into `vault/.obsidian/plugins/paperforge/`. +3. Open Obsidian -> Settings -> Community Plugins -> Enable `PaperForge`. +4. Open PaperForge settings and click `Open Installation Wizard`. +5. Follow the 5-step guided setup: Overview -> Directory Config -> Agent & Keys -> Install -> Done. + +> The wizard auto-detects Python, Zotero, and Better BibTeX before installation starts. + +--- + +## Method 2: CLI via pip (Developers) + +```bash +pip install git+https://github.com/LLLin000/PaperForge.git +``` + +Then run the headless setup: + +```bash +python -m paperforge setup --headless --agent opencode --paddleocr-key +``` + +For the latest stable release, replace `master` with a version tag: + +```bash +pip install git+https://github.com/LLLin000/PaperForge.git@v1.4.17rc3 +``` + +--- + +## Method 3: AI Agent Setup + +Copy the following to your AI agent for a guided headless install: + +- English: [docs/ai-agent-setup-guide.md](docs/ai-agent-setup-guide.md) +- 中文: [docs/ai-agent-setup-guide-zh.md](docs/ai-agent-setup-guide-zh.md) + +--- + +## Prerequisites + +| Software | Purpose | Get it | +|----------|---------|--------| +| Python 3.10+ | Run PaperForge CLI and backend tasks | https://python.org | +| Zotero | Literature management | https://zotero.org | +| Better BibTeX | Auto-export metadata as JSON | https://retorque.re/zotero-better-bibtex/ | +| PaddleOCR Key | OCR text and layout extraction | https://aistudio.baidu.com/paddleocr | + +--- + +## Post-Installation + +After installing, see [AGENTS.md](AGENTS.md) for first-use workflow and detailed command reference. diff --git a/README.en.md b/README.en.md index a2caa553..18e9428a 100644 --- a/README.en.md +++ b/README.en.md @@ -41,32 +41,7 @@ PaperForge connects the full path from source literature to structured insight. ## Install -### Obsidian Plugin (Recommended) - -1. Download the plugin files from the [latest release](https://github.com/LLLin000/PaperForge/releases/latest). -2. Copy them into `vault/.obsidian/plugins/paperforge/`. -3. Enable `PaperForge` in Obsidian community plugins. -4. Open the settings page and click `打开安装向导`. -5. Follow the 5-step wizard: Overview → Directory Config → Agent & Keys → Install → Done. - -> The wizard auto-detects Python, Zotero, and Better BibTeX before installation starts. - -### Prerequisites - -| Software | Purpose | Get it | -|----------|---------|--------| -| Python 3.10+ | Run PaperForge CLI and backend tasks | https://python.org | -| Zotero | Literature management | https://zotero.org | -| Better BibTeX | Auto-export metadata as JSON | https://retorque.re/zotero-better-bibtex/ | -| PaddleOCR Key | OCR text and layout extraction | https://aistudio.baidu.com/paddleocr | - -### CLI (Advanced) - -```bash -cd /path/to/your/vault -pip install git+https://github.com/LLLin000/PaperForge.git -python -m paperforge setup --headless --agent opencode --paddleocr-key -``` +See [INSTALLATION.md](INSTALLATION.md) for the complete installation guide. ## Usage @@ -131,7 +106,7 @@ python -m paperforge setup --headless --agent opencode --paddleocr-key | Document | Content | |----------|---------| | [Setup Guide](docs/setup-guide.md) | Full setup walkthrough | -| [Quick Install](docs/INSTALLATION.md) | Short install path | +| [Installation Guide](INSTALLATION.md) | Full install reference | | [Post-Install Guide](AGENTS.md) | First-use workflow guide | | [Changelog](CHANGELOG.md) | Version history | | [Contributing](CONTRIBUTING.md) | Dev setup and conventions | diff --git a/README.md b/README.md index f3078b8d..ebf8defb 100644 --- a/README.md +++ b/README.md @@ -44,32 +44,7 @@ PaperForge 不是单纯的 OCR 工具,也不只是 Zotero 到 Obsidian 的搬 ## 安装 -### Obsidian 插件(推荐) - -1. 从 [最新 Release](https://github.com/LLLin000/PaperForge/releases/latest) 下载插件文件。 -2. 复制到 `vault/.obsidian/plugins/paperforge/`。 -3. 在 Obsidian 中启用 `PaperForge`。 -4. 打开设置页,点击 `打开安装向导`。 -5. 跟随 5 步向导完成安装:概览 → 目录配置 → Agent 与密钥 → 安装 → 完成。 - -> 安装向导会自动检测 Python、Zotero 和 Better BibTeX 是否就绪。 - -### 前置准备 - -| 软件 | 用途 | 获取 | -|------|------|------| -| Python 3.10+ | 运行 PaperForge CLI 与后端流程 | https://python.org | -| Zotero | 文献管理 | https://zotero.org | -| Better BibTeX | Zotero JSON 自动导出 | https://retorque.re/zotero-better-bibtex/ | -| PaddleOCR Key | OCR 文本与版面提取 | https://aistudio.baidu.com/paddleocr | - -### 命令行安装(开发者) - -```bash -cd /path/to/your/vault -pip install git+https://github.com/LLLin000/PaperForge.git -python -m paperforge setup --headless --agent opencode --paddleocr-key -``` +完整安装指南请参见 [INSTALLATION.md](INSTALLATION.md)。 ## 使用方式 diff --git a/docs/ai-agent-setup-guide-zh.md b/docs/ai-agent-setup-guide-zh.md index 52070395..00a890b5 100644 --- a/docs/ai-agent-setup-guide-zh.md +++ b/docs/ai-agent-setup-guide-zh.md @@ -1,6 +1,8 @@ # PaperForge Headless Setup — For LLM Agents (简体中文) > **IMPORTANT**: 直接读取此文件。不要用 WebFetch 或摘要——会丢失关键的 CLI 参数和决策逻辑。 +> +> 完整安装参考请见 [INSTALLATION.md](../INSTALLATION.md)。 --- diff --git a/docs/ai-agent-setup-guide.md b/docs/ai-agent-setup-guide.md index c151b1b9..f856e435 100644 --- a/docs/ai-agent-setup-guide.md +++ b/docs/ai-agent-setup-guide.md @@ -1,6 +1,8 @@ # PaperForge Headless Setup — For LLM Agents (English) > **IMPORTANT**: Read this file directly. Do NOT use WebFetch or summarization — you will lose critical CLI flags and decision logic. +> +> For the full installation reference, see [INSTALLATION.md](../INSTALLATION.md). --- diff --git a/manifest.json b/manifest.json index bd59966f..7512c58d 100644 --- a/manifest.json +++ b/manifest.json @@ -1,7 +1,7 @@ { "id": "paperforge", "name": "PaperForge", - "version": "1.4.17rc3", + "version": "1.4.17rc4", "minAppVersion": "1.9.0", "description": "PaperForge — Zotero literature pipeline. Sync PDFs, run OCR, and read with AI-assisted deep reading.", "author": "Lin Zhaoxuan", diff --git a/paperforge/__init__.py b/paperforge/__init__.py index 049a4d4e..7b11f4d9 100644 --- a/paperforge/__init__.py +++ b/paperforge/__init__.py @@ -1,3 +1,3 @@ """paperforge — PaperForge package.""" -__version__ = "1.4.17rc3" +__version__ = "1.4.17rc4" diff --git a/paperforge/adapters/__init__.py b/paperforge/adapters/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/paperforge/adapters/bbt.py b/paperforge/adapters/bbt.py new file mode 100644 index 00000000..ce8d2bb0 --- /dev/null +++ b/paperforge/adapters/bbt.py @@ -0,0 +1,216 @@ +from __future__ import annotations + +import logging +from pathlib import Path + +logger = logging.getLogger(__name__) + + +def _normalize_attachment_path(path: str, zotero_dir: Path | None = None) -> tuple[str, str, str]: + """Normalize a BBT attachment path to a consistent storage: format. + + Handles three real-world BBT export formats: + 1. Absolute Windows paths: D:\\...\\Zotero\\storage\\8CHARKEY\\filename.pdf + -> storage:8CHARKEY/filename.pdf + 2. storage: prefix: storage:KEY/filename.pdf -> pass through + 3. Bare relative: KEY/filename.pdf -> storage:KEY/filename.pdf + + Args: + path: Raw path from BBT JSON attachment. + zotero_dir: Optional absolute path to Zotero data directory for + validating absolute paths. + + Returns: + Tuple of (normalized_path, bbt_path_raw, zotero_storage_key). + normalized_path uses forward slashes and storage: prefix for + Zotero storage paths. bbt_path_raw preserves the original input + for debugging. zotero_storage_key is the 8-character Zotero key. + """ + raw = str(path or "").strip() + if not raw: + return ("", "", "") + + bbt_path_raw = raw + + # Format 2: Already has storage: prefix — pass through with slash normalization + if raw.startswith("storage:"): + storage_rel = raw[len("storage:") :].lstrip("/").lstrip("\\") + storage_rel = storage_rel.replace("\\", "/") + parts = storage_rel.split("/") + zotero_storage_key = parts[0] if parts else "" + return (f"storage:{storage_rel}", bbt_path_raw, zotero_storage_key) + + # Format 1: Absolute Windows path pointing to Zotero storage + candidate = Path(raw) + if candidate.is_absolute(): + norm_path = raw.replace("\\", "/") + # Detect Zotero storage pattern: .../storage/8CHARKEY/... + if "/storage/" in norm_path: + parts_after_storage = norm_path.split("/storage/", 1)[1] + parts = parts_after_storage.split("/") + if len(parts) >= 2 and len(parts[0]) == 8 and parts[0].isalnum(): + zotero_storage_key = parts[0] + filename = "/".join(parts[1:]) + return (f"storage:{zotero_storage_key}/{filename}", bbt_path_raw, zotero_storage_key) + # Absolute path but not in Zotero storage — mark as absolute + return (f"absolute:{raw}", bbt_path_raw, "") + + # Format 3: Bare relative path — prepend storage: prefix + norm = raw.replace("\\", "/") + parts = norm.split("/") + zotero_storage_key = parts[0] if parts else "" + return (f"storage:{norm}", bbt_path_raw, zotero_storage_key) + + +def _identify_main_pdf(attachments: list[dict]) -> tuple[dict | None, list[dict]]: + """Identify the main PDF and supplementary materials from attachments. + + Uses a hybrid three-priority strategy (Decision D-02): + 1. Primary: attachment.title == "PDF" AND contentType == "application/pdf" + 2. Fallback heuristic: largest file by size (if available), else shortest title + 3. Final fallback: first PDF attachment in the list + + Args: + attachments: List of attachment dicts from load_export_rows(). + + Returns: + Tuple of (main_pdf_attachment, supplementary_attachments). + main_pdf_attachment may be None if no PDFs found. + supplementary_attachments is a list of all other PDF attachments. + """ + pdf_attachments = [a for a in attachments if isinstance(a, dict) and a.get("contentType") == "application/pdf"] + + if not pdf_attachments: + return (None, []) + + # Priority 1: Title exactly equals "PDF" + for att in pdf_attachments: + if att.get("title") == "PDF": + main = att + supplementary = [a for a in pdf_attachments if a is not main] + return (main, supplementary) + + # Priority 2: Largest file by size (if size field is available and differentiated) + sized = [(a, a.get("size", 0) or 0) for a in pdf_attachments] + sized.sort(key=lambda x: x[1], reverse=True) + if sized and sized[0][1] > 0 and (len(sized) == 1 or sized[0][1] > sized[1][1]): + main = sized[0][0] + supplementary = [a for a in pdf_attachments if a is not main] + return (main, supplementary) + + # Priority 2b (sizes equal or unavailable): shortest title + titled = [(a, len(str(a.get("title", "")))) for a in pdf_attachments] + titled.sort(key=lambda x: x[1]) + main = titled[0][0] + supplementary = [a for a in pdf_attachments if a is not main] + return (main, supplementary) + + +def extract_authors(item: dict) -> list[str]: + authors = [] + for creator in item.get("creators", []): + if creator.get("creatorType") != "author": + continue + full_name = " ".join( + part for part in [creator.get("firstName", ""), creator.get("lastName", "")] if part + ).strip() + if full_name: + authors.append(full_name) + elif creator.get("name"): + authors.append(creator["name"]) + return authors + + +def collection_fields(collection_paths: list[str]) -> dict[str, str | list[str]]: + paths = [path for path in collection_paths if path] + primary = paths[0] if paths else "" + if paths: + primary = sorted(paths, key=lambda value: (value.count("/"), len(value), value), reverse=True)[0] + tags = [] + seen = set() + for path in paths: + for part in [segment.strip() for segment in path.split("/") if segment.strip()]: + if part not in seen: + seen.add(part) + tags.append(part) + group = primary + return {"collections": paths, "collection_tags": tags, "collection_group": [group] if group else []} + + +def resolve_item_collection_paths(item: dict, collection_lookup: dict) -> list[str]: + paths = [] + collection_keys = item.get("collections") or [] + if collection_keys: + for key in collection_keys: + paths.append(collection_lookup.get("path_by_key", {}).get(key, key)) + item_id = item.get("itemID") + if item_id is not None: + paths.extend(collection_lookup.get("paths_by_item_id", {}).get(item_id, [])) + return sorted({path for path in paths if path}, key=lambda value: (-value.count("/"), value)) + + +def load_export_rows(path: Path) -> list[dict]: + from paperforge.worker._domain import build_collection_lookup + from paperforge.worker._utils import _extract_year, read_json + + data = read_json(path) + if isinstance(data, list): + return data + if isinstance(data, dict) and isinstance(data.get("items"), list): + collection_lookup = build_collection_lookup(data.get("collections", {})) + rows = [] + for item in data["items"]: + if item.get("itemType") in {"attachment", "note", "annotation"}: + continue + attachments = [] + for attachment in item.get("attachments", []): + if not isinstance(attachment, dict): + continue + raw_path = attachment.get("path", "") + normalized_path, bbt_path_raw, zotero_storage_key = _normalize_attachment_path(raw_path) + # Preserve contentType from BBT if present; fallback to file extension + content_type = attachment.get("contentType", "") + if not content_type and str(normalized_path).lower().endswith(".pdf"): + content_type = "application/pdf" + attachments.append( + { + "path": normalized_path, + "contentType": content_type, + "title": attachment.get("title", ""), + "bbt_path_raw": bbt_path_raw, + "zotero_storage_key": zotero_storage_key, + "size": attachment.get("size", 0) or 0, + } + ) + main_pdf, supplementary_pdfs = _identify_main_pdf(attachments) + pdf_path = main_pdf["path"] if main_pdf else "" + bbt_path_raw = main_pdf["bbt_path_raw"] if main_pdf else "" + zotero_storage_key = main_pdf["zotero_storage_key"] if main_pdf else "" + path_error = "not_found" if not main_pdf else "" + supplementary = [a["path"] for a in supplementary_pdfs] if supplementary_pdfs else [] + attachment_count = len(attachments) + rows.append( + { + "key": item.get("key") or item.get("itemKey", ""), + "title": item.get("title", ""), + "authors": extract_authors(item), + "creators": item.get("creators", []), + "abstract": item.get("abstractNote", ""), + "journal": item.get("publicationTitle", ""), + "extra": item.get("extra", ""), + "year": _extract_year(item.get("date", "")), + "date": item.get("date", ""), + "doi": item.get("DOI", ""), + "pmid": item.get("PMID", ""), + "collections": resolve_item_collection_paths(item, collection_lookup), + "attachments": attachments, + "pdf_path": pdf_path, + "supplementary": supplementary, + "attachment_count": attachment_count, + "bbt_path_raw": bbt_path_raw, + "zotero_storage_key": zotero_storage_key, + "path_error": path_error, + } + ) + return rows + raise ValueError(f"Unsupported export format: {path}") diff --git a/paperforge/adapters/obsidian_frontmatter.py b/paperforge/adapters/obsidian_frontmatter.py new file mode 100644 index 00000000..20331c40 --- /dev/null +++ b/paperforge/adapters/obsidian_frontmatter.py @@ -0,0 +1,318 @@ +from __future__ import annotations + +import re +from pathlib import Path +from typing import Optional + + +def _yaml_quote(value: str) -> str: + if isinstance(value, bool): + return "true" if value else "false" + return '"' + str(value or "").replace("\\", "\\\\").replace('"', '\\"') + '"' + + +def _yaml_block(value: str) -> list[str]: + value = (value or "").strip() + if not value: + return ["abstract: |-", " "] + lines = ["abstract: |-"] + for line in value.splitlines(): + lines.append(f" {line}") + return lines + + +def compute_final_collection(row: dict) -> str: + user_raw = str(row.get("user_collection", "") or "").strip() + user_resolved = str(row.get("user_collection_resolved", "") or "").strip() + recommended = str(row.get("recommended_collection", "") or "").strip() + if user_raw: + return user_resolved + return recommended + + +DEEP_READING_HEADER = "## 🔍 精读" + + +def _read_frontmatter_bool_from_text(text: str, key: str, default: bool = False) -> bool: + match = re.search(rf"^{re.escape(key)}:\s*(?:[\"'])?(true|false)(?:[\"'])?\s*$", text, re.MULTILINE | re.IGNORECASE) + if not match: + return default + return match.group(1).lower() == "true" + + +def _read_frontmatter_optional_bool_from_text(text: str, key: str) -> Optional[bool]: + match = re.search(rf"^{re.escape(key)}:\s*(?:[\"'])?(true|false)(?:[\"'])?\s*$", text, re.MULTILINE | re.IGNORECASE) + if not match: + return None + return match.group(1).lower() == "true" + + +def _legacy_control_flags(paths: dict[str, Path], zotero_key: str) -> dict[str, Optional[bool]]: + records_root = paths.get("library_records") + if not records_root or not records_root.exists(): + return {"do_ocr": None, "analyze": None} + for record_path in records_root.rglob(f"{zotero_key}.md"): + try: + text = record_path.read_text(encoding="utf-8") + except Exception: + continue + return { + "do_ocr": _read_frontmatter_optional_bool_from_text(text, "do_ocr"), + "analyze": _read_frontmatter_optional_bool_from_text(text, "analyze"), + } + return {"do_ocr": None, "analyze": None} + + +def canonicalize_decision(value: str) -> str: + text = str(value or "").strip() + if text in {"", "待查"}: + return "待定" + if text in {"排除", "不纳入"}: + return "不纳入" + if text == "纳入": + return "纳入" + return "待定" + + +def candidate_markdown(row: dict) -> str: + row = dict(row) + row["final_collection"] = compute_final_collection(row) + row["decision"] = canonicalize_decision(row.get("decision", "")) + lines = ["---"] + ordered_keys = [ + "candidate_id", + "domain", + "title", + "authors", + "year", + "journal", + "doi", + "pmid", + "source", + "requester_skill", + "request_context", + "abstract_short", + "decision", + "recommended_collection", + "recommend_confidence", + "recommend_reason", + "user_collection", + "user_collection_resolved", + "final_collection", + "collection_resolution", + "duplicate_hint", + "existing_zotero_key", + "existing_collections", + "import_status", + "note", + "candidate_source_type", + "source_zotero_key", + "cited_ref_number", + "trigger_sentence", + "source_context", + "task_relevance_reason", + "harvest_priority", + "raw_reference", + "status", + ] + row.setdefault("status", "candidate") + for key in ordered_keys: + value = row.get(key, "") + if isinstance(value, list): + lines.append(f"{key}:") + for item in value: + lines.append(f" - {_yaml_quote(item)}") + elif value == "": + lines.append(f"{key}:") + elif "\n" in str(value): + lines.extend( + _yaml_block(str(value)).copy() + if key == "abstract" + else [f"{key}: |-"] + [f" {line}" for line in str(value).splitlines()] + ) + else: + lines.append(f"{key}: {_yaml_quote(value)}") + lines.extend( + [ + "---", + "", + f"# {row['candidate_id']}", + "", + "候选文献轻量记录,仅用于 Base 决策和 write-back 触发,不是正式文献卡片。", + "", + ] + ) + return "\n".join(lines) + + +def generate_review(candidates: list[dict]) -> str: + normalized = [] + for row in candidates: + copy = dict(row) + copy["decision"] = canonicalize_decision(copy.get("decision", "")) + normalized.append(copy) + include = [c for c in normalized if c.get("decision") == "纳入"] + exclude = [c for c in normalized if c.get("decision") == "不纳入"] + lines = [ + "# 本轮候选总览", + "", + "## 检索背景", + "", + f"- 候选数量:{len(normalized)}", + f"- 建议纳入:{len(include)}", + f"- 不纳入:{len(exclude)}", + "", + "## 总体判断", + "", + "- 当前候选池已经按决策状态分层,可直接进入 Base 处理。", + "", + "## 推荐优先纳入", + "", + ] + if include: + for row in include: + lines.extend( + [ + f"### {row['candidate_id']}", + "", + f"- 标题:{row['title']}", + f"- 推荐分类:`{compute_final_collection(row)}`", + f"- 理由:{row.get('recommend_reason', '')}", + "", + ] + ) + else: + lines.extend(["- 暂无", ""]) + lines.extend(["## 不纳入", ""]) + if exclude: + for row in exclude: + lines.extend( + [ + f"### {row['candidate_id']}", + "", + f"- 标题:{row['title']}", + f"- 理由:{row.get('recommend_reason', '')}", + "", + ] + ) + else: + lines.extend(["- 暂无", ""]) + lines.extend(["## 下一步", "", "1. 在 Base 中确认决策。", "2. 对纳入项执行 write-back。", "3. 刷新正式索引。", ""]) + return "\n".join(lines) + + +def extract_preserved_deep_reading(text: str) -> str: + if not text: + return "" + match = re.search("^## 🔍 精读\\s*$", text, re.MULTILINE) + if not match: + return "" + start = match.start() + preserved = text[start:].strip() + return preserved + + +def has_deep_reading_content(text: str) -> bool: + preserved = extract_preserved_deep_reading(text) + if not preserved: + return False + body = preserved.replace(DEEP_READING_HEADER, "").strip() + if not body: + return False + + clarity_ok = bool(re.search( + r'-\s*\*\*Clarity\*\*(清晰度):(.+)', body + )) and not re.search( + r'-\s*\*\*Clarity\*\*(清晰度):\s*$', body, re.MULTILINE + ) + + figure_sec = _extract_section(body, r'\*\*Figure 导读\*\*') + figure_ok = False + if figure_sec: + for line in figure_sec.splitlines(): + s = line.strip() + if s.startswith('- ') and ':' in s: + _, after = s.split(':', 1) + if after.strip() and after.strip() != '(待补充)': + figure_ok = True + break + + issue_sec = _extract_section(body, r'\*\*遗留问题\*\*') + if not issue_sec: + issue_sec = _extract_section(body, r'####\s*遗留问题') + issue_ok = False + if issue_sec: + dirty = [l.strip() for l in issue_sec.splitlines() if l.strip()] + substantive = [l for l in dirty if l not in ('-', '(待补充)', '', '**遗留问题**')] + issue_ok = bool(substantive) + + return clarity_ok and figure_ok and issue_ok + + +def _extract_section(body: str, section_header: str) -> str | None: + m = re.search( + section_header + r'\n*(.*?)(?=\n(?:#{1,3})\s|\Z)', + body, re.DOTALL, + ) + if m: + return m.group(1).strip() + return None + + +def _add_missing_frontmatter_fields(existing_content: str, new_fields: dict[str, str]) -> str: + if not existing_content.startswith("---"): + return existing_content + parts = existing_content.split("---", 2) + if len(parts) < 3: + return existing_content + frontmatter = parts[1] + body = parts[2] + lines_to_add = [] + for key, value in new_fields.items(): + pattern = "^" + re.escape(key) + "\\s*:" + if not re.search(pattern, frontmatter, re.MULTILINE): + lines_to_add.append(f"{key}: {_yaml_quote(value)}") + if not lines_to_add: + return existing_content + new_frontmatter = frontmatter.rstrip("\n") + "\n" + "\n".join(lines_to_add) + "\n" + return f"---{new_frontmatter}---{body}" + + +def update_frontmatter_field(content: str, key: str, value: str) -> str: + if not content.startswith("---"): + return content + pattern = "^" + re.escape(key) + "\\s*:.*$" + replacement = f"{key}: {_yaml_quote(value)}" + new_content, count = re.subn(pattern, replacement, content, flags=re.MULTILINE, count=1) + if count == 0: + new_content = _add_missing_frontmatter_fields(content, {key: value}) + return new_content + + +def read_frontmatter_dict(text: str) -> dict: + """Parse YAML frontmatter from markdown text using yaml.safe_load. + + Falls back to regex parsing if YAML parser fails (malformed frontmatter). + Returns empty dict if no frontmatter. + """ + import re + import yaml + + fm_match = re.match(r'^---\s*\n(.*?)\n---', text, re.DOTALL) + if not fm_match: + return {} + + try: + data = yaml.safe_load(fm_match.group(1)) + if isinstance(data, dict): + return data + except Exception: + pass + + result = {} + for line in fm_match.group(1).splitlines(): + line = line.strip() + if ':' in line: + key, _, val = line.partition(':') + result[key.strip()] = val.strip() + return result diff --git a/paperforge/adapters/zotero_paths.py b/paperforge/adapters/zotero_paths.py new file mode 100644 index 00000000..e2c5a239 --- /dev/null +++ b/paperforge/adapters/zotero_paths.py @@ -0,0 +1,62 @@ +from __future__ import annotations + +import os +from pathlib import Path + + +def obsidian_wikilink_for_pdf(pdf_path: str, vault_dir: Path, zotero_dir: Path | None = None) -> str: + text = str(pdf_path or "").strip() + if not text: + return "" + # Handle storage: prefix paths by resolving through zotero_dir + if text.startswith("storage:") and zotero_dir is not None: + storage_rel = text[len("storage:") :].lstrip("/").lstrip("\\") + absolute_pdf_path = zotero_dir / "storage" / storage_rel.replace("/", os.sep) + absolute_str = str(absolute_pdf_path) + else: + absolute_str = absolutize_vault_path(vault_dir, text, resolve_junction=True) + if not absolute_str: + return "" + absolute_path = Path(absolute_str) + try: + relative = absolute_path.relative_to(vault_dir) + except ValueError: + # Path outside vault — try to route through Zotero junction inside vault + if zotero_dir is not None and zotero_dir.exists(): + try: + from paperforge.pdf_resolver import resolve_junction + real_zotero = resolve_junction(zotero_dir) + if real_zotero != zotero_dir: + rel_to_zotero = absolute_path.relative_to(real_zotero) + via_junction = zotero_dir / rel_to_zotero + relative = via_junction.relative_to(vault_dir) + return f"[[{relative.as_posix()}]]" + except (ValueError, OSError): + pass + return f"[[{absolute_path.as_posix()}]]" + return f"[[{relative.as_posix()}]]" + + +def absolutize_vault_path(vault: Path, path: str, resolve_junction: bool = False) -> str: + text = str(path or "").strip() + if not text: + return "" + candidate = Path(text) + result = str(candidate) if candidate.is_absolute() else str((vault / text.replace("/", os.sep)).resolve()) + if resolve_junction: + from paperforge.pdf_resolver import resolve_junction + + result = str(resolve_junction(Path(result))) + return result + + +def obsidian_wikilink_for_path(vault: Path, path: str) -> str: + absolute = absolutize_vault_path(vault, path, resolve_junction=True) + if not absolute: + return "" + absolute_path = Path(absolute) + try: + relative = absolute_path.relative_to(vault) + except ValueError: + return f"[[{absolute_path.as_posix()}]]" + return f"[[{relative.as_posix()}]]" diff --git a/paperforge/cli.py b/paperforge/cli.py index 636c1c6b..37b6086b 100644 --- a/paperforge/cli.py +++ b/paperforge/cli.py @@ -181,6 +181,11 @@ def build_parser() -> argparse.ArgumentParser: action="store_true", help="Force full rebuild of the canonical asset index", ) + p_sync.add_argument( + "--json", + action="store_true", + help="Output result as JSON (PFResult envelope)", + ) # selection-sync (backward compat) sub.add_parser("selection-sync", help="Sync Zotero selection to library records") @@ -203,6 +208,11 @@ def build_parser() -> argparse.ArgumentParser: action="store_true", help="Diagnose OCR configuration without running", ) + p_ocr.add_argument( + "--json", + action="store_true", + help="Output result as JSON (PFResult envelope)", + ) p_ocr.add_argument( "--key", metavar="KEY", @@ -237,6 +247,10 @@ def build_parser() -> argparse.ArgumentParser: help="Output all entries in the canonical index (JSON array)", ) + # dashboard + p_dash = sub.add_parser("dashboard", help="Aggregated stats and permissions for the plugin dashboard") + p_dash.add_argument("--json", action="store_true", help="Output as PFResult JSON") + # base-refresh p_base = sub.add_parser("base-refresh", help="Refresh Obsidian Base view files") p_base.add_argument( @@ -247,7 +261,8 @@ def build_parser() -> argparse.ArgumentParser: ) # doctor - sub.add_parser("doctor", help="Validate PaperForge setup and configuration") + doctor_p = sub.add_parser("doctor", help="Validate PaperForge setup and configuration") + doctor_p.add_argument("--json", action="store_true", dest="json_output", help="Output JSON") # update sub.add_parser("update", help="Update PaperForge to the latest version") @@ -312,6 +327,11 @@ def build_parser() -> argparse.ArgumentParser: action="store_true", help="Skip environment checks (for testing/CI)", ) + p_setup.add_argument( + "--modular", + action="store_true", + help="Use modular setup components (v2.1+)", + ) return parser @@ -438,6 +458,11 @@ def main(argv: list[str] | None = None) -> int: return repair.run(args) + if args.command == "dashboard": + from paperforge.commands import dashboard + + return dashboard.run(args) + if args.command == "base-refresh": force = getattr(args, "force", False) paths = args.paths @@ -450,7 +475,12 @@ def main(argv: list[str] | None = None) -> int: if args.command == "doctor": from paperforge.worker.status import run_doctor - return run_doctor(vault) + kw = {} + if getattr(args, "verbose", False): + kw["verbose"] = True + if getattr(args, "json_output", False): + kw["json_output"] = True + return run_doctor(vault, **kw) if args.command == "update": from paperforge.worker.update import run_update @@ -458,7 +488,23 @@ def main(argv: list[str] | None = None) -> int: return run_update(vault) if args.command == "setup": - if getattr(args, "headless", False): + if getattr(args, "modular", False): + from paperforge.setup.plan import SetupPlan + + config = { + "system_dir": getattr(args, "system_dir", None) or "System", + "resources_dir": getattr(args, "resources_dir", None) or "Resources", + "literature_dir": getattr(args, "literature_dir", None) or "Literature", + "control_dir": getattr(args, "control_dir", None) or "LiteratureControl", + "base_dir": getattr(args, "base_dir", None) or "Bases", + } + plan = SetupPlan( + vault=vault, + config=config, + agent_type=getattr(args, "agent", "opencode"), + ) + return plan.execute(json_output=getattr(args, "json_output", False)) + elif getattr(args, "headless", False): from paperforge.setup_wizard import headless_setup return headless_setup( diff --git a/paperforge/commands/__init__.py b/paperforge/commands/__init__.py index 5102b498..4a570f2f 100644 --- a/paperforge/commands/__init__.py +++ b/paperforge/commands/__init__.py @@ -8,6 +8,7 @@ _COMMAND_REGISTRY: dict[str, str] = { "repair": "paperforge.commands.repair", "status": "paperforge.commands.status", "context": "paperforge.commands.context", + "dashboard": "paperforge.commands.dashboard", } diff --git a/paperforge/commands/dashboard.py b/paperforge/commands/dashboard.py new file mode 100644 index 00000000..fc1a1a70 --- /dev/null +++ b/paperforge/commands/dashboard.py @@ -0,0 +1,167 @@ +"""Dashboard command. + +Returns a PFResult with aggregated stats and permissions for the plugin dashboard. +""" + +from __future__ import annotations + +import logging +import os +import re +from pathlib import Path + +from paperforge import __version__ +from paperforge.config import load_vault_config, paperforge_paths, resolve_vault +from paperforge.core.errors import ErrorCode +from paperforge.core.result import PFError, PFResult + +logger = logging.getLogger(__name__) + + +def run(args) -> int: + """Run dashboard command and print PFResult JSON to stdout.""" + vault = getattr(args, "vault_path", None) + if vault is None: + try: + vault = resolve_vault(cli_vault=getattr(args, "vault", None)) + except FileNotFoundError as exc: + result = PFResult( + ok=False, + command="dashboard", + version=__version__, + error=PFError(code=ErrorCode.INTERNAL_ERROR, message=str(exc)), + ) + print(result.to_json()) + return 1 + + try: + data = _gather_dashboard_data(vault) + result = PFResult(ok=True, command="dashboard", version=__version__, data=data) + print(result.to_json()) + return 0 + except Exception as exc: + logger.exception("Dashboard gathering failed") + result = PFResult( + ok=False, + command="dashboard", + version=__version__, + error=PFError(code=ErrorCode.INTERNAL_ERROR, message=str(exc)), + ) + print(result.to_json()) + return 1 + + +def _gather_dashboard_data(vault: Path) -> dict: + """Gather stats and permissions for dashboard display.""" + cfg = load_vault_config(vault) + paths = paperforge_paths(vault, cfg) + + # ── Papers / formal note count ── + _skip_names = {"fulltext.md", "deep-reading.md", "discussion.md"} + record_count = 0 + if paths["literature"].exists(): + for p in paths["literature"].rglob("*.md"): + if p.name not in _skip_names: + record_count += 1 + + # ── Domain counts (first-level subdirs under literature) ── + domain_counts: dict[str, int] = {} + if paths["literature"].exists(): + for domain_dir in sorted(paths["literature"].iterdir()): + if domain_dir.is_dir(): + count = sum( + 1 for p in domain_dir.rglob("*.md") if p.name not in _skip_names + ) + if count > 0: + domain_counts[domain_dir.name] = count + + # ── PDF health & OCR health from frontmatter ── + pdf_healthy = 0 + pdf_broken = 0 + pdf_missing = 0 + ocr_pending = 0 + ocr_done = 0 + ocr_failed = 0 + + _path_error_pat = re.compile(r'^path_error:\s*"(.+?)"\s*$', re.MULTILINE) + _pdf_path_pat = re.compile(r'^pdf_path:\s*".*?"\s*$', re.MULTILINE) + _ocr_status_pat = re.compile(r'^ocr_status:\s*(\S+)', re.MULTILINE) + _do_ocr_pat = re.compile(r'^do_ocr:\s*true\s*$', re.MULTILINE) + + if paths["literature"].exists(): + for note_path in paths["literature"].rglob("*.md"): + if note_path.name in _skip_names: + continue + try: + text = note_path.read_text(encoding="utf-8") + except Exception: + continue + + # PDF health + path_error_m = _path_error_pat.search(text) + if path_error_m: + error_type = path_error_m.group(1) + if error_type == "not_found": + pdf_missing += 1 + else: + pdf_broken += 1 + elif _pdf_path_pat.search(text): + pdf_healthy += 1 + + # OCR health + ocr_status_m = _ocr_status_pat.search(text) + if ocr_status_m: + status = ocr_status_m.group(1).strip().lower().strip('"') + if status == "done": + ocr_done += 1 + elif status in ("pending", "queued", "processing", "running", ""): + ocr_pending += 1 + else: + ocr_failed += 1 + elif _do_ocr_pat.search(text): + ocr_pending += 1 + + # ── Permissions ── + export_files = sorted(paths["exports"].glob("*.json")) if paths["exports"].exists() else [] + can_sync = len(export_files) > 0 + + paddle_token = ( + os.environ.get("PADDLEOCR_API_TOKEN") + or os.environ.get("PADDLEOCR_API_KEY") + or os.environ.get("OCR_TOKEN") + ) + can_ocr = bool(paddle_token) + + can_copy_context = False + pf_dir = paths.get("paperforge", vault / cfg["system_dir"] / "PaperForge") + if pf_dir.exists(): + try: + pf_dir.parent.mkdir(parents=True, exist_ok=True) + test_file = pf_dir / ".write_test" + test_file.touch() + test_file.unlink() + can_copy_context = True + except (OSError, PermissionError): + pass + + return { + "stats": { + "papers": record_count, + "pdf_health": { + "healthy": pdf_healthy, + "broken": pdf_broken, + "missing": pdf_missing, + }, + "ocr_health": { + "pending": ocr_pending, + "done": ocr_done, + "failed": ocr_failed, + }, + "domain_counts": domain_counts, + }, + "permissions": { + "can_sync": can_sync, + "can_ocr": can_ocr, + "can_copy_context": can_copy_context, + }, + } diff --git a/paperforge/commands/ocr.py b/paperforge/commands/ocr.py index fb9fc997..f4a05917 100644 --- a/paperforge/commands/ocr.py +++ b/paperforge/commands/ocr.py @@ -4,10 +4,54 @@ import argparse import logging from pathlib import Path +from paperforge import __version__ +from paperforge.core.errors import ErrorCode +from paperforge.core.result import PFError, PFResult + logger = logging.getLogger(__name__) -def _diagnose(vault: Path, live: bool = False) -> int: +def _collect_ocr_queue_data(vault: Path) -> dict: + """Scan OCR meta files and build queue status data dict. + + Returns dict shaped as: + {queue: {pending: [...], processing: [...], done: [...], failed: [...]}, + total: N, done: N, failed: N, pending: N, processing: N} + """ + from paperforge.worker._utils import pipeline_paths, read_json + + paths = pipeline_paths(vault) + queue_lists = {"pending": [], "processing": [], "done": [], "failed": []} + ocr_root = paths.get("ocr") + if ocr_root and ocr_root.exists(): + for meta_path in sorted(ocr_root.glob("*/meta.json")): + try: + meta = read_json(meta_path) + except Exception: + continue + key = str(meta.get("zotero_key", "") or "").strip() + status = str(meta.get("ocr_status", "") or "").strip().lower() + if not key: + continue + if status == "done": + queue_lists["done"].append(key) + elif status in ("queued", "running", "processing"): + queue_lists["processing"].append(key) + elif status == "pending": + queue_lists["pending"].append(key) + else: + queue_lists["failed"].append(key) + return { + "queue": queue_lists, + "total": sum(len(v) for v in queue_lists.values()), + "done": len(queue_lists["done"]), + "failed": len(queue_lists["failed"]), + "pending": len(queue_lists["pending"]), + "processing": len(queue_lists["processing"]), + } + + +def _diagnose(vault: Path, live: bool = False, json_output: bool = False) -> int: """Run OCR diagnostics and print results.""" from paperforge.ocr_diagnostics import ocr_doctor @@ -15,6 +59,32 @@ def _diagnose(vault: Path, live: bool = False) -> int: level = result.get("level", 0) passed = result.get("passed", False) + if json_output: + queue_data = _collect_ocr_queue_data(vault) + pf_error = None + if not passed: + pf_error = PFError( + code=ErrorCode.OCR_TOKEN_MISSING if level == 1 else ErrorCode.INTERNAL_ERROR, + message=result.get("error", "OCR diagnosis failed"), + details={"level": level, "fix": result.get("fix", "")}, + ) + pf_result = PFResult( + ok=passed, + command="ocr-diagnose", + version=__version__, + data={ + "diagnosis": { + "level": level, + "passed": passed, + "message": result.get("message", result.get("error", "")), + }, + **queue_data, + }, + error=pf_error, + ) + print(pf_result.to_json()) + return 0 if passed else 1 + print(f"OCR Doctor — Level {level} diagnostic") print("-" * 40) if passed: @@ -64,11 +134,12 @@ def run(args: argparse.Namespace) -> int: diagnose_only = getattr(args, "diagnose", False) key = getattr(args, "key", None) live = getattr(args, "live", False) + json_output = getattr(args, "json", False) # Backward compat: if subcommand was "doctor", diagnose ocr_action = getattr(args, "ocr_action", None) if ocr_action == "doctor" or diagnose_only: - return _diagnose(vault, live=live) + return _diagnose(vault, live=live, json_output=json_output) if key: logger.info("Processing specific key: %s", key) diff --git a/paperforge/commands/sync.py b/paperforge/commands/sync.py index f4bb9b26..c57e9cfe 100644 --- a/paperforge/commands/sync.py +++ b/paperforge/commands/sync.py @@ -1,10 +1,14 @@ """Sync command — unifies selection-sync and index-refresh.""" import argparse +import json as _json import logging from pathlib import Path +from paperforge import __version__ from paperforge.config import migrate_paperforge_json +from paperforge.core.errors import ErrorCode +from paperforge.core.result import PFError, PFResult logger = logging.getLogger(__name__) @@ -74,8 +78,18 @@ def run(args: argparse.Namespace) -> int: index_only = getattr(args, "index", False) rebuild_index = getattr(args, "rebuild_index", False) domain = getattr(args, "domain", None) + json_output = getattr(args, "json", False) if dry_run: + if json_output: + result = PFResult( + ok=True, + command="sync", + version=__version__, + data={"dry_run": True, "selection": not index_only, "index": not selection_only}, + ) + print(result.to_json()) + return 0 print("[DRY-RUN] Would run sync operations") if not selection_only and not index_only: print(" - selection-sync") @@ -90,17 +104,39 @@ def run(args: argparse.Namespace) -> int: return 0 exit_code = 0 + sync_counts = {"new": 0, "updated": 0, "skipped": 0, "failed": 0, "errors": []} if not index_only: run_selection_sync = _get_run_selection_sync() - code = run_selection_sync(vault, verbose=getattr(args, "verbose", False)) - if code != 0: - exit_code = code + result = run_selection_sync(vault, verbose=getattr(args, "verbose", False), json_output=json_output) + if json_output: + sync_counts = result + elif result != 0: + exit_code = result if not selection_only: run_index_refresh = _get_run_index_refresh() - code = run_index_refresh(vault, verbose=getattr(args, "verbose", False), rebuild_index=rebuild_index) + code = run_index_refresh(vault, verbose=getattr(args, "verbose", False), rebuild_index=rebuild_index, json_output=json_output) if code != 0 and exit_code == 0: exit_code = code + if json_output: + errors = sync_counts.get("errors", []) + pf_error = None + if errors or exit_code != 0: + pf_error = PFError( + code=ErrorCode.SYNC_FAILED, + message=f"Sync completed with {len(errors)} error(s)", + details={"errors": errors, "exit_code": exit_code}, + ) + result = PFResult( + ok=exit_code == 0 and not errors, + command="sync", + version=__version__, + data=sync_counts, + error=pf_error, + ) + print(result.to_json()) + return 0 + return exit_code diff --git a/paperforge/core/__init__.py b/paperforge/core/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/paperforge/core/errors.py b/paperforge/core/errors.py new file mode 100644 index 00000000..a37e1641 --- /dev/null +++ b/paperforge/core/errors.py @@ -0,0 +1,19 @@ +from __future__ import annotations + +from enum import Enum + + +class ErrorCode(str, Enum): + """Centralized error codes for PFResult contracts.""" + + PYTHON_NOT_FOUND = "PYTHON_NOT_FOUND" + VERSION_MISMATCH = "VERSION_MISMATCH" + BBT_EXPORT_NOT_FOUND = "BBT_EXPORT_NOT_FOUND" + OCR_TOKEN_MISSING = "OCR_TOKEN_MISSING" + SYNC_FAILED = "SYNC_FAILED" + VALIDATION_ERROR = "VALIDATION_ERROR" + INTERNAL_ERROR = "INTERNAL_ERROR" + UNKNOWN = "UNKNOWN" + + def __str__(self) -> str: + return self.value diff --git a/paperforge/core/result.py b/paperforge/core/result.py new file mode 100644 index 00000000..5ab062ee --- /dev/null +++ b/paperforge/core/result.py @@ -0,0 +1,67 @@ +from __future__ import annotations + +import json +from dataclasses import dataclass, field +from typing import Any + +from paperforge.core.errors import ErrorCode + + +@dataclass +class PFError: + code: ErrorCode + message: str + details: dict = field(default_factory=dict) + + +@dataclass +class PFResult: + ok: bool + command: str + version: str + data: Any = None + error: PFError | None = None + + def __bool__(self) -> bool: + return self.ok + + def to_dict(self) -> dict: + raw: dict[str, Any] = { + "ok": self.ok, + "command": self.command, + "version": self.version, + } + if self.data is not None: + raw["data"] = self.data + else: + raw["data"] = None + if self.error is not None: + raw["error"] = { + "code": self.error.code.value, + "message": self.error.message, + "details": self.error.details, + } + else: + raw["error"] = None + return raw + + def to_json(self) -> str: + return json.dumps(self.to_dict(), indent=2, ensure_ascii=False) + + @classmethod + def from_dict(cls, data: dict) -> PFResult: + error = None + if data.get("error") is not None: + err_data = data["error"] + error = PFError( + code=ErrorCode(err_data["code"]), + message=err_data["message"], + details=err_data.get("details", {}), + ) + return cls( + ok=data["ok"], + command=data["command"], + version=data["version"], + data=data.get("data"), + error=error, + ) diff --git a/paperforge/core/state.py b/paperforge/core/state.py new file mode 100644 index 00000000..63123e4a --- /dev/null +++ b/paperforge/core/state.py @@ -0,0 +1,105 @@ +"""State machine enums and allowed transitions for PaperForge lifecycle.""" + +from __future__ import annotations + +from enum import Enum +from typing import ClassVar + + +class OcrStatus(str, Enum): + """OCR processing state for a paper.""" + PENDING = "pending" + PROCESSING = "processing" + DONE = "done" + FAILED = "failed" + + @classmethod + def from_legacy(cls, value: str) -> OcrStatus: + """Map legacy OCR state strings to canonical enum.""" + mapping = { + "pending": cls.PENDING, + "processing": cls.PROCESSING, + "queued": cls.PENDING, + "running": cls.PROCESSING, + "done": cls.DONE, + "failed": cls.FAILED, + "error": cls.FAILED, + "blocked": cls.FAILED, + "nopdf": cls.FAILED, + "done_incomplete": cls.DONE, + } + return mapping.get(value.strip().lower(), cls.PENDING) + + def __str__(self) -> str: + return self.value + + +class PdfStatus(str, Enum): + """PDF attachment health state.""" + HEALTHY = "healthy" + BROKEN = "broken" + MISSING = "missing" + + def __str__(self) -> str: + return self.value + + +class Lifecycle(str, Enum): + """Derived lifecycle stage summarizing the paper's overall progress.""" + PDF_READY = "pdf_ready" + OCR_READY = "ocr_ready" + ANALYZE_READY = "analyze_ready" + DEEP_READ_DONE = "deep_read_done" + ERROR_STATE = "error_state" + + def __str__(self) -> str: + return self.value + + +class ALLOWED_TRANSITIONS: + """Legal state migration tables. + + Workers MUST validate transitions via check() before writing state changes. + """ + + OCR_STATUS: ClassVar[dict[OcrStatus, list[OcrStatus]]] = { + OcrStatus.PENDING: [OcrStatus.PROCESSING, OcrStatus.FAILED], + OcrStatus.PROCESSING: [OcrStatus.DONE, OcrStatus.FAILED], + OcrStatus.DONE: [OcrStatus.PENDING], # re-run + OcrStatus.FAILED: [OcrStatus.PENDING], # retry + } + + DEEP_READING_STATUS: ClassVar[dict[str, list[str]]] = { + "pending": ["done"], + "done": ["pending"], # re-run with --force + } + + LIFECYCLE: ClassVar[dict[Lifecycle, list[Lifecycle]]] = { + Lifecycle.PDF_READY: [Lifecycle.OCR_READY, Lifecycle.ERROR_STATE], + Lifecycle.OCR_READY: [Lifecycle.ANALYZE_READY, Lifecycle.ERROR_STATE], + Lifecycle.ANALYZE_READY: [Lifecycle.DEEP_READ_DONE, Lifecycle.ERROR_STATE], + Lifecycle.DEEP_READ_DONE: [Lifecycle.PDF_READY, Lifecycle.ERROR_STATE], # re-run + Lifecycle.ERROR_STATE: [Lifecycle.PDF_READY], # recover + } + + @staticmethod + def check_ocr(current: OcrStatus, target: OcrStatus) -> tuple[bool, str]: + """Validate OCR status transition.""" + allowed = ALLOWED_TRANSITIONS.OCR_STATUS.get(current, []) + if target in allowed: + return True, "" + return False, ( + f"Illegal OCR transition: {current.value} -> {target.value}. " + f"Allowed from '{current.value}': {[s.value for s in allowed]}" + ) + + @staticmethod + def check_lifecycle(current: Lifecycle, target: Lifecycle) -> tuple[bool, str]: + """Validate lifecycle transition.""" + allowed = ALLOWED_TRANSITIONS.LIFECYCLE.get(current, []) + if target in allowed: + return True, "" + return False, ( + f"Illegal lifecycle transition: {current.value} -> {target.value}. " + f"Allowed from '{current.value}': {[s.value for s in allowed]}" + ) diff --git a/paperforge/doctor/__init__.py b/paperforge/doctor/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/paperforge/doctor/field_validator.py b/paperforge/doctor/field_validator.py new file mode 100644 index 00000000..cdc4e542 --- /dev/null +++ b/paperforge/doctor/field_validator.py @@ -0,0 +1,157 @@ +"""Field registry validation — used by paperforge doctor to detect state drift.""" + +from __future__ import annotations + +from pathlib import Path +from typing import Any + + +def validate_entry_fields( + entry: dict, + owner: str, + registry: dict, + entry_label: str = "", +) -> list[dict]: + """Validate a single entry's fields against the field registry. + + Returns list of issues, each with: + - severity: "error" | "warning" | "info" + - code: "MISSING_REQUIRED" | "UNKNOWN_FIELD" | "DRIFT" | "TYPE_MISMATCH" + - message: human-readable description + - field: field name + """ + issues = [] + owner_fields = registry.get(owner, {}) + + if not owner_fields: + return issues + + # Check 1: Required fields present + for field_name, meta in owner_fields.items(): + if meta.get("required", False) and field_name not in entry: + issues.append({ + "severity": "error", + "code": "MISSING_REQUIRED", + "field": field_name, + "message": f"Missing required field '{field_name}' in {owner} entry{(' (' + entry_label + ')') if entry_label else ''}", + "suggestion": f"Add '{field_name}' to the entry with appropriate value ({meta.get('type', 'str')})", + }) + elif not meta.get("required", True) and field_name not in entry: + issues.append({ + "severity": "info", + "code": "MISSING_OPTIONAL", + "field": field_name, + "message": f"Missing optional field '{field_name}' in {owner} entry{(' (' + entry_label + ')') if entry_label else ''}", + }) + + # Check 2: Unknown fields (drift detection) + known_fields = set(owner_fields.keys()) + for key in entry: + if key not in known_fields: + issues.append({ + "severity": "warning", + "code": "DRIFT", + "field": key, + "message": f"Unknown field '{key}' in {owner} entry{(' (' + entry_label + ')') if entry_label else ''} — not in field registry", + "suggestion": f"Either remove '{key}' or add it to field_registry.yaml under '{owner}'", + }) + + return issues + + +def validate_collection( + entries: list[dict], + owner: str, + registry: dict, +) -> dict: + """Validate a collection of entries against field registry. + + Returns aggregate diagnostics: + - total_entries: int + - entries_with_errors: int + - entries_with_warnings: int + - issues: list[dict] (all individual issues) + - summary: dict (per-code counts) + - summary_text: str (human-readable) + """ + all_issues = [] + error_entries = set() + warning_entries = set() + + for i, entry in enumerate(entries): + label = entry.get("zotero_key", f"entry_{i}") + issues = validate_entry_fields(entry, owner, registry, label) + has_error = False + has_warning = False + for issue in issues: + all_issues.append(issue) + if issue["severity"] == "error": + has_error = True + elif issue["severity"] == "warning": + has_warning = True + if has_error: + error_entries.add(label) + if has_warning: + warning_entries.add(label) + + # Summary + code_counts: dict[str, int] = {} + for issue in all_issues: + code = issue["code"] + code_counts[code] = code_counts.get(code, 0) + 1 + + lines = [] + if all_issues: + lines.append(f"Found {len(all_issues)} field issue(s) across {len(entries)} {owner} entries:") + lines.append(f" Entries with errors: {len(error_entries)}") + lines.append(f" Entries with warnings: {len(warning_entries)}") + for code, count in sorted(code_counts.items()): + lines.append(f" {code}: {count}") + else: + lines.append(f"All {len(entries)} {owner} entries are valid against field registry.") + + # Add per-entry details for actionable items + if all_issues: + lines.append("") + lines.append("Details:") + for issue in all_issues[:20]: # cap display + lines.append(f" [{issue['severity'].upper()}] {issue['message']}") + if issue.get("suggestion"): + lines.append(f" Suggestion: {issue['suggestion']}") + if len(all_issues) > 20: + lines.append(f" ... and {len(all_issues) - 20} more issues") + + return { + "total_entries": len(entries), + "entries_with_errors": len(error_entries), + "entries_with_warnings": len(warning_entries), + "issues": all_issues, + "summary": code_counts, + "summary_text": "\n".join(lines), + } + + +def validate_frontmatter_from_file( + file_path: Path, + registry: dict, +) -> list[dict]: + """Read a single formal note and validate its frontmatter against registry.""" + try: + text = file_path.read_text(encoding="utf-8") + except Exception: + return [{"severity": "error", "code": "READ_ERROR", "field": "", "message": f"Cannot read {file_path}"}] + + # Parse frontmatter + import re + fm_match = re.match(r"^---\s*\n(.*?)\n---", text, re.DOTALL) + if not fm_match: + return [{"severity": "warning", "code": "NO_FRONTMATTER", "field": "", "message": f"No frontmatter found in {file_path.name}"}] + + frontmatter = {} + for line in fm_match.group(1).splitlines(): + line = line.strip() + if ":" in line: + key, _, val = line.partition(":") + frontmatter[key.strip()] = val.strip() + + return validate_entry_fields(frontmatter, "frontmatter", registry, file_path.stem) diff --git a/paperforge/plugin/main.js b/paperforge/plugin/main.js index 99ae34d6..3d25c0d9 100644 --- a/paperforge/plugin/main.js +++ b/paperforge/plugin/main.js @@ -336,7 +336,6 @@ class PaperForgeStatusView extends ItemView { } _fetchStats(quiet) { - // Phase 25: Read canonical index JSON directly (D-05) if (!quiet && !this._cachedStats) { this._metricsEl.empty(); this._metricsEl.createEl('div', { cls: 'paperforge-status-loading', text: 'Loading...' }); @@ -346,6 +345,47 @@ class PaperForgeStatusView extends ItemView { const vp = this.app.vault.adapter.basePath; const plugin = this.app.plugins.plugins['paperforge']; + const { path: pythonExe, extraArgs = [] } = resolvePythonExecutable(vp, plugin?.settings); + + // Plan 57-04: Try paperforge dashboard --json first, fallback to file + execFile(pythonExe, [...extraArgs, '-m', 'paperforge', 'dashboard', '--json'], { cwd: vp, timeout: 30000 }, (err, stdout) => { + if (!err) { + try { + const body = JSON.parse(stdout); + if (body.ok && body.data) { + const d = this._normalizeDashboardData(body.data); + this._cachedStats = d; + this._metricsEl.empty(); + this._renderStats(d); + this._renderOcr(d); + this._dashboardPermissions = body.data.permissions || {}; + return; + } + } catch (_) { + // Fall through to file fallback + } + } + // Fallback: read formal-library.json directly + this._fallbackFetchStats(quiet, vp, plugin); + }); + } + + _normalizeDashboardData(data) { + const stats = data.stats || {}; + const ocrHealth = stats.ocr_health || {}; + const pdfHealth = stats.pdf_health || {}; + const ocrTotal = (ocrHealth.done || 0) + (ocrHealth.pending || 0) + (ocrHealth.failed || 0); + return { + total_papers: stats.papers || 0, + formal_notes: stats.papers || 0, + exports: 0, + bases: 0, + ocr: { total: ocrTotal, pending: ocrHealth.pending || 0, processing: 0, done: ocrHealth.done || 0, failed: ocrHealth.failed || 0 }, + path_errors: (pdfHealth.broken || 0) + (pdfHealth.missing || 0), + }; + } + + _fallbackFetchStats(quiet, vp, plugin) { const systemDir = plugin?.settings?.system_dir || 'System'; const indexPath = path.join(vp, systemDir, 'PaperForge', 'indexes', 'formal-library.json'); @@ -402,11 +442,10 @@ class PaperForgeStatusView extends ItemView { this._renderStats(this._cachedStats); this._renderOcr(this._cachedStats); } catch (err) { - // D-07: Fallback — spawn CLI if file is missing or corrupt + // D-07: Second fallback — spawn CLI if file is missing or corrupt if (!quiet && !this._cachedStats) { this._metricsEl.createEl('div', { cls: 'paperforge-status-loading', text: 'No index \u2014 trying CLI...' }); } - const plugin = this.app.plugins.plugins['paperforge']; const { path: pythonExe, extraArgs = [] } = resolvePythonExecutable(vp, plugin?.settings); execFile(pythonExe, [...extraArgs, '-m', 'paperforge', 'status', '--json'], { cwd: vp, timeout: 30000 }, (err2, stdout) => { if (err2) { @@ -1837,7 +1876,7 @@ class PaperForgeSettingTab extends PluginSettingTab { _syncRuntime(btn) { const vp = this.app.vault.adapter.basePath; const { path: pythonExe, extraArgs = [] } = resolvePythonExecutable(vp, this.plugin.settings); - const ver = this.plugin.manifest.version || '1.4.17rc2'; + const ver = this.plugin.manifest.version; const installCmd = buildRuntimeInstallCommand(pythonExe, ver, extraArgs); btn.setDisabled(true); @@ -2417,7 +2456,7 @@ class PaperForgeSetupModal extends Modal { if (!hasPaperforge) { this._log(t('install_bootstrapping')); - const ver = this.plugin.manifest.version || '1.4.17rc2'; + const ver = this.plugin.manifest.version; await runPython([ '-m', 'pip', 'install', '--upgrade', `git+https://github.com/LLLin000/PaperForge.git@${ver}`, @@ -2668,7 +2707,7 @@ module.exports = class PaperForgePlugin extends Plugin { _autoUpdate() { const vp = this.app.vault.adapter.basePath; const { path: pythonExe, extraArgs = [] } = resolvePythonExecutable(vp, this.settings); - const ver = this.manifest.version || '1.4.17rc2'; + const ver = this.manifest.version; const url = `git+https://github.com/LLLin000/PaperForge.git@${ver}`; // Check if installed package version matches plugin version diff --git a/paperforge/plugin/manifest.json b/paperforge/plugin/manifest.json index bd59966f..7512c58d 100644 --- a/paperforge/plugin/manifest.json +++ b/paperforge/plugin/manifest.json @@ -1,7 +1,7 @@ { "id": "paperforge", "name": "PaperForge", - "version": "1.4.17rc3", + "version": "1.4.17rc4", "minAppVersion": "1.9.0", "description": "PaperForge — Zotero literature pipeline. Sync PDFs, run OCR, and read with AI-assisted deep reading.", "author": "Lin Zhaoxuan", diff --git a/paperforge/plugin/versions.json b/paperforge/plugin/versions.json index b07b3407..4e06cb27 100644 --- a/paperforge/plugin/versions.json +++ b/paperforge/plugin/versions.json @@ -1,4 +1,5 @@ { - "1.4.3": "1.9.0", - "1.4.17rc3": "1.9.0" + "1.4.3": "1.0.0", + "1.4.17rc3": "1.9.0", + "1.4.17rc4": "1.9.0" } \ No newline at end of file diff --git a/paperforge/schema/__init__.py b/paperforge/schema/__init__.py new file mode 100644 index 00000000..c5e9b674 --- /dev/null +++ b/paperforge/schema/__init__.py @@ -0,0 +1,35 @@ +"""Field registry: schema definitions for all field-carrying structures.""" + +from __future__ import annotations + +from pathlib import Path + + +def load_field_registry(path: Path | None = None) -> dict: + """Load field registry from YAML file. + + Returns nested dict: {owner: {field: {type, required, public, description}}} + """ + import yaml + + if path is None: + path = Path(__file__).resolve().parent / "field_registry.yaml" + + if not path.exists(): + return {} + + with open(path, "r", encoding="utf-8") as f: + data = yaml.safe_load(f) + + return data if isinstance(data, dict) else {} + + +def get_owner_fields(registry: dict, owner: str) -> dict: + """Get all field definitions for a given owner.""" + return registry.get(owner, {}) + + +def get_field_info(registry: dict, owner: str, field_name: str) -> dict | None: + """Get metadata for a specific field in an owner.""" + owner_fields = get_owner_fields(registry, owner) + return owner_fields.get(field_name) diff --git a/paperforge/schema/field_registry.yaml b/paperforge/schema/field_registry.yaml new file mode 100644 index 00000000..27aafb30 --- /dev/null +++ b/paperforge/schema/field_registry.yaml @@ -0,0 +1,215 @@ +frontmatter: + zotero_key: + type: str + required: true + public: true + description: "Zotero citation key" + domain: + type: str + required: true + public: true + description: "Classification domain" + title: + type: str + required: true + public: true + description: "Paper title" + year: + type: str + required: false + public: true + description: "Publication year" + doi: + type: str + required: false + public: true + description: "Digital Object Identifier" + collection_path: + type: str + required: false + public: true + description: "Zotero collection path" + has_pdf: + type: bool + required: true + public: true + description: "Has PDF attachment" + pdf_path: + type: str + required: false + public: true + description: "Wikilink to main PDF" + supplementary: + type: list + required: false + public: true + description: "Wikilinks to supplementary PDFs" + fulltext_md_path: + type: str + required: false + public: true + description: "Wikilink to OCR fulltext" + recommend_analyze: + type: bool + required: false + public: true + description: "System recommends analysis" + analyze: + type: bool + required: false + public: true + description: "User requests deep reading" + do_ocr: + type: bool + required: false + public: true + description: "User requests OCR" + ocr_status: + type: str + required: false + public: true + description: "OCR processing status" + deep_reading_status: + type: str + required: false + public: true + description: "Deep reading status" + path_error: + type: str + required: false + public: true + description: "Path resolution error" + +index_entry: + file: + type: str + required: true + public: true + description: "Path to formal note" + zotero_key: + type: str + required: true + public: true + description: "Zotero citation key" + domain: + type: str + required: true + public: true + description: "Classification domain" + title: + type: str + required: true + public: true + description: "Paper title" + year: + type: str + required: false + public: true + description: "Publication year" + doi: + type: str + required: false + public: true + description: "DOI" + collection_path: + type: str + required: false + public: true + description: "Zotero collection path" + has_pdf: + type: bool + required: true + public: true + description: "Has PDF" + pdf_path: + type: str + required: false + public: true + description: "PDF path" + supplementary: + type: list + required: false + public: true + description: "Supplementary PDFs" + fulltext_md_path: + type: str + required: false + public: true + description: "Fulltext path" + recommend_analyze: + type: bool + required: false + public: true + description: "Recommend analysis" + analyze: + type: bool + required: false + public: true + description: "Analysis flag" + do_ocr: + type: bool + required: false + public: true + description: "OCR flag" + ocr_status: + type: str + required: false + public: true + description: "OCR status" + deep_reading_status: + type: str + required: false + public: true + description: "Deep reading status" + path_error: + type: str + required: false + public: true + description: "Path error" + lifecycle: + type: str + required: false + public: true + description: "Derived lifecycle stage" + health: + type: str + required: false + public: true + description: "Derived health dimension scores" + maturity: + type: str + required: false + public: true + description: "Derived maturity score" + next_step: + type: str + required: false + public: true + description: "Recommended next action" + +ocr_meta: + ocr_status: + type: str + required: true + public: true + description: "OCR processing status" + error: + type: str + required: false + public: true + description: "Error message if failed" + start_time: + type: str + required: false + public: false + description: "OCR start timestamp" + completion_time: + type: str + required: false + public: false + description: "OCR completion timestamp" + page_count: + type: int + required: false + public: false + description: "Number of pages processed" diff --git a/paperforge/services/__init__.py b/paperforge/services/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/paperforge/services/sync_service.py b/paperforge/services/sync_service.py new file mode 100644 index 00000000..786ed7b9 --- /dev/null +++ b/paperforge/services/sync_service.py @@ -0,0 +1,78 @@ +from __future__ import annotations + +import logging +from pathlib import Path +from typing import Any + +from paperforge.adapters.bbt import ( + extract_authors, + _identify_main_pdf, + _normalize_attachment_path, + load_export_rows, + resolve_item_collection_paths, +) +from paperforge.adapters.obsidian_frontmatter import ( + candidate_markdown, + read_frontmatter_dict, + update_frontmatter_field, +) +from paperforge.adapters.zotero_paths import ( + absolutize_vault_path, + obsidian_wikilink_for_path, + obsidian_wikilink_for_pdf, +) + +logger = logging.getLogger(__name__) + + +class SyncService: + """Orchestrates the sync lifecycle: load BBT exports, match entries, generate notes.""" + + def __init__(self, vault: Path): + self.vault = vault + self.paths: dict[str, Path] = {} + self._resolved = False + + def resolve_paths(self) -> dict[str, Path]: + """Resolve vault paths using pipeline_paths.""" + from paperforge.worker._utils import pipeline_paths + self.paths = pipeline_paths(self.vault) + self._resolved = True + return self.paths + + def load_exports(self) -> list[dict]: + """Load all BBT export JSON files.""" + from paperforge.config import load_vault_config + cfg = load_vault_config(self.vault) + system_dir = self.vault / cfg.get("system_dir", "99_System") + exports_dir = system_dir / "PaperForge" / "exports" + if not exports_dir.exists(): + logger.warning("Exports directory not found: %s", exports_dir) + return [] + + rows = [] + for f in sorted(exports_dir.glob("*.json")): + try: + rows.extend(load_export_rows(f)) + except Exception as e: + logger.error("Failed to load %s: %s", f.name, e) + return rows + + def build_candidates(self, rows: list[dict]) -> list[dict]: + """Generate candidate markdown entries from BBT rows.""" + candidates = [] + for row in rows: + try: + markdown = candidate_markdown(row) + candidates.append({"row": row, "markdown": markdown}) + except Exception as e: + logger.error("Failed to build candidate: %s", e) + return candidates + + def run_sync(self, verbose: bool = False, json_output: bool = False) -> int | dict: + """Full sync orchestration. Delegates to run_selection_sync logic. + + Returns int (exit code) for CLI mode, dict for json_output mode. + """ + from paperforge.worker.sync import run_selection_sync + return run_selection_sync(self.vault, verbose=verbose, json_output=json_output) diff --git a/paperforge/setup/__init__.py b/paperforge/setup/__init__.py new file mode 100644 index 00000000..c6981f74 --- /dev/null +++ b/paperforge/setup/__init__.py @@ -0,0 +1,24 @@ +"""PaperForge setup package — modular setup components.""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any + + +@dataclass +class SetupStepResult: + """Result of a single setup step.""" + step: str + ok: bool = True + message: str = "" + error: str | None = None + details: dict[str, Any] = field(default_factory=dict) + + def to_dict(self) -> dict[str, Any]: + return { + "step": self.step, + "ok": self.ok, + "message": self.message, + "error": self.error, + } diff --git a/paperforge/setup/agent.py b/paperforge/setup/agent.py new file mode 100644 index 00000000..156e7d5f --- /dev/null +++ b/paperforge/setup/agent.py @@ -0,0 +1,133 @@ +"""AgentInstaller — deploys skill files and agent configs.""" + +from __future__ import annotations + +import shutil +from pathlib import Path + +from paperforge.setup import SetupStepResult + + +class AgentInstaller: + """Deploy agent skill files, command files, and rules.""" + + def __init__(self, vault: Path, agent_type: str = "opencode"): + self.vault = vault + self.agent_type = agent_type + self._script_dir = Path(__file__).resolve().parent.parent + + def _get_skills_dir(self) -> Path: + """Get the target skills directory based on agent type.""" + if self.agent_type == "opencode": + base = Path.home() / ".config" / "opencode" + elif self.agent_type == "claude": + base = Path.home() / ".claude" + elif self.agent_type == "codex": + base = Path.home() / ".codex" + else: + base = self.vault / ".agents" + return base / "skills" + + def deploy_skills(self) -> SetupStepResult: + """Deploy literature-qa skill directory to agent config.""" + source_skills = self._script_dir / "skills" / "literature-qa" + if not source_skills.exists(): + return SetupStepResult( + step="agent_installer", + ok=False, + message="Skill source directory not found", + error=f"Not found: {source_skills}", + ) + + target_dir = self._get_skills_dir() / "literature-qa" + target_dir.mkdir(parents=True, exist_ok=True) + + try: + if source_skills.is_dir(): + shutil.copytree(source_skills, target_dir, dirs_exist_ok=True) + else: + shutil.copy2(source_skills, target_dir) + + return SetupStepResult( + step="agent_installer", + ok=True, + message=f"Deployed literature-qa skill to {target_dir}", + details={"source": str(source_skills), "target": str(target_dir)}, + ) + except Exception as e: + return SetupStepResult( + step="agent_installer", + ok=False, + message="Failed to deploy skills", + error=str(e), + ) + + def deploy_commands(self) -> SetupStepResult: + """Deploy command files to vault agent config dir.""" + source_commands = self._script_dir / "command_files" + if not source_commands.exists(): + return SetupStepResult( + step="agent_installer", + ok=True, + message="No command files to deploy", + details={"skipped": True}, + ) + + agent_dir = self.vault / ".agents" + target_dir = agent_dir / "command_files" + target_dir.mkdir(parents=True, exist_ok=True) + + try: + for f in source_commands.iterdir(): + if f.is_file(): + shutil.copy2(f, target_dir / f.name) + + return SetupStepResult( + step="agent_installer", + ok=True, + message=f"Deployed command files to {target_dir}", + details={"source": str(source_commands), "target": str(target_dir)}, + ) + except Exception as e: + return SetupStepResult( + step="agent_installer", + ok=False, + message="Failed to deploy command files", + error=str(e), + ) + + def deploy_agent_config(self) -> SetupStepResult: + """Deploy AGENTS.md and other agent config files.""" + source_agents = self._script_dir.parent / "AGENTS.md" + if not source_agents.exists(): + return SetupStepResult( + step="agent_installer", + ok=True, + message="No AGENTS.md to deploy", + details={"skipped": True}, + ) + + try: + target = self.vault / "AGENTS.md" + shutil.copy2(source_agents, target) + return SetupStepResult( + step="agent_installer", + ok=True, + message=f"Deployed AGENTS.md to {target}", + details={"source": str(source_agents), "target": str(target)}, + ) + except Exception as e: + return SetupStepResult( + step="agent_installer", + ok=False, + message="Failed to deploy AGENTS.md", + error=str(e), + ) + + def run_all(self) -> list[SetupStepResult]: + """Run all deployment steps.""" + results = [] + results.append(self.deploy_skills()) + results.append(self.deploy_commands()) + results.append(self.deploy_agent_config()) + return results diff --git a/paperforge/setup/checker.py b/paperforge/setup/checker.py new file mode 100644 index 00000000..5f03b7a1 --- /dev/null +++ b/paperforge/setup/checker.py @@ -0,0 +1,68 @@ +"""SetupChecker — validates preconditions before setup begins.""" + +from __future__ import annotations + +import shutil +import subprocess +import sys +from pathlib import Path +from typing import Any + +from paperforge.setup import SetupStepResult + + +class SetupChecker: + """Validate all preconditions before any installation step.""" + + MIN_PYTHON = (3, 10) + + def __init__(self, vault: Path): + self.vault = vault + + def run(self) -> SetupStepResult: + """Run all precondition checks. Returns result with details.""" + issues: list[str] = [] + details: dict[str, Any] = {} + + # Check Python version + py_version = sys.version_info[:2] + details["python_version"] = f"{py_version[0]}.{py_version[1]}" + if py_version < self.MIN_PYTHON: + issues.append(f"Python {py_version[0]}.{py_version[1]} < {self.MIN_PYTHON[0]}.{self.MIN_PYTHON[1]}") + + # Check pip availability + pip_path = shutil.which("pip") or shutil.which("pip3") + details["pip_found"] = pip_path is not None + if not pip_path: + issues.append("pip not found in PATH") + + # Check vault directory + details["vault_exists"] = self.vault.exists() + if not self.vault.exists(): + issues.append(f"Vault directory not found: {self.vault}") + + # Check Zotero + zotero_path = shutil.which("zotero") or shutil.which("zotero.exe") + details["zotero_found"] = zotero_path is not None + + # Check Better BibTeX exports + system_dir = self.vault / "99_System" + exports_dir = system_dir / "PaperForge" / "exports" + bbt_exists = exports_dir.exists() and len(list(exports_dir.glob("*.json"))) > 0 + details["bbt_exports_found"] = bbt_exists + + if issues: + return SetupStepResult( + step="checker", + ok=False, + message=f"Precondition checks failed ({len(issues)} issue(s))", + error="; ".join(issues), + details=details, + ) + + return SetupStepResult( + step="checker", + ok=True, + message="All preconditions met", + details=details, + ) diff --git a/paperforge/setup/config_writer.py b/paperforge/setup/config_writer.py new file mode 100644 index 00000000..b3ed8e82 --- /dev/null +++ b/paperforge/setup/config_writer.py @@ -0,0 +1,80 @@ +"""ConfigWriter — atomic paperforge.json writer.""" + +from __future__ import annotations + +import json +import os +import tempfile +from pathlib import Path + +from paperforge.setup import SetupStepResult + + +class ConfigWriter: + """Write paperforge.json atomically using tempfile + os.replace.""" + + REQUIRED_KEYS = ["system_dir", "resources_dir", "literature_dir", "control_dir"] + + def __init__(self, vault: Path): + self.vault = vault + self.config_path = vault / "paperforge.json" + + def write(self, config: dict) -> SetupStepResult: + """Write paperforge.json atomically.""" + # Validate required keys + missing = [k for k in self.REQUIRED_KEYS if k not in config] + if missing: + return SetupStepResult( + step="config_writer", + ok=False, + message="Missing required config keys", + error=f"Missing: {', '.join(missing)}", + details={"missing_keys": missing}, + ) + + try: + # Write to temp file, then atomic replace + fd, tmp_path = tempfile.mkstemp( + suffix=".json", + prefix="paperforge_", + dir=str(self.vault), + ) + try: + with os.fdopen(fd, "w", encoding="utf-8") as f: + json.dump(config, f, indent=2, ensure_ascii=False) + os.replace(tmp_path, str(self.config_path)) + except Exception: + # Clean up temp file on failure + try: + os.unlink(tmp_path) + except Exception: + pass + raise + + return SetupStepResult( + step="config_writer", + ok=True, + message=f"paperforge.json written to {self.config_path}", + details={"path": str(self.config_path), "keys": list(config.keys())}, + ) + except Exception as e: + return SetupStepResult( + step="config_writer", + ok=False, + message="Failed to write config", + error=str(e), + ) + + def read(self) -> dict | None: + """Read existing config, return None if not exists.""" + if not self.config_path.exists(): + return None + try: + with open(self.config_path, "r", encoding="utf-8") as f: + return json.load(f) + except Exception: + return None + + def exists(self) -> bool: + """Check if config file exists.""" + return self.config_path.exists() diff --git a/paperforge/setup/plan.py b/paperforge/setup/plan.py new file mode 100644 index 00000000..836492b6 --- /dev/null +++ b/paperforge/setup/plan.py @@ -0,0 +1,97 @@ +"""SetupPlan — orchestrates all setup steps in sequence.""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path +from typing import Callable + +from paperforge.setup import SetupStepResult +from paperforge.setup.checker import SetupChecker +from paperforge.setup.config_writer import ConfigWriter +from paperforge.setup.vault import VaultInitializer +from paperforge.setup.runtime import RuntimeInstaller +from paperforge.setup.agent import AgentInstaller + +ProgressCallback = Callable[[str], None] + + +class SetupPlan: + """Orchestrate the setup lifecycle: check -> config -> vault -> install -> agent.""" + + def __init__( + self, + vault: Path, + config: dict | None = None, + env_values: dict[str, str] | None = None, + zotero_path: str | None = None, + agent_type: str = "opencode", + version: str | None = None, + progress_callback: ProgressCallback | None = None, + ): + self.vault = vault + self.config = config or {} + self.env_values = env_values or {} + self.zotero_path = zotero_path + self.agent_type = agent_type + self.version = version + self.progress_callback = progress_callback + + def _log(self, message: str) -> None: + if self.progress_callback: + self.progress_callback(message) + + def execute(self, json_output: bool = False) -> list[SetupStepResult] | int: + """Run all setup steps in sequence. + + Args: + json_output: If True, return results list as JSON string. + If False, return exit code (0 = success, 1 = failure). + """ + results: list[SetupStepResult] = [] + + # Step 1: Checker + self._log("Checking preconditions...") + checker = SetupChecker(self.vault) + results.append(checker.run()) + + # Step 2: Config writer + self._log("Writing config...") + writer = ConfigWriter(self.vault) + results.append(writer.write(self.config)) + + # Step 3: Vault initializer + self._log("Initializing vault structure...") + vault_init = VaultInitializer(self.vault, self.config) + results.append(vault_init.create_directories()) + results.append(vault_init.create_zotero_junction(self.zotero_path)) + results.append(vault_init.merge_env(self.env_values)) + + # Step 4: Runtime installer + self._log("Installing runtime...") + installer = RuntimeInstaller(self.vault, version=self.version, progress_callback=self._log) + results.append(installer.install()) + + # Step 5: Agent installer + self._log("Deploying agent config...") + agent = AgentInstaller(self.vault, agent_type=self.agent_type) + agent_results = agent.run_all() + results.extend(agent_results) + + if json_output: + output = [r.to_dict() for r in results] + print(json.dumps(output, indent=2, ensure_ascii=False)) + return 0 + + # Print summary + ok_count = sum(1 for r in results if r.ok) + total = len(results) + print(f"Setup complete: {ok_count}/{total} steps passed") + for r in results: + status = "PASS" if r.ok else "FAIL" + print(f" [{status}] {r.step}: {r.message}") + if not r.ok and r.error: + print(f" Error: {r.error}") + + return 0 if all(r.ok for r in results) else 1 diff --git a/paperforge/setup/runtime.py b/paperforge/setup/runtime.py new file mode 100644 index 00000000..12277fc7 --- /dev/null +++ b/paperforge/setup/runtime.py @@ -0,0 +1,103 @@ +"""RuntimeInstaller -- pip install with version pinning.""" + +from __future__ import annotations + +import subprocess +import sys +from pathlib import Path +from typing import Callable + +from paperforge.core.errors import ErrorCode +from paperforge.setup import SetupStepResult + + +ProgressCallback = Callable[[str], None] + + +class RuntimeInstaller: + """Install PaperForge Python package with version pinning.""" + + def __init__( + self, + vault: Path, + version: str | None = None, + progress_callback: ProgressCallback | None = None, + ): + self.vault = vault + self.version = version + self.progress_callback = progress_callback + + def _log(self, message: str) -> None: + if self.progress_callback: + self.progress_callback(message) + + def _pip_install( + self, package_spec: str + ) -> tuple[bool, str, str]: + """Run pip install and return (ok, stdout, stderr).""" + try: + result = subprocess.run( + [sys.executable, "-m", "pip", "install", package_spec], + capture_output=True, + text=True, + timeout=120, + ) + return (result.returncode == 0, result.stdout, result.stderr) + except subprocess.TimeoutExpired: + return (False, "", "pip install timed out after 120s") + except Exception as e: + return (False, "", str(e)) + + def install(self) -> SetupStepResult: + """Install paperforge via pip with optional version pin.""" + self._log("Installing PaperForge...") + + if self.version: + package_spec = ( + f"git+https://github.com/LLLin000/PaperForge.git@{self.version}" + ) + else: + package_spec = "git+https://github.com/LLLin000/PaperForge.git" + + ok, stdout, stderr = self._pip_install(package_spec) + + if ok: + return SetupStepResult( + step="runtime_installer", + ok=True, + message=( + f"PaperForge installed successfully{(' (' + self.version + ')') if self.version else ''}" + ), + details={"version": self.version or "latest", "stdout": stdout[:500]}, + ) + else: + error_code = ErrorCode.INTERNAL_ERROR + if "pip" in stderr.lower(): + error_code = ErrorCode.INTERNAL_ERROR + elif "connection" in stderr.lower() or "timeout" in stderr.lower(): + error_code = ErrorCode.INTERNAL_ERROR + + return SetupStepResult( + step="runtime_installer", + ok=False, + message="Failed to install PaperForge", + error=f"[{error_code.value}] {stderr[:500]}", + details={"version": self.version or "latest", "stderr": stderr[:500]}, + ) + + def check_installed(self) -> bool: + """Check if paperforge is already installed.""" + try: + result = subprocess.run( + [ + sys.executable, + "-c", + "from paperforge import __version__; print(__version__)", + ], + capture_output=True, + text=True, + timeout=10, + ) + return result.returncode == 0 and bool(result.stdout.strip()) + except Exception: + return False diff --git a/paperforge/setup/vault.py b/paperforge/setup/vault.py new file mode 100644 index 00000000..4a0da535 --- /dev/null +++ b/paperforge/setup/vault.py @@ -0,0 +1,150 @@ +"""VaultInitializer -- creates vault directory structure and env config.""" + +from __future__ import annotations + +import os +import subprocess +from pathlib import Path + +from paperforge.setup import SetupStepResult + + +class VaultInitializer: + """Create vault directory structure, Zotero junction, and .env file.""" + + DEFAULT_DIRS = { + "system_dir": "99_System", + "resources_dir": "03_Resources", + "literature_dir": "Literature", + "control_dir": "99_System/PaperForge", + } + + def __init__(self, vault: Path, config: dict): + self.vault = vault + self.config = config + + def create_directories(self) -> SetupStepResult: + """Create all required vault directories.""" + dirs_to_create = [ + self.vault / "paperforge.json", + ] + + for key in ("system_dir", "resources_dir", "literature_dir", "control_dir"): + rel = self.config.get(key, self.DEFAULT_DIRS.get(key, "")) + if rel: + dirs_to_create.append(self.vault / rel) + + created = [] + existing = [] + for d in dirs_to_create: + if d.suffix: + d.parent.mkdir(parents=True, exist_ok=True) + continue + if not d.exists(): + d.mkdir(parents=True, exist_ok=True) + created.append(str(d.relative_to(self.vault))) + else: + existing.append(str(d.relative_to(self.vault))) + + return SetupStepResult( + step="vault_initializer", + ok=True, + message=f"Created {len(created)} director(ies), {len(existing)} already exist", + details={"created": created, "existing": existing}, + ) + + def create_zotero_junction(self, zotero_path: str | None = None) -> SetupStepResult: + """Create Zotero junction/symlink to vault.""" + system_dir = self.vault / self.config.get( + "system_dir", self.DEFAULT_DIRS["system_dir"] + ) + zotero_link = system_dir / "Zotero" + + if zotero_link.exists() or zotero_link.is_symlink(): + return SetupStepResult( + step="vault_initializer", + ok=True, + message=f"Zotero link already exists at {zotero_link}", + details={"path": str(zotero_link)}, + ) + + if not zotero_path: + return SetupStepResult( + step="vault_initializer", + ok=True, + message="Zotero path not provided -- skipping junction creation", + details={"skipped": True}, + ) + + try: + if os.name == "nt": + result = subprocess.run( + ["cmd", "/c", "mklink", "/J", str(zotero_link), zotero_path], + capture_output=True, + text=True, + timeout=30, + ) + if result.returncode != 0: + return SetupStepResult( + step="vault_initializer", + ok=False, + message="Failed to create Zotero junction", + error=result.stderr.strip() or result.stdout.strip(), + ) + else: + zotero_link.symlink_to(zotero_path, target_is_directory=True) + + return SetupStepResult( + step="vault_initializer", + ok=True, + message=f"Zotero junction created: {zotero_link} -> {zotero_path}", + details={"source": str(zotero_link), "target": zotero_path}, + ) + except Exception as e: + return SetupStepResult( + step="vault_initializer", + ok=False, + message="Failed to create Zotero junction", + error=str(e), + ) + + def merge_env(self, env_values: dict[str, str]) -> SetupStepResult: + """Merge values into .env file.""" + env_path = self.vault / ".env" + + existing = {} + if env_path.exists(): + try: + for line in env_path.read_text(encoding="utf-8").splitlines(): + line = line.strip() + if line and not line.startswith("#") and "=" in line: + key, _, val = line.partition("=") + existing[key.strip()] = val.strip() + except Exception: + pass + + updated = dict(existing) + updated.update(env_values) + + try: + lines = [f"{k}={v}\n" for k, v in updated.items()] + env_path.write_text("".join(lines), encoding="utf-8") + + added = [k for k in env_values if k not in existing] + return SetupStepResult( + step="vault_initializer", + ok=True, + message=f"Created .env with {len(updated)} entries ({len(added)} new)", + details={ + "path": str(env_path), + "total_keys": len(updated), + "new_keys": added, + }, + ) + except Exception as e: + return SetupStepResult( + step="vault_initializer", + ok=False, + message="Failed to write .env", + error=str(e), + ) diff --git a/paperforge/setup_wizard.py b/paperforge/setup_wizard.py index 29665f9c..dd1dd243 100644 --- a/paperforge/setup_wizard.py +++ b/paperforge/setup_wizard.py @@ -23,6 +23,10 @@ from pathlib import Path from paperforge import __version__ +# Backward-compat imports (v2.1 modular setup) +from paperforge.setup.checker import SetupChecker +from paperforge.setup.config_writer import ConfigWriter + if sys.platform == "win32": import winreg else: diff --git a/paperforge/worker/asset_index.py b/paperforge/worker/asset_index.py index 483ba215..72c5fb21 100644 --- a/paperforge/worker/asset_index.py +++ b/paperforge/worker/asset_index.py @@ -396,7 +396,7 @@ def _build_entry(item: dict, vault: Path, paths: dict, domain: str, zotero_dir: } # ---- derived state (Phase 24: lifecycle, health, maturity, next-step) ---- - entry["lifecycle"] = compute_lifecycle(entry) + entry["lifecycle"] = str(compute_lifecycle(entry)) entry["health"] = compute_health(entry) entry["maturity"] = compute_maturity(entry) entry["next_step"] = compute_next_step(entry) diff --git a/paperforge/worker/asset_state.py b/paperforge/worker/asset_state.py index 2ca81af3..57603eb3 100644 --- a/paperforge/worker/asset_state.py +++ b/paperforge/worker/asset_state.py @@ -13,6 +13,8 @@ Exports: from __future__ import annotations +from paperforge.core.state import Lifecycle + def compute_lifecycle(entry: dict) -> str: """Derive the current lifecycle state from an index entry. @@ -23,6 +25,10 @@ def compute_lifecycle(entry: dict) -> str: - **fulltext_ready**: ocr_status == "done" (validated), deep_reading NOT done - **deep_read_done**: OCR done AND deep-reading done + Returns ``Lifecycle`` enum members where an equivalent exists, or plain + strings for states not yet represented in the enum (``"indexed"``, + ``"fulltext_ready"``). + Uses .get() with safe defaults for all field accesses. """ has_pdf = bool(entry.get("has_pdf", False)) @@ -36,11 +42,11 @@ def compute_lifecycle(entry: dict) -> str: # OCR validated done opens the door to fulltext_ready and beyond if ocr_status == "done": if deep_reading_status == "done": - return "deep_read_done" + return Lifecycle.DEEP_READ_DONE return "fulltext_ready" # pdf_ready: has PDF attachment but OCR not validated done - return "pdf_ready" + return Lifecycle.PDF_READY def compute_health(entry: dict) -> dict[str, str]: diff --git a/paperforge/worker/status.py b/paperforge/worker/status.py index d28da255..ced9d656 100644 --- a/paperforge/worker/status.py +++ b/paperforge/worker/status.py @@ -24,6 +24,8 @@ from paperforge.worker._utils import ( ) from paperforge.worker.base_views import ensure_base_views +from paperforge.core.result import PFResult + logger = logging.getLogger(__name__) @@ -356,7 +358,7 @@ def _query_resolved_module(interp: str, extra_args: list[str], module_name: str) return None -def run_doctor(vault: Path, verbose: bool = False) -> int: +def run_doctor(vault: Path, verbose: bool = False, json_output: bool = False) -> int: """Validate PaperForge setup and report by category. Returns: @@ -445,12 +447,12 @@ def run_doctor(vault: Path, verbose: bool = False) -> int: ver = mod_info_resolved.get("version") ver_str = f" ({ver})" if ver else "" - # PyYAML specific: warn if version < 6.0 + # PyYAML specific: fail if version < 6.0 (hard dependency in pyproject.toml) if mod_name == "yaml" and ver: try: ver_parts = str(ver).split(".") if int(ver_parts[0]) < 6: - add_check("Python 环境", "warn", + add_check("Python 环境", "fail", f"{mod_info['label']} {ver} (需要 >=6.0)", f"运行: {interp} -m pip install {mod_info['pip']}") continue @@ -589,6 +591,34 @@ def run_doctor(vault: Path, verbose: bool = False) -> int: check_pdf_paths(vault, paths, add_check) check_wikilink_format(vault, paths, add_check) + # --- Field registry validation (Phase 59) --- + try: + from paperforge.schema import load_field_registry + from paperforge.doctor.field_validator import validate_frontmatter_from_file + + registry = load_field_registry() + except Exception: + registry = {} + if registry: + literature_dir = paths.get("literature") + if literature_dir and literature_dir.exists(): + note_files = sorted(literature_dir.rglob("*.md")) + total_issues = 0 + for note_file in note_files: + if note_file.name in ("fulltext.md", "deep-reading.md", "discussion.md"): + continue + issues = validate_frontmatter_from_file(note_file, registry) + for issue in issues: + total_issues += 1 + add_check( + "字段注册表", + "fail" if issue["severity"] == "error" else "warn" if issue["severity"] == "warning" else "pass", + issue["message"], + issue.get("suggestion", ""), + ) + if total_issues == 0: + add_check("字段注册表", "pass", "所有 formal note frontmatter 与字段注册表一致") + ld_deep_script = paths.get("ld_deep_script") skill_dir = None if ld_deep_script: @@ -760,6 +790,58 @@ def run_doctor(vault: Path, verbose: bool = False) -> int: else: add_check("Index Health", "info", "No canonical index -- run `paperforge sync` to generate") + if json_output: + checklist_data = [ + {"category": cat, "status": st, "message": msg, "fix": fx} + for cat, st, msg, fx in checks + ] + status_counts: dict[str, int] = {} + for _, st, _, _ in checks: + status_counts[st] = status_counts.get(st, 0) + 1 + _has_fail = any(status == "fail" for _, status, _, _ in checks) + _has_warn = any(status == "warn" for _, status, _, _ in checks) + _verdict = "FAIL" if _has_fail else ("WARN" if _has_warn else "OK") + _fail_categories = set() + for cat, status, _, fx in checks: + if status == "fail" and fx: + _fail_categories.add(cat) + if "PaperForge 包" in _fail_categories and not bool(_fail_categories - {"PaperForge 包"}): + _next_action = "Recommended: run pip install via the resolved interpreter" + elif "Python 环境" in _fail_categories or "Python 环境 (插件)" in _fail_categories: + _next_action = "Recommended: install/upgrade Python to 3.10+" + elif "Vault 结构" in _fail_categories or "Config Migration" in _fail_categories: + _next_action = "Recommended: run `paperforge sync`" + elif "OCR 配置" in _fail_categories: + _next_action = "Recommended: configure PADDLEOCR_API_TOKEN in .env" + elif "BBT 导出" in _fail_categories: + _next_action = "Recommended: configure Better BibTeX auto-export" + elif "Zotero 链接" in _fail_categories or "Path Resolution" in _fail_categories: + _next_action = "Recommended: create Zotero junction" + elif _has_warn: + _next_action = "Recommended: `paperforge doctor` indicates warnings - review above" + else: + _next_action = "Recommended: `paperforge sync` to ensure index is current" + payload = { + "checks": checklist_data, + "summary": { + "total": len(checks), + "pass": status_counts.get("pass", 0), + "warn": status_counts.get("warn", 0), + "fail": status_counts.get("fail", 0), + "info": status_counts.get("info", 0), + }, + "verdict": _verdict, + "next_action": _next_action, + } + result = PFResult( + ok=not _has_fail, + command="doctor", + version=__import__("paperforge").__version__, + data=payload, + ) + print(result.to_json()) + return 0 if not _has_fail else 1 + print("PaperForge Doctor") print("=" * 40) current_category = "" @@ -971,7 +1053,7 @@ def run_status(vault: Path, verbose: bool = False, json_output: bool = False) -> maturity_distribution = summary["maturity_distribution"] if json_output: - data = { + payload = { "version": __import__("paperforge").__version__, "vault": str(vault), "system_dir": cfg["system_dir"], @@ -995,7 +1077,13 @@ def run_status(vault: Path, verbose: bool = False, json_output: bool = False) -> "health_aggregate": health_aggregate, "maturity_distribution": maturity_distribution, } - print(_json.dumps(data, indent=2, ensure_ascii=False)) + result = PFResult( + ok=True, + command="status", + version=__import__("paperforge").__version__, + data=payload, + ) + print(result.to_json()) return 0 print("PaperForge status") diff --git a/paperforge/worker/sync.py b/paperforge/worker/sync.py index 6a9086dc..ce905175 100644 --- a/paperforge/worker/sync.py +++ b/paperforge/worker/sync.py @@ -14,6 +14,25 @@ import requests from paperforge.config import load_vault_config, paperforge_paths from paperforge.worker._domain import build_collection_lookup, load_domain_config, load_domain_collections +from paperforge.adapters.zotero_paths import ( + absolutize_vault_path, + obsidian_wikilink_for_path, + obsidian_wikilink_for_pdf, +) +from paperforge.adapters.obsidian_frontmatter import ( + _add_missing_frontmatter_fields, + _extract_section, + _legacy_control_flags, + _read_frontmatter_bool_from_text, + _read_frontmatter_optional_bool_from_text, + canonicalize_decision, + candidate_markdown, + compute_final_collection, + extract_preserved_deep_reading, + generate_review, + has_deep_reading_content, + update_frontmatter_field, +) from paperforge.worker._utils import ( _extract_year, lookup_impact_factor, @@ -28,40 +47,20 @@ from paperforge.worker._utils import ( yaml_quote, ) -import paperforge.worker.asset_index as asset_index +from paperforge.adapters.bbt import ( + _identify_main_pdf, + _normalize_attachment_path, + collection_fields, + extract_authors, + load_export_rows, + resolve_item_collection_paths, +) + +import paperforge.worker.asset_index as asset_index logger = logging.getLogger(__name__) -def _read_frontmatter_bool_from_text(text: str, key: str, default: bool = False) -> bool: - match = re.search(rf"^{re.escape(key)}:\s*(?:[\"'])?(true|false)(?:[\"'])?\s*$", text, re.MULTILINE | re.IGNORECASE) - if not match: - return default - return match.group(1).lower() == "true" - - -def _read_frontmatter_optional_bool_from_text(text: str, key: str) -> Optional[bool]: - match = re.search(rf"^{re.escape(key)}:\s*(?:[\"'])?(true|false)(?:[\"'])?\s*$", text, re.MULTILINE | re.IGNORECASE) - if not match: - return None - return match.group(1).lower() == "true" - - -def _legacy_control_flags(paths: dict[str, Path], zotero_key: str) -> dict[str, Optional[bool]]: - records_root = paths.get("library_records") - if not records_root or not records_root.exists(): - return {"do_ocr": None, "analyze": None} - for record_path in records_root.rglob(f"{zotero_key}.md"): - try: - text = record_path.read_text(encoding="utf-8") - except Exception: - continue - return { - "do_ocr": _read_frontmatter_optional_bool_from_text(text, "do_ocr"), - "analyze": _read_frontmatter_optional_bool_from_text(text, "analyze"), - } - return {"do_ocr": None, "analyze": None} - def load_export_inventory(paths: dict[str, Path]) -> dict[str, dict]: inventory = {"doi": {}, "pmid": {}, "title": {}} @@ -180,533 +179,6 @@ def apply_existing_library_match(row: dict, inventory: dict[str, dict]) -> dict: return resolved -def resolve_item_collection_paths(item: dict, collection_lookup: dict) -> list[str]: - paths = [] - collection_keys = item.get("collections") or [] - if collection_keys: - for key in collection_keys: - paths.append(collection_lookup.get("path_by_key", {}).get(key, key)) - item_id = item.get("itemID") - if item_id is not None: - paths.extend(collection_lookup.get("paths_by_item_id", {}).get(item_id, [])) - return sorted({path for path in paths if path}, key=lambda value: (-value.count("/"), value)) - - -def obsidian_wikilink_for_pdf(pdf_path: str, vault_dir: Path, zotero_dir: Path | None = None) -> str: - text = str(pdf_path or "").strip() - if not text: - return "" - # Handle storage: prefix paths by resolving through zotero_dir - if text.startswith("storage:") and zotero_dir is not None: - storage_rel = text[len("storage:") :].lstrip("/").lstrip("\\") - absolute_pdf_path = zotero_dir / "storage" / storage_rel.replace("/", os.sep) - absolute_str = str(absolute_pdf_path) - else: - absolute_str = absolutize_vault_path(vault_dir, text, resolve_junction=True) - if not absolute_str: - return "" - absolute_path = Path(absolute_str) - try: - relative = absolute_path.relative_to(vault_dir) - except ValueError: - # Path outside vault — try to route through Zotero junction inside vault - if zotero_dir is not None and zotero_dir.exists(): - try: - from paperforge.pdf_resolver import resolve_junction - real_zotero = resolve_junction(zotero_dir) - if real_zotero != zotero_dir: - rel_to_zotero = absolute_path.relative_to(real_zotero) - via_junction = zotero_dir / rel_to_zotero - relative = via_junction.relative_to(vault_dir) - return f"[[{relative.as_posix()}]]" - except (ValueError, OSError): - pass - return f"[[{absolute_path.as_posix()}]]" - return f"[[{relative.as_posix()}]]" - - -def absolutize_vault_path(vault: Path, path: str, resolve_junction: bool = False) -> str: - text = str(path or "").strip() - if not text: - return "" - candidate = Path(text) - result = str(candidate) if candidate.is_absolute() else str((vault / text.replace("/", os.sep)).resolve()) - if resolve_junction: - from paperforge.pdf_resolver import resolve_junction - - result = str(resolve_junction(Path(result))) - return result - - -def obsidian_wikilink_for_path(vault: Path, path: str) -> str: - absolute = absolutize_vault_path(vault, path, resolve_junction=True) - if not absolute: - return "" - absolute_path = Path(absolute) - try: - relative = absolute_path.relative_to(vault) - except ValueError: - return f"[[{absolute_path.as_posix()}]]" - return f"[[{relative.as_posix()}]]" - - -def collection_fields(collection_paths: list[str]) -> dict[str, str | list[str]]: - paths = [path for path in collection_paths if path] - primary = paths[0] if paths else "" - if paths: - primary = sorted(paths, key=lambda value: (value.count("/"), len(value), value), reverse=True)[0] - tags = [] - seen = set() - for path in paths: - for part in [segment.strip() for segment in path.split("/") if segment.strip()]: - if part not in seen: - seen.add(part) - tags.append(part) - group = primary - return {"collections": paths, "collection_tags": tags, "collection_group": [group] if group else []} - - -def extract_authors(item: dict) -> list[str]: - authors = [] - for creator in item.get("creators", []): - if creator.get("creatorType") != "author": - continue - full_name = " ".join( - part for part in [creator.get("firstName", ""), creator.get("lastName", "")] if part - ).strip() - if full_name: - authors.append(full_name) - elif creator.get("name"): - authors.append(creator["name"]) - return authors - - -def _normalize_attachment_path(path: str, zotero_dir: Path | None = None) -> tuple[str, str, str]: - """Normalize a BBT attachment path to a consistent storage: format. - - Handles three real-world BBT export formats: - 1. Absolute Windows paths: D:\\...\\Zotero\\storage\\8CHARKEY\\filename.pdf - -> storage:8CHARKEY/filename.pdf - 2. storage: prefix: storage:KEY/filename.pdf -> pass through - 3. Bare relative: KEY/filename.pdf -> storage:KEY/filename.pdf - - Args: - path: Raw path from BBT JSON attachment. - zotero_dir: Optional absolute path to Zotero data directory for - validating absolute paths. - - Returns: - Tuple of (normalized_path, bbt_path_raw, zotero_storage_key). - normalized_path uses forward slashes and storage: prefix for - Zotero storage paths. bbt_path_raw preserves the original input - for debugging. zotero_storage_key is the 8-character Zotero key. - """ - raw = str(path or "").strip() - if not raw: - return ("", "", "") - - bbt_path_raw = raw - - # Format 2: Already has storage: prefix — pass through with slash normalization - if raw.startswith("storage:"): - storage_rel = raw[len("storage:") :].lstrip("/").lstrip("\\") - storage_rel = storage_rel.replace("\\", "/") - parts = storage_rel.split("/") - zotero_storage_key = parts[0] if parts else "" - return (f"storage:{storage_rel}", bbt_path_raw, zotero_storage_key) - - # Format 1: Absolute Windows path pointing to Zotero storage - candidate = Path(raw) - if candidate.is_absolute(): - norm_path = raw.replace("\\", "/") - # Detect Zotero storage pattern: .../storage/8CHARKEY/... - if "/storage/" in norm_path: - parts_after_storage = norm_path.split("/storage/", 1)[1] - parts = parts_after_storage.split("/") - if len(parts) >= 2 and len(parts[0]) == 8 and parts[0].isalnum(): - zotero_storage_key = parts[0] - filename = "/".join(parts[1:]) - return (f"storage:{zotero_storage_key}/{filename}", bbt_path_raw, zotero_storage_key) - # Absolute path but not in Zotero storage — mark as absolute - return (f"absolute:{raw}", bbt_path_raw, "") - - # Format 3: Bare relative path — prepend storage: prefix - norm = raw.replace("\\", "/") - parts = norm.split("/") - zotero_storage_key = parts[0] if parts else "" - return (f"storage:{norm}", bbt_path_raw, zotero_storage_key) - - -def _identify_main_pdf(attachments: list[dict]) -> tuple[dict | None, list[dict]]: - """Identify the main PDF and supplementary materials from attachments. - - Uses a hybrid three-priority strategy (Decision D-02): - 1. Primary: attachment.title == "PDF" AND contentType == "application/pdf" - 2. Fallback heuristic: largest file by size (if available), else shortest title - 3. Final fallback: first PDF attachment in the list - - Args: - attachments: List of attachment dicts from load_export_rows(). - - Returns: - Tuple of (main_pdf_attachment, supplementary_attachments). - main_pdf_attachment may be None if no PDFs found. - supplementary_attachments is a list of all other PDF attachments. - """ - pdf_attachments = [a for a in attachments if isinstance(a, dict) and a.get("contentType") == "application/pdf"] - - if not pdf_attachments: - return (None, []) - - # Priority 1: Title exactly equals "PDF" - for att in pdf_attachments: - if att.get("title") == "PDF": - main = att - supplementary = [a for a in pdf_attachments if a is not main] - return (main, supplementary) - - # Priority 2: Largest file by size (if size field is available and differentiated) - sized = [(a, a.get("size", 0) or 0) for a in pdf_attachments] - sized.sort(key=lambda x: x[1], reverse=True) - if sized and sized[0][1] > 0 and (len(sized) == 1 or sized[0][1] > sized[1][1]): - main = sized[0][0] - supplementary = [a for a in pdf_attachments if a is not main] - return (main, supplementary) - - # Priority 2b (sizes equal or unavailable): shortest title - titled = [(a, len(str(a.get("title", "")))) for a in pdf_attachments] - titled.sort(key=lambda x: x[1]) - main = titled[0][0] - supplementary = [a for a in pdf_attachments if a is not main] - return (main, supplementary) - - -def load_export_rows(path: Path) -> list[dict]: - data = read_json(path) - if isinstance(data, list): - return data - if isinstance(data, dict) and isinstance(data.get("items"), list): - collection_lookup = build_collection_lookup(data.get("collections", {})) - rows = [] - for item in data["items"]: - if item.get("itemType") in {"attachment", "note", "annotation"}: - continue - attachments = [] - for attachment in item.get("attachments", []): - if not isinstance(attachment, dict): - continue - raw_path = attachment.get("path", "") - normalized_path, bbt_path_raw, zotero_storage_key = _normalize_attachment_path(raw_path) - # Preserve contentType from BBT if present; fallback to file extension - content_type = attachment.get("contentType", "") - if not content_type and str(normalized_path).lower().endswith(".pdf"): - content_type = "application/pdf" - attachments.append( - { - "path": normalized_path, - "contentType": content_type, - "title": attachment.get("title", ""), - "bbt_path_raw": bbt_path_raw, - "zotero_storage_key": zotero_storage_key, - "size": attachment.get("size", 0) or 0, - } - ) - main_pdf, supplementary_pdfs = _identify_main_pdf(attachments) - pdf_path = main_pdf["path"] if main_pdf else "" - bbt_path_raw = main_pdf["bbt_path_raw"] if main_pdf else "" - zotero_storage_key = main_pdf["zotero_storage_key"] if main_pdf else "" - path_error = "not_found" if not main_pdf else "" - supplementary = [a["path"] for a in supplementary_pdfs] if supplementary_pdfs else [] - attachment_count = len(attachments) - rows.append( - { - "key": item.get("key") or item.get("itemKey", ""), - "title": item.get("title", ""), - "authors": extract_authors(item), - "creators": item.get("creators", []), - "abstract": item.get("abstractNote", ""), - "journal": item.get("publicationTitle", ""), - "extra": item.get("extra", ""), - "year": _extract_year(item.get("date", "")), - "date": item.get("date", ""), - "doi": item.get("DOI", ""), - "pmid": item.get("PMID", ""), - "collections": resolve_item_collection_paths(item, collection_lookup), - "attachments": attachments, - "pdf_path": pdf_path, - "supplementary": supplementary, - "attachment_count": attachment_count, - "bbt_path_raw": bbt_path_raw, - "zotero_storage_key": zotero_storage_key, - "path_error": path_error, - } - ) - return rows - raise ValueError(f"Unsupported export format: {path}") - - -def compute_final_collection(row: dict) -> str: - user_raw = str(row.get("user_collection", "") or "").strip() - user_resolved = str(row.get("user_collection_resolved", "") or "").strip() - recommended = str(row.get("recommended_collection", "") or "").strip() - if user_raw: - return user_resolved - return recommended - - -def canonicalize_decision(value: str) -> str: - text = str(value or "").strip() - if text in {"", "待查"}: - return "待定" - if text in {"排除", "不纳入"}: - return "不纳入" - if text == "纳入": - return "纳入" - return "待定" - - -def candidate_markdown(row: dict) -> str: - row = dict(row) - row["final_collection"] = compute_final_collection(row) - row["decision"] = canonicalize_decision(row.get("decision", "")) - lines = ["---"] - ordered_keys = [ - "candidate_id", - "domain", - "title", - "authors", - "year", - "journal", - "doi", - "pmid", - "source", - "requester_skill", - "request_context", - "abstract_short", - "decision", - "recommended_collection", - "recommend_confidence", - "recommend_reason", - "user_collection", - "user_collection_resolved", - "final_collection", - "collection_resolution", - "duplicate_hint", - "existing_zotero_key", - "existing_collections", - "import_status", - "note", - "candidate_source_type", - "source_zotero_key", - "cited_ref_number", - "trigger_sentence", - "source_context", - "task_relevance_reason", - "harvest_priority", - "raw_reference", - "status", - ] - row.setdefault("status", "candidate") - for key in ordered_keys: - value = row.get(key, "") - if isinstance(value, list): - lines.append(f"{key}:") - for item in value: - lines.append(f" - {yaml_quote(item)}") - elif value == "": - lines.append(f"{key}:") - elif "\n" in str(value): - lines.extend( - yaml_block(str(value)).copy() - if key == "abstract" - else [f"{key}: |-"] + [f" {line}" for line in str(value).splitlines()] - ) - else: - lines.append(f"{key}: {yaml_quote(value)}") - lines.extend( - [ - "---", - "", - f"# {row['candidate_id']}", - "", - "候选文献轻量记录,仅用于 Base 决策和 write-back 触发,不是正式文献卡片。", - "", - ] - ) - return "\n".join(lines) - - -def generate_review(candidates: list[dict]) -> str: - normalized = [] - for row in candidates: - copy = dict(row) - copy["decision"] = canonicalize_decision(copy.get("decision", "")) - normalized.append(copy) - include = [c for c in normalized if c.get("decision") == "纳入"] - exclude = [c for c in normalized if c.get("decision") == "不纳入"] - lines = [ - "# 本轮候选总览", - "", - "## 检索背景", - "", - f"- 候选数量:{len(normalized)}", - f"- 建议纳入:{len(include)}", - f"- 不纳入:{len(exclude)}", - "", - "## 总体判断", - "", - "- 当前候选池已经按决策状态分层,可直接进入 Base 处理。", - "", - "## 推荐优先纳入", - "", - ] - if include: - for row in include: - lines.extend( - [ - f"### {row['candidate_id']}", - "", - f"- 标题:{row['title']}", - f"- 推荐分类:`{compute_final_collection(row)}`", - f"- 理由:{row.get('recommend_reason', '')}", - "", - ] - ) - else: - lines.extend(["- 暂无", ""]) - lines.extend(["## 不纳入", ""]) - if exclude: - for row in exclude: - lines.extend( - [ - f"### {row['candidate_id']}", - "", - f"- 标题:{row['title']}", - f"- 理由:{row.get('recommend_reason', '')}", - "", - ] - ) - else: - lines.extend(["- 暂无", ""]) - lines.extend(["## 下一步", "", "1. 在 Base 中确认决策。", "2. 对纳入项执行 write-back。", "3. 刷新正式索引。", ""]) - return "\n".join(lines) - - -DEEP_READING_HEADER = "## 🔍 精读" - - -def extract_preserved_deep_reading(text: str) -> str: - """Extract the `## 🔍 精读` section by matching it as a real markdown header. - - Uses regex to ensure we match `## 🔍 精读` at the start of a line, - avoiding false positives from prose text that merely mentions the string. - """ - if not text: - return "" - match = re.search("^## 🔍 精读\\s*$", text, re.MULTILINE) - if not match: - return "" - start = match.start() - preserved = text[start:].strip() - return preserved - - -def has_deep_reading_content(text: str) -> bool: - """Return True only if the deep-reading section passes a three-anchor check. - - For deep_reading_status to be "done", ALL three of these must have - substantive content (not just template scaffolding): - - 1. **Clarity**(清晰度): — text after the colon - 2. **Figure 导读** — at least one list item with content - 3. **遗留问题** — at least one non-empty entry - """ - preserved = extract_preserved_deep_reading(text) - if not preserved: - return False - body = preserved.replace(DEEP_READING_HEADER, "").strip() - if not body: - return False - - # Anchor 1: Clarity - clarity_ok = bool(re.search( - r'-\s*\*\*Clarity\*\*(清晰度):(.+)', body - )) and not re.search( - r'-\s*\*\*Clarity\*\*(清晰度):\s*$', body, re.MULTILINE - ) - - # Anchor 2: Figure 导读 — at least one list item with content after colon - figure_sec = _extract_section(body, r'\*\*Figure 导读\*\*') - figure_ok = False - if figure_sec: - for line in figure_sec.splitlines(): - s = line.strip() - if s.startswith('- ') and ':' in s: - _, after = s.split(':', 1) - if after.strip() and after.strip() != '(待补充)': - figure_ok = True - break - - # Anchor 3: 遗留问题 — at least one non-empty line beneath marker - issue_sec = _extract_section(body, r'\*\*遗留问题\*\*') - if not issue_sec: - issue_sec = _extract_section(body, r'####\s*遗留问题') - issue_ok = False - if issue_sec: - dirty = [l.strip() for l in issue_sec.splitlines() if l.strip()] - substantive = [l for l in dirty if l not in ('-', '(待补充)', '', '**遗留问题**')] - issue_ok = bool(substantive) - - return clarity_ok and figure_ok and issue_ok - - -def _extract_section(body: str, section_header: str) -> str | None: - """Return block of text from *section_header* to the next section break. - - Section break is either a ``##`` / ``###`` header or end of string. - """ - m = re.search( - section_header + r'\n*(.*?)(?=\n(?:#{1,3})\s|\Z)', - body, re.DOTALL, - ) - if m: - return m.group(1).strip() - return None - - - - -def _add_missing_frontmatter_fields(existing_content: str, new_fields: dict[str, str]) -> str: - """Surgically append missing fields to existing frontmatter without overwriting anything.""" - if not existing_content.startswith("---"): - return existing_content - parts = existing_content.split("---", 2) - if len(parts) < 3: - return existing_content - frontmatter = parts[1] - body = parts[2] - lines_to_add = [] - for key, value in new_fields.items(): - pattern = "^" + re.escape(key) + "\\s*:" - if not re.search(pattern, frontmatter, re.MULTILINE): - lines_to_add.append(f"{key}: {yaml_quote(value)}") - if not lines_to_add: - return existing_content - new_frontmatter = frontmatter.rstrip("\n") + "\n" + "\n".join(lines_to_add) + "\n" - return f"---{new_frontmatter}---{body}" - - -def update_frontmatter_field(content: str, key: str, value: str) -> str: - """Update an existing frontmatter field value, or add if missing.""" - if not content.startswith("---"): - return content - pattern = "^" + re.escape(key) + "\\s*:.*$" - replacement = f"{key}: {yaml_quote(value)}" - new_content, count = re.subn(pattern, replacement, content, flags=re.MULTILINE, count=1) - if count == 0: - new_content = _add_missing_frontmatter_fields(content, {key: value}) - return new_content - def load_control_actions(paths: dict[str, Path]) -> dict[str, dict]: actions = {} @@ -736,7 +208,7 @@ def load_control_actions(paths: dict[str, Path]) -> dict[str, dict]: return actions -def run_selection_sync(vault: Path, verbose: bool = False) -> int: +def run_selection_sync(vault: Path, verbose: bool = False, json_output: bool = False) -> int | dict: from paperforge.worker.base_views import ensure_base_views from paperforge.worker.ocr import validate_ocr_meta @@ -795,6 +267,8 @@ def run_selection_sync(vault: Path, verbose: bool = False) -> int: # Formal notes now carry workflow flags (do_ocr, analyze) directly. # Existing library-records are migrated via Phase 40 logic. updated += 1 + if json_output: + return {"new": written, "updated": updated, "skipped": 0, "failed": 0, "errors": []} print(f"selection-sync: wrote {written} records, updated {updated} records") return 0 @@ -1720,7 +1194,7 @@ def migrate_to_workspace(vault: Path, paths: dict) -> int: return migrated -def run_index_refresh(vault: Path, verbose: bool = False, rebuild_index: bool = False) -> int: +def run_index_refresh(vault: Path, verbose: bool = False, rebuild_index: bool = False, json_output: bool = False) -> int: """Refresh the canonical asset index. Default behavior: full rebuild. This is the safe default because @@ -1732,6 +1206,7 @@ def run_index_refresh(vault: Path, verbose: bool = False, rebuild_index: bool = vault: Path to the vault root. verbose: If True, print detailed progress. rebuild_index: If True, force full rebuild (default: True for sync). + json_output: If True, suppress human-readable print output. """ from paperforge.worker.base_views import ensure_base_views from paperforge.worker.ocr import validate_ocr_meta @@ -1800,7 +1275,7 @@ def run_index_refresh(vault: Path, verbose: bool = False, rebuild_index: bool = deleted_count += 1 except Exception: pass - if deleted_count > 0: + if deleted_count > 0 and not json_output: print(f"index-refresh: cleaned {deleted_count} orphaned records in {domain}") # Clean up flat notes: delete only if confirmed by canonical index + workspace @@ -1831,10 +1306,10 @@ def run_index_refresh(vault: Path, verbose: bool = False, rebuild_index: bool = cleaned_flat += 1 except Exception: pass - if cleaned_flat > 0: + if cleaned_flat > 0 and not json_output: print(f"index-refresh: cleaned {cleaned_flat} flat note(s) (migrated to workspace)") - if control_records_dir.exists(): + if control_records_dir.exists() and not json_output: total = sum(1 for _ in control_records_dir.rglob("*.md")) print(f"index-refresh: {total} formal note(s) in literature") diff --git a/pyproject.toml b/pyproject.toml index de660006..72fc765c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -70,13 +70,14 @@ paperforge = [ [tool.pytest.ini_options] addopts = "--ignore=tests/sandbox/00_TestVault/ --strict-markers" -testpaths = ["tests/unit", "tests/cli", "tests/e2e", "tests/journey", "tests/chaos"] +testpaths = ["tests/unit", "tests/cli", "tests/e2e", "tests/journey", "tests/chaos", "tests/audit"] markers = [ "unit: Unit tests (Level 1) — fast, isolated", "cli: CLI contract tests (Level 2) — subprocess boundary", "e2e: End-to-end tests (Level 4) — temp vault", "journey: User journey tests (Level 5) — full workflows", "chaos: Destructive tests (Level 6) — abnormal scenarios", + "audit: Consistency audit tests — validate L1 mocks against L4 real pipeline output", "slow: Tests that take >30s (skip during development)", "snapshot: Tests that use snapshot comparison", ] diff --git a/scripts/check_version_sync.py b/scripts/check_version_sync.py new file mode 100644 index 00000000..d631c9c8 --- /dev/null +++ b/scripts/check_version_sync.py @@ -0,0 +1,159 @@ +#!/usr/bin/env python3 +"""Version consistency gate script. + +Checks that all version declarations across the codebase match +the canonical version in paperforge.__version__. + +Declaration locations checked: +1. paperforge/__init__.py (source of truth) +2. paperforge/plugin/manifest.json +3. paperforge/plugin/versions.json +4. pyproject.toml (dynamic version config) +5. paperforge/plugin/main.js (version fallback strings) + +Exit code: 0 if all match, 1 if any mismatch. +""" + +from __future__ import annotations + +import json +import re +import sys +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent + + +def get_canonical_version() -> str: + """Read version from paperforge.__init__.""" + init_path = REPO_ROOT / "paperforge" / "__init__.py" + content = init_path.read_text(encoding="utf-8") + m = re.search(r'__version__\s*=\s*"([^"]+)"', content) + if not m: + print("[FAIL] Cannot parse __version__ from paperforge/__init__.py") + sys.exit(1) + return m.group(1) + + +def check_init(canonical: str) -> list[str]: + errors = [] + init_path = REPO_ROOT / "paperforge" / "__init__.py" + content = init_path.read_text(encoding="utf-8") + m = re.search(r'__version__\s*=\s*"([^"]+)"', content) + if m and m.group(1) != canonical: + errors.append(f' paperforge/__init__.py: __version__ = "{m.group(1)}" (expected "{canonical}")') + return errors + + +def check_plugin_manifest(canonical: str) -> list[str]: + errors = [] + path = REPO_ROOT / "paperforge" / "plugin" / "manifest.json" + try: + data = json.loads(path.read_text(encoding="utf-8")) + ver = data.get("version") + if ver != canonical: + errors.append(f' paperforge/plugin/manifest.json: version = "{ver}" (expected "{canonical}")') + except (json.JSONDecodeError, FileNotFoundError) as e: + errors.append(f" paperforge/plugin/manifest.json: failed to read — {e}") + return errors + + +def check_root_manifest(canonical: str) -> list[str]: + errors = [] + path = REPO_ROOT / "manifest.json" + try: + data = json.loads(path.read_text(encoding="utf-8")) + ver = data.get("version") + if ver != canonical: + errors.append(f' manifest.json: version = "{ver}" (expected "{canonical}")') + except (json.JSONDecodeError, FileNotFoundError) as e: + errors.append(f" manifest.json: failed to read — {e}") + return errors + + +def check_versions_json(canonical: str) -> list[str]: + errors = [] + path = REPO_ROOT / "paperforge" / "plugin" / "versions.json" + try: + data = json.loads(path.read_text(encoding="utf-8")) + if canonical not in data: + errors.append(f' paperforge/plugin/versions.json: missing entry for version "{canonical}"') + except (json.JSONDecodeError, FileNotFoundError) as e: + errors.append(f" paperforge/plugin/versions.json: failed to read — {e}") + return errors + + +def check_pyproject_toml() -> list[str]: + errors = [] + path = REPO_ROOT / "pyproject.toml" + if not path.exists(): + errors.append(" pyproject.toml: not found") + return errors + content = path.read_text(encoding="utf-8") + if 'dynamic = ["version"]' not in content: + errors.append(' pyproject.toml: missing dynamic = ["version"]') + if 'version = {attr = "paperforge.__version__"}' not in content: + errors.append(" pyproject.toml: missing version attr reference") + return errors + + +def check_main_js(canonical: str) -> list[str]: + errors = [] + path = REPO_ROOT / "paperforge" / "plugin" / "main.js" + if not path.exists(): + errors.append(" paperforge/plugin/main.js: not found") + return errors + content = path.read_text(encoding="utf-8") + # Find hardcoded version strings in fallback patterns + fallback_pattern = re.compile(r"['\"](\d+\.\d+\.\d+\w*)['\"]") + for i, line in enumerate(content.splitlines(), 1): + stripped = line.strip() + # Skip comments and non-fallback lines + if "manifest.version" not in stripped and "this.manifest" not in stripped: + continue + for m in fallback_pattern.finditer(stripped): + found = m.group(1) + if found == canonical: + continue + errors.append( + f' paperforge/plugin/main.js:{i}: hardcoded version "{found}"' + f' (expected "{canonical}" or dynamic from manifest)' + ) + return errors + + +def main() -> int: + canonical = get_canonical_version() + print(f"Canonical version: {canonical}") + print() + + checks: list[tuple[str, list[str]]] = [ + ("paperforge/__init__.py", check_init(canonical)), + ("paperforge/plugin/manifest.json", check_plugin_manifest(canonical)), + ("manifest.json", check_root_manifest(canonical)), + ("paperforge/plugin/versions.json", check_versions_json(canonical)), + ("pyproject.toml (dynamic version config)", check_pyproject_toml()), + ("paperforge/plugin/main.js (fallback strings)", check_main_js(canonical)), + ] + + total_errors = 0 + for label, errors in checks: + if not errors: + print(f"[PASS] {label}") + else: + print(f"[FAIL] {label}") + for e in errors: + print(e) + total_errors += len(errors) + + print() + if total_errors == 0: + print("All version declarations are consistent.") + return 0 + else: + print(f"Found {total_errors} version mismatch(es).") + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/cli/snapshots/test_json_contracts/test_paths_json_snapshot/paths_json/default_config.json b/tests/cli/snapshots/test_json_contracts/test_paths_json_snapshot/paths_json/default_config.json deleted file mode 100644 index 162bba60..00000000 --- a/tests/cli/snapshots/test_json_contracts/test_paths_json_snapshot/paths_json/default_config.json +++ /dev/null @@ -1,5 +0,0 @@ -{ - "vault": "", - "worker_script": "D:\\L\\Med\\Research\\99_System\\LiteraturePipeline\\github-release\\paperforge\\worker\\__init__.py", - "ld_deep_script": "D:\\L\\Med\\Research\\99_System\\LiteraturePipeline\\github-release\\paperforge\\skills\\literature-qa\\scripts\\ld_deep.py" -} \ No newline at end of file diff --git a/tests/cli/snapshots/test_json_contracts/test_status_json_snapshot/status_json/empty_vault.json b/tests/cli/snapshots/test_json_contracts/test_status_json_snapshot/status_json/empty_vault.json deleted file mode 100644 index 0b8e5941..00000000 --- a/tests/cli/snapshots/test_json_contracts/test_status_json_snapshot/status_json/empty_vault.json +++ /dev/null @@ -1,23 +0,0 @@ -{ - "version": "", - "vault": "", - "system_dir": "System", - "resources_dir": "Resources", - "exports": 0, - "domains": 0, - "total_papers": 0, - "formal_notes": 0, - "bases": 1, - "ocr": { - "total": 0, - "pending": 0, - "processing": 0, - "done": 0, - "failed": 0 - }, - "path_errors": 0, - "env_configured": true, - "lifecycle_level_counts": {}, - "health_aggregate": {}, - "maturity_distribution": {} -} \ No newline at end of file diff --git a/tests/cli/test_contract_helpers.py b/tests/cli/test_contract_helpers.py index c0081471..269e2760 100644 --- a/tests/cli/test_contract_helpers.py +++ b/tests/cli/test_contract_helpers.py @@ -25,16 +25,12 @@ def normalize_snapshot(output: str, vault_path: str | None = None) -> str: """ result = output - # Normalize vault paths (absolute paths containing /tmp/ or \\tmp\\) + # Normalize vault paths — use literal replace (not regex) to handle JSON-escaped paths if vault_path: - escaped = re.escape(str(vault_path)) - result = re.sub(escaped, "", result) - - # Fallback: any path containing /tmp/pf_vault_ (Unix) or pf_vault_ (Windows) + result = result.replace(str(vault_path), "") + # Fallback: any path matching common temp vault patterns + result = re.sub(r'[A-Za-z]:[\\/][^\s"\']*pf_vault_[a-z0-9_]+', "", result) result = re.sub(r'/tmp/pf_vault_[^/\s"\']+', "", result) - result = re.sub(r'\\\\temp\\\\pf_vault_[^\\\\\s"\']+', "", result) - # Broader: any string ending in pf_vault_XXXXX as a path component - result = re.sub(r'[A-Za-z]:[\\/][^\s"\']*pf_vault_[a-z0-9]+', "", result) # Normalize ISO timestamps: 2026-05-08T12:34:56.789012+00:00 result = re.sub( diff --git a/tests/cli/test_json_contracts.py b/tests/cli/test_json_contracts.py index 39a4ea1d..d01d480b 100644 --- a/tests/cli/test_json_contracts.py +++ b/tests/cli/test_json_contracts.py @@ -14,29 +14,21 @@ from .test_contract_helpers import ( class TestPathsJson: - """Contract: 'paperforge paths --json' returns stable JSON.""" + """Contract: 'paperforge paths --json' returns valid JSON with correct value types.""" REQUIRED_KEYS = {"vault", "worker_script", "ld_deep_script"} - def test_paths_json_valid(self, cli_invoker): - """paths --json outputs valid JSON with expected keys.""" + def test_paths_json_value_semantics(self, cli_invoker): + """paths --json returns paths with correct value types (all non-empty strings).""" result = cli_invoker(["paths", "--json"]) assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr}" data = assert_valid_json(result.stdout) assert_json_shape(data, self.REQUIRED_KEYS) - - def test_paths_json_snapshot(self, cli_invoker, snapshot): - """paths --json matches snapshot (normalized).""" - result = cli_invoker(["paths", "--json"]) - assert result.returncode == 0 - - # Parse and re-serialize for consistent formatting - data = json.loads(result.stdout) - vault = data.get("vault", "") - normalized = normalize_snapshot( - json.dumps(data, indent=2, ensure_ascii=False), vault - ) - snapshot.assert_match(normalized, "paths_json/default_config.json") + for key in self.REQUIRED_KEYS: + val = data[key] + assert isinstance(val, str), f"{key} should be string, got {type(val).__name__}" + assert len(val) > 0, f"{key} should not be empty" + assert data.get("vault", "").startswith("<") or len(data.get("vault", "")) > 0 def test_paths_no_json_text_output(self, cli_invoker): """paths without --json outputs human-readable text.""" @@ -48,7 +40,7 @@ class TestPathsJson: class TestStatusJson: - """Contract: 'paperforge status --json' returns stable JSON.""" + """Contract: 'paperforge status --json' returns JSON with correct value semantics.""" REQUIRED_KEYS = {"total_papers", "version", "vault", "system_dir", "resources_dir"} OPTIONAL_KEYS = { @@ -57,23 +49,30 @@ class TestStatusJson: "lifecycle_level_counts", "health_aggregate", "maturity_distribution", } - def test_status_json_on_empty_vault(self, cli_invoker): - """status --json on empty vault returns valid JSON with all contract keys.""" + def test_status_json_value_semantics(self, cli_invoker): + """status --json returns valid JSON with all contract keys and correct value types.""" result = cli_invoker(["status", "--json"]) assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr}" data = assert_valid_json(result.stdout) assert_json_shape(data, self.REQUIRED_KEYS, self.OPTIONAL_KEYS) - # Empty vault should have 0 papers - assert data.get("total_papers", -1) >= 0 + total = data.get("total_papers", -1) + assert isinstance(total, int) and total >= 0, f"total_papers should be non-negative int, got {total}" + ver = data.get("version", "") + assert isinstance(ver, str) and len(ver) > 0, f"version should be non-empty string, got {ver}" + assert data.get("system_dir", "") == "System", f"system_dir should be 'System', got {data.get('system_dir')}" + assert data.get("resources_dir", "") == "Resources", f"resources_dir should be 'Resources', got {data.get('resources_dir')}" + if "ocr" in data: + ocr = data["ocr"] + for field in ("total", "pending", "processing", "done", "failed"): + assert isinstance(ocr.get(field), int), f"ocr.{field} should be int, got {type(ocr.get(field)).__name__}" - def test_status_json_snapshot(self, cli_invoker, snapshot): - """status --json matches snapshot (normalized).""" + def test_status_json_snapshot_regression(self, cli_invoker, snapshot): + """status --json snapshot for regression detection (value semantics tested separately).""" result = cli_invoker(["status", "--json"]) assert result.returncode == 0 try: data = json.loads(result.stdout) except json.JSONDecodeError: - # status --json may not output JSON in all vault states pytest.skip("status --json did not produce valid JSON") vault = data.get("vault", "") normalized = normalize_snapshot( diff --git a/tests/unit/__init__.py b/tests/unit/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/unit/adapters/__init__.py b/tests/unit/adapters/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/unit/adapters/test_bbt.py b/tests/unit/adapters/test_bbt.py new file mode 100644 index 00000000..1d608ae9 --- /dev/null +++ b/tests/unit/adapters/test_bbt.py @@ -0,0 +1,391 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from paperforge.adapters.bbt import ( + _identify_main_pdf, + _normalize_attachment_path, + collection_fields, + extract_authors, + load_export_rows, + resolve_item_collection_paths, +) + + +class TestNormalizeAttachmentPath: + def test_storage_prefix_passthrough(self) -> None: + result = _normalize_attachment_path("storage:ABCD1234/file.pdf") + assert result == ("storage:ABCD1234/file.pdf", "storage:ABCD1234/file.pdf", "ABCD1234") + + def test_storage_prefix_backslash_normalization(self) -> None: + result = _normalize_attachment_path("storage:ABCD1234\\file.pdf") + assert result == ("storage:ABCD1234/file.pdf", "storage:ABCD1234\\file.pdf", "ABCD1234") + + def test_absolute_windows_path_with_zotero_storage(self) -> None: + result = _normalize_attachment_path(r"D:\Zotero\storage\ABCD1234\paper.pdf") + assert result == ("storage:ABCD1234/paper.pdf", r"D:\Zotero\storage\ABCD1234\paper.pdf", "ABCD1234") + + def test_absolute_windows_path_not_in_zotero(self) -> None: + result = _normalize_attachment_path(r"D:\other\paper.pdf") + assert result == ("absolute:D:\\other\\paper.pdf", r"D:\other\paper.pdf", "") + + def test_bare_relative_path(self) -> None: + result = _normalize_attachment_path("EFGH5678/doc.pdf") + assert result == ("storage:EFGH5678/doc.pdf", "EFGH5678/doc.pdf", "EFGH5678") + + def test_chinese_filename(self) -> None: + result = _normalize_attachment_path(r"D:\Zotero\storage\ABCD1234\中文论文.pdf") + assert result == ("storage:ABCD1234/中文论文.pdf", r"D:\Zotero\storage\ABCD1234\中文论文.pdf", "ABCD1234") + + def test_path_with_spaces(self) -> None: + result = _normalize_attachment_path(r"D:\Zotero\storage\ABCD1234\my paper.pdf") + assert result == ("storage:ABCD1234/my paper.pdf", r"D:\Zotero\storage\ABCD1234\my paper.pdf", "ABCD1234") + + def test_empty_path_returns_empty(self) -> None: + assert _normalize_attachment_path("") == ("", "", "") + assert _normalize_attachment_path(None) == ("", "", "") + assert _normalize_attachment_path(" ") == ("", "", "") + + +class TestIdentifyMainPdf: + def test_single_pdf_becomes_main(self) -> None: + atts = [{"contentType": "application/pdf", "title": "Full Text", "path": "storage:K/key.pdf"}] + main, supp = _identify_main_pdf(atts) + assert main is atts[0] + assert supp == [] + + def test_title_pdf_is_selected(self) -> None: + atts = [ + {"contentType": "application/pdf", "title": "PDF", "path": "storage:K/main.pdf"}, + {"contentType": "application/pdf", "title": "Supplement", "path": "storage:K/supp.pdf"}, + ] + main, supp = _identify_main_pdf(atts) + assert main["title"] == "PDF" + assert len(supp) == 1 + assert supp[0]["title"] == "Supplement" + + def test_largest_file_by_size(self) -> None: + atts = [ + {"contentType": "application/pdf", "title": "Small", "size": 100}, + {"contentType": "application/pdf", "title": "Large", "size": 999999}, + ] + main, supp = _identify_main_pdf(atts) + assert main["title"] == "Large" + assert len(supp) == 1 + + def test_shortest_title_when_sizes_equal(self) -> None: + atts = [ + {"contentType": "application/pdf", "title": "A Longer Title Here", "size": 500}, + {"contentType": "application/pdf", "title": "Short", "size": 500}, + ] + main, supp = _identify_main_pdf(atts) + assert main["title"] == "Short" + + def test_no_attachments_returns_none(self) -> None: + main, supp = _identify_main_pdf([]) + assert main is None + assert supp == [] + + def test_no_pdf_attachments(self) -> None: + atts = [{"contentType": "text/html", "title": "Snapshot"}] + main, supp = _identify_main_pdf(atts) + assert main is None + assert supp == [] + + def test_no_size_field_falls_back_to_shortest_title(self) -> None: + atts = [ + {"contentType": "application/pdf", "title": "Long Title Document"}, + {"contentType": "application/pdf", "title": "AB"}, + ] + main, supp = _identify_main_pdf(atts) + assert main["title"] == "AB" + + def test_first_pdf_fallback(self) -> None: + atts = [ + {"contentType": "application/pdf", "title": "First"}, + {"contentType": "application/pdf", "title": "Second"}, + ] + main, supp = _identify_main_pdf(atts) + assert main["title"] == "First" + + def test_non_dict_item_skipped(self) -> None: + atts = [ + {"contentType": "application/pdf", "title": "PDF", "path": "storage:K/main.pdf"}, + "not a dict", + ] + main, supp = _identify_main_pdf(atts) + assert main is not None + assert main["title"] == "PDF" + + +class TestExtractAuthors: + def test_single_author(self) -> None: + item = {"creators": [{"creatorType": "author", "firstName": "John", "lastName": "Doe"}]} + assert extract_authors(item) == ["John Doe"] + + def test_multiple_authors(self) -> None: + item = { + "creators": [ + {"creatorType": "author", "firstName": "Alice", "lastName": "Smith"}, + {"creatorType": "author", "firstName": "Bob", "lastName": "Jones"}, + ] + } + assert extract_authors(item) == ["Alice Smith", "Bob Jones"] + + def test_skips_non_author_creators(self) -> None: + item = { + "creators": [ + {"creatorType": "author", "firstName": "Only", "lastName": "Author"}, + {"creatorType": "editor", "firstName": "Not", "lastName": "Included"}, + ] + } + assert extract_authors(item) == ["Only Author"] + + def test_empty_creators(self) -> None: + assert extract_authors({"creators": []}) == [] + + def test_missing_creators_key(self) -> None: + assert extract_authors({}) == [] + + def test_author_with_no_first_name(self) -> None: + item = {"creators": [{"creatorType": "author", "lastName": "Doe"}]} + assert extract_authors(item) == ["Doe"] + + def test_author_with_no_last_name(self) -> None: + item = {"creators": [{"creatorType": "author", "firstName": "John"}]} + assert extract_authors(item) == ["John"] + + def test_author_with_name_field(self) -> None: + item = {"creators": [{"creatorType": "author", "name": "Institutional Author"}]} + assert extract_authors(item) == ["Institutional Author"] + + +class TestCollectionFields: + def test_single_path(self) -> None: + result = collection_fields(["骨科/脊柱"]) + assert result["collections"] == ["骨科/脊柱"] + assert "骨科" in result["collection_tags"] + assert "脊柱" in result["collection_tags"] + assert result["collection_group"] == ["骨科/脊柱"] + + def test_multiple_paths(self) -> None: + paths = ["骨科/脊柱/颈椎", "骨科/脊柱"] + result = collection_fields(paths) + assert len(result["collections"]) == 2 + assert "骨科" in result["collection_tags"] + assert "脊柱" in result["collection_tags"] + assert "颈椎" in result["collection_tags"] + + def test_empty_list(self) -> None: + result = collection_fields([]) + assert result["collections"] == [] + assert result["collection_tags"] == [] + assert result["collection_group"] == [] + + def test_nested_path(self) -> None: + result = collection_fields(["骨科/脊柱/颈椎/后路"]) + assert result["collections"] == ["骨科/脊柱/颈椎/后路"] + assert result["collection_tags"] == ["骨科", "脊柱", "颈椎", "后路"] + + def test_duplicate_tags_deduplicated(self) -> None: + result = collection_fields(["骨科/脊柱", "骨科/创伤"]) + assert result["collection_tags"] == ["骨科", "脊柱", "创伤"] + + def test_primary_is_deepest_path(self) -> None: + result = collection_fields(["运动医学/膝关节", "运动医学"]) + assert result["collection_group"] == ["运动医学/膝关节"] + + def test_whitespace_in_paths(self) -> None: + result = collection_fields([" 骨科 / 脊柱 "]) + assert "骨科" in result["collection_tags"] + assert "脊柱" in result["collection_tags"] + + +class TestResolveItemCollectionPaths: + def test_single_collection(self) -> None: + item = {"collections": ["col1"]} + lookup = {"path_by_key": {"col1": "骨科/脊柱"}, "paths_by_item_id": {}} + assert resolve_item_collection_paths(item, lookup) == ["骨科/脊柱"] + + def test_multiple_collections(self) -> None: + item = {"collections": ["col1", "col2"]} + lookup = { + "path_by_key": {"col1": "骨科/脊柱", "col2": "运动医学/膝关节"}, + "paths_by_item_id": {}, + } + result = resolve_item_collection_paths(item, lookup) + assert "骨科/脊柱" in result + assert "运动医学/膝关节" in result + + def test_item_id_paths(self) -> None: + item = {"itemID": 42} + lookup = {"path_by_key": {}, "paths_by_item_id": {42: ["骨科/创伤"]}} + assert resolve_item_collection_paths(item, lookup) == ["骨科/创伤"] + + def test_combined_collections_and_item_id(self) -> None: + item = {"collections": ["col1"], "itemID": 42} + lookup = { + "path_by_key": {"col1": "骨科/脊柱"}, + "paths_by_item_id": {42: ["骨科/脊柱/颈椎"]}, + } + result = resolve_item_collection_paths(item, lookup) + assert "骨科/脊柱" in result + assert "骨科/脊柱/颈椎" in result + + def test_empty_item(self) -> None: + item = {} + lookup = {"path_by_key": {}, "paths_by_item_id": {}} + assert resolve_item_collection_paths(item, lookup) == [] + + def test_unknown_key_falls_back_to_key(self) -> None: + item = {"collections": ["unknown_key"]} + lookup = {"path_by_key": {}, "paths_by_item_id": {}} + assert resolve_item_collection_paths(item, lookup) == ["unknown_key"] + + def test_deduplication(self) -> None: + item = {"collections": ["col1"], "itemID": 1} + lookup = { + "path_by_key": {"col1": "骨科/脊柱"}, + "paths_by_item_id": {1: ["骨科/脊柱"]}, + } + assert resolve_item_collection_paths(item, lookup) == ["骨科/脊柱"] + + def test_no_collections_key(self) -> None: + item = {"key": "ABC123"} + lookup = {"path_by_key": {}, "paths_by_item_id": {}} + assert resolve_item_collection_paths(item, lookup) == [] + + +class TestLoadExportRows: + def test_valid_json_with_items(self, tmp_path: Path) -> None: + data = { + "items": [ + { + "key": "ABC123", + "itemType": "journalArticle", + "title": "Test Article", + "creators": [{"creatorType": "author", "firstName": "John", "lastName": "Doe"}], + "date": "2023", + "DOI": "10.1234/test", + "PMID": "12345678", + "publicationTitle": "Test Journal", + "abstractNote": "An abstract.", + "collections": ["col1"], + "attachments": [ + { + "path": "storage:ABCD1234/paper.pdf", + "contentType": "application/pdf", + "title": "PDF", + } + ], + } + ], + "collections": {"col1": {"name": "骨科", "parent": "", "items": []}}, + } + path = tmp_path / "export.json" + path.write_text(json.dumps(data), encoding="utf-8") + rows = load_export_rows(path) + assert len(rows) == 1 + assert rows[0]["key"] == "ABC123" + assert rows[0]["title"] == "Test Article" + assert rows[0]["authors"] == ["John Doe"] + assert rows[0]["year"] == "2023" + assert rows[0]["doi"] == "10.1234/test" + assert rows[0]["pmid"] == "12345678" + assert rows[0]["pdf_path"] == "storage:ABCD1234/paper.pdf" + assert rows[0]["collections"] == ["骨科"] + + def test_list_format_data(self, tmp_path: Path) -> None: + data = [ + {"key": "ITEM1", "title": "Item 1"}, + {"key": "ITEM2", "title": "Item 2"}, + ] + path = tmp_path / "list.json" + path.write_text(json.dumps(data), encoding="utf-8") + rows = load_export_rows(path) + assert len(rows) == 2 + assert rows[0]["key"] == "ITEM1" + assert rows[1]["key"] == "ITEM2" + + def test_empty_json_list(self, tmp_path: Path) -> None: + path = tmp_path / "empty.json" + path.write_text("[]", encoding="utf-8") + assert load_export_rows(path) == [] + + def test_file_not_found(self, tmp_path: Path) -> None: + path = tmp_path / "nonexistent.json" + with pytest.raises(FileNotFoundError): + load_export_rows(path) + + def test_skips_attachment_notes_and_annotations(self, tmp_path: Path) -> None: + data = { + "items": [ + {"key": "ART1", "itemType": "journalArticle", "title": "Real Article", "creators": [], "date": "", "collections": [], "attachments": []}, + {"key": "ATT1", "itemType": "attachment", "title": "An Attachment"}, + {"key": "NOT1", "itemType": "note", "title": "A Note"}, + {"key": "ANN1", "itemType": "annotation", "title": "An Annotation"}, + ], + "collections": {}, + } + path = tmp_path / "mixed.json" + path.write_text(json.dumps(data), encoding="utf-8") + rows = load_export_rows(path) + assert len(rows) == 1 + assert rows[0]["key"] == "ART1" + + def test_unsupported_format_raises_error(self, tmp_path: Path) -> None: + path = tmp_path / "bad.json" + path.write_text('"just a string"', encoding="utf-8") + with pytest.raises(ValueError, match="Unsupported export format"): + load_export_rows(path) + + def test_pdf_content_type_fallback(self, tmp_path: Path) -> None: + data = { + "items": [ + { + "key": "ABC123", + "itemType": "journalArticle", + "title": "Test", + "date": "", + "creators": [], + "collections": [], + "attachments": [ + { + "path": "storage:KEY/file.pdf", + } + ], + } + ], + "collections": {}, + } + path = tmp_path / "ctype.json" + path.write_text(json.dumps(data), encoding="utf-8") + rows = load_export_rows(path) + assert rows[0]["attachments"][0]["contentType"] == "application/pdf" + + def test_supplementary_pdfs_separated(self, tmp_path: Path) -> None: + data = { + "items": [ + { + "key": "ABC123", + "itemType": "journalArticle", + "title": "Test", + "date": "", + "creators": [], + "collections": [], + "attachments": [ + {"path": "storage:K/main.pdf", "contentType": "application/pdf", "title": "PDF"}, + {"path": "storage:K/supp1.pdf", "contentType": "application/pdf", "title": "Supplement 1"}, + ], + } + ], + "collections": {}, + } + path = tmp_path / "supp.json" + path.write_text(json.dumps(data), encoding="utf-8") + rows = load_export_rows(path) + assert rows[0]["pdf_path"] == "storage:K/main.pdf" + assert rows[0]["supplementary"] == ["storage:K/supp1.pdf"] diff --git a/tests/unit/adapters/test_obsidian_frontmatter.py b/tests/unit/adapters/test_obsidian_frontmatter.py new file mode 100644 index 00000000..0f32e311 --- /dev/null +++ b/tests/unit/adapters/test_obsidian_frontmatter.py @@ -0,0 +1,249 @@ +from __future__ import annotations + +from pathlib import Path + +from paperforge.adapters.obsidian_frontmatter import ( + _add_missing_frontmatter_fields, + _extract_section, + _legacy_control_flags, + _read_frontmatter_bool_from_text, + _read_frontmatter_optional_bool_from_text, + canonicalize_decision, + candidate_markdown, + compute_final_collection, + extract_preserved_deep_reading, + generate_review, + has_deep_reading_content, + read_frontmatter_dict, + update_frontmatter_field, +) + + +class TestReadFrontmatterBoolFromText: + def test_true_returns_true(self) -> None: + text = "some_key: true\nother: thing" + assert _read_frontmatter_bool_from_text(text, "some_key") is True + + def test_false_returns_false(self) -> None: + text = "some_key: false\nother: thing" + assert _read_frontmatter_bool_from_text(text, "some_key") is False + + def test_missing_returns_default(self) -> None: + text = "other: thing" + assert _read_frontmatter_bool_from_text(text, "missing") is False + assert _read_frontmatter_bool_from_text(text, "missing", True) is True + + +class TestReadFrontmatterOptionalBoolFromText: + def test_true_returns_true(self) -> None: + text = "do_ocr: true" + assert _read_frontmatter_optional_bool_from_text(text, "do_ocr") is True + + def test_false_returns_false(self) -> None: + text = "do_ocr: false" + assert _read_frontmatter_optional_bool_from_text(text, "do_ocr") is False + + def test_missing_returns_none(self) -> None: + text = "other: thing" + assert _read_frontmatter_optional_bool_from_text(text, "missing") is None + + +class TestReadFrontmatterDict: + def test_valid_yaml(self) -> None: + text = "---\nkey: value\nnum: 42\n---\nbody" + result = read_frontmatter_dict(text) + assert result == {"key": "value", "num": 42} + + def test_malformed_yaml_fallback(self) -> None: + text = "---\nkey: value\n: : malformed\n---\nbody" + result = read_frontmatter_dict(text) + assert "key" not in result or result.get("key") is not None + + def test_no_frontmatter_returns_empty(self) -> None: + text = "plain text body\nno frontmatter here" + result = read_frontmatter_dict(text) + assert result == {} + + def test_nested_values(self) -> None: + text = "---\nnested:\n inner: val\n num: 42\n---" + result = read_frontmatter_dict(text) + assert result["nested"]["inner"] == "val" + assert result["nested"]["num"] == 42 + + +class TestLegacyControlFlags: + def test_no_library_records_path(self, tmp_path: Path) -> None: + paths = {"library_records": tmp_path / "nonexistent"} + result = _legacy_control_flags(paths, "ABCDEFG") + assert result == {"do_ocr": None, "analyze": None} + + def test_no_record_found(self, tmp_path: Path) -> None: + records_dir = tmp_path / "records" + records_dir.mkdir() + paths = {"library_records": records_dir} + result = _legacy_control_flags(paths, "NOKEY") + assert result == {"do_ocr": None, "analyze": None} + + def test_record_read_success(self, tmp_path: Path) -> None: + records_dir = tmp_path / "records" + records_dir.mkdir() + record_file = records_dir / "ABCDEFG.md" + record_file.write_text("---\ndo_ocr: true\nanalyze: false\n---", encoding="utf-8") + paths = {"library_records": records_dir} + result = _legacy_control_flags(paths, "ABCDEFG") + assert result == {"do_ocr": True, "analyze": False} + + +class TestComputeFinalCollection: + def test_user_collection_used_when_present(self) -> None: + row = {"user_collection": "骨科/脊柱", "user_collection_resolved": "骨科/脊柱/颈椎", "recommended_collection": "骨科/创伤"} + assert compute_final_collection(row) == "骨科/脊柱/颈椎" + + def test_recommended_fallback(self) -> None: + row = {"user_collection": "", "user_collection_resolved": "", "recommended_collection": "骨科/创伤"} + assert compute_final_collection(row) == "骨科/创伤" + + def test_empty_returns_empty(self) -> None: + row = {"user_collection": "", "user_collection_resolved": "", "recommended_collection": ""} + assert compute_final_collection(row) == "" + + +class TestCanonicalizeDecision: + def test_empty_returns_daidin(self) -> None: + assert canonicalize_decision("") == "待定" + + def test_daicha_returns_daidin(self) -> None: + assert canonicalize_decision("待查") == "待定" + + def test_paichu_returns_bunaron(self) -> None: + assert canonicalize_decision("排除") == "不纳入" + + def test_bunaron_returns_bunaron(self) -> None: + assert canonicalize_decision("不纳入") == "不纳入" + + def test_naru_returns_naru(self) -> None: + assert canonicalize_decision("纳入") == "纳入" + + +class TestCandidateMarkdown: + def test_complete_row(self) -> None: + row = {"candidate_id": "C001", "domain": "骨科", "title": "Test Paper", "authors": ["Author A"], "year": 2024} + result = candidate_markdown(row) + assert result.startswith("---") + assert '"C001"' in result + assert '"Test Paper"' in result + + def test_minimum_row(self) -> None: + row = {"candidate_id": "C002", "title": "Minimal"} + result = candidate_markdown(row) + assert result.startswith("---") + assert "C002" in result + + def test_missing_optional_fields(self) -> None: + row = {"candidate_id": "C003", "title": "Partial", "doi": "10.1234/test"} + result = candidate_markdown(row) + assert "doi" in result + assert result.startswith("---") + + +class TestGenerateReview: + def test_include_and_exclude(self) -> None: + candidates = [ + {"candidate_id": "C001", "title": "Keep", "decision": "纳入", "recommended_collection": "骨科/脊柱"}, + {"candidate_id": "C002", "title": "Drop", "decision": "排除"}, + ] + result = generate_review(candidates) + assert "Keep" in result + assert "Drop" in result + assert "建议纳入:1" in result + assert "不纳入:1" in result + + def test_empty_list(self) -> None: + result = generate_review([]) + assert "候选数量:0" in result + + +class TestExtractPreservedDeepReading: + def test_with_section(self) -> None: + text = "some preamble\n## 🔍 精读\nDeep reading content here\nmore content" + result = extract_preserved_deep_reading(text) + assert "## 🔍 精读" in result + assert "Deep reading content" in result + + def test_without_section(self) -> None: + text = "some content\nno deep reading section" + result = extract_preserved_deep_reading(text) + assert result == "" + + def test_empty_text(self) -> None: + assert extract_preserved_deep_reading("") == "" + + +class TestHasDeepReadingContent: + def test_full_content_returns_true(self) -> None: + text = """preamble +## 🔍 精读 +- **Clarity**(清晰度):The paper clearly describes the method +- **Figure 导读** + - Figure 1:Shows the main result +- **遗留问题** + - Need more validation""" + assert has_deep_reading_content(text) is True + + def test_empty_body_returns_false(self) -> None: + text = "## 🔍 精读" + assert has_deep_reading_content(text) is False + + def test_no_deep_reading_section_returns_false(self) -> None: + text = "no section here" + assert has_deep_reading_content(text) is False + + +class TestExtractSection: + def test_section_extracted(self) -> None: + body = "Some text\n**Figure 导读**\n- Figure 1: content\n## Next section" + result = _extract_section(body, r'\*\*Figure 导读\*\*') + assert result is not None + assert "Figure 1" in result + + def test_no_match_returns_none(self) -> None: + body = "no matching header here" + result = _extract_section(body, r'\*\*Missing\*\*') + assert result is None + + +class TestAddMissingFrontmatterFields: + def test_fields_added(self) -> None: + content = "---\nexisting: val\n---\nbody" + result = _add_missing_frontmatter_fields(content, {"new_key": "new_val"}) + assert "new_key:" in result + assert "existing: val" in result + + def test_all_present_no_change(self) -> None: + content = "---\nexisting: val\n---\nbody" + result = _add_missing_frontmatter_fields(content, {"existing": "val"}) + assert result == content + + def test_no_frontmatter_returns_unchanged(self) -> None: + content = "no frontmatter here" + result = _add_missing_frontmatter_fields(content, {"key": "val"}) + assert result == content + + +class TestUpdateFrontmatterField: + def test_overwrite_existing(self) -> None: + content = "---\nkey: old\nother: val\n---\nbody" + result = update_frontmatter_field(content, "key", "new") + assert '"new"' in result + assert "old" not in result + + def test_add_new_key(self) -> None: + content = "---\none: val\n---\nbody" + result = update_frontmatter_field(content, "new_key", "new_val") + assert "new_key:" in result + assert "one: val" in result + + def test_no_frontmatter_returns_unchanged(self) -> None: + content = "no frontmatter" + result = update_frontmatter_field(content, "key", "val") + assert result == content diff --git a/tests/unit/adapters/test_zotero_paths.py b/tests/unit/adapters/test_zotero_paths.py new file mode 100644 index 00000000..1f03d2a7 --- /dev/null +++ b/tests/unit/adapters/test_zotero_paths.py @@ -0,0 +1,133 @@ +from __future__ import annotations + +from pathlib import Path +from unittest.mock import patch + +from paperforge.adapters.zotero_paths import ( + absolutize_vault_path, + obsidian_wikilink_for_path, + obsidian_wikilink_for_pdf, +) + + +class TestObsidianWikilinkForPdf: + """Tests for obsidian_wikilink_for_pdf path resolution.""" + + def test_empty_path_returns_empty(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + assert obsidian_wikilink_for_pdf("", vault) == "" + assert obsidian_wikilink_for_pdf(None, vault) == "" + assert obsidian_wikilink_for_pdf(" ", vault) == "" + + def test_storage_prefix_resolves_through_zotero_dir(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + zotero = tmp_path / "zotero_data" + zotero.mkdir() + result = obsidian_wikilink_for_pdf("storage:ABCD1234/paper.pdf", vault, zotero) + assert result == f"[[{(zotero / 'storage' / 'ABCD1234' / 'paper.pdf').as_posix()}]]" + + def test_absolute_windows_path_outside_vault(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + abs_path = tmp_path / "outside" / "file.pdf" + abs_path.parent.mkdir(parents=True) + abs_path.write_text("content") + result = obsidian_wikilink_for_pdf(str(abs_path), vault) + assert "outside" in result + assert "file.pdf" in result + assert result.startswith("[[") + assert result.endswith("]]") + + def test_cjk_filename_with_spaces(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + zotero = tmp_path / "zotero_data" + zotero.mkdir() + result = obsidian_wikilink_for_pdf("storage:KEY/中文 文件.pdf", vault, zotero) + assert "中文" in result + assert "文件" in result + + def test_bare_key_no_prefix_resolves_relative_to_vault(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + result = obsidian_wikilink_for_pdf("ABCD1234/paper.pdf", vault) + assert result == "[[ABCD1234/paper.pdf]]" + + def test_storage_prefix_without_zotero_dir_falls_back(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + result = obsidian_wikilink_for_pdf("storage:ABCD/paper.pdf", vault) + assert "ABCD" in result + + +class TestAbsolutizeVaultPath: + """Tests for absolutize_vault_path.""" + + def test_empty_path_returns_empty(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + assert absolutize_vault_path(vault, "") == "" + assert absolutize_vault_path(vault, " ") == "" + + def test_relative_path_resolution(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + result = absolutize_vault_path(vault, "subdir/file.pdf") + expected = str((vault / "subdir" / "file.pdf").resolve()) + assert result == expected + + def test_relative_with_backslashes(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + result = absolutize_vault_path(vault, "subdir\\file.pdf") + expected = str((vault / "subdir" / "file.pdf").resolve()) + assert result == expected + + def test_absolute_path_passthrough(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + abs_path = str(tmp_path / "outside" / "file.pdf") + result = absolutize_vault_path(vault, abs_path) + assert result == abs_path + + def test_junction_resolution_applied_when_flag_set(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + with patch("paperforge.pdf_resolver.resolve_junction") as mock_resolve: + mock_resolve.return_value = tmp_path / "resolved" / "file.pdf" + result = absolutize_vault_path(vault, "sub/file.pdf", resolve_junction=True) + assert "resolved" in result + + def test_junction_not_applied_when_flag_false(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + with patch("paperforge.pdf_resolver.resolve_junction") as mock_resolve: + result = absolutize_vault_path(vault, "sub/file.pdf", resolve_junction=False) + mock_resolve.assert_not_called() + + +class TestObsidianWikilinkForPath: + """Tests for obsidian_wikilink_for_path.""" + + def test_empty_path_returns_empty(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + assert obsidian_wikilink_for_path(vault, "") == "" + assert obsidian_wikilink_for_path(vault, " ") == "" + + def test_relative_path_to_wikilink(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + result = obsidian_wikilink_for_path(vault, "subdir/paper.pdf") + assert result == "[[subdir/paper.pdf]]" + + def test_absolute_path_outside_vault(self, tmp_path: Path) -> None: + vault = tmp_path / "vault" + vault.mkdir() + abs_path = tmp_path / "outside" / "file.pdf" + abs_path.parent.mkdir(parents=True) + abs_path.write_text("content") + result = obsidian_wikilink_for_path(vault, str(abs_path)) + assert result == f"[[{abs_path.as_posix()}]]" diff --git a/tests/unit/core/__init__.py b/tests/unit/core/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/unit/core/test_errors.py b/tests/unit/core/test_errors.py new file mode 100644 index 00000000..8bfb7323 --- /dev/null +++ b/tests/unit/core/test_errors.py @@ -0,0 +1,28 @@ +"""Unit tests for ErrorCode enum.""" +from __future__ import annotations + +from paperforge.core.errors import ErrorCode + + +class TestErrorCodeMembers: + """ErrorCode enum members match expected string values.""" + + def test_members_and_values(self) -> None: + assert ErrorCode.PYTHON_NOT_FOUND.value == "PYTHON_NOT_FOUND" + assert ErrorCode.VERSION_MISMATCH.value == "VERSION_MISMATCH" + assert ErrorCode.BBT_EXPORT_NOT_FOUND.value == "BBT_EXPORT_NOT_FOUND" + assert ErrorCode.OCR_TOKEN_MISSING.value == "OCR_TOKEN_MISSING" + assert ErrorCode.SYNC_FAILED.value == "SYNC_FAILED" + assert ErrorCode.VALIDATION_ERROR.value == "VALIDATION_ERROR" + assert ErrorCode.INTERNAL_ERROR.value == "INTERNAL_ERROR" + assert ErrorCode.UNKNOWN.value == "UNKNOWN" + + def test_str_returns_value(self) -> None: + assert str(ErrorCode.PYTHON_NOT_FOUND) == "PYTHON_NOT_FOUND" + assert str(ErrorCode.UNKNOWN) == "UNKNOWN" + + def test_member_count(self) -> None: + assert len(ErrorCode) == 8 + + def test_is_str_enum(self) -> None: + assert isinstance(ErrorCode.PYTHON_NOT_FOUND, str) diff --git a/tests/unit/core/test_result.py b/tests/unit/core/test_result.py new file mode 100644 index 00000000..eef9d6d0 --- /dev/null +++ b/tests/unit/core/test_result.py @@ -0,0 +1,105 @@ +"""Unit tests for PFResult / PFError round-trip serialization.""" +from __future__ import annotations + +import json + +from paperforge.core.errors import ErrorCode +from paperforge.core.result import PFError, PFResult + + +class TestPFResultBool: + """PFResult truthiness follows ok field.""" + + def test_ok_is_true(self) -> None: + result = PFResult(ok=True, command="sync", version="1.0.0") + assert bool(result) is True + + def test_not_ok_is_false(self) -> None: + error = PFError(code=ErrorCode.SYNC_FAILED, message="sync failed") + result = PFResult(ok=False, command="sync", version="1.0.0", error=error) + assert bool(result) is False + + +class TestPFResultRoundTrip: + """PFResult serialization round-trip: to_dict -> from_dict.""" + + def test_ok_with_data(self) -> None: + original = PFResult( + ok=True, + command="sync", + version="1.4.17rc3", + data={"created": 5, "updated": 2}, + ) + d = original.to_dict() + reconstructed = PFResult.from_dict(d) + assert reconstructed.ok == original.ok + assert reconstructed.command == original.command + assert reconstructed.version == original.version + assert reconstructed.data == original.data + assert reconstructed.error is None + + def test_with_error(self) -> None: + error = PFError( + code=ErrorCode.SYNC_FAILED, + message="Export file not found", + details={"path": "/tmp/library.json"}, + ) + original = PFResult( + ok=False, + command="sync", + version="1.4.17rc3", + data=None, + error=error, + ) + d = original.to_dict() + reconstructed = PFResult.from_dict(d) + assert reconstructed.ok is False + assert reconstructed.command == "sync" + assert reconstructed.data is None + assert reconstructed.error is not None + assert reconstructed.error.code == ErrorCode.SYNC_FAILED + assert reconstructed.error.message == "Export file not found" + assert reconstructed.error.details == {"path": "/tmp/library.json"} + + def test_none_data_serializes_as_null(self) -> None: + result = PFResult(ok=True, command="status", version="1.4.17rc3", data=None) + d = result.to_dict() + assert d["data"] is None + assert d["ok"] is True + + def test_no_error_serializes_as_null(self) -> None: + result = PFResult(ok=True, command="status", version="1.4.17rc3") + d = result.to_dict() + assert d["error"] is None + + def test_to_json_round_trip(self) -> None: + original = PFResult( + ok=True, + command="sync", + version="1.4.17rc3", + data={"count": 42}, + ) + json_str = original.to_json() + parsed = json.loads(json_str) + reconstructed = PFResult.from_dict(parsed) + assert reconstructed == original + + def test_to_json_indent_two(self) -> None: + result = PFResult(ok=True, command="ping", version="1.0.0") + json_str = result.to_json() + lines = json_str.splitlines() + for line in lines: + if line.strip().startswith('"'): + assert line.startswith(" ") or line.startswith("{") + + def test_from_dict_missing_error_returns_none(self) -> None: + data = { + "ok": True, + "command": "test", + "version": "1.0.0", + "data": None, + "error": None, + } + result = PFResult.from_dict(data) + assert result.error is None + assert result.ok is True diff --git a/tests/unit/core/test_state.py b/tests/unit/core/test_state.py new file mode 100644 index 00000000..17bc1c25 --- /dev/null +++ b/tests/unit/core/test_state.py @@ -0,0 +1,223 @@ +"""Unit tests for paperforge.core.state — state machine enums and transitions.""" +from __future__ import annotations + +from paperforge.core.state import ( + ALLOWED_TRANSITIONS, + Lifecycle, + OcrStatus, + PdfStatus, +) + + +class TestOcrStatus: + """OcrStatus enum values and legacy mapping.""" + + def test_values_match_expected_strings(self) -> None: + assert OcrStatus.PENDING.value == "pending" + assert OcrStatus.PROCESSING.value == "processing" + assert OcrStatus.DONE.value == "done" + assert OcrStatus.FAILED.value == "failed" + + def test_from_legacy_canonical_pending(self) -> None: + assert OcrStatus.from_legacy("pending") is OcrStatus.PENDING + + def test_from_legacy_canonical_processing(self) -> None: + assert OcrStatus.from_legacy("processing") is OcrStatus.PROCESSING + + def test_from_legacy_canonical_done(self) -> None: + assert OcrStatus.from_legacy("done") is OcrStatus.DONE + + def test_from_legacy_canonical_failed(self) -> None: + assert OcrStatus.from_legacy("failed") is OcrStatus.FAILED + + def test_from_legacy_queued_maps_to_pending(self) -> None: + assert OcrStatus.from_legacy("queued") is OcrStatus.PENDING + + def test_from_legacy_running_maps_to_processing(self) -> None: + assert OcrStatus.from_legacy("running") is OcrStatus.PROCESSING + + def test_from_legacy_error_maps_to_failed(self) -> None: + assert OcrStatus.from_legacy("error") is OcrStatus.FAILED + + def test_from_legacy_blocked_maps_to_failed(self) -> None: + assert OcrStatus.from_legacy("blocked") is OcrStatus.FAILED + + def test_from_legacy_nopdf_maps_to_failed(self) -> None: + assert OcrStatus.from_legacy("nopdf") is OcrStatus.FAILED + + def test_from_legacy_done_incomplete_maps_to_done(self) -> None: + assert OcrStatus.from_legacy("done_incomplete") is OcrStatus.DONE + + def test_from_legacy_unknown_maps_to_pending(self) -> None: + assert OcrStatus.from_legacy("nonexistent") is OcrStatus.PENDING + + def test_from_legacy_case_and_whitespace_insensitive(self) -> None: + assert OcrStatus.from_legacy(" DONE ") is OcrStatus.DONE + assert OcrStatus.from_legacy("Queued") is OcrStatus.PENDING + + def test_str_returns_value(self) -> None: + assert str(OcrStatus.PENDING) == "pending" + assert str(OcrStatus.DONE) == "done" + + +class TestPdfStatus: + """PdfStatus enum values.""" + + def test_values_match_expected_strings(self) -> None: + assert PdfStatus.HEALTHY.value == "healthy" + assert PdfStatus.BROKEN.value == "broken" + assert PdfStatus.MISSING.value == "missing" + + def test_str_returns_value(self) -> None: + assert str(PdfStatus.HEALTHY) == "healthy" + assert str(PdfStatus.BROKEN) == "broken" + assert str(PdfStatus.MISSING) == "missing" + + +class TestLifecycle: + """Lifecycle enum values.""" + + def test_values_match_expected_strings(self) -> None: + assert Lifecycle.PDF_READY.value == "pdf_ready" + assert Lifecycle.OCR_READY.value == "ocr_ready" + assert Lifecycle.ANALYZE_READY.value == "analyze_ready" + assert Lifecycle.DEEP_READ_DONE.value == "deep_read_done" + assert Lifecycle.ERROR_STATE.value == "error_state" + + def test_str_returns_value(self) -> None: + assert str(Lifecycle.PDF_READY) == "pdf_ready" + + def test_enum_equals_string_value(self) -> None: + """str enum members compare equal to their string value for backward compat.""" + assert Lifecycle.PDF_READY == "pdf_ready" + assert Lifecycle.DEEP_READ_DONE == "deep_read_done" + + +class TestAllowedTransitionsOcr: + """ALLOWED_TRANSITIONS.check_ocr validation.""" + + def test_pending_to_processing_valid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_ocr(OcrStatus.PENDING, OcrStatus.PROCESSING) + assert ok is True + assert msg == "" + + def test_pending_to_failed_valid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_ocr(OcrStatus.PENDING, OcrStatus.FAILED) + assert ok is True + assert msg == "" + + def test_processing_to_done_valid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_ocr(OcrStatus.PROCESSING, OcrStatus.DONE) + assert ok is True + assert msg == "" + + def test_processing_to_failed_valid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_ocr(OcrStatus.PROCESSING, OcrStatus.FAILED) + assert ok is True + assert msg == "" + + def test_done_to_pending_valid_rerun(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_ocr(OcrStatus.DONE, OcrStatus.PENDING) + assert ok is True + assert msg == "" + + def test_failed_to_pending_valid_retry(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_ocr(OcrStatus.FAILED, OcrStatus.PENDING) + assert ok is True + assert msg == "" + + def test_pending_to_done_invalid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_ocr(OcrStatus.PENDING, OcrStatus.DONE) + assert ok is False + assert "Illegal OCR transition" in msg + + def test_done_to_processing_invalid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_ocr(OcrStatus.DONE, OcrStatus.PROCESSING) + assert ok is False + assert "Illegal OCR transition" in msg + + def test_failed_to_done_invalid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_ocr(OcrStatus.FAILED, OcrStatus.DONE) + assert ok is False + assert "Illegal OCR transition" in msg + + def test_invalid_transition_includes_allowed_list(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_ocr(OcrStatus.PENDING, OcrStatus.DONE) + assert ok is False + assert "processing" in msg + assert "failed" in msg + + +class TestAllowedTransitionsLifecycle: + """ALLOWED_TRANSITIONS.check_lifecycle validation.""" + + def test_pdf_ready_to_ocr_ready_valid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_lifecycle( + Lifecycle.PDF_READY, Lifecycle.OCR_READY + ) + assert ok is True + assert msg == "" + + def test_pdf_ready_to_error_valid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_lifecycle( + Lifecycle.PDF_READY, Lifecycle.ERROR_STATE + ) + assert ok is True + assert msg == "" + + def test_ocr_ready_to_analyze_ready_valid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_lifecycle( + Lifecycle.OCR_READY, Lifecycle.ANALYZE_READY + ) + assert ok is True + assert msg == "" + + def test_analyze_ready_to_deep_read_done_valid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_lifecycle( + Lifecycle.ANALYZE_READY, Lifecycle.DEEP_READ_DONE + ) + assert ok is True + assert msg == "" + + def test_deep_read_done_to_pdf_ready_valid_rerun(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_lifecycle( + Lifecycle.DEEP_READ_DONE, Lifecycle.PDF_READY + ) + assert ok is True + assert msg == "" + + def test_error_state_to_pdf_ready_valid_recover(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_lifecycle( + Lifecycle.ERROR_STATE, Lifecycle.PDF_READY + ) + assert ok is True + assert msg == "" + + def test_pdf_ready_to_deep_read_done_invalid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_lifecycle( + Lifecycle.PDF_READY, Lifecycle.DEEP_READ_DONE + ) + assert ok is False + assert "Illegal lifecycle transition" in msg + + def test_ocr_ready_to_deep_read_done_invalid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_lifecycle( + Lifecycle.OCR_READY, Lifecycle.DEEP_READ_DONE + ) + assert ok is False + assert "Illegal lifecycle transition" in msg + + def test_deep_read_done_to_analyze_ready_invalid(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_lifecycle( + Lifecycle.DEEP_READ_DONE, Lifecycle.ANALYZE_READY + ) + assert ok is False + assert "Illegal lifecycle transition" in msg + + def test_invalid_transition_includes_allowed_list(self) -> None: + ok, msg = ALLOWED_TRANSITIONS.check_lifecycle( + Lifecycle.PDF_READY, Lifecycle.DEEP_READ_DONE + ) + assert ok is False + assert "ocr_ready" in msg + assert "error_state" in msg diff --git a/tests/unit/doctor/__init__.py b/tests/unit/doctor/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/unit/doctor/test_field_validator.py b/tests/unit/doctor/test_field_validator.py new file mode 100644 index 00000000..3712bd83 --- /dev/null +++ b/tests/unit/doctor/test_field_validator.py @@ -0,0 +1,106 @@ +from pathlib import Path + +from paperforge.doctor.field_validator import ( + validate_entry_fields, + validate_collection, + validate_frontmatter_from_file, +) + +SAMPLE_REGISTRY = { + "frontmatter": { + "zotero_key": {"type": "str", "required": True, "public": True}, + "domain": {"type": "str", "required": True, "public": True}, + "title": {"type": "str", "required": True, "public": True}, + "year": {"type": "str", "required": False, "public": True}, + "doi": {"type": "str", "required": False, "public": True}, + "has_pdf": {"type": "bool", "required": True, "public": True}, + "analyze": {"type": "bool", "required": False, "public": True}, + "ocr_status": {"type": "str", "required": False, "public": True}, + }, +} + + +def test_complete_entry_no_issues(): + entry = { + "zotero_key": "ABCDEF", + "domain": "骨科", + "title": "A Study", + "year": "2024", + "doi": "10.1234/test", + "has_pdf": "true", + "analyze": "false", + "ocr_status": "pending", + } + issues = validate_entry_fields(entry, "frontmatter", SAMPLE_REGISTRY) + assert issues == [] + + +def test_missing_required_field(): + entry = { + "domain": "骨科", + "title": "A Study", + "has_pdf": "true", + } + issues = validate_entry_fields(entry, "frontmatter", SAMPLE_REGISTRY, "ABCDEF") + required_errors = [i for i in issues if i["code"] == "MISSING_REQUIRED"] + assert len(required_errors) == 1 + assert required_errors[0]["severity"] == "error" + assert required_errors[0]["field"] == "zotero_key" + + +def test_unknown_field_drift_warning(): + entry = { + "zotero_key": "ABCDEF", + "domain": "骨科", + "title": "A Study", + "has_pdf": "true", + "unknown_field": "something", + } + issues = validate_entry_fields(entry, "frontmatter", SAMPLE_REGISTRY) + drift = [i for i in issues if i["code"] == "DRIFT"] + assert len(drift) == 1 + assert drift[0]["severity"] == "warning" + assert drift[0]["field"] == "unknown_field" + + +def test_missing_optional_field(): + entry = { + "zotero_key": "ABCDEF", + "domain": "骨科", + "title": "A Study", + "has_pdf": "true", + } + issues = validate_entry_fields(entry, "frontmatter", SAMPLE_REGISTRY) + optional_missing = [i for i in issues if i["code"] == "MISSING_OPTIONAL"] + assert len(optional_missing) > 0 + assert all(i["severity"] == "info" for i in optional_missing) + + +def test_validate_collection_mixed(): + entries = [ + {"zotero_key": "A1", "domain": "骨科", "title": "T1", "has_pdf": "true"}, + {"zotero_key": "A2", "domain": "骨科", "has_pdf": "true"}, + ] + result = validate_collection(entries, "frontmatter", SAMPLE_REGISTRY) + assert result["total_entries"] == 2 + assert result["entries_with_errors"] == 1 + assert "MISSING_REQUIRED" in result["summary"] + + +def test_validate_frontmatter_valid(tmp_path): + note = tmp_path / "ABCDEF - Test.md" + note.write_text( + "---\nzotero_key: ABCDEF\ndomain: 骨科\ntitle: Test\nhas_pdf: true\n---\n\nSome content", + encoding="utf-8", + ) + issues = validate_frontmatter_from_file(note, SAMPLE_REGISTRY) + required_errors = [i for i in issues if i["severity"] == "error"] + assert len(required_errors) == 0 + + +def test_validate_frontmatter_no_frontmatter(tmp_path): + note = tmp_path / "NoFrontmatter.md" + note.write_text("Just content without frontmatter", encoding="utf-8") + issues = validate_frontmatter_from_file(note, SAMPLE_REGISTRY) + assert len(issues) == 1 + assert issues[0]["code"] == "NO_FRONTMATTER" diff --git a/tests/unit/schema/__init__.py b/tests/unit/schema/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/unit/schema/test_field_registry.py b/tests/unit/schema/test_field_registry.py new file mode 100644 index 00000000..a7fa5bc6 --- /dev/null +++ b/tests/unit/schema/test_field_registry.py @@ -0,0 +1,90 @@ +from __future__ import annotations + +from pathlib import Path + +from paperforge.schema import ( + get_field_info, + get_owner_fields, + load_field_registry, +) + +HERE = Path(__file__).resolve().parent +REGISTRY_PATH = HERE.parent.parent.parent / "paperforge" / "schema" / "field_registry.yaml" + + +class TestLoadFieldRegistry: + def test_valid_path_returns_non_empty(self) -> None: + reg = load_field_registry(REGISTRY_PATH) + assert isinstance(reg, dict) + assert len(reg) > 0 + + def test_nonexistent_path_returns_empty(self) -> None: + reg = load_field_registry(Path("/nonexistent/path.yaml")) + assert reg == {} + + def test_default_path_works(self) -> None: + reg = load_field_registry() + assert isinstance(reg, dict) + assert len(reg) > 0 + + +class TestRegistryOwners: + def test_has_all_three_owners(self) -> None: + reg = load_field_registry(REGISTRY_PATH) + assert "frontmatter" in reg + assert "index_entry" in reg + assert "ocr_meta" in reg + assert len(reg) == 3 + + +class TestFrontmatterFields: + def test_has_required_fields(self) -> None: + reg = load_field_registry(REGISTRY_PATH) + fm = get_owner_fields(reg, "frontmatter") + required = {"zotero_key", "domain", "title", "has_pdf"} + for field in required: + assert field in fm, f"Missing required frontmatter field: {field}" + assert fm[field]["required"] is True, f"{field} should be required" + + def test_all_expected_fields_present(self) -> None: + reg = load_field_registry(REGISTRY_PATH) + fm = get_owner_fields(reg, "frontmatter") + expected = { + "zotero_key", "domain", "title", "year", "doi", + "collection_path", "has_pdf", "pdf_path", "supplementary", + "fulltext_md_path", "recommend_analyze", "analyze", + "do_ocr", "ocr_status", "deep_reading_status", "path_error", + } + assert set(fm.keys()) == expected + + +class TestGetOwnerFields: + def test_returns_correct_owner(self) -> None: + reg = load_field_registry(REGISTRY_PATH) + fm = get_owner_fields(reg, "frontmatter") + assert "zotero_key" in fm + assert fm["zotero_key"]["type"] == "str" + + def test_unknown_owner_returns_empty(self) -> None: + reg = load_field_registry(REGISTRY_PATH) + assert get_owner_fields(reg, "nonexistent") == {} + + +class TestGetFieldInfo: + def test_known_field_returns_metadata(self) -> None: + reg = load_field_registry(REGISTRY_PATH) + info = get_field_info(reg, "frontmatter", "zotero_key") + assert info is not None + assert info["type"] == "str" + assert info["required"] is True + assert info["public"] is True + + def test_unknown_field_returns_none(self) -> None: + reg = load_field_registry(REGISTRY_PATH) + info = get_field_info(reg, "frontmatter", "nonexistent_field") + assert info is None + + def test_unknown_owner_field_returns_none(self) -> None: + reg = load_field_registry(REGISTRY_PATH) + info = get_field_info(reg, "nonexistent", "zotero_key") + assert info is None diff --git a/tests/unit/services/__init__.py b/tests/unit/services/__init__.py new file mode 100644 index 00000000..e69de29b