lllin000_PaperForge/paperforge/embedding/status.py
LLLin000 0dfbf2a68d feat: PR 8 — Object Vector Retrieval (paperforge_objects collection)
New paperforge_objects collection in ChromaDB for figure/table caption
vector search, completing the memory layer retrieval triad:

  paperforge_fulltext → legacy_chunk
  paperforge_body     → body_unit
  paperforge_objects  → object_unit (NEW)

Changes:
- manifest.py: compute_object_units_hash(), manifest records both hashes
- _chroma.py: _COLLECTION_NAMES includes paperforge_objects
- builder.py: get_object_units_for_embedding() + embed_object_units()
- status.py: object_chunk_count, total_chunks
- state_snapshot.py: write_vector_runtime() extended with new fields
- search.py: 3rd collection, dict-based source mapping, object_kind/label
- __init__.py: exports for new functions
- commands/embed.py: _has_object_units_in_db, body+object structured
  path routing with dual-hash resume, extended counters
- Test fix: test_build_paper_manifest data complete for hash functions

Verification: 20 object vectors embedded, all per-paper counts match,
status total consistent (58266 = 57435 + 811 + 20),
object caption queries return object_unit results via merge_retrieve.
Unit tests: 153 pass, 1 skip.
2026-07-06 19:58:46 +08:00

91 lines
2.6 KiB
Python

from __future__ import annotations
import logging
from pathlib import Path
from paperforge.embedding._chroma import get_collection, get_vector_db_path
from paperforge.embedding._config import get_api_model
from paperforge.embedding.backends import get_vector_backend
logger = logging.getLogger(__name__)
def _module_available(name: str) -> bool:
"""Check whether *name* can be imported without side effects."""
import importlib
try:
importlib.import_module(name)
return True
except ImportError:
return False
def get_available_backends() -> dict[str, dict]:
"""Return a dict of known backends with installation and selection status."""
return {
"chroma": {
"installed": True,
"selected": True,
"supports_hybrid": False,
"supports_multimodal": False,
},
"lancedb": {
"installed": _module_available("lancedb"),
"selected": False,
"supports_hybrid": True,
"supports_multimodal": True,
},
}
def get_embed_status(vault: Path) -> dict:
"""Get vector DB status across both collections."""
db_path = get_vector_db_path(vault)
exists = db_path.exists()
chunk_count = 0
object_chunk_count = 0
body_chunk_count = 0
error = ""
corrupted = False
if exists:
# Count paperforge_fulltext
try:
col = get_collection(vault, name="paperforge_fulltext")
chunk_count = col.count()
except Exception:
pass
# Count paperforge_body
try:
col_body = get_collection(vault, name="paperforge_body")
body_chunk_count = col_body.count()
except Exception:
pass
# Count paperforge_objects
try:
col_o = get_collection(vault, name="paperforge_objects")
object_chunk_count = col_o.count()
except Exception:
pass
# Backend health (checks primary collection)
backend = get_vector_backend(vault)
health = backend.health()
healthy = health["healthy"]
error = health.get("error", "")
corrupted = health.get("corrupted", False)
model = get_api_model(vault)
return {
"db_exists": exists,
"chunk_count": chunk_count,
"body_chunk_count": body_chunk_count,
"object_chunk_count": object_chunk_count,
"total_chunks": chunk_count + body_chunk_count + object_chunk_count,
"model": model,
"mode": "api",
"healthy": healthy,
"corrupted": corrupted,
"error": error,
}