fix(avatar): OCR image-only knowledge PDFs
This commit is contained in:
@@ -8,7 +8,8 @@ import threading
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from database import SessionLocal
|
||||
from models import KnowledgeChunk, KnowledgeDoc
|
||||
from models import Avatar, KnowledgeChunk, KnowledgeDoc
|
||||
from services.pdf_ocr_service import extract_scanned_pdf_text
|
||||
import embeddings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -76,7 +77,23 @@ class KnowledgeVectorizer:
|
||||
|
||||
self._set_progress(db, doc, "extracting", 8)
|
||||
text = embeddings.extract_text(path, f".{doc.file_type}")
|
||||
self._set_progress(db, doc, "chunking", 22)
|
||||
if doc.file_type == "pdf" and not text.strip():
|
||||
avatar = db.get(Avatar, doc.avatar_id)
|
||||
if not avatar:
|
||||
raise ValueError("文档所属分身不存在")
|
||||
|
||||
def ocr_progress(done: int, total: int):
|
||||
percent = 8 + int((done / max(1, total)) * 20)
|
||||
self._set_progress(db, doc, "ocr", min(percent, 28))
|
||||
|
||||
self._set_progress(db, doc, "ocr", 8)
|
||||
text = extract_scanned_pdf_text(
|
||||
db,
|
||||
avatar,
|
||||
path,
|
||||
on_progress=ocr_progress,
|
||||
)
|
||||
self._set_progress(db, doc, "chunking", 29)
|
||||
chunks = embeddings.chunk_text(text)
|
||||
if not chunks:
|
||||
raise ValueError("文档没有可建立索引的文字内容")
|
||||
|
||||
@@ -0,0 +1,130 @@
|
||||
"""OCR fallback for image-only PDF knowledge documents."""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from typing import Callable
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from models import Avatar
|
||||
from services.chat_model_config import get_chat_model_config
|
||||
from services.token_billing import (
|
||||
estimate_fallback_usage,
|
||||
release_reservation,
|
||||
reserve_avatar_tokens,
|
||||
settle_reservation,
|
||||
)
|
||||
from services.vision_service import call_vision_model, prepare_image
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
PDF_OCR_PROMPT = (
|
||||
"请逐字转录这一页扫描文档中的全部可见文字和表格,只输出转录内容,不要解释,不要使用 Markdown 代码块。"
|
||||
"保留标题、段落、项目编号、数值和自然换行;看不清的内容写作[无法辨认],不要猜测、纠错或补全。"
|
||||
)
|
||||
|
||||
|
||||
def _positive_int(name: str, default: int, minimum: int, maximum: int) -> int:
|
||||
try:
|
||||
value = int(os.getenv(name, str(default)))
|
||||
except ValueError:
|
||||
value = default
|
||||
return max(minimum, min(maximum, value))
|
||||
|
||||
|
||||
def extract_scanned_pdf_text(
|
||||
db: Session,
|
||||
avatar: Avatar,
|
||||
path: str,
|
||||
*,
|
||||
on_progress: Callable[[int, int], None] | None = None,
|
||||
) -> str:
|
||||
"""Render and OCR an image-only PDF while preserving page order."""
|
||||
try:
|
||||
import pymupdf
|
||||
except ImportError as exc:
|
||||
raise RuntimeError("扫描型 PDF 识别组件未安装") from exc
|
||||
|
||||
max_pages = _positive_int("KNOWLEDGE_PDF_OCR_MAX_PAGES", 80, 1, 300)
|
||||
render_dpi = _positive_int("KNOWLEDGE_PDF_OCR_DPI", 144, 96, 200)
|
||||
max_attempts = _positive_int("KNOWLEDGE_PDF_OCR_ATTEMPTS", 3, 1, 5)
|
||||
model_config = get_chat_model_config()
|
||||
model = model_config.ocr_model or model_config.vision_model
|
||||
if not model_config.api_key or not model:
|
||||
raise RuntimeError("扫描型 PDF 需要配置视觉 OCR 模型")
|
||||
|
||||
texts: list[str] = []
|
||||
with pymupdf.open(path) as document:
|
||||
total_pages = document.page_count
|
||||
if total_pages <= 0:
|
||||
raise ValueError("PDF 没有可识别页面")
|
||||
if total_pages > max_pages:
|
||||
raise ValueError(
|
||||
f"扫描型 PDF 共 {total_pages} 页,超过单次 OCR 上限 {max_pages} 页,请拆分后上传"
|
||||
)
|
||||
|
||||
scale = render_dpi / 72
|
||||
for page_index in range(total_pages):
|
||||
page = document.load_page(page_index)
|
||||
pixmap = page.get_pixmap(
|
||||
matrix=pymupdf.Matrix(scale, scale),
|
||||
colorspace=pymupdf.csRGB,
|
||||
alpha=False,
|
||||
)
|
||||
prepared = prepare_image(pixmap.tobytes("jpeg", jpg_quality=88))
|
||||
estimate_messages = [{
|
||||
"role": "user",
|
||||
"content": f"[扫描 PDF 第 {page_index + 1}/{total_pages} 页]\n{PDF_OCR_PROMPT}",
|
||||
}]
|
||||
reservation = reserve_avatar_tokens(
|
||||
db,
|
||||
avatar,
|
||||
"knowledge_pdf_ocr",
|
||||
model,
|
||||
estimate_messages,
|
||||
model_config.vision_max_tokens,
|
||||
)
|
||||
try:
|
||||
result = None
|
||||
for attempt in range(1, max_attempts + 1):
|
||||
try:
|
||||
result = call_vision_model(
|
||||
prepared,
|
||||
model_config,
|
||||
model=model,
|
||||
prompt=PDF_OCR_PROMPT,
|
||||
json_output=False,
|
||||
)
|
||||
break
|
||||
except RuntimeError:
|
||||
if attempt == max_attempts:
|
||||
raise
|
||||
time.sleep(min(4, attempt))
|
||||
content = str((result or {}).get("content") or "").strip()
|
||||
if not content:
|
||||
raise RuntimeError("扫描型 PDF 页面识别结果为空")
|
||||
settle_reservation(
|
||||
db,
|
||||
reservation,
|
||||
(result or {}).get("usage"),
|
||||
fallback_total=estimate_fallback_usage(estimate_messages, content),
|
||||
)
|
||||
except Exception as exc:
|
||||
release_reservation(db, reservation, str(exc))
|
||||
raise RuntimeError(
|
||||
f"扫描型 PDF 第 {page_index + 1}/{total_pages} 页识别失败:{exc}"
|
||||
) from exc
|
||||
|
||||
texts.append(f"[第 {page_index + 1} 页]\n{content}")
|
||||
if on_progress:
|
||||
on_progress(page_index + 1, total_pages)
|
||||
logger.info(
|
||||
"Scanned PDF OCR completed avatar=%s page=%s/%s",
|
||||
avatar.id,
|
||||
page_index + 1,
|
||||
total_pages,
|
||||
)
|
||||
|
||||
return "\n\n".join(texts).strip()
|
||||
Reference in New Issue
Block a user