Files
huihuiSquare/digital-avatar-app/backend/services/pdf_ocr_service.py
T

131 lines
4.7 KiB
Python

"""OCR fallback for image-only PDF knowledge documents."""
import logging
import os
import time
from typing import Callable
from sqlalchemy.orm import Session
from models import Avatar
from services.chat_model_config import get_chat_model_config
from services.token_billing import (
estimate_fallback_usage,
release_reservation,
reserve_avatar_tokens,
settle_reservation,
)
from services.vision_service import call_vision_model, prepare_image
logger = logging.getLogger(__name__)
PDF_OCR_PROMPT = (
"请逐字转录这一页扫描文档中的全部可见文字和表格,只输出转录内容,不要解释,不要使用 Markdown 代码块。"
"保留标题、段落、项目编号、数值和自然换行;看不清的内容写作[无法辨认],不要猜测、纠错或补全。"
)
def _positive_int(name: str, default: int, minimum: int, maximum: int) -> int:
try:
value = int(os.getenv(name, str(default)))
except ValueError:
value = default
return max(minimum, min(maximum, value))
def extract_scanned_pdf_text(
db: Session,
avatar: Avatar,
path: str,
*,
on_progress: Callable[[int, int], None] | None = None,
) -> str:
"""Render and OCR an image-only PDF while preserving page order."""
try:
import pymupdf
except ImportError as exc:
raise RuntimeError("扫描型 PDF 识别组件未安装") from exc
max_pages = _positive_int("KNOWLEDGE_PDF_OCR_MAX_PAGES", 80, 1, 300)
render_dpi = _positive_int("KNOWLEDGE_PDF_OCR_DPI", 144, 96, 200)
max_attempts = _positive_int("KNOWLEDGE_PDF_OCR_ATTEMPTS", 3, 1, 5)
model_config = get_chat_model_config()
model = model_config.ocr_model or model_config.vision_model
if not model_config.api_key or not model:
raise RuntimeError("扫描型 PDF 需要配置视觉 OCR 模型")
texts: list[str] = []
with pymupdf.open(path) as document:
total_pages = document.page_count
if total_pages <= 0:
raise ValueError("PDF 没有可识别页面")
if total_pages > max_pages:
raise ValueError(
f"扫描型 PDF 共 {total_pages} 页,超过单次 OCR 上限 {max_pages} 页,请拆分后上传"
)
scale = render_dpi / 72
for page_index in range(total_pages):
page = document.load_page(page_index)
pixmap = page.get_pixmap(
matrix=pymupdf.Matrix(scale, scale),
colorspace=pymupdf.csRGB,
alpha=False,
)
prepared = prepare_image(pixmap.tobytes("jpeg", jpg_quality=88))
estimate_messages = [{
"role": "user",
"content": f"[扫描 PDF 第 {page_index + 1}/{total_pages} 页]\n{PDF_OCR_PROMPT}",
}]
reservation = reserve_avatar_tokens(
db,
avatar,
"knowledge_pdf_ocr",
model,
estimate_messages,
model_config.vision_max_tokens,
)
try:
result = None
for attempt in range(1, max_attempts + 1):
try:
result = call_vision_model(
prepared,
model_config,
model=model,
prompt=PDF_OCR_PROMPT,
json_output=False,
)
break
except RuntimeError:
if attempt == max_attempts:
raise
time.sleep(min(4, attempt))
content = str((result or {}).get("content") or "").strip()
if not content:
raise RuntimeError("扫描型 PDF 页面识别结果为空")
settle_reservation(
db,
reservation,
(result or {}).get("usage"),
fallback_total=estimate_fallback_usage(estimate_messages, content),
)
except Exception as exc:
release_reservation(db, reservation, str(exc))
raise RuntimeError(
f"扫描型 PDF 第 {page_index + 1}/{total_pages} 页识别失败:{exc}"
) from exc
texts.append(f"[第 {page_index + 1} 页]\n{content}")
if on_progress:
on_progress(page_index + 1, total_pages)
logger.info(
"Scanned PDF OCR completed avatar=%s page=%s/%s",
avatar.id,
page_index + 1,
total_pages,
)
return "\n\n".join(texts).strip()