131 lines
4.7 KiB
Python
131 lines
4.7 KiB
Python
"""OCR fallback for image-only PDF knowledge documents."""
|
|
|
|
import logging
|
|
import os
|
|
import time
|
|
from typing import Callable
|
|
|
|
from sqlalchemy.orm import Session
|
|
|
|
from models import Avatar
|
|
from services.chat_model_config import get_chat_model_config
|
|
from services.token_billing import (
|
|
estimate_fallback_usage,
|
|
release_reservation,
|
|
reserve_avatar_tokens,
|
|
settle_reservation,
|
|
)
|
|
from services.vision_service import call_vision_model, prepare_image
|
|
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
PDF_OCR_PROMPT = (
|
|
"请逐字转录这一页扫描文档中的全部可见文字和表格,只输出转录内容,不要解释,不要使用 Markdown 代码块。"
|
|
"保留标题、段落、项目编号、数值和自然换行;看不清的内容写作[无法辨认],不要猜测、纠错或补全。"
|
|
)
|
|
|
|
|
|
def _positive_int(name: str, default: int, minimum: int, maximum: int) -> int:
|
|
try:
|
|
value = int(os.getenv(name, str(default)))
|
|
except ValueError:
|
|
value = default
|
|
return max(minimum, min(maximum, value))
|
|
|
|
|
|
def extract_scanned_pdf_text(
|
|
db: Session,
|
|
avatar: Avatar,
|
|
path: str,
|
|
*,
|
|
on_progress: Callable[[int, int], None] | None = None,
|
|
) -> str:
|
|
"""Render and OCR an image-only PDF while preserving page order."""
|
|
try:
|
|
import pymupdf
|
|
except ImportError as exc:
|
|
raise RuntimeError("扫描型 PDF 识别组件未安装") from exc
|
|
|
|
max_pages = _positive_int("KNOWLEDGE_PDF_OCR_MAX_PAGES", 80, 1, 300)
|
|
render_dpi = _positive_int("KNOWLEDGE_PDF_OCR_DPI", 144, 96, 200)
|
|
max_attempts = _positive_int("KNOWLEDGE_PDF_OCR_ATTEMPTS", 3, 1, 5)
|
|
model_config = get_chat_model_config()
|
|
model = model_config.ocr_model or model_config.vision_model
|
|
if not model_config.api_key or not model:
|
|
raise RuntimeError("扫描型 PDF 需要配置视觉 OCR 模型")
|
|
|
|
texts: list[str] = []
|
|
with pymupdf.open(path) as document:
|
|
total_pages = document.page_count
|
|
if total_pages <= 0:
|
|
raise ValueError("PDF 没有可识别页面")
|
|
if total_pages > max_pages:
|
|
raise ValueError(
|
|
f"扫描型 PDF 共 {total_pages} 页,超过单次 OCR 上限 {max_pages} 页,请拆分后上传"
|
|
)
|
|
|
|
scale = render_dpi / 72
|
|
for page_index in range(total_pages):
|
|
page = document.load_page(page_index)
|
|
pixmap = page.get_pixmap(
|
|
matrix=pymupdf.Matrix(scale, scale),
|
|
colorspace=pymupdf.csRGB,
|
|
alpha=False,
|
|
)
|
|
prepared = prepare_image(pixmap.tobytes("jpeg", jpg_quality=88))
|
|
estimate_messages = [{
|
|
"role": "user",
|
|
"content": f"[扫描 PDF 第 {page_index + 1}/{total_pages} 页]\n{PDF_OCR_PROMPT}",
|
|
}]
|
|
reservation = reserve_avatar_tokens(
|
|
db,
|
|
avatar,
|
|
"knowledge_pdf_ocr",
|
|
model,
|
|
estimate_messages,
|
|
model_config.vision_max_tokens,
|
|
)
|
|
try:
|
|
result = None
|
|
for attempt in range(1, max_attempts + 1):
|
|
try:
|
|
result = call_vision_model(
|
|
prepared,
|
|
model_config,
|
|
model=model,
|
|
prompt=PDF_OCR_PROMPT,
|
|
json_output=False,
|
|
)
|
|
break
|
|
except RuntimeError:
|
|
if attempt == max_attempts:
|
|
raise
|
|
time.sleep(min(4, attempt))
|
|
content = str((result or {}).get("content") or "").strip()
|
|
if not content:
|
|
raise RuntimeError("扫描型 PDF 页面识别结果为空")
|
|
settle_reservation(
|
|
db,
|
|
reservation,
|
|
(result or {}).get("usage"),
|
|
fallback_total=estimate_fallback_usage(estimate_messages, content),
|
|
)
|
|
except Exception as exc:
|
|
release_reservation(db, reservation, str(exc))
|
|
raise RuntimeError(
|
|
f"扫描型 PDF 第 {page_index + 1}/{total_pages} 页识别失败:{exc}"
|
|
) from exc
|
|
|
|
texts.append(f"[第 {page_index + 1} 页]\n{content}")
|
|
if on_progress:
|
|
on_progress(page_index + 1, total_pages)
|
|
logger.info(
|
|
"Scanned PDF OCR completed avatar=%s page=%s/%s",
|
|
avatar.id,
|
|
page_index + 1,
|
|
total_pages,
|
|
)
|
|
|
|
return "\n\n".join(texts).strip()
|