"""OCR fallback for image-only PDF knowledge documents.""" import logging import os import time from typing import Callable from sqlalchemy.orm import Session from models import Avatar from services.chat_model_config import get_chat_model_config from services.token_billing import ( estimate_fallback_usage, release_reservation, reserve_avatar_tokens, settle_reservation, ) from services.vision_service import call_vision_model, prepare_image logger = logging.getLogger(__name__) PDF_OCR_PROMPT = ( "请逐字转录这一页扫描文档中的全部可见文字和表格,只输出转录内容,不要解释,不要使用 Markdown 代码块。" "保留标题、段落、项目编号、数值和自然换行;看不清的内容写作[无法辨认],不要猜测、纠错或补全。" ) def _positive_int(name: str, default: int, minimum: int, maximum: int) -> int: try: value = int(os.getenv(name, str(default))) except ValueError: value = default return max(minimum, min(maximum, value)) def extract_scanned_pdf_text( db: Session, avatar: Avatar, path: str, *, on_progress: Callable[[int, int], None] | None = None, ) -> str: """Render and OCR an image-only PDF while preserving page order.""" try: import pymupdf except ImportError as exc: raise RuntimeError("扫描型 PDF 识别组件未安装") from exc max_pages = _positive_int("KNOWLEDGE_PDF_OCR_MAX_PAGES", 80, 1, 300) render_dpi = _positive_int("KNOWLEDGE_PDF_OCR_DPI", 144, 96, 200) max_attempts = _positive_int("KNOWLEDGE_PDF_OCR_ATTEMPTS", 3, 1, 5) model_config = get_chat_model_config() model = model_config.ocr_model or model_config.vision_model if not model_config.api_key or not model: raise RuntimeError("扫描型 PDF 需要配置视觉 OCR 模型") texts: list[str] = [] with pymupdf.open(path) as document: total_pages = document.page_count if total_pages <= 0: raise ValueError("PDF 没有可识别页面") if total_pages > max_pages: raise ValueError( f"扫描型 PDF 共 {total_pages} 页,超过单次 OCR 上限 {max_pages} 页,请拆分后上传" ) scale = render_dpi / 72 for page_index in range(total_pages): page = document.load_page(page_index) pixmap = page.get_pixmap( matrix=pymupdf.Matrix(scale, scale), colorspace=pymupdf.csRGB, alpha=False, ) prepared = prepare_image(pixmap.tobytes("jpeg", jpg_quality=88)) estimate_messages = [{ "role": "user", "content": f"[扫描 PDF 第 {page_index + 1}/{total_pages} 页]\n{PDF_OCR_PROMPT}", }] reservation = reserve_avatar_tokens( db, avatar, "knowledge_pdf_ocr", model, estimate_messages, model_config.vision_max_tokens, ) try: result = None for attempt in range(1, max_attempts + 1): try: result = call_vision_model( prepared, model_config, model=model, prompt=PDF_OCR_PROMPT, json_output=False, ) break except RuntimeError: if attempt == max_attempts: raise time.sleep(min(4, attempt)) content = str((result or {}).get("content") or "").strip() if not content: raise RuntimeError("扫描型 PDF 页面识别结果为空") settle_reservation( db, reservation, (result or {}).get("usage"), fallback_total=estimate_fallback_usage(estimate_messages, content), ) except Exception as exc: release_reservation(db, reservation, str(exc)) raise RuntimeError( f"扫描型 PDF 第 {page_index + 1}/{total_pages} 页识别失败:{exc}" ) from exc texts.append(f"[第 {page_index + 1} 页]\n{content}") if on_progress: on_progress(page_index + 1, total_pages) logger.info( "Scanned PDF OCR completed avatar=%s page=%s/%s", avatar.id, page_index + 1, total_pages, ) return "\n\n".join(texts).strip()