fix(avatar): ground replies in recognized images
This commit is contained in:
@@ -47,6 +47,25 @@ QA_SEMANTIC_THRESHOLD = 0.72
|
||||
QA_MATCH_MARGIN = 0.06
|
||||
KNOWLEDGE_MIN_SCORE = float(os.getenv("KNOWLEDGE_MIN_SCORE", "0.42"))
|
||||
|
||||
_IMAGE_ACCESS_DENIAL_PATTERNS = (
|
||||
re.compile(
|
||||
r"(?:我|目前|暂时|这里|本身|系统)?\s*(?:无法|不能|没法|不支持)\s*"
|
||||
r"(?:直接)?\s*(?:查看|看到|看见|识别|读取|访问|打开|分析|理解)"
|
||||
r"(?:\s*(?:或|、|/)\s*(?:查看|看到|看见|识别|读取|访问|打开|分析|理解))*\s*"
|
||||
r"(?:你(?:发|提供|上传)的|这张|该|当前)?\s*(?:图片|图像|照片|影像|文件)"
|
||||
),
|
||||
re.compile(
|
||||
r"(?:我|这里|目前|暂时)?\s*(?:看不到|看不见|未看到|没有看到|没收到|未收到)\s*"
|
||||
r"(?:你(?:发|提供|上传)的|这张|该|当前)?\s*(?:图片|图像|照片|影像)"
|
||||
),
|
||||
re.compile(
|
||||
r"\b(?:i\s+)?(?:can(?:not|'t)|am\s+unable\s+to)\s+(?:directly\s+)?"
|
||||
r"(?:view|see|access|read|analy[sz]e|recogni[sz]e)\s+"
|
||||
r"(?:the\s+|this\s+|your\s+)?(?:image|photo|picture|scan)\b",
|
||||
re.IGNORECASE,
|
||||
),
|
||||
)
|
||||
|
||||
_WRITING_SYSTEM_PATTERNS = {
|
||||
"han": re.compile(r"[\u3400-\u4dbf\u4e00-\u9fff]"),
|
||||
"latin": re.compile(r"[A-Za-z\u00c0-\u024f]"),
|
||||
@@ -178,6 +197,75 @@ def _image_retrieval_question(question: str, image_contexts: list[dict]) -> str:
|
||||
return "\n".join(part for part in parts if part).strip()
|
||||
|
||||
|
||||
def _answer_denies_available_image(answer: str) -> bool:
|
||||
"""Reject only whole-image access denials, not uncertainty about one field."""
|
||||
value = re.sub(r"\s+", " ", answer or "").strip()
|
||||
return any(pattern.search(value) for pattern in _IMAGE_ACCESS_DENIAL_PATTERNS)
|
||||
|
||||
|
||||
def _compact_context_text(value: Any, limit: int) -> str:
|
||||
lines = [re.sub(r"\s+", " ", line).strip() for line in str(value or "").splitlines()]
|
||||
text = "\n".join(line for line in lines if line).strip()
|
||||
return text[:limit].rstrip()
|
||||
|
||||
|
||||
def _grounded_image_fallback(question: str, image_contexts: list[dict]) -> str:
|
||||
"""Build a safe answer from completed vision data when the chat model contradicts it."""
|
||||
summaries: list[str] = []
|
||||
facts: list[str] = []
|
||||
excerpts: list[str] = []
|
||||
warnings: list[str] = []
|
||||
for context in image_contexts:
|
||||
summary = _compact_context_text(context.get("summary"), 500)
|
||||
if summary:
|
||||
summaries.append(summary)
|
||||
structured = context.get("structuredData") or {}
|
||||
if isinstance(structured, dict):
|
||||
for fact in structured.get("key_facts") or []:
|
||||
value = _compact_context_text(fact, 300)
|
||||
if value:
|
||||
facts.append(value)
|
||||
extracted = _compact_context_text(context.get("extractedText"), 900)
|
||||
if extracted:
|
||||
excerpts.append(extracted)
|
||||
warning = _compact_context_text(context.get("warning"), 300)
|
||||
if warning:
|
||||
warnings.append(warning)
|
||||
|
||||
summaries = list(dict.fromkeys(summaries))
|
||||
facts = list(dict.fromkeys(facts))[:6]
|
||||
excerpts = list(dict.fromkeys(excerpts))
|
||||
warnings = list(dict.fromkeys(warnings))
|
||||
writing_system = _dominant_writing_system(question)
|
||||
|
||||
if writing_system == "latin":
|
||||
parts = []
|
||||
if summaries:
|
||||
parts.append("From the image, I can confirm: " + " ".join(summaries))
|
||||
if facts:
|
||||
parts.append("Key details:\n" + "\n".join(
|
||||
f"{index}. {fact}" for index, fact in enumerate(facts, 1)
|
||||
))
|
||||
elif excerpts:
|
||||
parts.append("Visible text:\n" + excerpts[0])
|
||||
if warnings:
|
||||
parts.append("Please note: " + " ".join(warnings))
|
||||
return "\n".join(parts).strip() or "The image is available, but there is not enough clear detail to confirm more."
|
||||
|
||||
parts = []
|
||||
if summaries:
|
||||
parts.append("从这张图中可以确认:" + ";".join(summaries).rstrip("。;") + "。")
|
||||
if facts:
|
||||
parts.append("其中比较明确的信息有:\n" + "\n".join(
|
||||
f"{index}. {fact}" for index, fact in enumerate(facts, 1)
|
||||
))
|
||||
elif excerpts:
|
||||
parts.append("图中可见的主要文字是:\n" + excerpts[0])
|
||||
if warnings:
|
||||
parts.append("需要注意:" + ";".join(warnings).rstrip("。;") + "。")
|
||||
return "\n".join(parts).strip() or "这张图已经看到了,但目前能确认的清晰信息比较有限。"
|
||||
|
||||
|
||||
def _run_billed_vision_call(
|
||||
db: Session,
|
||||
avatar: Avatar,
|
||||
@@ -553,8 +641,10 @@ def _build_prompt(
|
||||
if image_contexts:
|
||||
image_material = json.dumps(image_contexts, ensure_ascii=False, default=str)
|
||||
system += (
|
||||
"\n以下是当前会话图片经过视觉识别后得到的资料:\n"
|
||||
"\n当前会话图片已经成功读取并完成内容识别,以下资料就是可直接使用的图片内容:\n"
|
||||
f"{image_material}"
|
||||
"\n必须直接依据这些图片内容回答当前问题。禁止声称无法查看、看不到、未收到、无法识别、"
|
||||
"无法读取或不能访问图片,也不要要求对方重新上传;只有资料明确标记读取失败时才可以请对方重发。"
|
||||
"\n图片资料可能包含 OCR 错字、模糊内容或用户尚未确认的信息,只能按可见内容谨慎表达。"
|
||||
"标准答题对中的事实优先级高于图片资料,知识库事实优先级高于模型推测;发生冲突时遵循更高优先级资料,"
|
||||
"并自然提醒对方核对原图。不得声称看到了图片中不存在的内容。"
|
||||
@@ -815,6 +905,14 @@ def _resolve_reply(
|
||||
except Exception as exc:
|
||||
release_reservation(db, reservation, str(exc))
|
||||
raise
|
||||
answer = str(answer or "").strip()
|
||||
if image_contexts and _answer_denies_available_image(answer):
|
||||
logger.warning(
|
||||
"chat model contradicted ready image context avatar=%s source=%s",
|
||||
avatar.id,
|
||||
usage_source,
|
||||
)
|
||||
answer = _grounded_image_fallback(question, image_contexts)
|
||||
result = {
|
||||
"answer": answer,
|
||||
"source": "qa" if matched else (
|
||||
|
||||
Reference in New Issue
Block a user