feat(avatar): 多文件知识库上传与进度展示 #17

Merged
stefanfeng merged 1 commits from codex/avatar-upload-progress-20260904 into main 2026-09-04 17:33:06 +08:00
9 changed files with 129 additions and 30 deletions
+2
View File
@@ -54,6 +54,8 @@ def init_db():
("knowledge_docs", "chunk_count", "INTEGER DEFAULT 0"),
("knowledge_docs", "vectorized_at", "TIMESTAMP"),
("knowledge_docs", "error_message", "VARCHAR DEFAULT ''"),
("knowledge_docs", "index_stage", "VARCHAR DEFAULT ''"),
("knowledge_docs", "index_progress", "INTEGER DEFAULT 0"),
("avatars", "owner_id", "VARCHAR DEFAULT ''"),
("authorizations", "takeover_enabled", "BOOLEAN DEFAULT 0"),
("authorizations", "takeover_mode", "VARCHAR DEFAULT 'immediate'"),
+8 -2
View File
@@ -51,7 +51,7 @@ def _hash_embedding(texts, dim=EMBED_DIM):
return vecs
def embed(texts):
def embed(texts, on_progress=None):
"""返回 list[list[float]],与输入顺序一致。"""
if not texts:
return []
@@ -64,6 +64,7 @@ def embed(texts):
except ValueError:
batch_size = 10
embeddings = []
total = len(texts)
for start in range(0, len(texts), batch_size):
batch = texts[start:start + batch_size]
payload = json.dumps({"input": batch, "model": model}).encode("utf-8")
@@ -84,8 +85,13 @@ def embed(texts):
if len(items) != len(batch):
raise ValueError("embedding response count does not match request")
embeddings.extend(item["embedding"] for item in items)
if on_progress:
on_progress(len(embeddings), total)
return embeddings
return _hash_embedding(texts)
vectors = _hash_embedding(texts)
if on_progress:
on_progress(len(vectors), len(texts))
return vectors
def cosine(a, b):
+4
View File
@@ -191,6 +191,8 @@ class KnowledgeDoc(Base):
file_url = Column(String, default="")
status = Column(String, default="uploaded") # uploaded | parsing | ready | failed
error_message = Column(String, default="") # 建立索引失败原因
index_stage = Column(String, default="") # queued | extracting | chunking | embedding | ready | failed
index_progress = Column(Integer, default=0) # 0-100
vectorized = Column(Boolean, default=False) # 是否已向量化
embedding_model = Column(String, default="") # 向量模型标识
chunk_count = Column(Integer, default=0) # 切片数量
@@ -207,6 +209,8 @@ class KnowledgeDoc(Base):
"fileUrl": self.file_url,
"status": self.status,
"errorMessage": self.error_message or "",
"indexStage": self.index_stage or "",
"indexProgress": int(self.index_progress or 0),
"vectorized": bool(self.vectorized),
"embeddingModel": self.embedding_model,
"chunkCount": self.chunk_count,
@@ -102,6 +102,8 @@ async def upload_doc(avatar_id: str, file: UploadFile = File(...), authorization
file_size=file_size,
file_url=f"/api/files/{avatar_id}/{stored}",
status="parsing",
index_stage="queued",
index_progress=0,
)
# Persist and acknowledge the upload first. Extraction and embeddings may take
@@ -134,6 +136,8 @@ def retry_doc(avatar_id: str, doc_id: str, authorization: str = Header(None), db
doc.chunk_count = 0
doc.vectorized_at = None
doc.error_message = ""
doc.index_stage = "queued"
doc.index_progress = 0
db.commit()
db.refresh(doc)
knowledge_vectorizer.enqueue(doc.id)
@@ -74,11 +74,19 @@ class KnowledgeVectorizer:
if not stored_name or not os.path.isfile(path):
raise FileNotFoundError("原文件不可用,请重新上传")
self._set_progress(db, doc, "extracting", 8)
text = embeddings.extract_text(path, f".{doc.file_type}")
self._set_progress(db, doc, "chunking", 22)
chunks = embeddings.chunk_text(text)
if not chunks:
raise ValueError("文档没有可建立索引的文字内容")
vectors = embeddings.embed(chunks)
self._set_progress(db, doc, "embedding", 30)
def embedding_progress(done: int, total: int):
percent = 30 + int((done / max(1, total)) * 65)
self._set_progress(db, doc, "embedding", min(percent, 95))
vectors = embeddings.embed(chunks, on_progress=embedding_progress)
if len(vectors) != len(chunks):
raise ValueError("向量服务返回数量与文档分段不一致")
@@ -103,6 +111,8 @@ class KnowledgeVectorizer:
doc.vectorized_at = datetime.now(timezone.utc)
doc.status = "ready"
doc.error_message = ""
doc.index_stage = "ready"
doc.index_progress = 100
db.commit()
logger.info("Knowledge document %s indexed with %s chunks", doc.id, len(chunks))
except Exception as exc:
@@ -116,10 +126,18 @@ class KnowledgeVectorizer:
failed_doc.chunk_count = 0
failed_doc.vectorized_at = None
failed_doc.error_message = str(exc)[:300] or "建立知识索引失败"
failed_doc.index_stage = "failed"
failed_doc.index_progress = 0
db.commit()
logger.exception("Knowledge vectorization failed for %s: %s", doc_id, exc)
finally:
db.close()
@staticmethod
def _set_progress(db, doc, stage: str, progress: int):
doc.index_stage = stage
doc.index_progress = progress
db.commit()
knowledge_vectorizer = KnowledgeVectorizer()
@@ -49,6 +49,7 @@ class RemoteEmbeddingTests(unittest.TestCase):
texts = [f"chunk-{index}" for index in range(14)]
batch_sizes = []
requested_urls = []
progress_updates = []
def fake_urlopen(request, timeout):
self.assertEqual(timeout, 30)
@@ -68,7 +69,10 @@ class RemoteEmbeddingTests(unittest.TestCase):
"EMBEDDING_MODEL": "text-embedding-v4",
"EMBEDDING_BATCH_SIZE": "10",
}), patch("embeddings.urllib.request.urlopen", side_effect=fake_urlopen):
result = embeddings.embed(texts)
result = embeddings.embed(
texts,
on_progress=lambda completed, total: progress_updates.append((completed, total)),
)
self.assertEqual(batch_sizes, [10, 4])
self.assertEqual(requested_urls, [
@@ -76,6 +80,7 @@ class RemoteEmbeddingTests(unittest.TestCase):
"https://embedding.example/v1/embeddings",
])
self.assertEqual(result, [[float(index)] for index in range(14)])
self.assertEqual(progress_updates, [(10, 14), (14, 14)])
def test_full_embedding_endpoint_is_not_modified(self):
self.assertEqual(
@@ -116,6 +116,8 @@ def test_background_vectorizer_commits_ready_document_and_chunks_together(
assert stored.status == "ready"
assert stored.vectorized is True
assert stored.chunk_count == 1
assert stored.index_stage == "ready"
assert stored.index_progress == 100
assert db.query(KnowledgeChunk).filter(KnowledgeChunk.doc_id == stored.id).count() == 1
db.query(KnowledgeChunk).filter(KnowledgeChunk.doc_id == stored.id).delete()
db.delete(stored)
+10 -2
View File
@@ -306,6 +306,8 @@ export interface KnowledgeDoc {
embeddingModel?: string
chunkCount?: number
errorMessage?: string
indexStage?: string
indexProgress?: number
createdAt: string
}
@@ -332,12 +334,18 @@ export const getKnowledgeDocs = (avatarId: string) =>
request.get<KnowledgeDoc[]>(`/avatar/${avatarId}/knowledge/docs`)
// 上传文档(支持 md/txt/pdf/doc/docx/xlsx)
export const uploadKnowledgeDoc = (avatarId: string, file: File) => {
export const uploadKnowledgeDoc = (
avatarId: string,
file: File,
onUploadProgress?: (loaded: number, total: number) => void
) => {
const form = new FormData()
form.append('file', file)
return request.post<KnowledgeDoc>(`/avatar/${avatarId}/knowledge/docs`, form, {
headers: { 'Content-Type': 'multipart/form-data' },
timeout: 120000
// A slow mobile uplink must not be mistaken for a failed upload.
timeout: 10 * 60 * 1000,
onUploadProgress: (event) => onUploadProgress?.(event.loaded, event.total || file.size)
})
}
@@ -15,7 +15,7 @@
<template v-else>
<div class="tab-switcher" role="tablist" aria-label="知识库类型">
<button class="tab-btn" :class="{ active: activeTab === 'docs' }" role="tab" :aria-selected="activeTab === 'docs'" @click="activeTab = 'docs'">文档知识库 <b>{{ docs.length }}</b></button>
<button class="tab-btn" :class="{ active: activeTab === 'docs' }" role="tab" :aria-selected="activeTab === 'docs'" @click="activeTab = 'docs'">文档知识库 <b>{{ displayDocs.length }}</b></button>
<button class="tab-btn" :class="{ active: activeTab === 'qa' }" role="tab" :aria-selected="activeTab === 'qa'" @click="activeTab = 'qa'">标准问答对 <b>{{ qaPairs.length }}</b></button>
</div>
@@ -25,14 +25,14 @@
<div class="upload-icon">📥</div>
<p class="upload-title"><span class="upload-link">点击上传</span></p>
<p class="upload-hint">支持 MD / TXT / PDF / DOC / DOCX / XLSX,上传后自动向量化</p>
<input ref="fileInput" type="file" accept=".md,.txt,.pdf,.doc,.docx,.xlsx" class="hidden-input" @change="onFileChange" />
<input ref="fileInput" type="file" multiple accept=".md,.txt,.pdf,.doc,.docx,.xlsx" class="hidden-input" @change="onFileChange" />
</div>
<p v-if="uploading" class="uploading-text">文件上传中…</p>
<p v-if="uploading" class="uploading-text">{{ pendingUploads.length }} 个文件正在上传</p>
<p v-if="uploadError" class="error-text">{{ uploadError }}</p>
</div>
<div v-if="docs.length" class="mobile-card-list">
<article v-for="doc in docs" :key="doc.id" class="knowledge-card">
<div v-if="displayDocs.length" class="mobile-card-list">
<article v-for="doc in displayDocs" :key="doc.id" class="knowledge-card">
<div class="card-icon">{{ fileEmoji(doc.fileType) }}</div>
<div class="card-content">
<div class="card-title-row">
@@ -41,10 +41,13 @@
</div>
<p class="card-meta">{{ doc.fileType.toUpperCase() }} · {{ formatSize(doc.fileSize) }} · {{ formatDate(doc.createdAt) }}</p>
<p class="card-detail">{{ documentState(doc).detail }}</p>
<div v-if="documentState(doc).progress !== undefined" class="progress-track" :aria-label="`${documentState(doc).label} ${documentState(doc).progress}%`">
<span class="progress-fill" :style="{ width: `${documentState(doc).progress}%` }"></span>
</div>
</div>
<div class="card-actions">
<button v-if="documentState(doc).tone === 'failed'" class="card-retry" @click="retryDoc(doc.id)">重新索引</button>
<button class="card-delete" @click="removeDoc(doc.id)">删除</button>
<button v-if="!doc.localUploading" class="card-delete" @click="removeDoc(doc.id)">{{ doc.localOnly ? '移除' : '删除' }}</button>
</div>
</article>
</div>
@@ -106,8 +109,9 @@ const avatarId = computed(() => pickScopedAvatarId(route.params.avatarId, store.
const activeTab = ref<'docs' | 'qa'>('docs')
const docs = ref<any[]>([])
const pendingUploads = ref<any[]>([])
const qaPairs = ref<any[]>([])
const uploading = ref(false)
const uploading = computed(() => pendingUploads.value.some((doc) => doc.localUploading))
const uploadError = ref('')
const dragOver = ref(false)
const fileInput = ref<HTMLInputElement | null>(null)
@@ -118,7 +122,15 @@ const searching = ref(false)
const searched = ref(false)
const searchResults = ref<any[]>([])
const displayDocs = computed(() => [...pendingUploads.value, ...docs.value])
const documentState = (doc: any) => {
if (doc.localUploading) {
return { tone: 'pending', label: '上传中', detail: `正在上传 ${doc.uploadProgress || 0}%`, progress: doc.uploadProgress || 0 }
}
if (doc.localOnly) {
return { tone: 'failed', label: '上传失败', detail: doc.errorMessage || '文件未上传成功,请移除后重试' }
}
if (doc.filePresent === false) {
return { tone: 'missing', label: '文件缺失', detail: '原文件不可用,请删除后重新上传' }
}
@@ -126,7 +138,12 @@ const documentState = (doc: any) => {
return { tone: 'ready', label: '已入库', detail: `已切分 ${doc.chunkCount || 0} 段,可用于对话` }
}
if (['uploaded', 'parsing'].includes(String(doc.status || '').toLowerCase())) {
return { tone: 'pending', label: '处理中', detail: '正在解析并建立知识索引' }
const stage = String(doc.indexStage || 'queued').toLowerCase()
const labels: Record<string, string> = {
queued: '等待处理', extracting: '解析文档', chunking: '切分文本', embedding: '向量化中'
}
const progress = Math.max(0, Math.min(99, Number(doc.indexProgress || 0)))
return { tone: 'pending', label: labels[stage] || '处理中', detail: `${labels[stage] || '正在建立知识索引'} ${progress}%`, progress }
}
return { tone: 'failed', label: '处理失败', detail: doc.errorMessage || '未能建立知识索引,请重新索引或重新上传' }
}
@@ -174,36 +191,62 @@ const loadQA = async () => {
const triggerFile = () => fileInput.value?.click()
const onFileChange = (e: Event) => {
const f = (e.target as HTMLInputElement).files?.[0]
if (f) doUpload(f)
const files = Array.from((e.target as HTMLInputElement).files || [])
if (files.length) uploadFiles(files)
;(e.target as HTMLInputElement).value = ''
}
const onDrop = (e: DragEvent) => {
dragOver.value = false
const f = e.dataTransfer?.files?.[0]
if (f) doUpload(f)
const files = Array.from(e.dataTransfer?.files || [])
if (files.length) uploadFiles(files)
}
const doUpload = async (file: File) => {
const uploadFiles = (files: File[]) => {
uploadError.value = ''
const ext = '.' + (file.name.split('.').pop() || '').toLowerCase()
if (!['.md', '.txt', '.pdf', '.doc', '.docx', '.xlsx'].includes(ext)) {
uploadError.value = `不支持的类型:${ext},仅支持 md/txt/pdf/doc/docx/xlsx`
return
}
if (!avatarId.value) {
uploadError.value = '请先创建数字分身'
return
}
uploading.value = true
for (const file of files) {
const ext = '.' + (file.name.split('.').pop() || '').toLowerCase()
if (!['.md', '.txt', '.pdf', '.doc', '.docx', '.xlsx'].includes(ext)) {
uploadError.value = `不支持的类型:${ext},仅支持 md/txt/pdf/doc/docx/xlsx`
continue
}
void uploadOne(file, ext)
}
}
const uploadOne = async (file: File, ext: string) => {
if (!avatarId.value) return
const localId = `upload-${Date.now()}-${Math.random().toString(16).slice(2)}`
const card = {
id: localId,
filename: file.name,
fileType: ext.slice(1),
fileSize: file.size,
createdAt: new Date().toISOString(),
localUploading: true,
localOnly: true,
uploadProgress: 0,
errorMessage: ''
}
pendingUploads.value.unshift(card)
try {
await uploadKnowledgeDoc(avatarId.value, file)
await loadDocs()
const created: any = await uploadKnowledgeDoc(avatarId.value, file, (loaded, total) => {
const current = pendingUploads.value.find((doc) => doc.id === localId)
if (current) current.uploadProgress = Math.min(99, Math.round((loaded / Math.max(1, total)) * 100))
})
pendingUploads.value = pendingUploads.value.filter((doc) => doc.id !== localId)
docs.value = [created, ...docs.value.filter((doc) => doc.id !== created.id)]
startDocumentPolling()
} catch (e: any) {
uploadError.value = e?.message || '上传失败'
} finally {
uploading.value = false
const current = pendingUploads.value.find((doc) => doc.id === localId)
if (current) {
current.localUploading = false
current.errorMessage = e?.message || '上传失败'
}
}
}
@@ -219,6 +262,11 @@ const retryDoc = async (id: string) => {
}
const removeDoc = async (id: string) => {
const local = pendingUploads.value.find((doc) => doc.id === id)
if (local?.localOnly) {
pendingUploads.value = pendingUploads.value.filter((doc) => doc.id !== id)
return
}
if (!avatarId.value) return
await deleteKnowledgeDoc(avatarId.value, id)
await loadDocs()
@@ -338,6 +386,8 @@ onUnmounted(stopDocumentPolling)
.status-pill.missing { color: #B91C1C; background: #FEF2F2; }
.status-pill.failed { color: #B91C1C; background: #FEF2F2; }
.card-meta, .card-detail { margin: 5px 0 0; color: #9398AE; font-size: 11px; line-height: 1.4; }.card-detail { color: #8B6B58; }
.progress-track { width: 100%; height: 4px; margin-top: 8px; overflow: hidden; border-radius: 999px; background: #FDE7D1; }
.progress-fill { display: block; height: 100%; border-radius: inherit; background: linear-gradient(90deg, #FB923C, #F97316); transition: width .25s ease; }
.card-actions { flex: 0 0 auto; display: flex; flex-direction: column; align-items: stretch; gap: 6px; }
.card-delete, .card-retry { align-self: center; border: 0; border-radius: 8px; padding: 7px 9px; font-size: 12px; cursor: pointer; white-space: nowrap; }
.card-delete { color: #EF4444; background: #FEF2F2; }