Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e71267cf86 | ||
|
|
359e558dbe | ||
|
|
3edf92c7cc | ||
|
|
97c4c73b58 | ||
|
|
95f91450d0 |
@@ -54,6 +54,8 @@ def init_db():
|
||||
("knowledge_docs", "chunk_count", "INTEGER DEFAULT 0"),
|
||||
("knowledge_docs", "vectorized_at", "TIMESTAMP"),
|
||||
("knowledge_docs", "error_message", "VARCHAR DEFAULT ''"),
|
||||
("knowledge_docs", "index_stage", "VARCHAR DEFAULT ''"),
|
||||
("knowledge_docs", "index_progress", "INTEGER DEFAULT 0"),
|
||||
("avatars", "owner_id", "VARCHAR DEFAULT ''"),
|
||||
("authorizations", "takeover_enabled", "BOOLEAN DEFAULT 0"),
|
||||
("authorizations", "takeover_mode", "VARCHAR DEFAULT 'immediate'"),
|
||||
|
||||
@@ -51,7 +51,7 @@ def _hash_embedding(texts, dim=EMBED_DIM):
|
||||
return vecs
|
||||
|
||||
|
||||
def embed(texts):
|
||||
def embed(texts, on_progress=None):
|
||||
"""返回 list[list[float]],与输入顺序一致。"""
|
||||
if not texts:
|
||||
return []
|
||||
@@ -64,6 +64,7 @@ def embed(texts):
|
||||
except ValueError:
|
||||
batch_size = 10
|
||||
embeddings = []
|
||||
total = len(texts)
|
||||
for start in range(0, len(texts), batch_size):
|
||||
batch = texts[start:start + batch_size]
|
||||
payload = json.dumps({"input": batch, "model": model}).encode("utf-8")
|
||||
@@ -84,8 +85,13 @@ def embed(texts):
|
||||
if len(items) != len(batch):
|
||||
raise ValueError("embedding response count does not match request")
|
||||
embeddings.extend(item["embedding"] for item in items)
|
||||
if on_progress:
|
||||
on_progress(len(embeddings), total)
|
||||
return embeddings
|
||||
return _hash_embedding(texts)
|
||||
vectors = _hash_embedding(texts)
|
||||
if on_progress:
|
||||
on_progress(len(vectors), len(texts))
|
||||
return vectors
|
||||
|
||||
|
||||
def cosine(a, b):
|
||||
|
||||
@@ -191,6 +191,8 @@ class KnowledgeDoc(Base):
|
||||
file_url = Column(String, default="")
|
||||
status = Column(String, default="uploaded") # uploaded | parsing | ready | failed
|
||||
error_message = Column(String, default="") # 建立索引失败原因
|
||||
index_stage = Column(String, default="") # queued | extracting | chunking | embedding | ready | failed
|
||||
index_progress = Column(Integer, default=0) # 0-100
|
||||
vectorized = Column(Boolean, default=False) # 是否已向量化
|
||||
embedding_model = Column(String, default="") # 向量模型标识
|
||||
chunk_count = Column(Integer, default=0) # 切片数量
|
||||
@@ -207,6 +209,8 @@ class KnowledgeDoc(Base):
|
||||
"fileUrl": self.file_url,
|
||||
"status": self.status,
|
||||
"errorMessage": self.error_message or "",
|
||||
"indexStage": self.index_stage or "",
|
||||
"indexProgress": int(self.index_progress or 0),
|
||||
"vectorized": bool(self.vectorized),
|
||||
"embeddingModel": self.embedding_model,
|
||||
"chunkCount": self.chunk_count,
|
||||
|
||||
@@ -1,4 +1,7 @@
|
||||
import os
|
||||
import json
|
||||
import shutil
|
||||
import time
|
||||
import uuid
|
||||
|
||||
from fastapi import APIRouter, UploadFile, File, Depends, Header, HTTPException
|
||||
@@ -19,6 +22,9 @@ os.makedirs(UPLOAD_DIR, exist_ok=True)
|
||||
ALLOWED_EXT = {".md", ".txt", ".pdf", ".doc", ".docx", ".xlsx"}
|
||||
MAX_UPLOAD_BYTES = 50 * 1024 * 1024
|
||||
UPLOAD_CHUNK_BYTES = 1024 * 1024
|
||||
MULTIPART_CHUNK_BYTES = 5 * 1024 * 1024
|
||||
MULTIPART_ROOT = ".multipart"
|
||||
MULTIPART_TTL_SECONDS = 24 * 60 * 60
|
||||
|
||||
|
||||
class QAIn(BaseModel):
|
||||
@@ -31,6 +37,74 @@ class EnabledIn(BaseModel):
|
||||
enabled: bool = True
|
||||
|
||||
|
||||
class MultipartUploadIn(BaseModel):
|
||||
filename: str
|
||||
fileSize: int
|
||||
totalChunks: int
|
||||
|
||||
|
||||
def _validate_document(filename: str, file_size: int):
|
||||
ext = os.path.splitext(filename or "")[1].lower()
|
||||
if ext not in ALLOWED_EXT:
|
||||
return None, f"不支持的文件类型:{ext or '空'},仅支持 md/txt/pdf/doc/docx/xlsx"
|
||||
if file_size <= 0:
|
||||
return None, "文件内容不能为空"
|
||||
if file_size > MAX_UPLOAD_BYTES:
|
||||
return None, "文件不能超过 50MB"
|
||||
return ext, ""
|
||||
|
||||
|
||||
def _multipart_dir(avatar_id: str, upload_id: str) -> str:
|
||||
safe_avatar_id = os.path.basename(avatar_id)
|
||||
safe_upload_id = os.path.basename(upload_id)
|
||||
if (
|
||||
safe_avatar_id != avatar_id
|
||||
or safe_upload_id != upload_id
|
||||
or len(upload_id) != 32
|
||||
or any(character not in "0123456789abcdef" for character in upload_id)
|
||||
):
|
||||
raise HTTPException(status_code=400, detail="上传标识无效")
|
||||
return os.path.join(UPLOAD_DIR, MULTIPART_ROOT, safe_avatar_id, safe_upload_id)
|
||||
|
||||
|
||||
def _purge_stale_multipart_uploads(avatar_id: str):
|
||||
avatar_upload_root = os.path.join(UPLOAD_DIR, MULTIPART_ROOT, os.path.basename(avatar_id))
|
||||
if not os.path.isdir(avatar_upload_root):
|
||||
return
|
||||
cutoff = time.time() - MULTIPART_TTL_SECONDS
|
||||
for entry in os.scandir(avatar_upload_root):
|
||||
if entry.is_dir(follow_symlinks=False) and entry.stat(follow_symlinks=False).st_mtime < cutoff:
|
||||
shutil.rmtree(entry.path, ignore_errors=True)
|
||||
|
||||
|
||||
def _read_multipart_metadata(avatar_id: str, upload_id: str) -> tuple[str, dict]:
|
||||
upload_dir = _multipart_dir(avatar_id, upload_id)
|
||||
metadata_path = os.path.join(upload_dir, "metadata.json")
|
||||
if not os.path.isfile(metadata_path):
|
||||
raise HTTPException(status_code=404, detail="上传任务不存在或已过期")
|
||||
with open(metadata_path, "r", encoding="utf-8") as stream:
|
||||
return upload_dir, json.load(stream)
|
||||
|
||||
|
||||
def _create_knowledge_doc(db: Session, avatar_id: str, filename: str, ext: str, file_size: int, stored: str):
|
||||
doc = KnowledgeDoc(
|
||||
id=uuid.uuid4().hex,
|
||||
avatar_id=avatar_id,
|
||||
filename=filename,
|
||||
file_type=ext.lstrip("."),
|
||||
file_size=file_size,
|
||||
file_url=f"/api/files/{avatar_id}/{stored}",
|
||||
status="parsing",
|
||||
index_stage="queued",
|
||||
index_progress=0,
|
||||
)
|
||||
db.add(doc)
|
||||
db.commit()
|
||||
db.refresh(doc)
|
||||
knowledge_vectorizer.enqueue(doc.id)
|
||||
return doc
|
||||
|
||||
|
||||
def _doc_payload(doc: KnowledgeDoc) -> dict:
|
||||
payload = doc.to_dict()
|
||||
stored_name = os.path.basename(doc.file_url or "")
|
||||
@@ -74,9 +148,9 @@ def list_docs(avatar_id: str, authorization: str = Header(None), db: Session = D
|
||||
@router.post("/avatar/{avatar_id}/knowledge/docs")
|
||||
async def upload_doc(avatar_id: str, file: UploadFile = File(...), authorization: str = Header(None), db: Session = Depends(get_db)):
|
||||
_require_owned_avatar(db, avatar_id, authorization)
|
||||
ext = os.path.splitext(file.filename or "")[1].lower()
|
||||
if ext not in ALLOWED_EXT:
|
||||
return fail(f"不支持的文件类型:{ext or '空'},仅支持 md/txt/pdf/doc/docx/xlsx", code=400)
|
||||
ext, validation_error = _validate_document(file.filename or "", 1)
|
||||
if validation_error:
|
||||
return fail(validation_error, code=400)
|
||||
avatar_dir = os.path.join(UPLOAD_DIR, avatar_id)
|
||||
os.makedirs(avatar_dir, exist_ok=True)
|
||||
stored = f"{uuid.uuid4().hex}{ext}"
|
||||
@@ -94,23 +168,124 @@ async def upload_doc(avatar_id: str, file: UploadFile = File(...), authorization
|
||||
if os.path.exists(path):
|
||||
os.remove(path)
|
||||
return fail(str(exc), code=400)
|
||||
doc = KnowledgeDoc(
|
||||
id=uuid.uuid4().hex,
|
||||
avatar_id=avatar_id,
|
||||
filename=file.filename,
|
||||
file_type=ext.lstrip("."),
|
||||
file_size=file_size,
|
||||
file_url=f"/api/files/{avatar_id}/{stored}",
|
||||
status="parsing",
|
||||
if file_size == 0:
|
||||
if os.path.exists(path):
|
||||
os.remove(path)
|
||||
return fail("文件内容不能为空", code=400)
|
||||
|
||||
doc = _create_knowledge_doc(db, avatar_id, file.filename or stored, ext, file_size, stored)
|
||||
return ok(_doc_payload(doc))
|
||||
|
||||
|
||||
@router.post("/avatar/{avatar_id}/knowledge/uploads")
|
||||
def create_multipart_upload(
|
||||
avatar_id: str,
|
||||
body: MultipartUploadIn,
|
||||
authorization: str = Header(None),
|
||||
db: Session = Depends(get_db),
|
||||
):
|
||||
_require_owned_avatar(db, avatar_id, authorization)
|
||||
ext, validation_error = _validate_document(body.filename, body.fileSize)
|
||||
if validation_error:
|
||||
return fail(validation_error, code=400)
|
||||
expected_chunks = (body.fileSize + MULTIPART_CHUNK_BYTES - 1) // MULTIPART_CHUNK_BYTES
|
||||
if body.totalChunks != expected_chunks:
|
||||
return fail("文件分片数量不正确", code=400)
|
||||
|
||||
_purge_stale_multipart_uploads(avatar_id)
|
||||
upload_id = uuid.uuid4().hex
|
||||
upload_dir = _multipart_dir(avatar_id, upload_id)
|
||||
os.makedirs(upload_dir, exist_ok=False)
|
||||
metadata = {
|
||||
"filename": body.filename,
|
||||
"fileSize": body.fileSize,
|
||||
"totalChunks": body.totalChunks,
|
||||
"extension": ext,
|
||||
}
|
||||
with open(os.path.join(upload_dir, "metadata.json"), "w", encoding="utf-8") as stream:
|
||||
json.dump(metadata, stream, ensure_ascii=False)
|
||||
return ok({"uploadId": upload_id, "chunkSize": MULTIPART_CHUNK_BYTES})
|
||||
|
||||
|
||||
@router.post("/avatar/{avatar_id}/knowledge/uploads/{upload_id}/chunks/{chunk_index}")
|
||||
async def upload_multipart_chunk(
|
||||
avatar_id: str,
|
||||
upload_id: str,
|
||||
chunk_index: int,
|
||||
file: UploadFile = File(...),
|
||||
authorization: str = Header(None),
|
||||
db: Session = Depends(get_db),
|
||||
):
|
||||
_require_owned_avatar(db, avatar_id, authorization)
|
||||
upload_dir, metadata = _read_multipart_metadata(avatar_id, upload_id)
|
||||
total_chunks = int(metadata["totalChunks"])
|
||||
if chunk_index < 0 or chunk_index >= total_chunks:
|
||||
return fail("文件分片序号不正确", code=400)
|
||||
|
||||
expected_size = min(
|
||||
MULTIPART_CHUNK_BYTES,
|
||||
int(metadata["fileSize"]) - chunk_index * MULTIPART_CHUNK_BYTES,
|
||||
)
|
||||
part_path = os.path.join(upload_dir, f"{chunk_index}.part")
|
||||
temporary_path = f"{part_path}.uploading"
|
||||
received = 0
|
||||
try:
|
||||
with open(temporary_path, "wb") as stream:
|
||||
while chunk := await file.read(UPLOAD_CHUNK_BYTES):
|
||||
received += len(chunk)
|
||||
if received > expected_size:
|
||||
raise ValueError("文件分片大小不正确")
|
||||
stream.write(chunk)
|
||||
if received != expected_size:
|
||||
raise ValueError("文件分片大小不正确")
|
||||
os.replace(temporary_path, part_path)
|
||||
except ValueError as exc:
|
||||
if os.path.exists(temporary_path):
|
||||
os.remove(temporary_path)
|
||||
return fail(str(exc), code=400)
|
||||
return ok({"chunkIndex": chunk_index, "uploadedBytes": received})
|
||||
|
||||
# Persist and acknowledge the upload first. Extraction and embeddings may take
|
||||
# minutes for a PDF and must never consume the browser request timeout.
|
||||
db.add(doc)
|
||||
db.commit()
|
||||
db.refresh(doc)
|
||||
knowledge_vectorizer.enqueue(doc.id)
|
||||
|
||||
@router.post("/avatar/{avatar_id}/knowledge/uploads/{upload_id}/complete")
|
||||
def complete_multipart_upload(
|
||||
avatar_id: str,
|
||||
upload_id: str,
|
||||
authorization: str = Header(None),
|
||||
db: Session = Depends(get_db),
|
||||
):
|
||||
_require_owned_avatar(db, avatar_id, authorization)
|
||||
upload_dir, metadata = _read_multipart_metadata(avatar_id, upload_id)
|
||||
total_chunks = int(metadata["totalChunks"])
|
||||
part_paths = [os.path.join(upload_dir, f"{index}.part") for index in range(total_chunks)]
|
||||
if not all(os.path.isfile(path) for path in part_paths):
|
||||
return fail("文件分片尚未上传完整", code=400)
|
||||
if sum(os.path.getsize(path) for path in part_paths) != int(metadata["fileSize"]):
|
||||
return fail("文件分片总大小不正确", code=400)
|
||||
|
||||
avatar_dir = os.path.join(UPLOAD_DIR, avatar_id)
|
||||
os.makedirs(avatar_dir, exist_ok=True)
|
||||
stored = f"{uuid.uuid4().hex}{metadata['extension']}"
|
||||
final_path = os.path.join(avatar_dir, stored)
|
||||
temporary_path = f"{final_path}.assembling"
|
||||
try:
|
||||
with open(temporary_path, "wb") as output:
|
||||
for part_path in part_paths:
|
||||
with open(part_path, "rb") as source:
|
||||
shutil.copyfileobj(source, output, UPLOAD_CHUNK_BYTES)
|
||||
os.replace(temporary_path, final_path)
|
||||
doc = _create_knowledge_doc(
|
||||
db,
|
||||
avatar_id,
|
||||
metadata["filename"],
|
||||
metadata["extension"],
|
||||
int(metadata["fileSize"]),
|
||||
stored,
|
||||
)
|
||||
except Exception:
|
||||
if os.path.exists(temporary_path):
|
||||
os.remove(temporary_path)
|
||||
raise
|
||||
shutil.rmtree(upload_dir, ignore_errors=True)
|
||||
return ok(_doc_payload(doc))
|
||||
|
||||
|
||||
@@ -134,6 +309,8 @@ def retry_doc(avatar_id: str, doc_id: str, authorization: str = Header(None), db
|
||||
doc.chunk_count = 0
|
||||
doc.vectorized_at = None
|
||||
doc.error_message = ""
|
||||
doc.index_stage = "queued"
|
||||
doc.index_progress = 0
|
||||
db.commit()
|
||||
db.refresh(doc)
|
||||
knowledge_vectorizer.enqueue(doc.id)
|
||||
|
||||
@@ -74,11 +74,19 @@ class KnowledgeVectorizer:
|
||||
if not stored_name or not os.path.isfile(path):
|
||||
raise FileNotFoundError("原文件不可用,请重新上传")
|
||||
|
||||
self._set_progress(db, doc, "extracting", 8)
|
||||
text = embeddings.extract_text(path, f".{doc.file_type}")
|
||||
self._set_progress(db, doc, "chunking", 22)
|
||||
chunks = embeddings.chunk_text(text)
|
||||
if not chunks:
|
||||
raise ValueError("文档没有可建立索引的文字内容")
|
||||
vectors = embeddings.embed(chunks)
|
||||
self._set_progress(db, doc, "embedding", 30)
|
||||
|
||||
def embedding_progress(done: int, total: int):
|
||||
percent = 30 + int((done / max(1, total)) * 65)
|
||||
self._set_progress(db, doc, "embedding", min(percent, 95))
|
||||
|
||||
vectors = embeddings.embed(chunks, on_progress=embedding_progress)
|
||||
if len(vectors) != len(chunks):
|
||||
raise ValueError("向量服务返回数量与文档分段不一致")
|
||||
|
||||
@@ -103,6 +111,8 @@ class KnowledgeVectorizer:
|
||||
doc.vectorized_at = datetime.now(timezone.utc)
|
||||
doc.status = "ready"
|
||||
doc.error_message = ""
|
||||
doc.index_stage = "ready"
|
||||
doc.index_progress = 100
|
||||
db.commit()
|
||||
logger.info("Knowledge document %s indexed with %s chunks", doc.id, len(chunks))
|
||||
except Exception as exc:
|
||||
@@ -116,10 +126,18 @@ class KnowledgeVectorizer:
|
||||
failed_doc.chunk_count = 0
|
||||
failed_doc.vectorized_at = None
|
||||
failed_doc.error_message = str(exc)[:300] or "建立知识索引失败"
|
||||
failed_doc.index_stage = "failed"
|
||||
failed_doc.index_progress = 0
|
||||
db.commit()
|
||||
logger.exception("Knowledge vectorization failed for %s: %s", doc_id, exc)
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
@staticmethod
|
||||
def _set_progress(db, doc, stage: str, progress: int):
|
||||
doc.index_stage = stage
|
||||
doc.index_progress = progress
|
||||
db.commit()
|
||||
|
||||
|
||||
knowledge_vectorizer = KnowledgeVectorizer()
|
||||
|
||||
@@ -49,6 +49,7 @@ class RemoteEmbeddingTests(unittest.TestCase):
|
||||
texts = [f"chunk-{index}" for index in range(14)]
|
||||
batch_sizes = []
|
||||
requested_urls = []
|
||||
progress_updates = []
|
||||
|
||||
def fake_urlopen(request, timeout):
|
||||
self.assertEqual(timeout, 30)
|
||||
@@ -68,7 +69,10 @@ class RemoteEmbeddingTests(unittest.TestCase):
|
||||
"EMBEDDING_MODEL": "text-embedding-v4",
|
||||
"EMBEDDING_BATCH_SIZE": "10",
|
||||
}), patch("embeddings.urllib.request.urlopen", side_effect=fake_urlopen):
|
||||
result = embeddings.embed(texts)
|
||||
result = embeddings.embed(
|
||||
texts,
|
||||
on_progress=lambda completed, total: progress_updates.append((completed, total)),
|
||||
)
|
||||
|
||||
self.assertEqual(batch_sizes, [10, 4])
|
||||
self.assertEqual(requested_urls, [
|
||||
@@ -76,6 +80,7 @@ class RemoteEmbeddingTests(unittest.TestCase):
|
||||
"https://embedding.example/v1/embeddings",
|
||||
])
|
||||
self.assertEqual(result, [[float(index)] for index in range(14)])
|
||||
self.assertEqual(progress_updates, [(10, 14), (14, 14)])
|
||||
|
||||
def test_full_embedding_endpoint_is_not_modified(self):
|
||||
self.assertEqual(
|
||||
|
||||
@@ -87,6 +87,84 @@ def test_upload_rejects_oversize_file_before_queuing_indexing(
|
||||
assert not list((tmp_path / context["avatar"].id).glob("*"))
|
||||
|
||||
|
||||
def test_multipart_upload_reassembles_file_before_queuing_indexing(
|
||||
tmp_path: Path,
|
||||
authorization_context,
|
||||
):
|
||||
context = authorization_context
|
||||
avatar_id = context["avatar"].id
|
||||
content = b"0123456789"
|
||||
with (
|
||||
patch("routers.knowledge.UPLOAD_DIR", str(tmp_path)),
|
||||
patch("routers.knowledge.MULTIPART_CHUNK_BYTES", 4),
|
||||
patch("routers.knowledge.knowledge_vectorizer.enqueue") as enqueue,
|
||||
):
|
||||
created = client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads",
|
||||
headers=context["owner_headers"],
|
||||
json={"filename": "large.pdf", "fileSize": len(content), "totalChunks": 3},
|
||||
).json()["data"]
|
||||
|
||||
for index, chunk in enumerate((content[:4], content[4:8], content[8:])):
|
||||
response = client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads/{created['uploadId']}/chunks/{index}",
|
||||
headers=context["owner_headers"],
|
||||
files={"file": (f"chunk-{index}", chunk, "application/octet-stream")},
|
||||
)
|
||||
assert response.json()["code"] == 200
|
||||
|
||||
completed = client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads/{created['uploadId']}/complete",
|
||||
headers=context["owner_headers"],
|
||||
).json()["data"]
|
||||
|
||||
assert completed["status"] == "parsing"
|
||||
assert completed["fileSize"] == len(content)
|
||||
enqueue.assert_called_once_with(completed["id"])
|
||||
stored_path = tmp_path / avatar_id / Path(completed["fileUrl"]).name
|
||||
assert stored_path.read_bytes() == content
|
||||
assert not (tmp_path / ".multipart" / avatar_id / created["uploadId"]).exists()
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
stored = db.query(KnowledgeDoc).filter(KnowledgeDoc.id == completed["id"]).one()
|
||||
db.delete(stored)
|
||||
db.commit()
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def test_multipart_upload_rejects_incomplete_parts(
|
||||
tmp_path: Path,
|
||||
authorization_context,
|
||||
):
|
||||
context = authorization_context
|
||||
avatar_id = context["avatar"].id
|
||||
with (
|
||||
patch("routers.knowledge.UPLOAD_DIR", str(tmp_path)),
|
||||
patch("routers.knowledge.MULTIPART_CHUNK_BYTES", 4),
|
||||
patch("routers.knowledge.knowledge_vectorizer.enqueue") as enqueue,
|
||||
):
|
||||
created = client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads",
|
||||
headers=context["owner_headers"],
|
||||
json={"filename": "large.pdf", "fileSize": 6, "totalChunks": 2},
|
||||
).json()["data"]
|
||||
client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads/{created['uploadId']}/chunks/0",
|
||||
headers=context["owner_headers"],
|
||||
files={"file": ("chunk-0", b"0123", "application/octet-stream")},
|
||||
)
|
||||
response = client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads/{created['uploadId']}/complete",
|
||||
headers=context["owner_headers"],
|
||||
)
|
||||
|
||||
assert response.json()["code"] == 400
|
||||
assert response.json()["message"] == "文件分片尚未上传完整"
|
||||
enqueue.assert_not_called()
|
||||
|
||||
|
||||
def test_background_vectorizer_commits_ready_document_and_chunks_together(
|
||||
tmp_path: Path,
|
||||
authorization_context,
|
||||
@@ -116,6 +194,8 @@ def test_background_vectorizer_commits_ready_document_and_chunks_together(
|
||||
assert stored.status == "ready"
|
||||
assert stored.vectorized is True
|
||||
assert stored.chunk_count == 1
|
||||
assert stored.index_stage == "ready"
|
||||
assert stored.index_progress == 100
|
||||
assert db.query(KnowledgeChunk).filter(KnowledgeChunk.doc_id == stored.id).count() == 1
|
||||
db.query(KnowledgeChunk).filter(KnowledgeChunk.doc_id == stored.id).delete()
|
||||
db.delete(stored)
|
||||
|
||||
@@ -306,6 +306,8 @@ export interface KnowledgeDoc {
|
||||
embeddingModel?: string
|
||||
chunkCount?: number
|
||||
errorMessage?: string
|
||||
indexStage?: string
|
||||
indexProgress?: number
|
||||
createdAt: string
|
||||
}
|
||||
|
||||
@@ -331,13 +333,83 @@ export interface SearchResult {
|
||||
export const getKnowledgeDocs = (avatarId: string) =>
|
||||
request.get<KnowledgeDoc[]>(`/avatar/${avatarId}/knowledge/docs`)
|
||||
|
||||
// 上传文档(支持 md/txt/pdf/doc/docx/xlsx)
|
||||
export const uploadKnowledgeDoc = (avatarId: string, file: File) => {
|
||||
const KNOWLEDGE_UPLOAD_CHUNK_SIZE = 5 * 1024 * 1024
|
||||
|
||||
const uploadKnowledgeChunk = async (
|
||||
avatarId: string,
|
||||
uploadId: string,
|
||||
chunkIndex: number,
|
||||
chunk: Blob,
|
||||
onProgress?: (loaded: number) => void
|
||||
) => {
|
||||
const form = new FormData()
|
||||
form.append('file', chunk, `chunk-${chunkIndex}`)
|
||||
let reportedLoaded = 0
|
||||
for (let attempt = 1; attempt <= 3; attempt += 1) {
|
||||
try {
|
||||
await request.post(
|
||||
`/avatar/${avatarId}/knowledge/uploads/${uploadId}/chunks/${chunkIndex}`,
|
||||
form,
|
||||
{
|
||||
headers: { 'Content-Type': 'multipart/form-data' },
|
||||
timeout: 2 * 60 * 1000,
|
||||
onUploadProgress: (event) => {
|
||||
reportedLoaded = Math.max(reportedLoaded, Math.min(event.loaded, chunk.size))
|
||||
onProgress?.(reportedLoaded)
|
||||
}
|
||||
}
|
||||
)
|
||||
return
|
||||
} catch (error: any) {
|
||||
const status = Number(error?.response?.status || 0)
|
||||
const retryable = !status || status === 408 || status === 429 || status >= 500
|
||||
if (!retryable || attempt === 3) throw error
|
||||
await new Promise((resolve) => window.setTimeout(resolve, attempt * 800))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 大文件拆成 5MB 分片,避免生产代理的请求体限制拦截整个文件。
|
||||
export const uploadKnowledgeDoc = async (
|
||||
avatarId: string,
|
||||
file: File,
|
||||
onUploadProgress?: (loaded: number, total: number) => void
|
||||
) => {
|
||||
if (file.size > KNOWLEDGE_UPLOAD_CHUNK_SIZE) {
|
||||
const totalChunks = Math.ceil(file.size / KNOWLEDGE_UPLOAD_CHUNK_SIZE)
|
||||
const upload: any = await request.post(`/avatar/${avatarId}/knowledge/uploads`, {
|
||||
filename: file.name,
|
||||
fileSize: file.size,
|
||||
totalChunks
|
||||
})
|
||||
let uploadedBytes = 0
|
||||
for (let index = 0; index < totalChunks; index += 1) {
|
||||
const start = index * KNOWLEDGE_UPLOAD_CHUNK_SIZE
|
||||
const chunk = file.slice(start, Math.min(start + KNOWLEDGE_UPLOAD_CHUNK_SIZE, file.size))
|
||||
await uploadKnowledgeChunk(
|
||||
avatarId,
|
||||
upload.uploadId,
|
||||
index,
|
||||
chunk,
|
||||
(chunkLoaded) => onUploadProgress?.(uploadedBytes + chunkLoaded, file.size)
|
||||
)
|
||||
uploadedBytes += chunk.size
|
||||
onUploadProgress?.(uploadedBytes, file.size)
|
||||
}
|
||||
return request.post<KnowledgeDoc>(
|
||||
`/avatar/${avatarId}/knowledge/uploads/${upload.uploadId}/complete`,
|
||||
undefined,
|
||||
{ timeout: 2 * 60 * 1000 }
|
||||
)
|
||||
}
|
||||
|
||||
const form = new FormData()
|
||||
form.append('file', file)
|
||||
return request.post<KnowledgeDoc>(`/avatar/${avatarId}/knowledge/docs`, form, {
|
||||
headers: { 'Content-Type': 'multipart/form-data' },
|
||||
timeout: 120000
|
||||
// A slow mobile uplink must not be mistaken for a failed upload.
|
||||
timeout: 10 * 60 * 1000,
|
||||
onUploadProgress: (event) => onUploadProgress?.(event.loaded, event.total || file.size)
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
|
||||
<template v-else>
|
||||
<div class="tab-switcher" role="tablist" aria-label="知识库类型">
|
||||
<button class="tab-btn" :class="{ active: activeTab === 'docs' }" role="tab" :aria-selected="activeTab === 'docs'" @click="activeTab = 'docs'">文档知识库 <b>{{ docs.length }}</b></button>
|
||||
<button class="tab-btn" :class="{ active: activeTab === 'docs' }" role="tab" :aria-selected="activeTab === 'docs'" @click="activeTab = 'docs'">文档知识库 <b>{{ displayDocs.length }}</b></button>
|
||||
<button class="tab-btn" :class="{ active: activeTab === 'qa' }" role="tab" :aria-selected="activeTab === 'qa'" @click="activeTab = 'qa'">标准问答对 <b>{{ qaPairs.length }}</b></button>
|
||||
</div>
|
||||
|
||||
@@ -25,14 +25,14 @@
|
||||
<div class="upload-icon">📥</div>
|
||||
<p class="upload-title"><span class="upload-link">点击上传</span></p>
|
||||
<p class="upload-hint">支持 MD / TXT / PDF / DOC / DOCX / XLSX,上传后自动向量化</p>
|
||||
<input ref="fileInput" type="file" accept=".md,.txt,.pdf,.doc,.docx,.xlsx" class="hidden-input" @change="onFileChange" />
|
||||
<input ref="fileInput" type="file" multiple accept=".md,.txt,.pdf,.doc,.docx,.xlsx" class="hidden-input" @change="onFileChange" />
|
||||
</div>
|
||||
<p v-if="uploading" class="uploading-text">文件上传中…</p>
|
||||
<p v-if="uploading" class="uploading-text">{{ pendingUploads.length }} 个文件正在上传</p>
|
||||
<p v-if="uploadError" class="error-text">{{ uploadError }}</p>
|
||||
</div>
|
||||
|
||||
<div v-if="docs.length" class="mobile-card-list">
|
||||
<article v-for="doc in docs" :key="doc.id" class="knowledge-card">
|
||||
<div v-if="displayDocs.length" class="mobile-card-list">
|
||||
<article v-for="doc in displayDocs" :key="doc.id" class="knowledge-card">
|
||||
<div class="card-icon">{{ fileEmoji(doc.fileType) }}</div>
|
||||
<div class="card-content">
|
||||
<div class="card-title-row">
|
||||
@@ -41,10 +41,13 @@
|
||||
</div>
|
||||
<p class="card-meta">{{ doc.fileType.toUpperCase() }} · {{ formatSize(doc.fileSize) }} · {{ formatDate(doc.createdAt) }}</p>
|
||||
<p class="card-detail">{{ documentState(doc).detail }}</p>
|
||||
<div v-if="documentState(doc).progress !== undefined" class="progress-track" :aria-label="`${documentState(doc).label} ${documentState(doc).progress}%`">
|
||||
<span class="progress-fill" :style="{ width: `${documentState(doc).progress}%` }"></span>
|
||||
</div>
|
||||
</div>
|
||||
<div class="card-actions">
|
||||
<button v-if="documentState(doc).tone === 'failed'" class="card-retry" @click="retryDoc(doc.id)">重新索引</button>
|
||||
<button class="card-delete" @click="removeDoc(doc.id)">删除</button>
|
||||
<button v-if="!doc.localUploading" class="card-delete" @click="removeDoc(doc.id)">{{ doc.localOnly ? '移除' : '删除' }}</button>
|
||||
</div>
|
||||
</article>
|
||||
</div>
|
||||
@@ -106,8 +109,9 @@ const avatarId = computed(() => pickScopedAvatarId(route.params.avatarId, store.
|
||||
const activeTab = ref<'docs' | 'qa'>('docs')
|
||||
|
||||
const docs = ref<any[]>([])
|
||||
const pendingUploads = ref<any[]>([])
|
||||
const qaPairs = ref<any[]>([])
|
||||
const uploading = ref(false)
|
||||
const uploading = computed(() => pendingUploads.value.some((doc) => doc.localUploading))
|
||||
const uploadError = ref('')
|
||||
const dragOver = ref(false)
|
||||
const fileInput = ref<HTMLInputElement | null>(null)
|
||||
@@ -118,7 +122,15 @@ const searching = ref(false)
|
||||
const searched = ref(false)
|
||||
const searchResults = ref<any[]>([])
|
||||
|
||||
const displayDocs = computed(() => [...pendingUploads.value, ...docs.value])
|
||||
|
||||
const documentState = (doc: any) => {
|
||||
if (doc.localUploading) {
|
||||
return { tone: 'pending', label: '上传中', detail: `正在上传 ${doc.uploadProgress || 0}%`, progress: doc.uploadProgress || 0 }
|
||||
}
|
||||
if (doc.localOnly) {
|
||||
return { tone: 'failed', label: '上传失败', detail: doc.errorMessage || '文件未上传成功,请移除后重试' }
|
||||
}
|
||||
if (doc.filePresent === false) {
|
||||
return { tone: 'missing', label: '文件缺失', detail: '原文件不可用,请删除后重新上传' }
|
||||
}
|
||||
@@ -126,7 +138,12 @@ const documentState = (doc: any) => {
|
||||
return { tone: 'ready', label: '已入库', detail: `已切分 ${doc.chunkCount || 0} 段,可用于对话` }
|
||||
}
|
||||
if (['uploaded', 'parsing'].includes(String(doc.status || '').toLowerCase())) {
|
||||
return { tone: 'pending', label: '处理中', detail: '正在解析并建立知识索引' }
|
||||
const stage = String(doc.indexStage || 'queued').toLowerCase()
|
||||
const labels: Record<string, string> = {
|
||||
queued: '等待处理', extracting: '解析文档', chunking: '切分文本', embedding: '向量化中'
|
||||
}
|
||||
const progress = Math.max(0, Math.min(99, Number(doc.indexProgress || 0)))
|
||||
return { tone: 'pending', label: labels[stage] || '处理中', detail: `${labels[stage] || '正在建立知识索引'} ${progress}%`, progress }
|
||||
}
|
||||
return { tone: 'failed', label: '处理失败', detail: doc.errorMessage || '未能建立知识索引,请重新索引或重新上传' }
|
||||
}
|
||||
@@ -174,36 +191,62 @@ const loadQA = async () => {
|
||||
const triggerFile = () => fileInput.value?.click()
|
||||
|
||||
const onFileChange = (e: Event) => {
|
||||
const f = (e.target as HTMLInputElement).files?.[0]
|
||||
if (f) doUpload(f)
|
||||
const files = Array.from((e.target as HTMLInputElement).files || [])
|
||||
if (files.length) uploadFiles(files)
|
||||
;(e.target as HTMLInputElement).value = ''
|
||||
}
|
||||
|
||||
const onDrop = (e: DragEvent) => {
|
||||
dragOver.value = false
|
||||
const f = e.dataTransfer?.files?.[0]
|
||||
if (f) doUpload(f)
|
||||
const files = Array.from(e.dataTransfer?.files || [])
|
||||
if (files.length) uploadFiles(files)
|
||||
}
|
||||
|
||||
const doUpload = async (file: File) => {
|
||||
const uploadFiles = (files: File[]) => {
|
||||
uploadError.value = ''
|
||||
const ext = '.' + (file.name.split('.').pop() || '').toLowerCase()
|
||||
if (!['.md', '.txt', '.pdf', '.doc', '.docx', '.xlsx'].includes(ext)) {
|
||||
uploadError.value = `不支持的类型:${ext},仅支持 md/txt/pdf/doc/docx/xlsx`
|
||||
return
|
||||
}
|
||||
if (!avatarId.value) {
|
||||
uploadError.value = '请先创建数字分身'
|
||||
return
|
||||
}
|
||||
uploading.value = true
|
||||
for (const file of files) {
|
||||
const ext = '.' + (file.name.split('.').pop() || '').toLowerCase()
|
||||
if (!['.md', '.txt', '.pdf', '.doc', '.docx', '.xlsx'].includes(ext)) {
|
||||
uploadError.value = `不支持的类型:${ext},仅支持 md/txt/pdf/doc/docx/xlsx`
|
||||
continue
|
||||
}
|
||||
void uploadOne(file, ext)
|
||||
}
|
||||
}
|
||||
|
||||
const uploadOne = async (file: File, ext: string) => {
|
||||
if (!avatarId.value) return
|
||||
const localId = `upload-${Date.now()}-${Math.random().toString(16).slice(2)}`
|
||||
const card = {
|
||||
id: localId,
|
||||
filename: file.name,
|
||||
fileType: ext.slice(1),
|
||||
fileSize: file.size,
|
||||
createdAt: new Date().toISOString(),
|
||||
localUploading: true,
|
||||
localOnly: true,
|
||||
uploadProgress: 0,
|
||||
errorMessage: ''
|
||||
}
|
||||
pendingUploads.value.unshift(card)
|
||||
try {
|
||||
await uploadKnowledgeDoc(avatarId.value, file)
|
||||
await loadDocs()
|
||||
const created: any = await uploadKnowledgeDoc(avatarId.value, file, (loaded, total) => {
|
||||
const current = pendingUploads.value.find((doc) => doc.id === localId)
|
||||
if (current) current.uploadProgress = Math.min(99, Math.round((loaded / Math.max(1, total)) * 100))
|
||||
})
|
||||
pendingUploads.value = pendingUploads.value.filter((doc) => doc.id !== localId)
|
||||
docs.value = [created, ...docs.value.filter((doc) => doc.id !== created.id)]
|
||||
startDocumentPolling()
|
||||
} catch (e: any) {
|
||||
uploadError.value = e?.message || '上传失败'
|
||||
} finally {
|
||||
uploading.value = false
|
||||
const current = pendingUploads.value.find((doc) => doc.id === localId)
|
||||
if (current) {
|
||||
current.localUploading = false
|
||||
current.errorMessage = e?.message || '上传失败'
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -219,6 +262,11 @@ const retryDoc = async (id: string) => {
|
||||
}
|
||||
|
||||
const removeDoc = async (id: string) => {
|
||||
const local = pendingUploads.value.find((doc) => doc.id === id)
|
||||
if (local?.localOnly) {
|
||||
pendingUploads.value = pendingUploads.value.filter((doc) => doc.id !== id)
|
||||
return
|
||||
}
|
||||
if (!avatarId.value) return
|
||||
await deleteKnowledgeDoc(avatarId.value, id)
|
||||
await loadDocs()
|
||||
@@ -338,6 +386,8 @@ onUnmounted(stopDocumentPolling)
|
||||
.status-pill.missing { color: #B91C1C; background: #FEF2F2; }
|
||||
.status-pill.failed { color: #B91C1C; background: #FEF2F2; }
|
||||
.card-meta, .card-detail { margin: 5px 0 0; color: #9398AE; font-size: 11px; line-height: 1.4; }.card-detail { color: #8B6B58; }
|
||||
.progress-track { width: 100%; height: 4px; margin-top: 8px; overflow: hidden; border-radius: 999px; background: #FDE7D1; }
|
||||
.progress-fill { display: block; height: 100%; border-radius: inherit; background: linear-gradient(90deg, #FB923C, #F97316); transition: width .25s ease; }
|
||||
.card-actions { flex: 0 0 auto; display: flex; flex-direction: column; align-items: stretch; gap: 6px; }
|
||||
.card-delete, .card-retry { align-self: center; border: 0; border-radius: 8px; padding: 7px 9px; font-size: 12px; cursor: pointer; white-space: nowrap; }
|
||||
.card-delete { color: #EF4444; background: #FEF2F2; }
|
||||
|
||||
Reference in New Issue
Block a user