fix(avatar): upload knowledge files in chunks
This commit is contained in:
@@ -87,6 +87,84 @@ def test_upload_rejects_oversize_file_before_queuing_indexing(
|
||||
assert not list((tmp_path / context["avatar"].id).glob("*"))
|
||||
|
||||
|
||||
def test_multipart_upload_reassembles_file_before_queuing_indexing(
|
||||
tmp_path: Path,
|
||||
authorization_context,
|
||||
):
|
||||
context = authorization_context
|
||||
avatar_id = context["avatar"].id
|
||||
content = b"0123456789"
|
||||
with (
|
||||
patch("routers.knowledge.UPLOAD_DIR", str(tmp_path)),
|
||||
patch("routers.knowledge.MULTIPART_CHUNK_BYTES", 4),
|
||||
patch("routers.knowledge.knowledge_vectorizer.enqueue") as enqueue,
|
||||
):
|
||||
created = client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads",
|
||||
headers=context["owner_headers"],
|
||||
json={"filename": "large.pdf", "fileSize": len(content), "totalChunks": 3},
|
||||
).json()["data"]
|
||||
|
||||
for index, chunk in enumerate((content[:4], content[4:8], content[8:])):
|
||||
response = client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads/{created['uploadId']}/chunks/{index}",
|
||||
headers=context["owner_headers"],
|
||||
files={"file": (f"chunk-{index}", chunk, "application/octet-stream")},
|
||||
)
|
||||
assert response.json()["code"] == 200
|
||||
|
||||
completed = client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads/{created['uploadId']}/complete",
|
||||
headers=context["owner_headers"],
|
||||
).json()["data"]
|
||||
|
||||
assert completed["status"] == "parsing"
|
||||
assert completed["fileSize"] == len(content)
|
||||
enqueue.assert_called_once_with(completed["id"])
|
||||
stored_path = tmp_path / avatar_id / Path(completed["fileUrl"]).name
|
||||
assert stored_path.read_bytes() == content
|
||||
assert not (tmp_path / ".multipart" / avatar_id / created["uploadId"]).exists()
|
||||
|
||||
db = SessionLocal()
|
||||
try:
|
||||
stored = db.query(KnowledgeDoc).filter(KnowledgeDoc.id == completed["id"]).one()
|
||||
db.delete(stored)
|
||||
db.commit()
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
def test_multipart_upload_rejects_incomplete_parts(
|
||||
tmp_path: Path,
|
||||
authorization_context,
|
||||
):
|
||||
context = authorization_context
|
||||
avatar_id = context["avatar"].id
|
||||
with (
|
||||
patch("routers.knowledge.UPLOAD_DIR", str(tmp_path)),
|
||||
patch("routers.knowledge.MULTIPART_CHUNK_BYTES", 4),
|
||||
patch("routers.knowledge.knowledge_vectorizer.enqueue") as enqueue,
|
||||
):
|
||||
created = client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads",
|
||||
headers=context["owner_headers"],
|
||||
json={"filename": "large.pdf", "fileSize": 6, "totalChunks": 2},
|
||||
).json()["data"]
|
||||
client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads/{created['uploadId']}/chunks/0",
|
||||
headers=context["owner_headers"],
|
||||
files={"file": ("chunk-0", b"0123", "application/octet-stream")},
|
||||
)
|
||||
response = client.post(
|
||||
f"/api/avatar/{avatar_id}/knowledge/uploads/{created['uploadId']}/complete",
|
||||
headers=context["owner_headers"],
|
||||
)
|
||||
|
||||
assert response.json()["code"] == 400
|
||||
assert response.json()["message"] == "文件分片尚未上传完整"
|
||||
enqueue.assert_not_called()
|
||||
|
||||
|
||||
def test_background_vectorizer_commits_ready_document_and_chunks_together(
|
||||
tmp_path: Path,
|
||||
authorization_context,
|
||||
|
||||
Reference in New Issue
Block a user