diff --git a/backend/package/yuxi/knowledge/base.py b/backend/package/yuxi/knowledge/base.py index 278c8e8c..f2d74519 100644 --- a/backend/package/yuxi/knowledge/base.py +++ b/backend/package/yuxi/knowledge/base.py @@ -351,6 +351,18 @@ class KnowledgeBase(ABC): await self._persist_file(file_id) + async def _mark_file_unparsed(self, file_id: str, operator_id: str | None = None) -> None: + if file_id not in self.files_meta: + return + + self.files_meta[file_id]["status"] = FileStatus.UPLOADED + self.files_meta[file_id].pop("markdown_file", None) + self.files_meta[file_id].pop("error", None) + self.files_meta[file_id]["updated_at"] = utc_isoformat() + if operator_id: + self.files_meta[file_id]["updated_by"] = operator_id + await self._persist_file(file_id) + async def _save_markdown_to_minio(self, db_id: str, file_id: str, content: str) -> str: """Save markdown content to MinIO and return HTTP URL""" from yuxi.storage.minio import get_minio_client diff --git a/backend/package/yuxi/knowledge/implementations/lightrag.py b/backend/package/yuxi/knowledge/implementations/lightrag.py index d0e22971..9e135c0f 100644 --- a/backend/package/yuxi/knowledge/implementations/lightrag.py +++ b/backend/package/yuxi/knowledge/implementations/lightrag.py @@ -348,6 +348,7 @@ class LightRagKB(KnowledgeBase): # Check markdown file exists if not file_meta.get("markdown_file"): + await self._mark_file_unparsed(file_id, operator_id) raise ValueError("File has not been parsed yet (no markdown_file)") # Clear previous error if any diff --git a/backend/package/yuxi/knowledge/implementations/milvus.py b/backend/package/yuxi/knowledge/implementations/milvus.py index 35b82a2d..5ba23cf2 100644 --- a/backend/package/yuxi/knowledge/implementations/milvus.py +++ b/backend/package/yuxi/knowledge/implementations/milvus.py @@ -333,6 +333,7 @@ class MilvusKB(KnowledgeBase): # Check markdown file exists if not file_meta.get("markdown_file"): + await self._mark_file_unparsed(file_id, operator_id) raise ValueError("File has not been parsed yet (no markdown_file)") # Clear previous error if any diff --git a/backend/test/unit/backends/test_lightrag_serialization.py b/backend/test/unit/backends/test_lightrag_serialization.py index e3de87d8..c6a01982 100644 --- a/backend/test/unit/backends/test_lightrag_serialization.py +++ b/backend/test/unit/backends/test_lightrag_serialization.py @@ -52,6 +52,36 @@ async def _fake_read_markdown(*_args, **_kwargs) -> str: return "mock markdown" +async def test_index_file_marks_file_unparsed_when_markdown_file_is_missing( + light_rag_kb: LightRagKB, + monkeypatch: pytest.MonkeyPatch, +) -> None: + persisted: list[str] = [] + db_id = "kb_missing_markdown" + file_id = "file-missing" + light_rag_kb.databases_meta[db_id] = {"metadata": {}} + light_rag_kb.files_meta[file_id] = { + "status": FileStatus.PARSED, + "path": "/tmp/file.md", + "filename": "file.md", + "processing_params": {}, + } + + async def fake_persist_file(saved_file_id: str) -> None: + persisted.append(saved_file_id) + + monkeypatch.setattr(light_rag_kb, "_persist_file", fake_persist_file) + monkeypatch.setattr(light_rag_kb, "_get_lightrag_instance", _make_async_return(SimpleNamespace())) + + with pytest.raises(ValueError, match="no markdown_file"): + await light_rag_kb.index_file(db_id, file_id, operator_id="user-1") + + assert light_rag_kb.files_meta[file_id]["status"] == FileStatus.UPLOADED + assert "markdown_file" not in light_rag_kb.files_meta[file_id] + assert light_rag_kb.files_meta[file_id]["updated_by"] == "user-1" + assert persisted == [file_id] + + async def test_index_file_serializes_writes_within_same_database( light_rag_kb: LightRagKB, monkeypatch: pytest.MonkeyPatch, @@ -151,9 +181,15 @@ async def test_add_documents_auto_index_uses_latest_parsed_metadata(monkeypatch: ) -> None: calls.append(("update_params", file_id)) - async def index_file(self, _db_id: str, file_id: str, operator_id: str | None = None) -> dict: + async def index_file( + self, + _db_id: str, + file_id: str, + operator_id: str | None = None, + params: dict | None = None, + ) -> dict: calls.append(("index", file_id)) - return {"file_id": file_id, "status": FileStatus.INDEXED, "operator_id": operator_id} + return {"file_id": file_id, "status": FileStatus.INDEXED, "operator_id": operator_id, "params": params} class FakeTaskContext: async def set_message(self, _message: str) -> None: diff --git a/docs/develop-guides/roadmap.md b/docs/develop-guides/roadmap.md index 0ea12f66..93bded10 100644 --- a/docs/develop-guides/roadmap.md +++ b/docs/develop-guides/roadmap.md @@ -37,6 +37,7 @@ ### 0.6.2 开发记录 +- 修复知识库文档入库状态回退:当已解析文件缺失 `markdown_file` 解析产物时,索引流程会将文件状态恢复为未解析,便于重新解析而不是停留在索引失败。 - 优化 Agent 输入框文件 mention:用户级 workspace 文件候选改为从独立 workspace API 递归加载,不再依赖 active thread;插入时仍转换为 `/home/gem/user-data/workspace/` 沙盒虚拟路径,并修复附件上传后未立即刷新 mention 候选的问题。 - 调整知识库思维导图后端结构:将思维导图路由文件重命名为知识库语义更明确的 router,并把文件列表整理、提示词构建、AI JSON 解析等纯逻辑下沉到知识库 utils。 - 新增个人工作区预览与管理:提供独立于对话 thread 的用户级 workspace API,并增加“工作区”页面,用于浏览个人 workspace 文件、预览 Markdown/文本/代码/图片/PDF;支持新建文件夹、上传文件、下载文件、删除文件/文件夹和多选删除;工作区预览支持 Markdown/TXT 在右侧预览框内切换编辑并保存,其他格式和非工作区预览默认只读;知识库与团队空间入口先展示到占位层级;默认创建 `agents/AGENTS.md`,并在 Agent 执行时将其内容追加到系统提示词。