fix: 禁用URL文档上传与解析功能
出于安全考虑,移除所有URL文档上传与解析相关功能: 1. 在知识库路由层添加URL上传校验 2. 移除URL元数据生成逻辑 3. 禁用URL转markdown功能 4. 在各知识库实现中移除URL处理逻辑 5. 在前端禁用URL上传选项并添加提示
This commit is contained in:
parent
55f2e020b5
commit
0ff771dc19
@ -210,6 +210,10 @@ async def add_documents(
|
|||||||
|
|
||||||
content_type = params.get("content_type", "file")
|
content_type = params.get("content_type", "file")
|
||||||
|
|
||||||
|
# 禁止 URL 解析与入库
|
||||||
|
if content_type == "url":
|
||||||
|
raise HTTPException(status_code=400, detail="URL 文档上传与解析已禁用")
|
||||||
|
|
||||||
# 安全检查:验证文件路径
|
# 安全检查:验证文件路径
|
||||||
if content_type == "file":
|
if content_type == "file":
|
||||||
from src.knowledge.utils.kb_utils import validate_file_path
|
from src.knowledge.utils.kb_utils import validate_file_path
|
||||||
|
|||||||
@ -9,7 +9,7 @@ from chromadb.config import Settings
|
|||||||
from chromadb.utils.embedding_functions import OpenAIEmbeddingFunction
|
from chromadb.utils.embedding_functions import OpenAIEmbeddingFunction
|
||||||
|
|
||||||
from src.knowledge.base import KnowledgeBase
|
from src.knowledge.base import KnowledgeBase
|
||||||
from src.knowledge.indexing import process_file_to_markdown, process_url_to_markdown
|
from src.knowledge.indexing import process_file_to_markdown
|
||||||
from src.knowledge.utils.kb_utils import (
|
from src.knowledge.utils.kb_utils import (
|
||||||
get_embedding_config,
|
get_embedding_config,
|
||||||
prepare_item_metadata,
|
prepare_item_metadata,
|
||||||
@ -204,10 +204,9 @@ class ChromaKB(KnowledgeBase):
|
|||||||
params["db_id"] = db_id
|
params["db_id"] = db_id
|
||||||
|
|
||||||
# 根据内容类型处理内容
|
# 根据内容类型处理内容
|
||||||
if content_type == "file":
|
if content_type != "file":
|
||||||
markdown_content = await process_file_to_markdown(item, params=params)
|
raise ValueError("URL 内容解析已禁用")
|
||||||
else: # URL
|
markdown_content = await process_file_to_markdown(item, params=params)
|
||||||
markdown_content = await process_url_to_markdown(item, params=params)
|
|
||||||
|
|
||||||
# 分割文本成块
|
# 分割文本成块
|
||||||
chunks = self._split_text_into_chunks(markdown_content, file_id, filename, params)
|
chunks = self._split_text_into_chunks(markdown_content, file_id, filename, params)
|
||||||
@ -296,10 +295,9 @@ class ChromaKB(KnowledgeBase):
|
|||||||
self._save_metadata()
|
self._save_metadata()
|
||||||
|
|
||||||
# 重新解析文件为 markdown
|
# 重新解析文件为 markdown
|
||||||
if content_type == "file":
|
if content_type != "file":
|
||||||
markdown_content = await process_file_to_markdown(file_path, params=params)
|
raise ValueError("URL 内容解析已禁用")
|
||||||
else:
|
markdown_content = await process_file_to_markdown(file_path, params=params)
|
||||||
markdown_content = await process_url_to_markdown(file_path, params=params)
|
|
||||||
|
|
||||||
# 先删除现有的 ChromaDB 数据(仅删除chunks,保留元数据)
|
# 先删除现有的 ChromaDB 数据(仅删除chunks,保留元数据)
|
||||||
await self.delete_file_chunks_only(db_id, file_id)
|
await self.delete_file_chunks_only(db_id, file_id)
|
||||||
|
|||||||
@ -10,7 +10,7 @@ from pymilvus import connections, utility
|
|||||||
|
|
||||||
from src import config
|
from src import config
|
||||||
from src.knowledge.base import KnowledgeBase
|
from src.knowledge.base import KnowledgeBase
|
||||||
from src.knowledge.indexing import process_file_to_markdown, process_url_to_markdown
|
from src.knowledge.indexing import process_file_to_markdown
|
||||||
from src.knowledge.utils.kb_utils import get_embedding_config, prepare_item_metadata
|
from src.knowledge.utils.kb_utils import get_embedding_config, prepare_item_metadata
|
||||||
from src.utils import hashstr, logger
|
from src.utils import hashstr, logger
|
||||||
from src.utils.datetime_utils import shanghai_now
|
from src.utils.datetime_utils import shanghai_now
|
||||||
@ -243,12 +243,11 @@ class LightRagKB(KnowledgeBase):
|
|||||||
params["db_id"] = db_id
|
params["db_id"] = db_id
|
||||||
|
|
||||||
# 根据内容类型处理内容
|
# 根据内容类型处理内容
|
||||||
if content_type == "file":
|
if content_type != "file":
|
||||||
markdown_content = await process_file_to_markdown(item, params=params)
|
raise ValueError("URL 内容解析已禁用")
|
||||||
markdown_content_lines = markdown_content[:100].replace("\n", " ")
|
markdown_content = await process_file_to_markdown(item, params=params)
|
||||||
logger.info(f"Markdown content: {markdown_content_lines}...")
|
markdown_content_lines = markdown_content[:100].replace("\n", " ")
|
||||||
else: # URL
|
logger.info(f"Markdown content: {markdown_content_lines}...")
|
||||||
markdown_content = await process_url_to_markdown(item, params=params)
|
|
||||||
|
|
||||||
# 使用 LightRAG 插入内容
|
# 使用 LightRAG 插入内容
|
||||||
await rag.ainsert(input=markdown_content, ids=file_id, file_paths=item_path)
|
await rag.ainsert(input=markdown_content, ids=file_id, file_paths=item_path)
|
||||||
@ -313,12 +312,11 @@ class LightRagKB(KnowledgeBase):
|
|||||||
self._save_metadata()
|
self._save_metadata()
|
||||||
|
|
||||||
# 重新解析文件为 markdown
|
# 重新解析文件为 markdown
|
||||||
if content_type == "file":
|
if content_type != "file":
|
||||||
markdown_content = await process_file_to_markdown(file_path, params=params)
|
raise ValueError("URL 内容解析已禁用")
|
||||||
markdown_content_lines = markdown_content[:100].replace("\n", " ")
|
markdown_content = await process_file_to_markdown(file_path, params=params)
|
||||||
logger.info(f"Markdown content: {markdown_content_lines}...")
|
markdown_content_lines = markdown_content[:100].replace("\n", " ")
|
||||||
else:
|
logger.info(f"Markdown content: {markdown_content_lines}...")
|
||||||
markdown_content = await process_url_to_markdown(file_path, params=params)
|
|
||||||
|
|
||||||
# 先删除现有的 LightRAG 数据(仅删除chunks,保留元数据)
|
# 先删除现有的 LightRAG 数据(仅删除chunks,保留元数据)
|
||||||
await self.delete_file_chunks_only(db_id, file_id)
|
await self.delete_file_chunks_only(db_id, file_id)
|
||||||
|
|||||||
@ -8,7 +8,7 @@ from typing import Any
|
|||||||
from pymilvus import Collection, CollectionSchema, DataType, FieldSchema, connections, db, utility
|
from pymilvus import Collection, CollectionSchema, DataType, FieldSchema, connections, db, utility
|
||||||
|
|
||||||
from src.knowledge.base import KnowledgeBase
|
from src.knowledge.base import KnowledgeBase
|
||||||
from src.knowledge.indexing import process_file_to_markdown, process_url_to_markdown
|
from src.knowledge.indexing import process_file_to_markdown
|
||||||
from src.knowledge.utils.kb_utils import (
|
from src.knowledge.utils.kb_utils import (
|
||||||
get_embedding_config,
|
get_embedding_config,
|
||||||
prepare_item_metadata,
|
prepare_item_metadata,
|
||||||
@ -247,10 +247,9 @@ class MilvusKB(KnowledgeBase):
|
|||||||
params = {}
|
params = {}
|
||||||
params["db_id"] = db_id
|
params["db_id"] = db_id
|
||||||
|
|
||||||
if content_type == "file":
|
if content_type != "file":
|
||||||
markdown_content = await process_file_to_markdown(item, params=params)
|
raise ValueError("URL 内容解析已禁用")
|
||||||
else:
|
markdown_content = await process_file_to_markdown(item, params=params)
|
||||||
markdown_content = await process_url_to_markdown(item, params=params)
|
|
||||||
|
|
||||||
chunks = self._split_text_into_chunks(markdown_content, file_id, filename, params)
|
chunks = self._split_text_into_chunks(markdown_content, file_id, filename, params)
|
||||||
logger.info(f"Split {filename} into {len(chunks)} chunks")
|
logger.info(f"Split {filename} into {len(chunks)} chunks")
|
||||||
@ -342,10 +341,9 @@ class MilvusKB(KnowledgeBase):
|
|||||||
self._save_metadata()
|
self._save_metadata()
|
||||||
|
|
||||||
# 重新解析文件为 markdown
|
# 重新解析文件为 markdown
|
||||||
if content_type == "file":
|
if content_type != "file":
|
||||||
markdown_content = await process_file_to_markdown(file_path, params=params)
|
raise ValueError("URL 内容解析已禁用")
|
||||||
else:
|
markdown_content = await process_file_to_markdown(file_path, params=params)
|
||||||
markdown_content = await process_url_to_markdown(file_path, params=params)
|
|
||||||
|
|
||||||
# 先删除现有的 Milvus 数据(仅删除chunks,保留元数据)
|
# 先删除现有的 Milvus 数据(仅删除chunks,保留元数据)
|
||||||
await self.delete_file_chunks_only(db_id, file_id)
|
await self.delete_file_chunks_only(db_id, file_id)
|
||||||
|
|||||||
@ -551,24 +551,4 @@ def _replace_image_links(markdown_content: str, images: list[dict]) -> str:
|
|||||||
|
|
||||||
|
|
||||||
async def process_url_to_markdown(url: str, params: dict | None = None) -> str:
|
async def process_url_to_markdown(url: str, params: dict | None = None) -> str:
|
||||||
"""
|
raise NotImplementedError("URL 解析功能已禁用")
|
||||||
将URL转换为markdown格式
|
|
||||||
|
|
||||||
Args:
|
|
||||||
url: URL地址
|
|
||||||
params: 处理参数
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
markdown格式内容
|
|
||||||
"""
|
|
||||||
import requests
|
|
||||||
from bs4 import BeautifulSoup
|
|
||||||
|
|
||||||
try:
|
|
||||||
response = requests.get(url, timeout=30)
|
|
||||||
soup = BeautifulSoup(response.content, "html.parser")
|
|
||||||
text_content = soup.get_text()
|
|
||||||
return f"# {url}\n\n{text_content}"
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Failed to process URL {url}: {e}")
|
|
||||||
return f"# {url}\n\nFailed to process URL: {e}"
|
|
||||||
|
|||||||
@ -150,12 +150,8 @@ def prepare_item_metadata(item: str, content_type: str, db_id: str, params: dict
|
|||||||
content_hash = calculate_content_hash(file_path)
|
content_hash = calculate_content_hash(file_path)
|
||||||
except Exception as exc: # noqa: BLE001
|
except Exception as exc: # noqa: BLE001
|
||||||
logger.warning(f"Failed to calculate content hash for {file_path}: {exc}")
|
logger.warning(f"Failed to calculate content hash for {file_path}: {exc}")
|
||||||
else: # URL
|
else:
|
||||||
file_id = f"url_{hashstr(item + str(time.time()), 6)}"
|
raise ValueError("URL 元数据生成已禁用")
|
||||||
file_type = "url"
|
|
||||||
filename = f"webpage_{hashstr(item, 6)}.md"
|
|
||||||
item_path = item
|
|
||||||
content_hash = None
|
|
||||||
|
|
||||||
metadata = {
|
metadata = {
|
||||||
"database_id": db_id,
|
"database_id": db_id,
|
||||||
|
|||||||
@ -12,7 +12,7 @@
|
|||||||
type="primary"
|
type="primary"
|
||||||
@click="chunkData"
|
@click="chunkData"
|
||||||
:loading="chunkLoading"
|
:loading="chunkLoading"
|
||||||
:disabled="(uploadMode === 'file' && fileList.length === 0) || (uploadMode === 'url' && !urlList.trim())"
|
:disabled="fileList.length === 0"
|
||||||
>
|
>
|
||||||
添加到知识库
|
添加到知识库
|
||||||
</a-button>
|
</a-button>
|
||||||
@ -26,6 +26,7 @@
|
|||||||
:options="uploadModeOptions"
|
:options="uploadModeOptions"
|
||||||
size="large"
|
size="large"
|
||||||
class="source-segmented"
|
class="source-segmented"
|
||||||
|
:disabled="true"
|
||||||
/>
|
/>
|
||||||
</div>
|
</div>
|
||||||
<div class="config-controls">
|
<div class="config-controls">
|
||||||
@ -102,23 +103,6 @@
|
|||||||
</div>
|
</div>
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
<!-- URL 输入区域 -->
|
|
||||||
<div class="url-input" v-if="uploadMode === 'url'">
|
|
||||||
<a-form layout="vertical">
|
|
||||||
<a-form-item label="网页链接 (每行一个URL)">
|
|
||||||
<a-textarea
|
|
||||||
v-model:value="urlList"
|
|
||||||
placeholder="请输入网页链接,每行一个"
|
|
||||||
:rows="6"
|
|
||||||
:disabled="chunkLoading"
|
|
||||||
/>
|
|
||||||
</a-form-item>
|
|
||||||
</a-form>
|
|
||||||
<p class="url-hint">
|
|
||||||
支持添加网页内容,系统会自动抓取网页文本并进行分块。请确保URL格式正确且可以公开访问。
|
|
||||||
</p>
|
|
||||||
</div>
|
|
||||||
</div>
|
</div>
|
||||||
</a-modal>
|
</a-modal>
|
||||||
|
|
||||||
@ -139,7 +123,7 @@
|
|||||||
|
|
||||||
<script setup>
|
<script setup>
|
||||||
import { ref, computed, onMounted, watch } from 'vue';
|
import { ref, computed, onMounted, watch } from 'vue';
|
||||||
import { message, Upload } from 'ant-design-vue';
|
import { message, Upload, Tooltip } from 'ant-design-vue';
|
||||||
import { useUserStore } from '@/stores/user';
|
import { useUserStore } from '@/stores/user';
|
||||||
import { useDatabaseStore } from '@/stores/database';
|
import { useDatabaseStore } from '@/stores/database';
|
||||||
import { ocrApi } from '@/apis/system_api';
|
import { ocrApi } from '@/apis/system_api';
|
||||||
@ -269,10 +253,12 @@ const uploadModeOptions = computed(() => [
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
value: 'url',
|
value: 'url',
|
||||||
label: h('div', { class: 'segmented-option' }, [
|
label: h(Tooltip, { title: 'URL 文档上传与解析功能已禁用,出于安全考虑,当前版本仅支持文件上传' }, {
|
||||||
h(LinkOutlined, { class: 'option-icon' }),
|
default: () => h('div', { class: 'segmented-option' }, [
|
||||||
h('span', { class: 'option-text' }, '输入网址'),
|
h(LinkOutlined, { class: 'option-icon' }),
|
||||||
]),
|
h('span', { class: 'option-text' }, '输入网址'),
|
||||||
|
])
|
||||||
|
}),
|
||||||
},
|
},
|
||||||
]);
|
]);
|
||||||
|
|
||||||
@ -280,8 +266,7 @@ const uploadModeOptions = computed(() => [
|
|||||||
const fileList = ref([]);
|
const fileList = ref([]);
|
||||||
|
|
||||||
|
|
||||||
// URL列表
|
// URL相关功能已移除
|
||||||
const urlList = ref('');
|
|
||||||
|
|
||||||
// OCR服务健康状态
|
// OCR服务健康状态
|
||||||
const ocrHealthStatus = ref({
|
const ocrHealthStatus = ref({
|
||||||
@ -330,14 +315,7 @@ const isOcrEnabled = computed(() => {
|
|||||||
return chunkParams.value.enable_ocr !== 'disable';
|
return chunkParams.value.enable_ocr !== 'disable';
|
||||||
});
|
});
|
||||||
|
|
||||||
watch(uploadMode, (mode, previous) => {
|
// 上传模式切换相关逻辑已移除
|
||||||
if (mode === 'url') {
|
|
||||||
previousOcrSelection.value = chunkParams.value.enable_ocr;
|
|
||||||
chunkParams.value.enable_ocr = 'disable';
|
|
||||||
} else if (mode === 'file' && previous === 'url') {
|
|
||||||
chunkParams.value.enable_ocr = previousOcrSelection.value || 'disable';
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
// 计算属性:是否有PDF或图片文件
|
// 计算属性:是否有PDF或图片文件
|
||||||
const hasPdfOrImageFiles = computed(() => {
|
const hasPdfOrImageFiles = computed(() => {
|
||||||
@ -630,31 +608,11 @@ const chunkData = async () => {
|
|||||||
} finally {
|
} finally {
|
||||||
store.state.chunkLoading = false;
|
store.state.chunkLoading = false;
|
||||||
}
|
}
|
||||||
} else if (uploadMode.value === 'url') {
|
|
||||||
const urls = urlList.value.split('\n')
|
|
||||||
.map(url => url.trim())
|
|
||||||
.filter(url => url.length > 0 && (url.startsWith('http://') || url.startsWith('https://')));
|
|
||||||
|
|
||||||
if (urls.length === 0) {
|
|
||||||
message.error('请输入有效的网页链接(必须以http://或https://开头)');
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
try {
|
|
||||||
store.state.chunkLoading = true;
|
|
||||||
success = await store.addFiles({ items: urls, contentType: 'url', params: chunkParams.value });
|
|
||||||
} catch (error) {
|
|
||||||
console.error('URL上传失败:', error);
|
|
||||||
message.error('URL上传失败: ' + (error.message || '未知错误'));
|
|
||||||
} finally {
|
|
||||||
store.state.chunkLoading = false;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
if (success) {
|
if (success) {
|
||||||
emit('update:visible', false);
|
emit('update:visible', false);
|
||||||
fileList.value = [];
|
fileList.value = [];
|
||||||
urlList.value = '';
|
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user