refactor: 更新文档,优化批量上传和解析脚本说明;移除不再支持的转换功能
This commit is contained in:
parent
d56cec5002
commit
59727d6d7e
@ -60,7 +60,7 @@ docker compose up paddlex --build
|
|||||||
|
|
||||||
## 批量处理脚本
|
## 批量处理脚本
|
||||||
|
|
||||||
系统提供便捷的批量处理脚本,支持文件上传和解析操作。
|
系统提供便捷的批量处理脚本,用于高效批量上传文档。
|
||||||
|
|
||||||
### 文件上传脚本
|
### 文件上传脚本
|
||||||
|
|
||||||
@ -90,26 +90,6 @@ uv run scripts/batch_upload.py upload \
|
|||||||
|
|
||||||
提示:系统按“内容哈希”进行去重;同一知识库已存在相同内容的文件会被拒绝(409)。
|
提示:系统按“内容哈希”进行去重;同一知识库已存在相同内容的文件会被拒绝(409)。
|
||||||
|
|
||||||
### 文件解析脚本
|
|
||||||
|
|
||||||
使用 `scripts/batch_upload.py trans` 将文件解析为 Markdown:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
# 批量解析文档
|
|
||||||
uv run scripts/batch_upload.py trans \
|
|
||||||
--db-id your_kb_id \
|
|
||||||
--directory path/to/your/data \
|
|
||||||
--output-dir path/to/output_markdown \
|
|
||||||
--pattern "*.docx" \
|
|
||||||
--base-url http://127.0.0.1:5050/api \
|
|
||||||
--username your_username \
|
|
||||||
--password your_password \
|
|
||||||
--concurrency 4 \
|
|
||||||
--recursive
|
|
||||||
```
|
|
||||||
|
|
||||||
**输出结果**: 解析后的 Markdown 文件将保存到指定输出目录。
|
|
||||||
|
|
||||||
### 脚本功能
|
### 脚本功能
|
||||||
|
|
||||||
- **进度跟踪**: 实时显示处理进度
|
- **进度跟踪**: 实时显示处理进度
|
||||||
|
|||||||
@ -32,7 +32,6 @@
|
|||||||
|
|
||||||
- 批量上传与转换示例?
|
- 批量上传与转换示例?
|
||||||
- 上传入库:`uv run scripts/batch_upload.py upload --db-id <id> --directory <dir> --username <u> --password <p> --base-url http://127.0.0.1:5050/api`
|
- 上传入库:`uv run scripts/batch_upload.py upload --db-id <id> --directory <dir> --username <u> --password <p> --base-url http://127.0.0.1:5050/api`
|
||||||
- 转 Markdown:`uv run scripts/batch_upload.py trans --db-id <id> --directory <dir> --username <u> --password <p>`
|
|
||||||
- 参考:高级配置 → 文档解析
|
- 参考:高级配置 → 文档解析
|
||||||
|
|
||||||
- 登录失败被锁定?
|
- 登录失败被锁定?
|
||||||
|
|||||||
@ -58,7 +58,6 @@ LightRAG 知识库可在知识库详情中可视化,但不支持在侧边栏
|
|||||||
### 批量脚本
|
### 批量脚本
|
||||||
|
|
||||||
- 上传并入库:参见 `scripts/batch_upload.py upload`
|
- 上传并入库:参见 `scripts/batch_upload.py upload`
|
||||||
- 转换为 Markdown:参见 `scripts/batch_upload.py trans`
|
|
||||||
|
|
||||||
## 知识图谱
|
## 知识图谱
|
||||||
|
|
||||||
|
|||||||
@ -230,74 +230,6 @@ def save_processed_files(record_file: pathlib.Path, processed_files: set[str]):
|
|||||||
console.print(f"[bold red]Error: Could not save processed files record: {e}[/bold red]")
|
console.print(f"[bold red]Error: Could not save processed files record: {e}[/bold red]")
|
||||||
|
|
||||||
|
|
||||||
async def convert_to_markdown(
|
|
||||||
client: httpx.AsyncClient,
|
|
||||||
base_url: str,
|
|
||||||
db_id: str,
|
|
||||||
server_file_path: str,
|
|
||||||
) -> str | None:
|
|
||||||
"""Calls the file-to-markdown conversion endpoint."""
|
|
||||||
try:
|
|
||||||
response = await client.post(
|
|
||||||
f"{base_url}/knowledge/files/markdown",
|
|
||||||
json={"db_id": db_id, "file_path": server_file_path},
|
|
||||||
timeout=600, # 10 minutes timeout for conversion
|
|
||||||
)
|
|
||||||
response.raise_for_status()
|
|
||||||
result = response.json()
|
|
||||||
if result.get("status") == "success":
|
|
||||||
return result.get("markdown_content")
|
|
||||||
else:
|
|
||||||
console.print(f"[bold red]Failed to convert {server_file_path}: {result.get('message')}[/bold red]")
|
|
||||||
return None
|
|
||||||
except httpx.HTTPStatusError as e:
|
|
||||||
console.print(
|
|
||||||
f"[bold red]Failed to convert {server_file_path}: {e.response.status_code} - {e.response.text}[/bold red]"
|
|
||||||
)
|
|
||||||
return None
|
|
||||||
except httpx.RequestError as e:
|
|
||||||
console.print(f"[bold red]Request failed for {server_file_path}: {e}[/bold red]")
|
|
||||||
return None
|
|
||||||
|
|
||||||
|
|
||||||
async def trans_worker(
|
|
||||||
semaphore: asyncio.Semaphore,
|
|
||||||
client: httpx.AsyncClient,
|
|
||||||
base_url: str,
|
|
||||||
db_id: str,
|
|
||||||
file_path: pathlib.Path,
|
|
||||||
output_dir: pathlib.Path,
|
|
||||||
progress: Progress,
|
|
||||||
task_id: int,
|
|
||||||
):
|
|
||||||
"""A worker task that uploads a file and converts it to markdown."""
|
|
||||||
async with semaphore:
|
|
||||||
# 1. Upload file
|
|
||||||
server_file_path = await upload_file(client, base_url, db_id, file_path)
|
|
||||||
if not server_file_path:
|
|
||||||
progress.update(task_id, advance=1, postfix=f"[red]Upload failed for {file_path.name}[/red]")
|
|
||||||
return file_path, "upload_failed"
|
|
||||||
|
|
||||||
# 2. Convert file to markdown
|
|
||||||
markdown_content = await convert_to_markdown(client, base_url, db_id, server_file_path)
|
|
||||||
if not markdown_content:
|
|
||||||
progress.update(task_id, advance=1, postfix=f"[yellow]Conversion failed for {file_path.name}[/yellow]")
|
|
||||||
return file_path, "conversion_failed"
|
|
||||||
|
|
||||||
# 3. Save markdown content to output directory
|
|
||||||
try:
|
|
||||||
output_path = output_dir / file_path.with_suffix(".md").name
|
|
||||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
with open(output_path, "w", encoding="utf-8") as f:
|
|
||||||
f.write(markdown_content)
|
|
||||||
progress.update(task_id, advance=1, postfix=f"[green]Converted {file_path.name}[/green]")
|
|
||||||
return file_path, "success"
|
|
||||||
except OSError as e:
|
|
||||||
console.print(f"[bold red]Error saving markdown for {file_path.name}: {e}[/bold red]")
|
|
||||||
progress.update(task_id, advance=1, postfix=f"[red]Save failed for {file_path.name}[/red]")
|
|
||||||
return file_path, "save_failed"
|
|
||||||
|
|
||||||
|
|
||||||
@app.command()
|
@app.command()
|
||||||
def upload(
|
def upload(
|
||||||
db_id: str = typer.Option(..., help="The ID of the knowledge base."),
|
db_id: str = typer.Option(..., help="The ID of the knowledge base."),
|
||||||
@ -447,91 +379,6 @@ def upload(
|
|||||||
asyncio.run(run())
|
asyncio.run(run())
|
||||||
|
|
||||||
|
|
||||||
@app.command()
|
|
||||||
def trans(
|
|
||||||
db_id: str = typer.Option(..., help="The ID of the knowledge base (for temporary file upload)."),
|
|
||||||
directory: pathlib.Path = typer.Option(
|
|
||||||
..., help="The directory containing files to convert.", exists=True, file_okay=False
|
|
||||||
),
|
|
||||||
output_dir: pathlib.Path = typer.Option("output_markdown", help="The directory to save converted markdown files."),
|
|
||||||
pattern: str = typer.Option("*.docx", help="The glob pattern for files to convert (e.g., '*.pdf', '*.docx')."),
|
|
||||||
base_url: str = typer.Option("http://127.0.0.1:5050/api", help="The base URL of the API server."),
|
|
||||||
username: str = typer.Option(..., help="Admin username for login."),
|
|
||||||
password: str = typer.Option(..., help="Admin password for login."),
|
|
||||||
concurrency: int = typer.Option(4, help="The number of concurrent conversion tasks."),
|
|
||||||
recursive: bool = typer.Option(False, "--recursive", "-r", help="Search for files recursively in subdirectories."),
|
|
||||||
):
|
|
||||||
"""
|
|
||||||
Batch convert files to Markdown format.
|
|
||||||
"""
|
|
||||||
console.print(f"[bold green]Starting batch conversion for files in: {directory}[/bold green]")
|
|
||||||
output_dir.mkdir(parents=True, exist_ok=True)
|
|
||||||
|
|
||||||
# Discover files
|
|
||||||
glob_method = directory.rglob if recursive else directory.glob
|
|
||||||
files_to_convert = list(glob_method(pattern))
|
|
||||||
if not files_to_convert:
|
|
||||||
console.print(f"[bold yellow]No files found in '{directory}' matching '{pattern}'. Aborting.[/bold yellow]")
|
|
||||||
raise typer.Exit()
|
|
||||||
|
|
||||||
# 过滤掉macos的隐藏文件
|
|
||||||
files_to_convert = [f for f in files_to_convert if not f.name.startswith("._")]
|
|
||||||
|
|
||||||
console.print(f"Found {len(files_to_convert)} files to convert.")
|
|
||||||
|
|
||||||
async def run():
|
|
||||||
async with httpx.AsyncClient() as client:
|
|
||||||
# Login
|
|
||||||
token = await login(client, base_url, username, password)
|
|
||||||
if not token:
|
|
||||||
raise typer.Exit(code=1)
|
|
||||||
|
|
||||||
client.headers = {"Authorization": f"Bearer {token}"}
|
|
||||||
|
|
||||||
# Setup concurrency and tasks
|
|
||||||
semaphore = asyncio.Semaphore(concurrency)
|
|
||||||
tasks = []
|
|
||||||
|
|
||||||
with Progress(
|
|
||||||
SpinnerColumn(),
|
|
||||||
TextColumn("[progress.description]{task.description}"),
|
|
||||||
BarColumn(),
|
|
||||||
TextColumn("[progress.percentage]{task.percentage:>3.0f}%"),
|
|
||||||
TimeElapsedColumn(),
|
|
||||||
TextColumn("{task.fields[postfix]}"),
|
|
||||||
console=console,
|
|
||||||
transient=True,
|
|
||||||
) as progress:
|
|
||||||
task_id = progress.add_task("[bold blue]Converting...", total=len(files_to_convert), postfix="")
|
|
||||||
|
|
||||||
for file_path in files_to_convert:
|
|
||||||
task = asyncio.create_task(
|
|
||||||
trans_worker(semaphore, client, base_url, db_id, file_path, output_dir, progress, task_id)
|
|
||||||
)
|
|
||||||
tasks.append(task)
|
|
||||||
|
|
||||||
results = await asyncio.gather(*tasks)
|
|
||||||
|
|
||||||
# Summarize results
|
|
||||||
successful_files = []
|
|
||||||
failed_files = []
|
|
||||||
|
|
||||||
for file_path, status in results:
|
|
||||||
if status == "success":
|
|
||||||
successful_files.append(file_path)
|
|
||||||
else:
|
|
||||||
failed_files.append((file_path, status))
|
|
||||||
|
|
||||||
console.print("[bold green]Batch conversion complete.[/bold green]")
|
|
||||||
console.print(f" - [green]Successful:[/green] {len(successful_files)}")
|
|
||||||
console.print(f" - [red]Failed:[/red] {len(failed_files)}")
|
|
||||||
if failed_files:
|
|
||||||
for f, status in failed_files:
|
|
||||||
console.print(f" - {f} (Reason: {status})")
|
|
||||||
|
|
||||||
asyncio.run(run())
|
|
||||||
|
|
||||||
|
|
||||||
"""
|
"""
|
||||||
# Example for upload
|
# Example for upload
|
||||||
uv run scripts/batch_upload.py upload \
|
uv run scripts/batch_upload.py upload \
|
||||||
@ -544,18 +391,6 @@ uv run scripts/batch_upload.py upload \
|
|||||||
--concurrency 4 \
|
--concurrency 4 \
|
||||||
--recursive \
|
--recursive \
|
||||||
--record-file scripts/tmp/batch_processed_files.txt
|
--record-file scripts/tmp/batch_processed_files.txt
|
||||||
|
|
||||||
# Example for trans
|
|
||||||
uv run scripts/batch_upload.py trans \
|
|
||||||
--db-id your_kb_id \
|
|
||||||
--directory path/to/your/data \
|
|
||||||
--output-dir path/to/output_markdown \
|
|
||||||
--pattern "*.docx" \
|
|
||||||
--base-url http://127.0.0.1:5050/api \
|
|
||||||
--username your_username \
|
|
||||||
--password your_password \
|
|
||||||
--concurrency 4 \
|
|
||||||
--recursive
|
|
||||||
"""
|
"""
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
app()
|
app()
|
||||||
|
|||||||
@ -51,8 +51,10 @@ class MinIOClient:
|
|||||||
host_ip = host_ip.split("://")[-1]
|
host_ip = host_ip.split("://")[-1]
|
||||||
host_ip = host_ip.rstrip("/")
|
host_ip = host_ip.rstrip("/")
|
||||||
self.public_endpoint = f"{host_ip}:9000"
|
self.public_endpoint = f"{host_ip}:9000"
|
||||||
|
logger.debug(f"Docker MinIOClient public_endpoint: {self.public_endpoint}")
|
||||||
else:
|
else:
|
||||||
self.public_endpoint = "localhost:9000"
|
self.public_endpoint = "localhost:9000"
|
||||||
|
logger.debug(f"Default_client: {self.public_endpoint}")
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def client(self) -> Minio:
|
def client(self) -> Minio:
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user