ForcePilot/backend/test/unit/channels/test_text_chunker_smoke.py
Kris 3264900bc9 test: 新增多渠道单元测试用例并配置测试环境变量
新增了Twitch、Telegram、Discord、Slack、Mattermost、WeChat、Zalo等多渠道的单元测试用例,覆盖了令牌处理、速率限制、消息去重、会话解析、格式转换、安全策略等模块
同时在测试配置中添加了测试用的OpenAI API密钥环境变量
2026-05-12 00:56:47 +08:00

76 lines
2.6 KiB
Python

"""Smoke tests for TextChunker"""
from yuxi.channels.infra.text_chunker import TextChunker, ChunkMode
def test_markdown_code_block_intact_when_chunked():
tc = TextChunker(max_length=30, mode=ChunkMode.MARKDOWN)
md_text = (
"Short intro\n"
"```python\n"
'print("hello world this is long")\n'
'print("another long line here too")\n'
"```\n"
"After code block."
)
result = tc.chunk(md_text)
print("MD code block (chunked):")
for i, c in enumerate(result):
print(f" Chunk {i}: {repr(c)}")
# Code block should not be split mid-block
for c in result:
if c.startswith("```"):
assert "```" in c.rsplit("\n", 1)[-1] or c.count("```") >= 2, (
f"Code block should not be split mid-block: {repr(c)}"
)
def test_markdown_table_intact():
tc = TextChunker(max_length=200, mode=ChunkMode.MARKDOWN)
table_text = "Header\n| col1 | col2 |\n|------|------|\n| a | b |\n| c | d |\n\nAfter table."
result = tc.chunk(table_text)
print("\nMD (table intact):")
for i, c in enumerate(result):
print(f" Chunk {i}: {repr(c)}")
assert any("|------|------|" in c for c in result), "Table should be in a chunk"
def test_markdown_headings_separate_when_over_limit():
tc = TextChunker(max_length=30, mode=ChunkMode.MARKDOWN)
heading_text = "## Hello\nSome content under heading.\n\n## World\nDifferent heading text here."
result = tc.chunk(heading_text)
print("\nMD (headings, over limit):")
for i, c in enumerate(result):
print(f" Chunk {i}: {repr(c)}")
assert len(result) >= 2, "Over-limit headings should separate content"
def test_empty_text():
tc = TextChunker(max_length=200, mode=ChunkMode.NEWLINE)
assert tc.chunk("") == []
assert tc.chunk_as_objects("") == []
def test_single_chunk():
tc = TextChunker(max_length=200, mode=ChunkMode.LENGTH)
result = tc.chunk("short text")
assert result == ["short text"]
def test_chunk_with_index():
tc = TextChunker(max_length=15, mode=ChunkMode.LENGTH)
results = list(tc.chunk_with_index("Part one here. Part two also here. Part three goes on."))
assert len(results) > 1
assert results[0] == (1, len(results), results[0][2])
assert results[-1] == (len(results), len(results), results[-1][2])
if __name__ == "__main__":
test_markdown_code_block_intact_when_chunked()
test_markdown_table_intact()
test_markdown_headings_separate_when_over_limit()
test_empty_text()
test_single_chunk()
test_chunk_with_index()
print("\nALL SMOKE TESTS PASSED")