ForcePilot/backend/package/yuxi/channels/adapters/telegram/format.py
Kris 71fec609bb feat(telegram-adapter): 实现完整的Telegram适配器基础代码
新增了Telegram适配器的全套基础模块,包括:
1.  核心适配器入口与会话工具
2.  账号管理、认证与配置系统
3.  连接相关的轮询、Webhook、更新偏移管理
4.  话题路由、管理与缓存系统
5.  消息反抖动、超时配置与工具类
6.  响应式UI与命令交互系统
7.  反应表情与通知系统
8.  审批与安全审计模块
9.  健康检查与状态监控
10. 贴纸缓存与视觉工具
11. 流式响应与协作功能
12. 群组迁移与目标归一化处理
2026-05-12 00:49:52 +08:00

111 lines
3.3 KiB
Python

from __future__ import annotations
import re
from html import escape
_ESCAPE_CHARS = {"&": "&amp;", "<": "&lt;", ">": "&gt;"}
_TAG_PATTERN = r"<(?!/?(?:b|strong|i|em|u|ins|s|strike|del|code|pre|a|tg-spoiler|tg-emoji)\b)[^>]*>"
_TAG_PLACEHOLDER_PATTERN = re.compile(_TAG_PATTERN)
_CODE_BLOCK_RE = re.compile(r"```(?:\w+)?\n.+?```", re.DOTALL)
_INLINE_CODE_RE = re.compile(r"`([^`\n]+?)`")
_PLACEHOLDER_PREFIX = "\x00MDPLACEHOLDER"
def markdown_to_html(text: str) -> str:
code_blocks: dict[str, str] = {}
text = _protect_code_spans(text, _CODE_BLOCK_RE, code_blocks)
text = _protect_code_spans(text, _INLINE_CODE_RE, code_blocks)
text = _convert_bold(text)
text = _convert_italic(text)
text = _convert_strikethrough(text)
text = _convert_underline(text)
text = _convert_links(text)
text = _convert_spoiler(text)
text = _restore_code_spans(text, code_blocks)
text = _sanitize_html(text)
return text
def strip_all_tags(text: str) -> str:
return re.sub(r"<[^>]+>", "", text)
def strip_html_tags(text: str) -> str:
return strip_all_tags(text)
def _protect_code_spans(text: str, pattern: re.Pattern, storage: dict[str, str]) -> str:
result = []
idx = 0
for m in pattern.finditer(text):
result.append(text[idx : m.start()])
placeholder = f"{_PLACEHOLDER_PREFIX}{len(storage)}"
storage[placeholder] = m.group(0)
result.append(placeholder)
idx = m.end()
result.append(text[idx:])
return "".join(result)
def _restore_code_spans(text: str, storage: dict[str, str]) -> str:
result = text
for placeholder, code_text in storage.items():
if placeholder.startswith(f"{_PLACEHOLDER_PREFIX}") and "```" in code_text:
inner = code_text[3:].rstrip("`").lstrip("\n")
lang_end = inner.find("\n")
inner = inner[lang_end + 1 :] if lang_end >= 0 else inner
safe = escape(inner, quote=False)
result = result.replace(placeholder, f"<pre>{safe}</pre>")
elif placeholder in result:
inner = code_text.strip("`")
safe = escape(inner, quote=False)
result = result.replace(placeholder, f"<code>{safe}</code>")
return result
def _convert_bold(text: str) -> str:
return re.sub(r"\*\*(.+?)\*\*", r"<b>\1</b>", text)
def _convert_italic(text: str) -> str:
text = re.sub(r"(?<!\*)\*([^*\n]+?)\*(?!\*)", r"<i>\1</i>", text)
text = re.sub(r"(?<!_)_([^_\n]+?)_(?!_)", r"<i>\1</i>", text)
return text
def _convert_strikethrough(text: str) -> str:
return re.sub(r"~~(.+?)~~", r"<s>\1</s>", text)
def _convert_underline(text: str) -> str:
return re.sub(r"__(\w.*?\w)__", r"<u>\1</u>", text)
def _convert_links(text: str) -> str:
return re.sub(r"\[(.+?)\]\((.+?)\)", r'<a href="\2">\1</a>', text)
def _convert_spoiler(text: str) -> str:
return re.sub(r"\|\|(.+?)\|\|", r"<tg-spoiler>\1</tg-spoiler>", text)
def _sanitize_html(text: str) -> str:
result = []
depth = 0
for ch in text:
if ch == "<":
depth += 1
elif ch == ">":
depth = max(0, depth - 1)
elif depth == 0:
if ch in _ESCAPE_CHARS:
result.append(_ESCAPE_CHARS[ch])
continue
result.append(ch)
return "".join(result)