ForcePilot/backend/package/yuxi/channel/extensions/twitter/format.py

63 lines
1.7 KiB
Python
Raw Normal View History

from __future__ import annotations
import re
TEXT_CHUNK_LIMIT = 10000
_MD_IMAGE_RE = re.compile(r"!\[.*?\]\([^)]+\)")
_MD_LINK_RE = re.compile(r"\[([^\]]+)\]\(([^)]+)\)")
_MD_BOLD_RE = re.compile(r"\*\*(.+?)\*\*")
_MD_ITALIC_RE = re.compile(r"(?<!\w)[*_](.+?)[*_](?!\w)")
_MD_CODE_RE = re.compile(r"`([^`]+)`")
_MD_CODE_BLOCK_RE = re.compile(r"```[\s\S]*?```")
def markdown_to_plain_text(md_text: str) -> str:
text = md_text
text = _MD_IMAGE_RE.sub("[图片]", text)
text = _MD_LINK_RE.sub(r"\1 (\2)", text)
text = _MD_BOLD_RE.sub(r"\1", text)
text = _MD_ITALIC_RE.sub(r"\1", text)
text = _MD_CODE_BLOCK_RE.sub("", text)
text = _MD_CODE_RE.sub(r"\1", text)
text = text.replace("\r\n", "\n").replace("\r", "\n")
text = re.sub(r"\n{3,}", "\n\n", text)
return text.strip()
def split_text_chunks(text: str, limit: int = TEXT_CHUNK_LIMIT) -> list[str]:
if len(text) <= limit:
return [text]
chunks: list[str] = []
paragraphs = text.split("\n\n")
current = ""
for para in paragraphs:
if len(current) + len(para) + 2 <= limit:
current = (current + "\n\n" + para) if current else para
else:
if current:
chunks.append(current)
if len(para) > limit:
for i in range(0, len(para), limit):
chunks.append(para[i : i + limit])
current = ""
else:
current = para
if current:
chunks.append(current)
return chunks if chunks else [text]
def markdown_to_native(md_text: str) -> str:
return markdown_to_plain_text(md_text)
def native_to_markdown(native_content: str) -> str:
return native_content