from __future__ import annotations import re TEXT_CHUNK_LIMIT = 10000 _MD_IMAGE_RE = re.compile(r"!\[.*?\]\([^)]+\)") _MD_LINK_RE = re.compile(r"\[([^\]]+)\]\(([^)]+)\)") _MD_BOLD_RE = re.compile(r"\*\*(.+?)\*\*") _MD_ITALIC_RE = re.compile(r"(? str: text = md_text text = _MD_IMAGE_RE.sub("[图片]", text) text = _MD_LINK_RE.sub(r"\1 (\2)", text) text = _MD_BOLD_RE.sub(r"\1", text) text = _MD_ITALIC_RE.sub(r"\1", text) text = _MD_CODE_BLOCK_RE.sub("", text) text = _MD_CODE_RE.sub(r"\1", text) text = text.replace("\r\n", "\n").replace("\r", "\n") text = re.sub(r"\n{3,}", "\n\n", text) return text.strip() def split_text_chunks(text: str, limit: int = TEXT_CHUNK_LIMIT) -> list[str]: if len(text) <= limit: return [text] chunks: list[str] = [] paragraphs = text.split("\n\n") current = "" for para in paragraphs: if len(current) + len(para) + 2 <= limit: current = (current + "\n\n" + para) if current else para else: if current: chunks.append(current) if len(para) > limit: for i in range(0, len(para), limit): chunks.append(para[i : i + limit]) current = "" else: current = para if current: chunks.append(current) return chunks if chunks else [text] def markdown_to_native(md_text: str) -> str: return markdown_to_plain_text(md_text) def native_to_markdown(native_content: str) -> str: return native_content