from __future__ import annotations import re _ALLOWED_TAGS = {"b", "i", "u", "s", "a", "br", "code", "pre", "blockquote", "img"} _UNSUPPORTED_REPLACEMENTS: dict[str, tuple[str, str]] = { "strong": ("", ""), "em": ("", ""), "strike": ("", ""), "del": ("", ""), "ins": ("", ""), "h1": ("", "\n"), "h2": ("", "\n"), "h3": ("", "\n"), "h4": ("", "\n"), "h5": ("", "\n"), "h6": ("", "\n"), "p": ("\n", ""), "ul": ("", ""), "ol": ("", ""), "li": ("• ", "\n"), "span": ("", ""), "div": ("\n", ""), "hr": ("\n---\n", ""), } _STRIP_TAGS = {"script", "style", "head", "meta", "link", "iframe", "object", "embed"} def render_html_to_imessage(html: str) -> str: if not html: return "" for tag in _STRIP_TAGS: html = re.sub(rf"<{tag}[^>]*>.*?", "", html, flags=re.DOTALL | re.IGNORECASE) html = re.sub(rf"<{tag}[^>]*/>", "", html, flags=re.IGNORECASE) for tag, (open_tag, close_tag) in _UNSUPPORTED_REPLACEMENTS.items(): html = re.sub(rf"<{tag}(\s[^>]*)?>", open_tag, html, flags=re.IGNORECASE) html = re.sub(rf"", close_tag, html, flags=re.IGNORECASE) html = _strip_unsafe_attributes(html) html = _normalize_links(html) html = html.replace("\r\n", "\n").replace("\r", "\n") return html.strip() def _strip_unsafe_attributes(html: str) -> str: def _clean_tag(tag_match: re.Match) -> str: full_tag = tag_match.group(0) tag_name_match = re.match(r"]+>", _clean_tag, html) def _normalize_links(text: str) -> str: url_pattern = re.compile(r"(?\"\)]+)") return url_pattern.sub(r'\1', text) def render_markdown_to_imessage(text: str) -> str: if not text: return "" codes: dict[str, str] = {} def _save_code(m: re.Match) -> str: key = f"\x00CODE{len(codes)}\x00" codes[key] = m.group(0) return key text = re.sub(r"`[^`]+`", _save_code, text) tokens = _tokenize_markdown(text) result = _render_tokens(tokens) for key, val in codes.items(): result = result.replace(key, val) return result def _tokenize_markdown(text: str) -> list[tuple[str, str]]: TOKEN_RE = re.compile(r"(\*\*|__|(? pos: tokens.append(("text", text[pos : m.start()])) tokens.append(("marker", m.group())) pos = m.end() if pos < len(text): tokens.append(("text", text[pos:])) return tokens def _render_tokens(tokens: list[tuple[str, str]]) -> str: stack: list[tuple[str, str]] = [] result: list[str] = [] MARKER_MAP = { "**": ("", ""), "__": ("", ""), "*": ("", ""), "~": ("", ""), "~~": ("", ""), } for kind, value in tokens: if kind == "text": result.append(value) elif kind == "marker": if value in MARKER_MAP: if stack and stack[-1][0] == value: stack.pop() result.append(MARKER_MAP[value][1]) else: stack.append((value, MARKER_MAP[value][0])) result.append(MARKER_MAP[value][0]) else: result.append(value) for marker, _ in reversed(stack): mapped = MARKER_MAP.get(marker) if mapped: result.append(mapped[1]) return "".join(result)