102 lines
2.3 KiB
Python
102 lines
2.3 KiB
Python
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import html
|
||
|
|
import re
|
||
|
|
from urllib.parse import urlparse
|
||
|
|
|
||
|
|
_ALLOWED_URL_SCHEMES: frozenset[str] = frozenset({
|
||
|
|
"http",
|
||
|
|
"https",
|
||
|
|
"ftp",
|
||
|
|
"ftps",
|
||
|
|
"mailto",
|
||
|
|
"tel",
|
||
|
|
})
|
||
|
|
|
||
|
|
_DISALLOWED_URL_SCHEMES: frozenset[str] = frozenset({
|
||
|
|
"javascript",
|
||
|
|
"data",
|
||
|
|
"vbscript",
|
||
|
|
"file",
|
||
|
|
"about",
|
||
|
|
"chrome",
|
||
|
|
"resource",
|
||
|
|
})
|
||
|
|
|
||
|
|
_HTML_TAG_RE = re.compile(r"<[^>]+>")
|
||
|
|
|
||
|
|
_HTML_ENTITY_RE = re.compile(r"&(?:#[0-9]+|#x[0-9a-fA-F]+|[a-zA-Z][a-zA-Z0-9]*);")
|
||
|
|
|
||
|
|
|
||
|
|
def strip_html_tags(text: str) -> str:
|
||
|
|
"""Remove all HTML tags from text."""
|
||
|
|
return _HTML_TAG_RE.sub("", text)
|
||
|
|
|
||
|
|
|
||
|
|
def escape_html(text: str) -> str:
|
||
|
|
"""Escape HTML special characters in text."""
|
||
|
|
return html.escape(text, quote=True)
|
||
|
|
|
||
|
|
|
||
|
|
def sanitize_url(url: str) -> str:
|
||
|
|
"""Sanitize a URL by removing dangerous schemes.
|
||
|
|
|
||
|
|
Returns an empty string if the URL uses a disallowed scheme.
|
||
|
|
Returns the original URL if it uses an allowed scheme or has no scheme.
|
||
|
|
"""
|
||
|
|
if not url:
|
||
|
|
return url
|
||
|
|
|
||
|
|
url = url.strip()
|
||
|
|
|
||
|
|
# Check for scheme-less URLs (relative paths)
|
||
|
|
if ":" not in url:
|
||
|
|
return url
|
||
|
|
|
||
|
|
parsed = urlparse(url)
|
||
|
|
scheme = parsed.scheme.lower()
|
||
|
|
|
||
|
|
if not scheme:
|
||
|
|
return url
|
||
|
|
|
||
|
|
if scheme in _DISALLOWED_URL_SCHEMES:
|
||
|
|
return ""
|
||
|
|
|
||
|
|
if scheme not in _ALLOWED_URL_SCHEMES:
|
||
|
|
# Unknown scheme - block by default for safety
|
||
|
|
return ""
|
||
|
|
|
||
|
|
return url
|
||
|
|
|
||
|
|
|
||
|
|
def sanitize_text_content(text: str) -> str:
|
||
|
|
"""Sanitize text content for safe rendering.
|
||
|
|
|
||
|
|
This function:
|
||
|
|
1. Strips HTML tags to prevent HTML injection
|
||
|
|
2. Escapes HTML special characters to prevent entity-based attacks
|
||
|
|
"""
|
||
|
|
if not text:
|
||
|
|
return text
|
||
|
|
|
||
|
|
# First strip any HTML tags
|
||
|
|
text = strip_html_tags(text)
|
||
|
|
|
||
|
|
# Then escape HTML special characters
|
||
|
|
text = escape_html(text)
|
||
|
|
|
||
|
|
return text
|
||
|
|
|
||
|
|
|
||
|
|
def sanitize_text_content_preserve_entities(text: str) -> str:
|
||
|
|
"""Sanitize text content by stripping HTML tags only, preserving entities.
|
||
|
|
|
||
|
|
Use this when the text will be rendered in a context that handles
|
||
|
|
its own escaping (e.g., JSON payloads where HTML entities should remain).
|
||
|
|
"""
|
||
|
|
if not text:
|
||
|
|
return text
|
||
|
|
|
||
|
|
# Only strip HTML tags, don't escape - let the downstream renderer handle escaping
|
||
|
|
return strip_html_tags(text)
|