diff --git a/declaude/markdown.py b/declaude/markdown.py index 4bd03d4..ec1e0f1 100644 --- a/declaude/markdown.py +++ b/declaude/markdown.py @@ -1,6 +1,7 @@ """Minimal Markdown → HTML renderer with code highlighting, tables and safe links.""" from __future__ import annotations +import html import re from typing import List @@ -9,6 +10,27 @@ from typing import List # MarkdownRenderer # --------------------------------------------------------------------------- +_SCHEME = re.compile(r"^([a-z][a-z0-9+.\-]*):") +_LINK_SCHEMES = ("http", "https", "mailto") + + +def _safe_url(url: str, image: bool = False) -> str: + """Allow only http(s)/mailto links and relative URLs; javascript:, data: and the like become "#". + + The text is already HTML-escaped at this point, so the scheme is checked on the unescaped form + with control characters and spaces removed (browsers ignore them inside the scheme). + """ + probe = re.sub(r"[\x00-\x20]", "", html.unescape(url)).lower() + m = _SCHEME.match(probe) + if m is None: + return url + if m.group(1) in _LINK_SCHEMES and not (image and m.group(1) == "mailto"): + return url + if image and probe.startswith("data:image/") and not probe.startswith("data:image/svg"): + return url + return "#" + + class MarkdownRenderer: """ Converts markdown text to HTML. @@ -300,10 +322,10 @@ class MarkdownRenderer: def _process_inline(self, text: str) -> str: # Images before links text = re.sub(r'!\[([^\]]*)\]\(([^)]+)\)', - r'\1', text) + lambda m: f'{m.group(1)}', text) # Links text = re.sub(r'\[([^\]]+)\]\(([^)]+)\)', - r'\1', text) + lambda m: f'{m.group(1)}', text) # Bold+italic text = re.sub(r'\*\*\*(.+?)\*\*\*', r'\1', text) text = re.sub(r'___(.+?)___', r'\1', text)