1
0
Fork 0
mirror of https://github.com/EDeev/declaude.git synced 2026-10-07 20:49:57 +03:00

Ссылки и картинки: только http(s), mailto и относительные адреса

Ссылка вида [текст](javascript:…) попадала в архив как есть и выполнялась при клике —
опасно, если чат содержит текст из чужого источника. Теперь javascript:, vbscript:, data:
(кроме растровых картинок) заменяются на «#», схема проверяется без учёта регистра и
пробелов.
This commit is contained in:
Egor Deev 2026-10-02 20:38:58 +00:00
parent 56bba831a9
commit 676ca1e926

View file

@ -1,6 +1,7 @@
"""Minimal Markdown → HTML renderer with code highlighting, tables and safe links."""
from __future__ import annotations
import html
import re
from typing import List
@ -9,6 +10,27 @@ from typing import List
# MarkdownRenderer
# ---------------------------------------------------------------------------
_SCHEME = re.compile(r"^([a-z][a-z0-9+.\-]*):")
_LINK_SCHEMES = ("http", "https", "mailto")
def _safe_url(url: str, image: bool = False) -> str:
"""Allow only http(s)/mailto links and relative URLs; javascript:, data: and the like become "#".
The text is already HTML-escaped at this point, so the scheme is checked on the unescaped form
with control characters and spaces removed (browsers ignore them inside the scheme).
"""
probe = re.sub(r"[\x00-\x20]", "", html.unescape(url)).lower()
m = _SCHEME.match(probe)
if m is None:
return url
if m.group(1) in _LINK_SCHEMES and not (image and m.group(1) == "mailto"):
return url
if image and probe.startswith("data:image/") and not probe.startswith("data:image/svg"):
return url
return "#"
class MarkdownRenderer:
"""
Converts markdown text to HTML.
@ -300,10 +322,10 @@ class MarkdownRenderer:
def _process_inline(self, text: str) -> str:
# Images before links
text = re.sub(r'!\[([^\]]*)\]\(([^)]+)\)',
r'<img alt="\1" src="\2" loading="lazy">', text)
lambda m: f'<img alt="{m.group(1)}" src="{_safe_url(m.group(2), image=True)}" loading="lazy">', text)
# Links
text = re.sub(r'\[([^\]]+)\]\(([^)]+)\)',
r'<a href="\2" target="_blank" rel="noopener">\1</a>', text)
lambda m: f'<a href="{_safe_url(m.group(2))}" target="_blank" rel="noopener">{m.group(1)}</a>', text)
# Bold+italic
text = re.sub(r'\*\*\*(.+?)\*\*\*', r'<strong><em>\1</em></strong>', text)
text = re.sub(r'___(.+?)___', r'<strong><em>\1</em></strong>', text)