mirror of
https://github.com/EDeev/declaude.git
synced 2026-10-08 04:59:30 +03:00
Ссылки и картинки: только http(s), mailto и относительные адреса
Ссылка вида [текст](javascript:…) попадала в архив как есть и выполнялась при клике — опасно, если чат содержит текст из чужого источника. Теперь javascript:, vbscript:, data: (кроме растровых картинок) заменяются на «#», схема проверяется без учёта регистра и пробелов.
This commit is contained in:
parent
56bba831a9
commit
676ca1e926
1 changed files with 24 additions and 2 deletions
|
|
@ -1,6 +1,7 @@
|
||||||
"""Minimal Markdown → HTML renderer with code highlighting, tables and safe links."""
|
"""Minimal Markdown → HTML renderer with code highlighting, tables and safe links."""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import html
|
||||||
import re
|
import re
|
||||||
from typing import List
|
from typing import List
|
||||||
|
|
||||||
|
|
@ -9,6 +10,27 @@ from typing import List
|
||||||
# MarkdownRenderer
|
# MarkdownRenderer
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
_SCHEME = re.compile(r"^([a-z][a-z0-9+.\-]*):")
|
||||||
|
_LINK_SCHEMES = ("http", "https", "mailto")
|
||||||
|
|
||||||
|
|
||||||
|
def _safe_url(url: str, image: bool = False) -> str:
|
||||||
|
"""Allow only http(s)/mailto links and relative URLs; javascript:, data: and the like become "#".
|
||||||
|
|
||||||
|
The text is already HTML-escaped at this point, so the scheme is checked on the unescaped form
|
||||||
|
with control characters and spaces removed (browsers ignore them inside the scheme).
|
||||||
|
"""
|
||||||
|
probe = re.sub(r"[\x00-\x20]", "", html.unescape(url)).lower()
|
||||||
|
m = _SCHEME.match(probe)
|
||||||
|
if m is None:
|
||||||
|
return url
|
||||||
|
if m.group(1) in _LINK_SCHEMES and not (image and m.group(1) == "mailto"):
|
||||||
|
return url
|
||||||
|
if image and probe.startswith("data:image/") and not probe.startswith("data:image/svg"):
|
||||||
|
return url
|
||||||
|
return "#"
|
||||||
|
|
||||||
|
|
||||||
class MarkdownRenderer:
|
class MarkdownRenderer:
|
||||||
"""
|
"""
|
||||||
Converts markdown text to HTML.
|
Converts markdown text to HTML.
|
||||||
|
|
@ -300,10 +322,10 @@ class MarkdownRenderer:
|
||||||
def _process_inline(self, text: str) -> str:
|
def _process_inline(self, text: str) -> str:
|
||||||
# Images before links
|
# Images before links
|
||||||
text = re.sub(r'!\[([^\]]*)\]\(([^)]+)\)',
|
text = re.sub(r'!\[([^\]]*)\]\(([^)]+)\)',
|
||||||
r'<img alt="\1" src="\2" loading="lazy">', text)
|
lambda m: f'<img alt="{m.group(1)}" src="{_safe_url(m.group(2), image=True)}" loading="lazy">', text)
|
||||||
# Links
|
# Links
|
||||||
text = re.sub(r'\[([^\]]+)\]\(([^)]+)\)',
|
text = re.sub(r'\[([^\]]+)\]\(([^)]+)\)',
|
||||||
r'<a href="\2" target="_blank" rel="noopener">\1</a>', text)
|
lambda m: f'<a href="{_safe_url(m.group(2))}" target="_blank" rel="noopener">{m.group(1)}</a>', text)
|
||||||
# Bold+italic
|
# Bold+italic
|
||||||
text = re.sub(r'\*\*\*(.+?)\*\*\*', r'<strong><em>\1</em></strong>', text)
|
text = re.sub(r'\*\*\*(.+?)\*\*\*', r'<strong><em>\1</em></strong>', text)
|
||||||
text = re.sub(r'___(.+?)___', r'<strong><em>\1</em></strong>', text)
|
text = re.sub(r'___(.+?)___', r'<strong><em>\1</em></strong>', text)
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue