Coverage for app/venv/lib/python3.14/site-packages/weblate/utils/markdown.py: 56%
66 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
4from __future__ import annotations
6import re
7import threading
8from functools import reduce
10import mistletoe
11from django.db.models import Q
12from django.utils.safestring import mark_safe
13from mistletoe import span_token
15from weblate.auth.models import User
17MENTION_RE = re.compile(r"(?<!\w)(@[\w.@+-]+)\b")
18MARKDOWN_LOCK = threading.Lock()
21def get_mention_users(text):
22 """Return IDs of users mentioned in the text."""
23 matches = MENTION_RE.findall(text)
24 if not matches: 24 ↛ 26line 24 didn't jump to line 26 because the condition on line 24 was always true
25 return User.objects.none()
26 return User.objects.filter(
27 reduce(lambda acc, x: acc | Q(username=x[1:]), matches, Q())
28 )
31class SkipHtmlSpan(span_token.HtmlSpan):
32 """A token that strips HTML tags from the content."""
34 pattern = re.compile(f"{span_token._open_tag}|{span_token._closing_tag}") # noqa: SLF001
35 parse_inner = False
36 content: str
38 def __init__(self, match) -> None:
39 self.content = ""
42class PlainAutoLink(span_token.AutoLink):
43 pattern = re.compile(
44 r"\b(https?://[A-Za-z0-9.!#$%&'*+/=?^_`{|})(~-]+[A-Za-z0-9})])(?=\W|$)"
45 )
48class SaferWeblateHtmlRenderer(mistletoe.HtmlRenderer):
49 """
50 A renderer which adds a layer of protection against malicious input.
52 1. Check if the URL is valid based on scheme and content
53 2. Strip HTML tags from the content.
54 """
56 _allowed_url_re = re.compile(r"^https?://", re.IGNORECASE)
58 def __init__(self, *args, **kwargs) -> None:
59 super().__init__(SkipHtmlSpan, PlainAutoLink, process_html_tokens=False)
61 def render_skip_html_span(self, token: SkipHtmlSpan) -> str:
62 """
63 Render a skip HTML span token.
65 Return the content of the token, without any HTML tags.
66 """
67 return token.content
69 def render_plain_auto_link(self, token: PlainAutoLink) -> str:
70 """
71 Render a skip HTML span token.
73 Return the content of the token, without any HTML tags.
74 """
75 return self.render_auto_link(token)
77 def convert_link(self, link: str) -> str:
78 return link.replace(' href="', ' rel="ugc" target="_blank" href="')
80 def render_link(self, token: span_token.Link) -> str:
81 """
82 Render a link token.
84 If the URL is valid, add the necessary attributes to make it open in a new tab.
85 """
86 if self.check_url(token.target):
87 return self.convert_link(super().render_link(token))
88 return self.escape_html_text(f"[{token.title}]({token.target})")
90 def render_auto_link(self, token: span_token.AutoLink | PlainAutoLink) -> str:
91 """
92 Render an auto link token.
94 If the URL is valid, render the auto link as usual.
95 Otherwise, escape the URL.
96 """
98 def valid_email(email: str) -> bool:
99 """Check if an email address is valid."""
100 pattern = re.compile(
101 r"(mailto:)?[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}"
102 )
103 return bool(pattern.match(email))
105 if self.check_url(token.target) or valid_email(token.target):
106 return self.convert_link(super().render_auto_link(token))
107 return self.escape_html_text(f"<{token.target}>")
109 def render_image(self, token: span_token.Image) -> str:
110 """
111 Render an image token.
113 If the URL is valid, add the necessary attributes to the image tag.
114 Otherwise, escape the URL.
115 """
116 if self.check_url(token.src):
117 return super().render_image(token)
118 return self.escape_html_text(f"")
120 def check_url(self, url: str) -> bool:
121 """Check if an url is valid or not the scheme."""
122 if url.startswith("/user/"):
123 return True
124 return bool(self._allowed_url_re.match(url))
127def render_markdown(text: str) -> str:
128 users = {u.username.lower(): u for u in get_mention_users(text)}
129 parts = MENTION_RE.split(text)
130 for pos, part in enumerate(parts):
131 if not part.startswith("@"): 131 ↛ 133line 131 didn't jump to line 133 because the condition on line 131 was always true
132 continue
133 username = part[1:].lower()
134 if username in users:
135 user = users[username]
136 parts[pos] = (
137 f'**[{part}]({user.get_absolute_url()} "{user.get_visible_name()}")**'
138 )
139 text = "".join(parts)
140 with MARKDOWN_LOCK, SaferWeblateHtmlRenderer() as renderer:
141 return mark_safe(renderer.render(mistletoe.Document(text))) # noqa: S308