Coverage for app/venv/lib/python3.14/site-packages/weblate/utils/html.py: 41%
90 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5from __future__ import annotations
7import re
8import threading
9from collections import defaultdict
10from typing import TYPE_CHECKING, Any
12import nh3
13from django.utils.html import format_html, format_html_join
14from django.utils.translation import pgettext
15from html2text import HTML2Text as _HTML2Text
16from lxml.etree import HTMLParser
18if TYPE_CHECKING: 18 ↛ 19line 18 didn't jump to line 19 because the condition on line 18 was never true
19 from collections.abc import Iterable
21 from django.utils.safestring import SafeString
22 from lxml.etree import ParserTarget
24 from weblate.checks.flags import Flags
25else:
26 ParserTarget = object
28MD_LINK = re.compile(
29 r"""
30 (?:
31 !? # Exclamation for images
32 \[((?:\[[^^\]]*\]|[^\[\]]|\](?=[^\[]*\]))*)\] # Link text
33 \(
34 \s*(<)?([\s\S]*?)(?(2)>) # URL
35 (?:\s+['"]([\s\S]*?)['"])?\s* # Title
36 \)
37 |
38 <(https?://[^>]+)> # URL
39 |
40 <([^>]+@[^>]+\.[^>]+)> # E-mail
41 )
42 """,
43 re.VERBOSE,
44)
45MD_BROKEN_LINK = re.compile(r"\] +\(")
46MD_REFLINK = re.compile(
47 r"!?\[(" # leading [
48 r"(?:\[[^^\]]*\]|[^\[\]]|\](?=[^\[]*\]))*" # link text
49 r")\]\s*\[([^^\]]*)\]" # trailing ] with optional target
50)
51MD_SYNTAX = re.compile(
52 r"""
53 (_{2})(?:[\s\S]+?)_{2}(?!_) # __word__
54 |
55 (\*{2})(?:[\s\S]+?)\*{2}(?!\*) # **word**
56 |
57 \b(_)(?:(?:__|[^_])+?)_\b # _word_
58 |
59 (\*)(?:(?:\*\*|[^\*])+?)\*(?!\*) # *word*
60 |
61 (`+)\s*(?:[\s\S]*?[^`])\s*\5(?!`) # `code`
62 |
63 (~~)(?=\S)(?:[\s\S]*?\S)~~ # ~~word~~
64 |
65 (<)(?:https?://[^>]+)> # URL
66 |
67 (<)(?:[^>]+@[^>]+\.[^>]+)> # E-mail
68 """,
69 re.VERBOSE,
70)
71MD_SYNTAX_GROUPS = 8
73IGNORE = {"body", "html"}
74CLEAN_CONTENT_TAGS = {"script", "style"}
76# Allow some chars:
77# - non breakable space
78SANE_CHARS = re.compile(r"[\xa0]")
79NH3_LOCK = threading.Lock()
82class MarkupExtractor(ParserTarget):
83 def __init__(self) -> None:
84 self.found_tags: set[str] = set()
85 self.found_attributes: dict[str, set[str]] = defaultdict(set)
87 def start(self, tag: str, attrs: dict[str, str]) -> None: # type: ignore[override]
88 if tag in IGNORE:
89 return
90 self.found_tags.add(tag)
91 self.found_attributes[tag].update(attrs.keys())
93 def close(self) -> None:
94 pass
97def extract_html_tags(text: str) -> tuple[set[str], dict[str, set[str]]]:
98 """Extract tags from text in a form suitable for HTML sanitization."""
99 extractor = MarkupExtractor()
100 if "<body" not in text.lower():
101 # Make sure we are in body, otherwise HTML parser migght halluciate we
102 # are in <head>
103 text = f"<body>{text}</body>"
104 parser = HTMLParser(collect_ids=False, target=extractor)
105 parser.feed(text)
106 return (extractor.found_tags, extractor.found_attributes)
109class HTMLSanitizer:
110 def __init__(self) -> None:
111 self.current = 0
112 self.replacements: dict[str, str] = {}
114 def clean(self, text: str, source: str, flags: Flags) -> str:
115 self.current = 0
116 self.replacements = {}
118 text = self.remove_special(text, flags)
120 tags, attributes = extract_html_tags(source)
122 with NH3_LOCK:
123 text = nh3.clean(
124 text,
125 link_rel=None,
126 tags=tags,
127 attributes=attributes,
128 clean_content_tags=CLEAN_CONTENT_TAGS - tags,
129 )
131 return self.add_back_special(text)
133 def handle_replace(self, match: re.Match) -> str:
134 self.current += 1
135 replacement = f"@@@@@weblate:{self.current}@@@@@"
136 self.replacements[replacement] = match.group(0)
137 return replacement
139 def remove_special(self, text: str, flags: Flags) -> str:
140 if "md-text" in flags:
141 text = MD_LINK.sub(self.handle_replace, text)
143 return SANE_CHARS.sub(self.handle_replace, text)
145 def add_back_special(self, text: str) -> str:
146 for replacement, original in self.replacements.items():
147 text = text.replace(replacement, original)
148 return text
151# Map tags to open and closing text
152WEBLATE_TAGS = {
153 # Word diff syntax for text changes
154 "ins": ("{+", "+}"),
155 "del": ("[-", "-]"),
156}
159class HTML2Text(_HTML2Text):
160 def __init__(self, bodywidth: int = 78) -> None:
161 super().__init__(bodywidth=bodywidth)
162 # Use Unicode characters instead of their ascii pseudo-replacements
163 self.unicode_snob = True
164 # Do not include any formatting for images
165 self.ignore_images = True
166 # Pad the cells to equal column width in tables
167 self.pad_tables = True
169 def handle_tag(self, tag: str, attrs: dict[str, str | None], start: bool) -> None:
170 # Special handling for certain tags
171 if tag in WEBLATE_TAGS:
172 self.o(WEBLATE_TAGS[tag][not start])
173 return
174 super().handle_tag(tag, attrs, start)
177def mail_quote_char(text: str) -> str | SafeString:
178 if text in {":", "."}:
179 return format_html("<span>{}</span>", text)
180 return text
183def mail_quote_value(text: str) -> str | SafeString:
184 """
185 Quote value to be used in e-mail notifications.
187 This tries to avoid automatic conversion to links by Gmail
188 and similar services.
190 Solution based on https://stackoverflow.com/a/23404042/225718
191 """
192 return format_html_join(
193 "",
194 "{}",
195 ((mail_quote_char(part),) for part in re.split(r"([.:])", text)),
196 )
199def format_html_join_comma(
200 format_string: str, args_generator: Iterable[Iterable[Any]]
201) -> SafeString:
202 return format_html_join(
203 pgettext("Joins a list of values", ", "), format_string, args_generator
204 )
207def list_to_tuples(strings: Iterable[Any]) -> Iterable[tuple[Any]]:
208 """Convert a list of strings into a list of single-element tuples."""
209 return ((s,) for s in strings)