Coverage for app/venv/lib/python3.14/site-packages/weblate/utils/html.py: 41%

90 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7import re 

8import threading 

9from collections import defaultdict 

10from typing import TYPE_CHECKING, Any 

11 

12import nh3 

13from django.utils.html import format_html, format_html_join 

14from django.utils.translation import pgettext 

15from html2text import HTML2Text as _HTML2Text 

16from lxml.etree import HTMLParser 

17 

18if TYPE_CHECKING: 18 ↛ 19line 18 didn't jump to line 19 because the condition on line 18 was never true

19 from collections.abc import Iterable 

20 

21 from django.utils.safestring import SafeString 

22 from lxml.etree import ParserTarget 

23 

24 from weblate.checks.flags import Flags 

25else: 

26 ParserTarget = object 

27 

28MD_LINK = re.compile( 

29 r""" 

30 (?: 

31 !? # Exclamation for images 

32 \[((?:\[[^^\]]*\]|[^\[\]]|\](?=[^\[]*\]))*)\] # Link text 

33 \( 

34 \s*(<)?([\s\S]*?)(?(2)>) # URL 

35 (?:\s+['"]([\s\S]*?)['"])?\s* # Title 

36 \) 

37 | 

38 <(https?://[^>]+)> # URL 

39 | 

40 <([^>]+@[^>]+\.[^>]+)> # E-mail 

41 ) 

42 """, 

43 re.VERBOSE, 

44) 

45MD_BROKEN_LINK = re.compile(r"\] +\(") 

46MD_REFLINK = re.compile( 

47 r"!?\[(" # leading [ 

48 r"(?:\[[^^\]]*\]|[^\[\]]|\](?=[^\[]*\]))*" # link text 

49 r")\]\s*\[([^^\]]*)\]" # trailing ] with optional target 

50) 

51MD_SYNTAX = re.compile( 

52 r""" 

53 (_{2})(?:[\s\S]+?)_{2}(?!_) # __word__ 

54 | 

55 (\*{2})(?:[\s\S]+?)\*{2}(?!\*) # **word** 

56 | 

57 \b(_)(?:(?:__|[^_])+?)_\b # _word_ 

58 | 

59 (\*)(?:(?:\*\*|[^\*])+?)\*(?!\*) # *word* 

60 | 

61 (`+)\s*(?:[\s\S]*?[^`])\s*\5(?!`) # `code` 

62 | 

63 (~~)(?=\S)(?:[\s\S]*?\S)~~ # ~~word~~ 

64 | 

65 (<)(?:https?://[^>]+)> # URL 

66 | 

67 (<)(?:[^>]+@[^>]+\.[^>]+)> # E-mail 

68 """, 

69 re.VERBOSE, 

70) 

71MD_SYNTAX_GROUPS = 8 

72 

73IGNORE = {"body", "html"} 

74CLEAN_CONTENT_TAGS = {"script", "style"} 

75 

76# Allow some chars: 

77# - non breakable space 

78SANE_CHARS = re.compile(r"[\xa0]") 

79NH3_LOCK = threading.Lock() 

80 

81 

82class MarkupExtractor(ParserTarget): 

83 def __init__(self) -> None: 

84 self.found_tags: set[str] = set() 

85 self.found_attributes: dict[str, set[str]] = defaultdict(set) 

86 

87 def start(self, tag: str, attrs: dict[str, str]) -> None: # type: ignore[override] 

88 if tag in IGNORE: 

89 return 

90 self.found_tags.add(tag) 

91 self.found_attributes[tag].update(attrs.keys()) 

92 

93 def close(self) -> None: 

94 pass 

95 

96 

97def extract_html_tags(text: str) -> tuple[set[str], dict[str, set[str]]]: 

98 """Extract tags from text in a form suitable for HTML sanitization.""" 

99 extractor = MarkupExtractor() 

100 if "<body" not in text.lower(): 

101 # Make sure we are in body, otherwise HTML parser migght halluciate we 

102 # are in <head> 

103 text = f"<body>{text}</body>" 

104 parser = HTMLParser(collect_ids=False, target=extractor) 

105 parser.feed(text) 

106 return (extractor.found_tags, extractor.found_attributes) 

107 

108 

109class HTMLSanitizer: 

110 def __init__(self) -> None: 

111 self.current = 0 

112 self.replacements: dict[str, str] = {} 

113 

114 def clean(self, text: str, source: str, flags: Flags) -> str: 

115 self.current = 0 

116 self.replacements = {} 

117 

118 text = self.remove_special(text, flags) 

119 

120 tags, attributes = extract_html_tags(source) 

121 

122 with NH3_LOCK: 

123 text = nh3.clean( 

124 text, 

125 link_rel=None, 

126 tags=tags, 

127 attributes=attributes, 

128 clean_content_tags=CLEAN_CONTENT_TAGS - tags, 

129 ) 

130 

131 return self.add_back_special(text) 

132 

133 def handle_replace(self, match: re.Match) -> str: 

134 self.current += 1 

135 replacement = f"@@@@@weblate:{self.current}@@@@@" 

136 self.replacements[replacement] = match.group(0) 

137 return replacement 

138 

139 def remove_special(self, text: str, flags: Flags) -> str: 

140 if "md-text" in flags: 

141 text = MD_LINK.sub(self.handle_replace, text) 

142 

143 return SANE_CHARS.sub(self.handle_replace, text) 

144 

145 def add_back_special(self, text: str) -> str: 

146 for replacement, original in self.replacements.items(): 

147 text = text.replace(replacement, original) 

148 return text 

149 

150 

151# Map tags to open and closing text 

152WEBLATE_TAGS = { 

153 # Word diff syntax for text changes 

154 "ins": ("{+", "+}"), 

155 "del": ("[-", "-]"), 

156} 

157 

158 

159class HTML2Text(_HTML2Text): 

160 def __init__(self, bodywidth: int = 78) -> None: 

161 super().__init__(bodywidth=bodywidth) 

162 # Use Unicode characters instead of their ascii pseudo-replacements 

163 self.unicode_snob = True 

164 # Do not include any formatting for images 

165 self.ignore_images = True 

166 # Pad the cells to equal column width in tables 

167 self.pad_tables = True 

168 

169 def handle_tag(self, tag: str, attrs: dict[str, str | None], start: bool) -> None: 

170 # Special handling for certain tags 

171 if tag in WEBLATE_TAGS: 

172 self.o(WEBLATE_TAGS[tag][not start]) 

173 return 

174 super().handle_tag(tag, attrs, start) 

175 

176 

177def mail_quote_char(text: str) -> str | SafeString: 

178 if text in {":", "."}: 

179 return format_html("<span>{}</span>", text) 

180 return text 

181 

182 

183def mail_quote_value(text: str) -> str | SafeString: 

184 """ 

185 Quote value to be used in e-mail notifications. 

186 

187 This tries to avoid automatic conversion to links by Gmail 

188 and similar services. 

189 

190 Solution based on https://stackoverflow.com/a/23404042/225718 

191 """ 

192 return format_html_join( 

193 "", 

194 "{}", 

195 ((mail_quote_char(part),) for part in re.split(r"([.:])", text)), 

196 ) 

197 

198 

199def format_html_join_comma( 

200 format_string: str, args_generator: Iterable[Iterable[Any]] 

201) -> SafeString: 

202 return format_html_join( 

203 pgettext("Joins a list of values", ", "), format_string, args_generator 

204 ) 

205 

206 

207def list_to_tuples(strings: Iterable[Any]) -> Iterable[tuple[Any]]: 

208 """Convert a list of strings into a list of single-element tuples.""" 

209 return ((s,) for s in strings)