Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/fluent/utils.py: 26%

110 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Henry Wilkes <henry@torproject.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7import re 

8from typing import TYPE_CHECKING 

9 

10from django.utils.html import escape, format_html, format_html_join 

11from django.utils.safestring import mark_safe 

12from translate.storage.fluent import ( 

13 FluentUnit, 

14) 

15 

16if TYPE_CHECKING: 16 ↛ 17line 16 didn't jump to line 17 because the condition on line 16 was never true

17 from collections.abc import Iterable, Iterator 

18 

19 from django.utils.safestring import SafeString 

20 from translate.storage.fluent import ( 

21 FluentPart, 

22 FluentReference, 

23 FluentSelectorBranch, 

24 ) 

25 

26 from weblate.checks.models import Check as CheckModel 

27 from weblate.trans.models.unit import Unit as TransUnitModel 

28 

29HighlightsType = list[tuple[int, int, str]] 

30 

31 

32def translation_from_check( 

33 check_model: CheckModel, 

34) -> tuple[TransUnitModel, str, str]: 

35 """Extract a translation unit, source and target from a Check Model.""" 

36 unit = check_model.unit 

37 # Do not currently support plurals. 

38 return (unit, unit.get_source_plurals()[0], unit.get_target_plurals()[0]) 

39 

40 

41def format_html_code( 

42 format_string: str, 

43 **kwargs: str, 

44) -> SafeString: 

45 """Wrap each keyword argument in a <code> tag.""" 

46 safe_kwargs = { 

47 key: format_html("<code>{value}</code>", value=value) 

48 for key, value in kwargs.items() 

49 } 

50 if safe_kwargs: 

51 return format_html(escape(format_string), **safe_kwargs) 

52 return mark_safe(escape(format_string)) # noqa: S308 

53 

54 

55def format_html_error_list(errors: Iterable[str]) -> SafeString: 

56 """Return a HTML SafeString with each given error on a new line.""" 

57 return format_html_join( 

58 mark_safe("<br />"), 

59 "{}", 

60 ((err,) for err in errors), 

61 ) 

62 

63 

64def variant_name(branches: list[FluentSelectorBranch]) -> str: 

65 """Get a variant name for the given branch path.""" 

66 if not branches: 

67 return "" 

68 return "[" + "][".join(branch.key for branch in branches) + "]" 

69 

70 

71class FluentPatterns: 

72 """Patterns from fluent EBNF.""" 

73 

74 BLANK = r"( |\n|\r\n)*" 

75 # Match string or number literals. 

76 ESCAPED_CHAR = r'(\\\\|\\"|\\u[0-9a-fA-F]{4}|\\U[0-9a-fA-F]{6})' 

77 STRING_LITERAL = r'"(' + ESCAPED_CHAR + r'|[^"\\])*"' 

78 NUMBER_LITERAL = r"-?[0-9]+(\.[0-9]+)?" 

79 IDENTIFIER = r"[a-zA-Z][a-zA-Z0-9_-]*" 

80 NAMED_ARGUMENT = ( 

81 IDENTIFIER 

82 + BLANK 

83 + r":" 

84 + BLANK 

85 + r"(" 

86 + STRING_LITERAL 

87 + r"|" 

88 + NUMBER_LITERAL 

89 + r")" 

90 ) 

91 NAMED_ARGUMENT_LIST = ( 

92 r"(" 

93 + NAMED_ARGUMENT 

94 + BLANK 

95 + r"," 

96 + BLANK 

97 + r")*" 

98 + r"(" 

99 + NAMED_ARGUMENT 

100 + r")?" 

101 ) 

102 NAMED_ARGUMENTS_CALL = BLANK + r"\(" + BLANK + NAMED_ARGUMENT_LIST + BLANK + r"\)" 

103 

104 @classmethod 

105 def placeable(cls, expression: str) -> str: 

106 """Wrap a fluent expression in placeable.""" 

107 return r"\{" + cls.BLANK + expression + cls.BLANK + r"\}" 

108 

109 @classmethod 

110 def reference(cls, ref: FluentReference) -> str: 

111 """Return a placeable pattern for the given reference.""" 

112 # NOTE: Technically a reference may appear as direct function argument 

113 # rather than placeables (with the curly braces). However, we don't 

114 # expect such references to be common, and they can always include the 

115 # surrounding braces within the function to achieve the same result. 

116 # This is most likely to come up for variables. E.g. NUMBER($num). 

117 # However, it becomes much more difficult to match this expression 

118 # for highlighting. 

119 if ref.type_name == "message": 

120 # Message and message attribute refs do not accept parameters. 

121 return cls.placeable(re.escape(ref.name)) 

122 if ref.type_name == "term": 

123 # Term references can accept parameters. 

124 # NOTE: Technically, positional parameters are allowed by the fluent 

125 # syntax, which can generally be quite complex expressions. However, 

126 # a Term would not be able to access this positional information. 

127 # Instead, we just build the regex to match named arguments, which 

128 # can only be string or number literals. 

129 return cls.placeable( 

130 r"-" + re.escape(ref.name) + r"(" + cls.NAMED_ARGUMENTS_CALL + ")?" 

131 ) 

132 if ref.type_name == "variable": 

133 return cls.placeable(r"\$" + re.escape(ref.name)) 

134 return "" 

135 

136 # Match a completed string or number placeable. 

137 LITERAL_PLACEABLE_REGEX = re.compile( 

138 # One or more "{". This should capture cases where the literal is 

139 # double-wrapped. E.f. { { "literal" } } 

140 r"(\{" 

141 + BLANK 

142 + r")+" 

143 + r"((?P<string>" 

144 + STRING_LITERAL 

145 + ")|(?P<number>" 

146 + NUMBER_LITERAL 

147 + r"))" 

148 # One of more "}". 

149 # NOTE: In order to be valid Fluent syntax, this should match 

150 # the same number of opening brackets. We assume the caller is 

151 # working on such a source with valid Fluent syntax. 

152 + r"(" 

153 + BLANK 

154 + r"\})+" 

155 ) 

156 

157 ESCAPED_CHAR_REGEX = re.compile(ESCAPED_CHAR) 

158 

159 @classmethod 

160 def split_literal_expressions( 

161 cls, source: str 

162 ) -> Iterator[tuple[int, str, str | None]]: 

163 """ 

164 Remove all Fluent literal expressions from the given source. 

165 

166 Returns the non-literal parts of the source as a list of 3-tuples of the 

167 part's starting offset in the source, its text, and the literal content 

168 stripped of any surrounding placeholder brackets. The literal content 

169 will be None for the final entry. 

170 """ 

171 pos = 0 

172 for literal_match in cls.LITERAL_PLACEABLE_REGEX.finditer(source): 

173 string_literal = literal_match.group("string") 

174 if string_literal is None: 

175 literal = literal_match.group("number") 

176 else: 

177 literal = "" 

178 for pos, chars in enumerate( 

179 # Remove outer quotes from string literal. 

180 cls.ESCAPED_CHAR_REGEX.split(string_literal[1:-1]) 

181 ): 

182 if not pos % 2: 

183 # Not an escaped character. 

184 literal += chars 

185 elif chars in {'\\"', "\\\\"}: 

186 # Unescape the character by removing the "\". 

187 literal += chars[1:] 

188 else: 

189 # Remove the leading "\u" or "\U" and convert hex 

190 # sequence to a number. 

191 unicode_point = int(chars[2:], 16) # noqa: FURB166 

192 try: 

193 # Try unescape the unicode sequence. 

194 literal += chr(unicode_point) 

195 except ValueError: 

196 # Number was too big (and not a valid unicode 

197 # character). 

198 # Just include the escaped sequence as it was. 

199 literal += chars 

200 yield ( 

201 pos, 

202 source[pos : literal_match.start()], 

203 literal, 

204 ) 

205 pos = literal_match.end() 

206 yield (pos, source[pos:], None) 

207 

208 @classmethod 

209 def highlight_source( 

210 cls, source: str, highlight_patterns: Iterable[str] 

211 ) -> HighlightsType: 

212 """ 

213 Generate a list of highlights for the given source. 

214 

215 Highlights all matches for the patterns in highlight_patterns, except 

216 within a literal expression. Returns the highlights as a list of 

217 3-tuples of the highlighted regions' starting positions, ending 

218 positions and text. 

219 """ 

220 unique_highlights = {p for p in highlight_patterns if p} 

221 if not unique_highlights: 

222 return [] 

223 regex = re.compile("|".join(unique_highlights), flags=re.MULTILINE) 

224 highlights: HighlightsType = [] 

225 for start, text, _ in cls.split_literal_expressions(source): 

226 # NOTE: text may be empty if two literals touch. 

227 highlights.extend( 

228 (start + match.start(), start + match.end(), match.group()) 

229 for match in regex.finditer(text) 

230 ) 

231 return highlights 

232 

233 

234class FluentUnitConverter: 

235 """Convert a translation unit into a FluentUnit.""" 

236 

237 def __init__(self, unit: TransUnitModel, source: str) -> None: 

238 self.unit = unit 

239 self.source = source 

240 

241 def fluent_type(self) -> str: 

242 """Get the fluent type of the given translation unit.""" 

243 flags = self.unit.all_flags 

244 if flags.has_value("fluent-type"): 

245 return flags.get_value("fluent-type") 

246 # Guess based on id. 

247 if self.unit.context and self.unit.context.startswith("-"): 

248 return "Term" 

249 return "Message" 

250 

251 def to_fluent_unit(self) -> FluentUnit | None: 

252 """Convert the given translation unit into a FluentUnit.""" 

253 if not self.source: 

254 return None 

255 fluent_type = self.fluent_type() 

256 try: 

257 return FluentUnit( 

258 source=self.source, unit_id=self.unit.context, fluent_type=fluent_type 

259 ) 

260 except ValueError: 

261 # Unexpected error. E.g. from invalid id. 

262 # We return a default Message unit instead. 

263 return FluentUnit(source=self.source) 

264 

265 def to_fluent_parts(self) -> list[FluentPart] | None: 

266 """Convert the given translation unit into fluent parts.""" 

267 unit = self.to_fluent_unit() 

268 if unit is None: 

269 return None 

270 return unit.get_parts() 

271 

272 def get_syntax_error(self) -> str | None: 

273 """Get the syntax error that would be produced for the unit.""" 

274 unit = self.to_fluent_unit() 

275 if unit is None: 

276 return None 

277 return unit.get_syntax_error()