Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/fluent/utils.py: 26%
110 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Henry Wilkes <henry@torproject.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5from __future__ import annotations
7import re
8from typing import TYPE_CHECKING
10from django.utils.html import escape, format_html, format_html_join
11from django.utils.safestring import mark_safe
12from translate.storage.fluent import (
13 FluentUnit,
14)
16if TYPE_CHECKING: 16 ↛ 17line 16 didn't jump to line 17 because the condition on line 16 was never true
17 from collections.abc import Iterable, Iterator
19 from django.utils.safestring import SafeString
20 from translate.storage.fluent import (
21 FluentPart,
22 FluentReference,
23 FluentSelectorBranch,
24 )
26 from weblate.checks.models import Check as CheckModel
27 from weblate.trans.models.unit import Unit as TransUnitModel
29HighlightsType = list[tuple[int, int, str]]
32def translation_from_check(
33 check_model: CheckModel,
34) -> tuple[TransUnitModel, str, str]:
35 """Extract a translation unit, source and target from a Check Model."""
36 unit = check_model.unit
37 # Do not currently support plurals.
38 return (unit, unit.get_source_plurals()[0], unit.get_target_plurals()[0])
41def format_html_code(
42 format_string: str,
43 **kwargs: str,
44) -> SafeString:
45 """Wrap each keyword argument in a <code> tag."""
46 safe_kwargs = {
47 key: format_html("<code>{value}</code>", value=value)
48 for key, value in kwargs.items()
49 }
50 if safe_kwargs:
51 return format_html(escape(format_string), **safe_kwargs)
52 return mark_safe(escape(format_string)) # noqa: S308
55def format_html_error_list(errors: Iterable[str]) -> SafeString:
56 """Return a HTML SafeString with each given error on a new line."""
57 return format_html_join(
58 mark_safe("<br />"),
59 "{}",
60 ((err,) for err in errors),
61 )
64def variant_name(branches: list[FluentSelectorBranch]) -> str:
65 """Get a variant name for the given branch path."""
66 if not branches:
67 return ""
68 return "[" + "][".join(branch.key for branch in branches) + "]"
71class FluentPatterns:
72 """Patterns from fluent EBNF."""
74 BLANK = r"( |\n|\r\n)*"
75 # Match string or number literals.
76 ESCAPED_CHAR = r'(\\\\|\\"|\\u[0-9a-fA-F]{4}|\\U[0-9a-fA-F]{6})'
77 STRING_LITERAL = r'"(' + ESCAPED_CHAR + r'|[^"\\])*"'
78 NUMBER_LITERAL = r"-?[0-9]+(\.[0-9]+)?"
79 IDENTIFIER = r"[a-zA-Z][a-zA-Z0-9_-]*"
80 NAMED_ARGUMENT = (
81 IDENTIFIER
82 + BLANK
83 + r":"
84 + BLANK
85 + r"("
86 + STRING_LITERAL
87 + r"|"
88 + NUMBER_LITERAL
89 + r")"
90 )
91 NAMED_ARGUMENT_LIST = (
92 r"("
93 + NAMED_ARGUMENT
94 + BLANK
95 + r","
96 + BLANK
97 + r")*"
98 + r"("
99 + NAMED_ARGUMENT
100 + r")?"
101 )
102 NAMED_ARGUMENTS_CALL = BLANK + r"\(" + BLANK + NAMED_ARGUMENT_LIST + BLANK + r"\)"
104 @classmethod
105 def placeable(cls, expression: str) -> str:
106 """Wrap a fluent expression in placeable."""
107 return r"\{" + cls.BLANK + expression + cls.BLANK + r"\}"
109 @classmethod
110 def reference(cls, ref: FluentReference) -> str:
111 """Return a placeable pattern for the given reference."""
112 # NOTE: Technically a reference may appear as direct function argument
113 # rather than placeables (with the curly braces). However, we don't
114 # expect such references to be common, and they can always include the
115 # surrounding braces within the function to achieve the same result.
116 # This is most likely to come up for variables. E.g. NUMBER($num).
117 # However, it becomes much more difficult to match this expression
118 # for highlighting.
119 if ref.type_name == "message":
120 # Message and message attribute refs do not accept parameters.
121 return cls.placeable(re.escape(ref.name))
122 if ref.type_name == "term":
123 # Term references can accept parameters.
124 # NOTE: Technically, positional parameters are allowed by the fluent
125 # syntax, which can generally be quite complex expressions. However,
126 # a Term would not be able to access this positional information.
127 # Instead, we just build the regex to match named arguments, which
128 # can only be string or number literals.
129 return cls.placeable(
130 r"-" + re.escape(ref.name) + r"(" + cls.NAMED_ARGUMENTS_CALL + ")?"
131 )
132 if ref.type_name == "variable":
133 return cls.placeable(r"\$" + re.escape(ref.name))
134 return ""
136 # Match a completed string or number placeable.
137 LITERAL_PLACEABLE_REGEX = re.compile(
138 # One or more "{". This should capture cases where the literal is
139 # double-wrapped. E.f. { { "literal" } }
140 r"(\{"
141 + BLANK
142 + r")+"
143 + r"((?P<string>"
144 + STRING_LITERAL
145 + ")|(?P<number>"
146 + NUMBER_LITERAL
147 + r"))"
148 # One of more "}".
149 # NOTE: In order to be valid Fluent syntax, this should match
150 # the same number of opening brackets. We assume the caller is
151 # working on such a source with valid Fluent syntax.
152 + r"("
153 + BLANK
154 + r"\})+"
155 )
157 ESCAPED_CHAR_REGEX = re.compile(ESCAPED_CHAR)
159 @classmethod
160 def split_literal_expressions(
161 cls, source: str
162 ) -> Iterator[tuple[int, str, str | None]]:
163 """
164 Remove all Fluent literal expressions from the given source.
166 Returns the non-literal parts of the source as a list of 3-tuples of the
167 part's starting offset in the source, its text, and the literal content
168 stripped of any surrounding placeholder brackets. The literal content
169 will be None for the final entry.
170 """
171 pos = 0
172 for literal_match in cls.LITERAL_PLACEABLE_REGEX.finditer(source):
173 string_literal = literal_match.group("string")
174 if string_literal is None:
175 literal = literal_match.group("number")
176 else:
177 literal = ""
178 for pos, chars in enumerate(
179 # Remove outer quotes from string literal.
180 cls.ESCAPED_CHAR_REGEX.split(string_literal[1:-1])
181 ):
182 if not pos % 2:
183 # Not an escaped character.
184 literal += chars
185 elif chars in {'\\"', "\\\\"}:
186 # Unescape the character by removing the "\".
187 literal += chars[1:]
188 else:
189 # Remove the leading "\u" or "\U" and convert hex
190 # sequence to a number.
191 unicode_point = int(chars[2:], 16) # noqa: FURB166
192 try:
193 # Try unescape the unicode sequence.
194 literal += chr(unicode_point)
195 except ValueError:
196 # Number was too big (and not a valid unicode
197 # character).
198 # Just include the escaped sequence as it was.
199 literal += chars
200 yield (
201 pos,
202 source[pos : literal_match.start()],
203 literal,
204 )
205 pos = literal_match.end()
206 yield (pos, source[pos:], None)
208 @classmethod
209 def highlight_source(
210 cls, source: str, highlight_patterns: Iterable[str]
211 ) -> HighlightsType:
212 """
213 Generate a list of highlights for the given source.
215 Highlights all matches for the patterns in highlight_patterns, except
216 within a literal expression. Returns the highlights as a list of
217 3-tuples of the highlighted regions' starting positions, ending
218 positions and text.
219 """
220 unique_highlights = {p for p in highlight_patterns if p}
221 if not unique_highlights:
222 return []
223 regex = re.compile("|".join(unique_highlights), flags=re.MULTILINE)
224 highlights: HighlightsType = []
225 for start, text, _ in cls.split_literal_expressions(source):
226 # NOTE: text may be empty if two literals touch.
227 highlights.extend(
228 (start + match.start(), start + match.end(), match.group())
229 for match in regex.finditer(text)
230 )
231 return highlights
234class FluentUnitConverter:
235 """Convert a translation unit into a FluentUnit."""
237 def __init__(self, unit: TransUnitModel, source: str) -> None:
238 self.unit = unit
239 self.source = source
241 def fluent_type(self) -> str:
242 """Get the fluent type of the given translation unit."""
243 flags = self.unit.all_flags
244 if flags.has_value("fluent-type"):
245 return flags.get_value("fluent-type")
246 # Guess based on id.
247 if self.unit.context and self.unit.context.startswith("-"):
248 return "Term"
249 return "Message"
251 def to_fluent_unit(self) -> FluentUnit | None:
252 """Convert the given translation unit into a FluentUnit."""
253 if not self.source:
254 return None
255 fluent_type = self.fluent_type()
256 try:
257 return FluentUnit(
258 source=self.source, unit_id=self.unit.context, fluent_type=fluent_type
259 )
260 except ValueError:
261 # Unexpected error. E.g. from invalid id.
262 # We return a default Message unit instead.
263 return FluentUnit(source=self.source)
265 def to_fluent_parts(self) -> list[FluentPart] | None:
266 """Convert the given translation unit into fluent parts."""
267 unit = self.to_fluent_unit()
268 if unit is None:
269 return None
270 return unit.get_parts()
272 def get_syntax_error(self) -> str | None:
273 """Get the syntax error that would be produced for the unit."""
274 unit = self.to_fluent_unit()
275 if unit is None:
276 return None
277 return unit.get_syntax_error()