Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/icu.py: 12%
255 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5from __future__ import annotations
7from collections import defaultdict
8from typing import TYPE_CHECKING
10from django.utils.translation import gettext, gettext_lazy
11from pyicumessageformat import Parser
13from weblate.checks.base import SourceCheck
14from weblate.checks.format import BaseFormatCheck
15from weblate.utils.html import format_html_join_comma, list_to_tuples
17if TYPE_CHECKING: 17 ↛ 18line 17 didn't jump to line 18 because the condition on line 17 was never true
18 from weblate.trans.models import Unit
20# Unique value for checking tags. Since types are
21# always strings, this will never be encountered.
22TAG_TYPE = -100
24# These types are to be considered numeric. Numeric placeholders
25# can be of any numeric type without triggering a warning from
26# the checker.
27NUMERIC_TYPES = {"number", "plural", "selectordinal"}
29# These types have their sub-messages checked to ensure that
30# sub-message selectors are valid.
31PLURAL_TYPES = {"plural", "selectordinal"}
33# ... and these are the valid selectors, along with selectors
34# for specific values, formatted such as: =0, =1, etc.
35PLURAL_SELECTORS = {"zero", "one", "two", "few", "many", "other"}
38# We construct two Parser instances, one for tags and one without.
39# Both parsers are configured to allow spaces inside formats, to not
40# require other (which we can do better ourselves), and to be
41# permissive about what types can have sub-messages.
42standard_parser = Parser(
43 {
44 "include_indices": True,
45 "loose_submessages": True,
46 "allow_format_spaces": True,
47 "require_other": False,
48 "allow_tags": False,
49 }
50)
52tag_parser = Parser(
53 {
54 "include_indices": True,
55 "loose_submessages": True,
56 "allow_format_spaces": True,
57 "require_other": False,
58 "allow_tags": True,
59 "strict_tags": False,
60 "tag_type": TAG_TYPE,
61 }
62)
65strict_tag_parser = Parser(
66 {
67 "include_indices": True,
68 "loose_submessages": True,
69 "allow_format_spaces": True,
70 "require_other": False,
71 "allow_tags": True,
72 "strict_tags": True,
73 "tag_type": TAG_TYPE,
74 }
75)
78def parse_icu(
79 source: str,
80 allow_tags: bool,
81 strict_tags: bool,
82 tag_prefix: str | None = None,
83 want_tokens=False,
84) -> tuple[list[str] | None, Exception | None, list[str] | None]:
85 """Parse an ICU MessageFormat message."""
86 ast = None
87 err: Exception | None = None
88 tokens: list[str] | None = [] if want_tokens else None
89 parser = standard_parser
90 if allow_tags:
91 parser = strict_tag_parser if strict_tags else tag_parser
93 parser.options["tag_prefix"] = tag_prefix
95 try:
96 ast = parser.parse(source, tokens)
97 except SyntaxError as e:
98 err = e
100 parser.options["tag_prefix"] = None
102 return ast, err, tokens
105def check_bad_plural_selector(selector):
106 if selector in PLURAL_SELECTORS:
107 return False
108 return selector[0] != "="
111def update_maybe_value(value, old):
112 """
113 Certain placeholder values can have one of four values.
115 `None`, `True`, `False`, or `0`.
117 `None` represents a value never set.
118 `True` or `False` represents a value selection.
119 `0` represents a set value with conflicting values.
121 This is useful if there are multiple placeholders with
122 conflicting type info.
123 """
124 if old is None or old == value:
125 return value
126 return 0
129def extract_highlights(token, source: str):
130 """Extract all placeholders from an AST selected for highlighting."""
131 if isinstance(token, str):
132 return
134 if isinstance(token, list):
135 for tok in token:
136 yield from extract_highlights(tok, source)
138 # Sanity check the token. They should always have
139 # start and end.
140 if "start" not in token or "end" not in token:
141 return
143 start = token["start"]
144 end = token["end"]
145 usable = start < len(source)
147 if token.get("hash"):
148 usable = False
150 if "options" in token:
151 usable = False
152 for subast in token["options"].values():
153 yield from extract_highlights(subast, source)
155 if "contents" in token:
156 usable = False
157 yield from extract_highlights(token["contents"], source)
159 if usable:
160 yield (start, end, source[start:end])
163def extract_placeholders(token, variables=None):
164 """Extract all placeholders from an AST and summarize their types."""
165 if variables is None:
166 variables = {}
168 if isinstance(token, str):
169 # Skip strings. Those aren't interesting.
170 return variables
172 if isinstance(token, list):
173 # If we have a list, then we have a list of tokens so iterate
174 # over the entire list.
175 for tok in token:
176 extract_placeholders(tok, variables)
178 return variables
180 if "name" not in token:
181 # There should always be a name. This is highly suspicious.
182 # Should this raise an exception?
183 return variables
185 name = token["name"]
186 ttype = token.get("type")
187 data = variables.setdefault(
188 name,
189 {
190 "name": name,
191 "types": set(),
192 "formats": set(),
193 "is_number": None,
194 "is_tag": None,
195 "is_empty": None,
196 },
197 )
199 if ttype:
200 is_tag = ttype is TAG_TYPE
201 data["is_tag"] = update_maybe_value(is_tag, data["is_tag"])
203 if is_tag:
204 data["is_empty"] = update_maybe_value(
205 "contents" not in token or not token["contents"], data["is_empty"]
206 )
207 else:
208 data["types"].add(ttype)
209 data["is_number"] = update_maybe_value(
210 ttype in NUMERIC_TYPES, data["is_number"]
211 )
212 if "format" in token:
213 data["formats"].add(token["format"])
215 elif name == "count":
216 # Assume count is a numeric type
217 data["types"].add("number")
218 data["is_number"] = update_maybe_value(True, data["is_number"])
220 if "options" in token:
221 choices = data.setdefault("choices", set())
223 # We need to do three things with options:
224 for selector, subast in token["options"].items():
225 # First, we log the selector for later comparison.
226 choices.add(selector)
228 # Second, we ensure the selector is valid if we're working
229 # with a plural/selectordinal type.
230 if ttype in PLURAL_TYPES and check_bad_plural_selector(selector):
231 data.setdefault("bad_plural", set()).add(selector)
233 # Finally, we process the sub-ast for this option.
234 extract_placeholders(subast, variables)
236 # Make sure we process the contents sub-ast if one exists.
237 if "contents" in token:
238 extract_placeholders(token["contents"], variables)
240 return variables
243class ICUCheckMixin:
244 def get_flags(self, unit: Unit):
245 if unit and unit.all_flags.has_value("icu-flags"):
246 return unit.all_flags.get_value("icu-flags")
247 return []
249 def get_tag_prefix(self, unit: Unit):
250 if unit and unit.all_flags.has_value("icu-tag-prefix"):
251 return unit.all_flags.get_value("icu-tag-prefix")
252 return None
255class ICUSourceCheck(ICUCheckMixin, SourceCheck):
256 """Check for ICU MessageFormat syntax."""
258 check_id = "icu_message_format_syntax"
259 name = gettext_lazy("ICU MessageFormat syntax")
260 description = gettext_lazy("Syntax errors in ICU MessageFormat strings.")
261 default_disabled = True
263 def __init__(self) -> None:
264 super().__init__()
265 self.enable_string = "icu-message-format"
266 self.ignore_string = f"ignore-{self.enable_string}"
268 def check_source_unit(self, sources: list[str], unit: Unit) -> bool:
269 """Checker for source strings. Only check for syntax issues."""
270 if not sources or not sources[0]:
271 return False
273 flags = self.get_flags(unit)
274 strict_tags = "strict-xml" in flags
275 allow_tags = strict_tags or "xml" in flags
276 tag_prefix = self.get_tag_prefix(unit)
278 _ast, src_err, _tokens = parse_icu(
279 sources[0], allow_tags, strict_tags, tag_prefix
280 )
281 return bool(src_err)
284class ICUMessageFormatCheck(ICUCheckMixin, BaseFormatCheck):
285 """Check for ICU MessageFormat string."""
287 check_id = "icu_message_format"
288 name = gettext_lazy("ICU MessageFormat")
289 description = gettext_lazy(
290 "Syntax errors and/or placeholder mismatches in ICU MessageFormat strings."
291 )
293 def check_format(self, source: str, target: str, ignore_missing, unit: Unit):
294 """Checker for ICU MessageFormat strings."""
295 if not target or not source:
296 return False
298 flags = self.get_flags(unit)
299 strict_tags = "strict-xml" in flags
300 allow_tags = strict_tags or "xml" in flags
301 tag_prefix = self.get_tag_prefix(unit)
303 result = defaultdict(list)
304 src_ast, src_err, _tokens = parse_icu(
305 source, allow_tags, strict_tags, tag_prefix
306 )
308 # Check to see if we're running on a source string only.
309 # If we are, then we can only run a syntax check on the
310 # source and be done.
311 if unit and unit.is_source:
312 if src_err:
313 result["syntax"].append(src_err)
314 return result
315 return False
317 tgt_ast, tgt_err, _tokens = parse_icu(
318 target, allow_tags, strict_tags, tag_prefix
319 )
320 if tgt_err:
321 result["syntax"].append(tgt_err)
323 if tgt_err:
324 return result
325 if src_err:
326 # We cannot run any further checks if the source
327 # string isn't valid, so just accept that the target
328 # string is valid for now.
329 return False
331 # Both strings are valid! Congratulations. Let's extract
332 # information on all the placeholders in both strings, and
333 # compare them to see if anything is wrong.
334 src_vars = extract_placeholders(src_ast)
335 tgt_vars = extract_placeholders(tgt_ast)
337 # First, we check all the variables in the target.
338 for name, data in tgt_vars.items():
339 self.check_for_other(result, name, data, flags)
341 if name in src_vars:
342 src_data = src_vars[name]
344 self.check_bad_plural(result, name, data, src_data, flags)
345 self.check_bad_submessage(result, name, data, src_data, flags)
346 self.check_wrong_type(result, name, data, src_data, flags)
348 if allow_tags:
349 self.check_tags(result, name, data, src_data, flags)
351 else:
352 self.check_bad_submessage(result, name, data, None, flags)
354 # The variable does not exist in the source,
355 # which suggests a mistake.
356 if "-extra" not in flags:
357 result["extra"].append(name)
359 # We also want to check for variables used in the
360 # source but not in the target.
361 self.check_missing(result, src_vars, tgt_vars, flags)
363 if result:
364 return result
365 return False
367 def check_missing(self, result, src_vars, tgt_vars, flags) -> None:
368 """Detect any variables in the target not in the source."""
369 if "-missing" in flags:
370 return
372 missing = [name for name in src_vars if name not in tgt_vars]
374 if missing:
375 result["missing"] = missing
377 def check_for_other(self, result, name, data, flags) -> None:
378 """Ensure types with sub-messages have other."""
379 if "-require_other" in flags:
380 return
382 choices = data.get("choices")
383 if choices and "other" not in choices:
384 result["no_other"].append(name)
386 def check_bad_plural(self, result, name, data, src_data, flags) -> None:
387 """Forward bad plural selectors detected during extraction."""
388 if "-plural_selectors" in flags:
389 return
391 if "bad_plural" in data:
392 result["bad_plural"].append([name, data["bad_plural"]])
394 def check_bad_submessage(self, result, name, data, src_data, flags) -> None:
395 """Detect any bad sub-message selectors."""
396 if "-submessage_selectors" in flags:
397 return
399 bad = set()
401 # We also want to check individual select choices.
402 if (
403 src_data
404 and "select" in data["types"]
405 and "select" in src_data["types"]
406 and "choices" in data
407 and "choices" in src_data
408 ):
409 choices = data["choices"]
410 src_choices = src_data["choices"]
412 for selector in choices:
413 if selector not in src_choices:
414 bad.add(selector)
416 if bad:
417 result["bad_submessage"].append([name, bad])
419 def check_wrong_type(self, result, name, data, src_data, flags) -> None:
420 """Ensure that types match, when possible."""
421 if "-types" in flags:
422 return
424 # If we're dealing with a number, we want to use
425 # special number logic, since numbers work with
426 # multiple types.
427 if isinstance(src_data["is_number"], bool) and src_data["is_number"]:
428 if src_data["is_number"] != data["is_number"]:
429 result["wrong_type"].append(name)
431 else:
432 for ttype in data["types"]:
433 if ttype not in src_data["types"]:
434 result["wrong_type"].append(name)
435 break
437 def check_tags(self, result, name, data, src_data, flags) -> None:
438 """Correct any erroneous XML tags."""
439 if "-tags" in flags:
440 return
442 if isinstance(src_data["is_tag"], bool) or data["is_tag"] is not None:
443 if src_data["is_tag"]:
444 if not data["is_tag"]:
445 result["should_be_tag"].append(name)
447 elif (
448 isinstance(src_data["is_empty"], bool)
449 and src_data["is_empty"] != data["is_empty"]
450 ):
451 if src_data["is_empty"]:
452 result["tag_not_empty"].append(name)
453 else:
454 result["tag_empty"].append(name)
456 elif data["is_tag"]:
457 result["not_tag"].append(name)
459 def format_result(self, result):
460 if result.get("syntax"):
461 yield gettext("Syntax error: %s") % format_html_join_comma(
462 "{}",
463 list_to_tuples(err.msg or "unknown error" for err in result["syntax"]),
464 )
466 if result.get("extra"):
467 yield gettext(
468 "One or more unknown placeholders in the translation: %s"
469 ) % format_html_join_comma("{}", list_to_tuples(result["extra"]))
471 if result.get("missing"):
472 yield gettext(
473 "One or more placeholders missing in the translation: %s"
474 ) % format_html_join_comma("{}", list_to_tuples(result["missing"]))
476 if result.get("wrong_type"):
477 yield gettext(
478 "One or more placeholder types are incorrect: %s"
479 ) % format_html_join_comma("{}", list_to_tuples(result["wrong_type"]))
481 if result.get("no_other"):
482 yield gettext("Missing other sub-message for: %s") % format_html_join_comma(
483 "{}", list_to_tuples(result["no_other"])
484 )
486 if result.get("bad_plural"):
487 yield gettext(
488 "Incorrect plural selectors for: %s"
489 ) % format_html_join_comma(
490 "{}", (f"{x[0]} ({', '.join(x[1])})" for x in result["bad_plural"])
491 )
493 if result.get("bad_submessage"):
494 yield gettext(
495 "Incorrect sub-message selectors for: %s"
496 ) % format_html_join_comma(
497 "{}", (f"{x[0]} ({', '.join(x[1])})" for x in result["bad_submessage"])
498 )
500 if result.get("should_be_tag"):
501 yield gettext(
502 "One or more placeholders should have "
503 "a corresponding XML tag in the translation: %s"
504 ) % format_html_join_comma("{}", list_to_tuples(result["should_be_tag"]))
506 if result.get("not_tag"):
507 yield gettext(
508 "One or more placeholders should not be "
509 "an XML tag in the translation: %s"
510 ) % format_html_join_comma("{}", list_to_tuples(result["not_tag"]))
512 if result.get("tag_not_empty"):
513 yield gettext(
514 "One or more XML tags has unexpected content in the translation: %s"
515 ) % format_html_join_comma("{}", list_to_tuples(result["tag_not_empty"]))
517 if result.get("tag_empty"):
518 yield gettext(
519 "One or more XML tags missing content in the translation: %s"
520 ) % format_html_join_comma("{}", list_to_tuples(result["tag_empty"]))
522 def check_highlight(self, source: str, unit: Unit):
523 if self.should_skip(unit): 523 ↛ 526line 523 didn't jump to line 526 because the condition on line 523 was always true
524 return
526 flags = self.get_flags(unit)
527 if "-highlight" in flags:
528 return
530 strict_tags = "strict-xml" in flags
531 allow_tags = strict_tags or "xml" in flags
532 tag_prefix = self.get_tag_prefix(unit)
534 ast, _err, _tokens = parse_icu(
535 source, allow_tags, strict_tags, tag_prefix, True
536 )
537 if not ast:
538 return
540 yield from extract_highlights(ast, source)