Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/chars.py: 34%
312 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5from __future__ import annotations
7import re
8import unicodedata
9from typing import TYPE_CHECKING, ClassVar
11from django.utils.translation import gettext_lazy
13from weblate.checks.base import CountingCheck, TargetCheck, TargetCheckParametrized
14from weblate.checks.markup import strip_entities
15from weblate.checks.parser import single_value_flag
16from weblate.checks.same import strip_format
18if TYPE_CHECKING: 18 ↛ 19line 18 didn't jump to line 19 because the condition on line 18 was never true
19 from collections.abc import Iterable
21 from weblate.trans.models import Unit
23 from .base import FixupType
25FRENCH_PUNCTUATION_NBSP = {":"}
26FRENCH_PUNCTUATION_NNBSP = {";", "?", "!"}
27FRENCH_PUNCTUATION = FRENCH_PUNCTUATION_NBSP.union(FRENCH_PUNCTUATION_NNBSP)
28FRENCH_PUNCTUATION_SPACING = {"Zs", "Ps", "Pe"}
29FRENCH_PUNCTUATION_FIXUP_RE_NBSP = "([ \u2009\u202f])([{}])".format(
30 "".join(FRENCH_PUNCTUATION_NBSP)
31)
32FRENCH_PUNCTUATION_FIXUP_RE_NNBSP = "([ \u00a0\u2009])([{}])".format(
33 "".join(FRENCH_PUNCTUATION_NNBSP)
34)
35FRENCH_PUNCTUATION_MISSING_RE_NBSP = "([^\u00a0])([{}])".format(
36 "".join(FRENCH_PUNCTUATION_NBSP)
37)
38FRENCH_PUNCTUATION_MISSING_RE_NNBSP = "([^\u202f])([{}])".format(
39 "".join(FRENCH_PUNCTUATION_NNBSP)
40)
41MY_QUESTION_MARK = "\u1038\u104b"
42INTERROBANGS = ("?!", "!?", "?!", "!?", "⁈", "⁉")
45class BeginNewlineCheck(TargetCheck):
46 """Check for newlines at beginning."""
48 check_id = "begin_newline"
49 name = gettext_lazy("Starting newline")
50 description = gettext_lazy(
51 "Source and translation do not both start with a newline."
52 )
54 def check_single(self, source: str, target: str, unit: Unit):
55 return self.check_chars(source, target, 0, {"\n"})
58class EndNewlineCheck(TargetCheck):
59 """Check for newlines at end."""
61 check_id = "end_newline"
62 name = gettext_lazy("Trailing newline")
63 description = gettext_lazy("Source and translation do not both end with a newline.")
65 def check_single(self, source: str, target: str, unit: Unit):
66 return self.check_chars(source, target, -1, {"\n"})
69class BeginSpaceCheck(TargetCheck):
70 """Whitespace check, starting whitespace usually is important for UI."""
72 check_id = "begin_space"
73 name = gettext_lazy("Starting spaces")
74 description = gettext_lazy(
75 "Source and translation do not both start with same number of spaces."
76 )
78 def check_single(self, source: str, target: str, unit: Unit):
79 # One letter things are usually decimal/thousand separators
80 if len(source) <= 1 and len(target) <= 1:
81 return False
83 stripped_target = target.lstrip(" ")
84 stripped_source = source.lstrip(" ")
86 # String translated to spaces only
87 if not stripped_target:
88 return False
90 # Count space chars in source and target
91 source_space = len(source) - len(stripped_source)
92 target_space = len(target) - len(stripped_target)
94 # Compare numbers
95 return source_space != target_space
97 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None:
98 source = unit.source_string
99 stripped_source = source.lstrip(" ")
100 spaces = len(source) - len(stripped_source)
101 replacement = source[:spaces] if spaces else ""
102 return [("regex", "^ *", replacement, "u")]
105class KabyleCharactersCheck(TargetCheck):
106 """Flag and suggest standard Kabyle characters instead of visually similar but incorrect ones."""
108 check_id = "kabyle-characters"
109 name = gettext_lazy("Non‑standard characters in Kabyle")
110 description = gettext_lazy(
111 "Use standardized Latin Kabyle characters (e.g. ɣ instead of Greek γ; ɛ instead of ε)."
112 )
114 confusable_to_standard: ClassVar[dict[str, str]] = {
115 "\u03b3": "\u0263",
116 "\u0393": "\u0194",
117 "\u03b5": "\u025b",
118 "\u0395": "\u0190",
119 "\u011f": "\u01e7",
120 "\u011e": "\u01e6",
121 }
123 def should_skip(self, unit: Unit) -> bool:
124 # Only run on Kabyle (covers 'kab' plus any variants)
125 if not unit.translation.language.is_base({"kab"}):
126 return True
127 return super().should_skip(unit)
129 def check_single(self, source: str, target: str, unit: Unit) -> bool:
130 # by now we know it's Kabyle, so just look for confusables
131 return any(char in target for char in self.confusable_to_standard)
133 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None:
134 return [
135 ("regex", re.escape(confusable), standard, "gu")
136 for confusable, standard in self.confusable_to_standard.items()
137 ]
140class EndSpaceCheck(TargetCheck):
141 """Whitespace check."""
143 check_id = "end_space"
144 name = gettext_lazy("Trailing space")
145 description = gettext_lazy("Source and translation do not both end with a space.")
147 def check_single(self, source: str, target: str, unit: Unit):
148 # One letter things are usually decimal/thousand separators
149 if len(source) <= 1 and len(target) <= 1:
150 return False
151 if not source or not target:
152 return False
154 stripped_target = target.rstrip(" ")
155 stripped_source = source.rstrip(" ")
157 # String translated to spaces only
158 if not stripped_target:
159 return False
161 # Count space chars in source and target
162 source_space = len(source) - len(stripped_source)
163 target_space = len(target) - len(stripped_target)
165 # Compare numbers
166 return source_space != target_space
168 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None:
169 source = unit.source_string
170 stripped_source = source.rstrip(" ")
171 spaces = len(source) - len(stripped_source)
172 replacement = source[-spaces:] if spaces else ""
173 return [("regex", " *$", replacement, "u")]
176class DoubleSpaceCheck(TargetCheck):
177 """Doublespace check."""
179 check_id = "double_space"
180 name = gettext_lazy("Double space")
181 description = gettext_lazy("Translation contains double space.")
183 def check_single(self, source: str, target: str, unit: Unit):
184 # One letter things are usually decimal/thousand separators
185 if len(source) <= 1 and len(target) <= 1:
186 return False
187 if not source or not target:
188 return False
189 if " " in source:
190 return False
191 # Check if target contains double space
192 return " " in target
194 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None:
195 return [("regex", " {2,}", " ", "u")]
198class EndStopCheck(TargetCheck):
199 """Check for final stop."""
201 check_id = "end_stop"
202 name = gettext_lazy("Mismatched full stop")
203 description = gettext_lazy(
204 "Source and translation do not both end with a full stop."
205 )
207 def _check_my(self, source: str, target: str):
208 if target.endswith(MY_QUESTION_MARK):
209 # Laeave this on the question mark check
210 return False
211 return self.check_chars(source, target, -1, {".", "။"})
213 def should_skip(self, unit: Unit) -> bool:
214 # Thai and Lojban does not have a full stop
215 if unit.translation.language.is_base({"th", "jbo"}):
216 return True
217 return super().should_skip(unit)
219 def check_single(self, source: str, target: str, unit: Unit):
220 if len(source) <= 4:
221 # Might need to use shortcut in translation
222 return False
223 if not target:
224 return False
225 # Allow ... to be translated into ellipsis
226 if source.endswith("...") and target[-1] == "…":
227 return False
228 if unit.translation.language.is_cjk() and source[-1] in {":", ";"}:
229 # Japanese sentence might need to end with full stop
230 # in case it's used before list.
231 return self.check_chars(source, target, -1, {";", ":", ":", ".", "。"})
232 if unit.translation.language.is_base({"hy"}):
233 return self.check_chars(
234 source,
235 target,
236 -1,
237 {".", "。", "।", "۔", "։", "·", "෴", "។", ":", "՝", "?", "!", "`"},
238 )
239 if unit.translation.language.is_base({"hi", "bn", "or"}):
240 # Using | instead of । is not typographically correct, but
241 # seems to be quite usual. \u0964 is correct, but \u09F7
242 # is also sometimes used instead in some popular editors.
243 return self.check_chars(source, target, -1, {".", "\u0964", "\u09f7", "|"})
244 if unit.translation.language.is_base({"sat"}):
245 # Santali uses "᱾" as full stop
246 return self.check_chars(source, target, -1, {".", "᱾"})
247 if unit.translation.language.is_base({"my"}):
248 return self._check_my(source, target)
249 return self.check_chars(
250 source, target, -1, {".", "。", "।", "۔", "։", "·", "෴", "។", "።"}
251 )
254class EndColonCheck(TargetCheck):
255 """Check for final colon."""
257 check_id = "end_colon"
258 name = gettext_lazy("Mismatched colon")
259 description = gettext_lazy("Source and translation do not both end with a colon.")
261 def should_skip(self, unit: Unit) -> bool:
262 # Thai and Lojban does not have a colon
263 if unit.translation.language.is_base({"th", "jbo"}):
264 return True
265 return super().should_skip(unit)
267 def _check_hy(self, source: str, target: str):
268 if source[-1] == ":":
269 return self.check_chars(source, target, -1, {":", "՝", "`"})
270 return False
272 def _check_ja(self, source: str, target: str):
273 # Japanese sentence might need to end with full stop
274 # in case it's used before list.
275 if source[-1] in {":", ";"}:
276 return self.check_chars(source, target, -1, {";", ":", ":", ".", "。"})
277 return False
279 def check_single(self, source: str, target: str, unit: Unit):
280 if not source or not target:
281 return False
282 if unit.translation.language.is_base({"hy"}):
283 return self._check_hy(source, target)
284 if unit.translation.language.is_cjk():
285 return self._check_ja(source, target)
286 return self.check_chars(source, target, -1, {":", ":", "៖"})
289class EndQuestionCheck(TargetCheck):
290 """Check for final question mark."""
292 check_id = "end_question"
293 name = gettext_lazy("Mismatched question mark")
294 description = gettext_lazy(
295 "Source and translation do not both end with a question mark."
296 )
297 question_el = ("?", ";", ";")
299 def should_skip(self, unit: Unit) -> bool:
300 # Thai and Lojban does not have a question mark
301 if unit.translation.language.is_base({"th", "jbo"}):
302 return True
303 return super().should_skip(unit)
305 def _check_hy(self, source: str, target: str):
306 if source[-1] == "?":
307 return self.check_chars(source, target, -1, {"?", "՞", "։"})
308 return False
310 def _check_el(self, source: str, target: str):
311 if source[-1] != "?":
312 return False
313 return target[-1] not in self.question_el
315 def _check_my(self, source: str, target: str):
316 return source.endswith("?") != target.endswith(MY_QUESTION_MARK)
318 def check_single(self, source: str, target: str, unit: Unit):
319 if not source or not target:
320 return False
321 if source.endswith(INTERROBANGS) or target.endswith(INTERROBANGS):
322 return False
323 if unit.translation.language.is_base({"hy"}):
324 return self._check_hy(source, target)
325 if unit.translation.language.is_base({"el"}):
326 return self._check_el(source, target)
327 if unit.translation.language.is_base({"my"}):
328 return self._check_my(source, target)
330 return self.check_chars(
331 source, target, -1, {"?", "՞", "؟", "⸮", "?", "፧", "꘏", "⳺"}
332 )
335class EndExclamationCheck(TargetCheck):
336 """Check for final exclamation mark."""
338 check_id = "end_exclamation"
339 name = gettext_lazy("Mismatched exclamation mark")
340 description = gettext_lazy(
341 "Source and translation do not both end with an exclamation mark."
342 )
344 def should_skip(self, unit: Unit) -> bool:
345 # Thai and Lojban and Armenian does not have an exclamation mark
346 if unit.translation.language.is_base({"hy", "th", "jbo"}):
347 return True
348 return super().should_skip(unit)
350 def check_single(self, source: str, target: str, unit: Unit):
351 if not source or not target:
352 return False
353 if source.endswith(INTERROBANGS) or target.endswith(INTERROBANGS):
354 return False
355 if (
356 unit.translation.language.is_base({"eu"})
357 and source[-1] == "!"
358 and "¡" in target
359 and "!" in target
360 ):
361 return False
362 if unit.translation.language.is_base({"my"}):
363 return self.check_chars(source, target, -1, {"!", "႟"})
364 if source.endswith("Texy!") or target.endswith("Texy!"):
365 return False
366 return self.check_chars(source, target, -1, {"!", "!", "՜", "᥄", "႟", "߹"})
369class EndInterrobangCheck(TargetCheck):
370 """Check for final interrobang expression."""
372 check_id = "end_interrobang"
373 name = gettext_lazy("Mismatched interrobang")
374 description = gettext_lazy(
375 "Source and translation do not both end with an interrobang expression."
376 )
378 def check_single(self, source: str, target: str, unit: Unit):
379 if not source or not target:
380 return False
382 return source.endswith(INTERROBANGS) != target.endswith(INTERROBANGS)
385class EndEllipsisCheck(TargetCheck):
386 """Check for ellipsis at the end of string."""
388 check_id = "end_ellipsis"
389 name = gettext_lazy("Mismatched ellipsis")
390 description = gettext_lazy(
391 "Source and translation do not both end with an ellipsis."
392 )
394 def should_skip(self, unit: Unit) -> bool:
395 # Thai and Lojban does not have a ellipsis
396 if unit.translation.language.is_base({"th", "jbo"}):
397 return True
398 return super().should_skip(unit)
400 def check_single(self, source: str, target: str, unit: Unit):
401 if not target:
402 return False
403 # Allow ... to be translated into ellipsis
404 if source.endswith("...") and target[-1] == "…":
405 return False
406 return self.check_chars(source, target, -1, {"…"})
409class EscapedNewlineCountingCheck(CountingCheck):
410 r"""Check whether there is same amount of escaped \n strings."""
412 string = "\\n"
413 check_id = "escaped_newline"
414 name = gettext_lazy("Mismatched \\n")
415 description = gettext_lazy(
416 "Number of \\n literals in translation does not match source."
417 )
419 ignore_re = re.compile(r"[A-Z]:\\\\[^\\ ]+(\\[^\\ ]+)+")
421 def check_single(self, source: str, target: str, unit: Unit):
422 if not target or not source:
423 return False
425 target = self.ignore_re.sub("", target)
426 source = self.ignore_re.sub("", source)
427 return super().check_single(source, target, unit)
430class NewLineCountCheck(CountingCheck):
431 """Check whether there is same amount of new lines."""
433 string = "\n"
434 check_id = "newline-count"
435 name = gettext_lazy("Mismatching line breaks")
436 description = gettext_lazy(
437 "Number of new lines in translation does not match source."
438 )
441class ZeroWidthSpaceCheck(TargetCheck):
442 """Check for zero width space char (<U+200B>)."""
444 check_id = "zero-width-space"
445 name = gettext_lazy("Zero-width space")
446 description = gettext_lazy("Translation contains extra zero-width space character.")
448 def check_single(self, source: str, target: str, unit: Unit):
449 if unit.translation.language.is_base({"km"}):
450 return False
451 if "\u200b" in source:
452 return False
453 return "\u200b" in target
455 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None:
456 return [("regex", "\u200b", "", "gu")]
459class MaxLengthCheck(TargetCheckParametrized):
460 """Check for maximum length of translation."""
462 check_id = "max-length"
463 name = gettext_lazy("Maximum length of translation")
464 description = gettext_lazy("Translation should not exceed given length.")
465 default_disabled = True
467 param_type = single_value_flag(int)
469 def check_target_params(
470 self, sources: list[str], targets: list[str], unit: Unit, value
471 ):
472 replace = self.get_replacement_function(unit)
473 return any(len(replace(target)) > value for target in targets)
476class EndSemicolonCheck(TargetCheck):
477 """Check for semicolon at end."""
479 check_id = "end_semicolon"
480 name = gettext_lazy("Mismatched semicolon")
481 description = gettext_lazy(
482 "Source and translation do not both end with a semicolon."
483 )
485 def check_single(self, source: str, target: str, unit: Unit):
486 if unit.translation.language.is_base({"el"}) and source and source[-1] == "?":
487 # Complement to question mark check
488 return False
489 return self.check_chars(
490 strip_entities(source), strip_entities(target), -1, {";"}
491 )
494class KashidaCheck(TargetCheck):
495 check_id = "kashida"
496 name = gettext_lazy("Kashida letter used")
497 description = gettext_lazy("The decorative kashida letters should not be used.")
499 kashida_regex = (
500 # Allow kashida after certain letters
501 "(?<![\u0628\u0643\u0644])"
502 # List of kashida letters to check
503 "[\u0640\ufcf2\ufcf3\ufcf4\ufe71\ufe77\ufe79\ufe7b\ufe7d\ufe7f]"
504 )
505 kashida_re = re.compile(kashida_regex)
507 def check_single(self, source: str, target: str, unit: Unit):
508 return self.kashida_re.search(target)
510 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None:
511 return [("regex", self.kashida_regex, "", "gu")]
514class PunctuationSpacingCheck(TargetCheck):
515 check_id = "punctuation_spacing"
516 name = gettext_lazy("Punctuation spacing")
517 description = gettext_lazy(
518 "Missing non breakable space before double punctuation sign."
519 )
521 def should_skip(self, unit: Unit) -> bool:
522 if (
523 not unit.translation.language.is_base({"fr"})
524 or unit.translation.language.code == "fr_CA"
525 ):
526 return True
527 return super().should_skip(unit)
529 def check_single(self, source: str, target: str, unit: Unit) -> bool:
530 # Remove possible markup
531 target = strip_format(target, unit.all_flags)
532 # Remove XML/HTML entities to simplify parsing
533 target = strip_entities(target)
535 whitespace = {" ", "\u00a0", "\u202f", "\u2009"}
537 total = len(target)
538 for i, char in enumerate(target):
539 if char in FRENCH_PUNCTUATION:
540 if i == 0:
541 # Trigger if punctionation at beginning of the string
542 return True
543 if (
544 i + 1 < total
545 and unicodedata.category(target[i + 1])
546 not in FRENCH_PUNCTUATION_SPACING
547 ):
548 # Ignore when not followed by space or open/close bracket
549 continue
550 prev_char = target[i - 1]
551 if prev_char not in whitespace and prev_char not in FRENCH_PUNCTUATION:
552 return True
553 return False
555 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None:
556 return [
557 # First fix possibly wrong whitespace
558 (
559 "regex",
560 FRENCH_PUNCTUATION_FIXUP_RE_NBSP,
561 "\u00a0$2",
562 "gu",
563 ),
564 (
565 "regex",
566 FRENCH_PUNCTUATION_FIXUP_RE_NNBSP,
567 "\u202f$2",
568 "gu",
569 ),
570 # Then add missing ones
571 (
572 "regex",
573 FRENCH_PUNCTUATION_MISSING_RE_NBSP,
574 "$1\u00a0$2",
575 "gu",
576 ),
577 (
578 "regex",
579 FRENCH_PUNCTUATION_MISSING_RE_NNBSP,
580 "$1\u202f$2",
581 "gu",
582 ),
583 ]