Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/duplicate.py: 28%
56 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5from __future__ import annotations
7import re
8from typing import TYPE_CHECKING
10from django.utils.html import format_html
11from django.utils.translation import gettext_lazy, ngettext
13from weblate.checks.base import TargetCheck
14from weblate.checks.same import replace_format_placeholder, strip_format
15from weblate.utils.html import format_html_join_comma
16from weblate.utils.unicodechars import NON_WORD_CHARS
18if TYPE_CHECKING: 18 ↛ 19line 18 didn't jump to line 19 because the condition on line 18 was never true
19 from weblate.trans.models import Unit
21# Regexp for non word chars
22NON_WORD = re.compile("[{}\\]]+".format("".join(NON_WORD_CHARS)))
24# Per language ignore list
25IGNORES = {
26 "fy": {"jo", "mei"},
27 "fr": {"vous", "nous"},
28 "hi": {"कर"},
29 "tr": {"tek", "adım", "gıcır", "sık"},
30 "sq": {"të"},
31 "vi": {"luôn", "song"},
32}
35class DuplicateCheck(TargetCheck):
36 """Check for duplicated tokens."""
38 check_id = "duplicate"
39 name = gettext_lazy("Consecutive duplicated words")
40 description = gettext_lazy("Text contains the same word twice in a row.")
42 def should_skip(self, unit: Unit) -> bool:
43 # Ignore the check for Toki Pona which often uses repeating words
44 if unit.translation.language.is_base({"tok"}):
45 return True
46 return super().should_skip(unit)
48 def extract_groups(
49 self, text: str, language_code: str
50 ) -> tuple[list[int], list[str]]:
51 previous = None
52 group = 1
53 groups: list[int] = []
54 words: list[str] = []
55 ignored = IGNORES.get(language_code, set())
56 for word in NON_WORD.split(text):
57 if not word:
58 continue
59 if word not in ignored and len(word) >= 2 and previous == word:
60 group += 1
61 elif group > 1 and previous is not None:
62 groups.append(group)
63 words.append(previous)
64 group = 1
65 previous = word
66 if group > 1 and previous is not None:
67 groups.append(group)
68 words.append(previous)
69 return groups, words
71 def check_single(self, source: str, target: str, unit: Unit):
72 source_code = unit.translation.component.source_language.base_code
73 lang_code = unit.translation.language.base_code
75 source_groups, source_words = self.extract_groups(
76 strip_format(
77 source, unit.all_flags, replacement=replace_format_placeholder
78 ),
79 source_code,
80 )
81 target_groups, target_words = self.extract_groups(
82 strip_format(
83 target, unit.all_flags, replacement=replace_format_placeholder
84 ),
85 lang_code,
86 )
88 # The same groups in source and target
89 if source_groups == target_groups:
90 return {}
92 return set(target_words) - set(source_words)
94 def get_description(self, check_obj):
95 duplicate = set()
96 unit = check_obj.unit
97 source = unit.source_string
98 for target in unit.get_target_plurals():
99 duplicate.update(self.check_single(source, target, unit))
100 return format_html(
101 "{} {}",
102 ngettext(
103 "The following word is duplicated:",
104 "The following words are duplicated:",
105 len(duplicate),
106 ),
107 format_html_join_comma(
108 "<code>{}</code>", ((word,) for word in sorted(duplicate))
109 ),
110 )