Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/duplicate.py: 28%

56 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7import re 

8from typing import TYPE_CHECKING 

9 

10from django.utils.html import format_html 

11from django.utils.translation import gettext_lazy, ngettext 

12 

13from weblate.checks.base import TargetCheck 

14from weblate.checks.same import replace_format_placeholder, strip_format 

15from weblate.utils.html import format_html_join_comma 

16from weblate.utils.unicodechars import NON_WORD_CHARS 

17 

18if TYPE_CHECKING: 18 ↛ 19line 18 didn't jump to line 19 because the condition on line 18 was never true

19 from weblate.trans.models import Unit 

20 

21# Regexp for non word chars 

22NON_WORD = re.compile("[{}\\]]+".format("".join(NON_WORD_CHARS))) 

23 

24# Per language ignore list 

25IGNORES = { 

26 "fy": {"jo", "mei"}, 

27 "fr": {"vous", "nous"}, 

28 "hi": {"कर"}, 

29 "tr": {"tek", "adım", "gıcır", "sık"}, 

30 "sq": {"të"}, 

31 "vi": {"luôn", "song"}, 

32} 

33 

34 

35class DuplicateCheck(TargetCheck): 

36 """Check for duplicated tokens.""" 

37 

38 check_id = "duplicate" 

39 name = gettext_lazy("Consecutive duplicated words") 

40 description = gettext_lazy("Text contains the same word twice in a row.") 

41 

42 def should_skip(self, unit: Unit) -> bool: 

43 # Ignore the check for Toki Pona which often uses repeating words 

44 if unit.translation.language.is_base({"tok"}): 

45 return True 

46 return super().should_skip(unit) 

47 

48 def extract_groups( 

49 self, text: str, language_code: str 

50 ) -> tuple[list[int], list[str]]: 

51 previous = None 

52 group = 1 

53 groups: list[int] = [] 

54 words: list[str] = [] 

55 ignored = IGNORES.get(language_code, set()) 

56 for word in NON_WORD.split(text): 

57 if not word: 

58 continue 

59 if word not in ignored and len(word) >= 2 and previous == word: 

60 group += 1 

61 elif group > 1 and previous is not None: 

62 groups.append(group) 

63 words.append(previous) 

64 group = 1 

65 previous = word 

66 if group > 1 and previous is not None: 

67 groups.append(group) 

68 words.append(previous) 

69 return groups, words 

70 

71 def check_single(self, source: str, target: str, unit: Unit): 

72 source_code = unit.translation.component.source_language.base_code 

73 lang_code = unit.translation.language.base_code 

74 

75 source_groups, source_words = self.extract_groups( 

76 strip_format( 

77 source, unit.all_flags, replacement=replace_format_placeholder 

78 ), 

79 source_code, 

80 ) 

81 target_groups, target_words = self.extract_groups( 

82 strip_format( 

83 target, unit.all_flags, replacement=replace_format_placeholder 

84 ), 

85 lang_code, 

86 ) 

87 

88 # The same groups in source and target 

89 if source_groups == target_groups: 

90 return {} 

91 

92 return set(target_words) - set(source_words) 

93 

94 def get_description(self, check_obj): 

95 duplicate = set() 

96 unit = check_obj.unit 

97 source = unit.source_string 

98 for target in unit.get_target_plurals(): 

99 duplicate.update(self.check_single(source, target, unit)) 

100 return format_html( 

101 "{} {}", 

102 ngettext( 

103 "The following word is duplicated:", 

104 "The following words are duplicated:", 

105 len(duplicate), 

106 ), 

107 format_html_join_comma( 

108 "<code>{}</code>", ((word,) for word in sorted(duplicate)) 

109 ), 

110 )