Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/same.py: 25%

101 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7import re 

8from typing import TYPE_CHECKING 

9 

10from django.utils.html import strip_tags 

11from django.utils.translation import gettext_lazy 

12from weblate_language_data.check_languages import LANGUAGES 

13 

14from weblate.checks.base import TargetCheck 

15from weblate.checks.data import IGNORE_WORDS 

16from weblate.checks.format import FLAG_RULES, PERCENT_MATCH 

17from weblate.checks.markup import BBCODE_MATCH 

18from weblate.checks.qt import QT_FORMAT_MATCH, QT_PLURAL_MATCH 

19from weblate.checks.ruby import RUBY_FORMAT_MATCH 

20 

21if TYPE_CHECKING: 21 ↛ 22line 21 didn't jump to line 22 because the condition on line 21 was never true

22 from collections.abc import Callable 

23 

24 from weblate.checks.flags import Flags 

25 from weblate.trans.models import Unit 

26 

27# Email address to ignore 

28EMAIL_RE = re.compile(r"[a-z0-9_.-]+@[a-z0-9_.-]+\.[a-z0-9-]{2,}", re.IGNORECASE) 

29 

30URL_RE = re.compile( 

31 r"(?:http|ftp)s?://" # http:// or https:// 

32 r"(?:(?:[A-Z0-9](?:[A-Z0-9-]{0,61}[A-Z0-9])?\.)+" 

33 r"(?:[A-Z]{2,6}\.?|[A-Z0-9-]{2,}\.?)|" # domain... 

34 r"localhost|" # localhost... 

35 r"\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3})" # ...or ip 

36 r"(?::\d+)?" # optional port 

37 r"(?:/?|[/?]\S+)$", 

38 re.IGNORECASE, 

39) 

40 

41HASH_RE = re.compile(r"#[A-Za-z0-9_-]*") 

42 

43DOMAIN_RE = re.compile( 

44 r"(?:[A-Z0-9](?:[A-Z0-9-]{0,61}[A-Z0-9])?\.)+" 

45 r"(?:[A-Z]{2,6}\.?|[A-Z0-9-]{2,}\.?)", 

46 re.IGNORECASE, 

47) 

48 

49PATH_RE = re.compile(r"(^|[ ])(/[a-zA-Z0-9=:?._-]+)+") 

50 

51TEMPLATE_RE = re.compile(r"{[a-z_-]+}|@[A-Z_]@", re.IGNORECASE) 

52 

53RST_MATCH = re.compile(r"(:[a-z:]+:`[^`]+`|``[^`]+``)") 

54 

55SPLIT_RE = re.compile( 

56 r"(?:\&(?:nbsp|rsaquo|lt|gt|amp|ldquo|rdquo|times|quot);|" 

57 r'[() ,.^`"\'\\/_<>!?;:|{}*^@%#&~=+\r\n✓—‑…\[\]0-9-])+', 

58 re.IGNORECASE, 

59) 

60 

61EMOJI_RE = re.compile(r"[\U00002600-\U000027bf]|[\U0001f000-\U0001fffd]") 

62 

63# Docbook tags to ignore 

64DB_TAGS = ("screen", "indexterm", "programlisting") 

65 

66 

67def replace_format_placeholder(match: re.Match) -> str: 

68 return f"x-weblate-{match.start(0)}" 

69 

70 

71def strip_format( 

72 msg: str, flags: Flags, replacement: str | Callable[[re.Match], str] = "" 

73) -> str: 

74 """ 

75 Remove format strings from the strings. 

76 

77 These are quite often not changed by translators. 

78 """ 

79 for format_flag, (regex, _is_position_based, _extract_string) in FLAG_RULES.items(): 

80 if format_flag in flags: 

81 return regex.sub("", msg) 

82 

83 if "qt-format" in flags: 

84 regex = QT_FORMAT_MATCH 

85 elif "qt-plural-format" in flags: 

86 regex = QT_PLURAL_MATCH 

87 elif "ruby-format" in flags: 

88 regex = RUBY_FORMAT_MATCH 

89 elif "rst-text" in flags: 

90 regex = RST_MATCH 

91 elif "percent-placeholders" in flags: 

92 regex = PERCENT_MATCH 

93 elif "bbcode-text" in flags: 

94 regex = BBCODE_MATCH 

95 else: 

96 return msg 

97 return regex.sub(replacement, msg) 

98 

99 

100def strip_string(msg: str) -> str: 

101 """Strip (usually) untranslated parts from the string.""" 

102 # Strip HTML markup 

103 stripped = strip_tags(msg) 

104 

105 # Remove emojis 

106 stripped = EMOJI_RE.sub(" ", stripped) 

107 

108 # Remove email addresses 

109 stripped = EMAIL_RE.sub("", stripped) 

110 

111 # Strip full URLs 

112 stripped = URL_RE.sub("", stripped) 

113 

114 # Strip hash tags / IRC channels 

115 stripped = HASH_RE.sub("", stripped) 

116 

117 # Strip domain names/URLs 

118 stripped = DOMAIN_RE.sub("", stripped) 

119 

120 # Strip file/URL paths 

121 stripped = PATH_RE.sub("", stripped) 

122 

123 # Strip template markup 

124 return TEMPLATE_RE.sub("", stripped) 

125 

126 

127def test_word(word, extra_ignore): 

128 """Test whether word should be ignored.""" 

129 return ( 

130 len(word) <= 2 

131 or word in IGNORE_WORDS 

132 or word in LANGUAGES 

133 or word in extra_ignore 

134 ) 

135 

136 

137def strip_placeholders(msg: str, unit: Unit) -> str: 

138 return re.sub( 

139 "|".join( 

140 re.escape(param) if isinstance(param, str) else param.pattern 

141 for param in unit.all_flags.get_value("placeholders") 

142 ), 

143 "", 

144 msg, 

145 ) 

146 

147 

148class SameCheck(TargetCheck): 

149 """Check for untranslated entries.""" 

150 

151 check_id = "same" 

152 name = gettext_lazy("Unchanged translation") 

153 description = gettext_lazy("Source and translation are identical.") 

154 

155 def should_ignore(self, source: str, unit: Unit) -> bool: 

156 """Check whether given unit should be ignored.""" 

157 from weblate.checks.flags import TYPED_FLAGS 

158 from weblate.glossary.models import get_glossary_terms 

159 

160 # Ignore some docbook tags 

161 if unit.note.startswith("Tag: ") and unit.note[5:] in DB_TAGS: 

162 return True 

163 

164 stripped = source 

165 flags = unit.all_flags 

166 

167 # Strip format strings 

168 stripped = strip_format(stripped, flags) 

169 

170 # Strip placeholder strings 

171 if "placeholders" in TYPED_FLAGS and "placeholders" in flags: 

172 stripped = strip_placeholders(stripped, unit) 

173 

174 if "strict-same" in flags: 

175 return not stripped 

176 

177 # Ignore name of the project 

178 extra_ignore = set( 

179 unit.translation.component.project.name.lower().split() 

180 + unit.translation.component.name.lower().split() 

181 ) 

182 

183 # Lower case source 

184 lower_source = source.lower() 

185 

186 # Check special things like 1:4 1/2 or copyright 

187 if ( 

188 len(source.strip("0123456789:/,.")) <= 1 

189 or "(c) copyright" in lower_source 

190 or "©" in source 

191 ): 

192 return True 

193 

194 # Strip glossary terms 

195 if "check-glossary" in flags: 

196 # Extract untranslatable terms 

197 terms = [ 

198 re.escape(term.source) 

199 for term in get_glossary_terms(unit, include_variants=False) 

200 if "read-only" in term.all_flags 

201 ] 

202 if terms: 

203 stripped = re.sub("|".join(terms), "", source, flags=re.IGNORECASE) 

204 

205 # Strip typically untranslatable parts 

206 stripped = strip_string(stripped) 

207 

208 # Ignore strings which don't contain any string to translate 

209 # or just single letter (usually unit or something like that) 

210 # or are whole uppercase (abbreviations) 

211 if len(stripped) <= 1 or stripped.isupper(): 

212 return True 

213 # Check if we have any word which is not in exceptions list 

214 # (words which are often same in foreign language) 

215 for word in SPLIT_RE.split(stripped.lower()): 

216 if not test_word(word, extra_ignore): 

217 return False 

218 

219 return True 

220 

221 def should_skip(self, unit: Unit) -> bool: 

222 # Skip read-only units and ignored check 

223 if unit.readonly or super().should_skip(unit): 

224 return True 

225 

226 source_language = unit.translation.component.source_language.base_code 

227 

228 return ( 

229 # Ignore the check for source language, 

230 unit.translation.language.is_base({source_language}) 

231 # English variants will have most things untranslated 

232 # Interlingua is also quite often similar to English 

233 or ( 

234 source_language == "en" 

235 and unit.translation.language.is_base({"en", "ia"}) 

236 ) 

237 ) 

238 

239 def check_single(self, source: str, target: str, unit: Unit): 

240 # One letter things are usually labels or decimal/thousand separators 

241 if len(source) <= 1 and len(target) <= 1: 

242 return False 

243 

244 # Check for ignoring 

245 if self.should_ignore(source, unit): 

246 return False 

247 

248 return source == target