Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/same.py: 25%
101 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5from __future__ import annotations
7import re
8from typing import TYPE_CHECKING
10from django.utils.html import strip_tags
11from django.utils.translation import gettext_lazy
12from weblate_language_data.check_languages import LANGUAGES
14from weblate.checks.base import TargetCheck
15from weblate.checks.data import IGNORE_WORDS
16from weblate.checks.format import FLAG_RULES, PERCENT_MATCH
17from weblate.checks.markup import BBCODE_MATCH
18from weblate.checks.qt import QT_FORMAT_MATCH, QT_PLURAL_MATCH
19from weblate.checks.ruby import RUBY_FORMAT_MATCH
21if TYPE_CHECKING: 21 ↛ 22line 21 didn't jump to line 22 because the condition on line 21 was never true
22 from collections.abc import Callable
24 from weblate.checks.flags import Flags
25 from weblate.trans.models import Unit
27# Email address to ignore
28EMAIL_RE = re.compile(r"[a-z0-9_.-]+@[a-z0-9_.-]+\.[a-z0-9-]{2,}", re.IGNORECASE)
30URL_RE = re.compile(
31 r"(?:http|ftp)s?://" # http:// or https://
32 r"(?:(?:[A-Z0-9](?:[A-Z0-9-]{0,61}[A-Z0-9])?\.)+"
33 r"(?:[A-Z]{2,6}\.?|[A-Z0-9-]{2,}\.?)|" # domain...
34 r"localhost|" # localhost...
35 r"\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3})" # ...or ip
36 r"(?::\d+)?" # optional port
37 r"(?:/?|[/?]\S+)$",
38 re.IGNORECASE,
39)
41HASH_RE = re.compile(r"#[A-Za-z0-9_-]*")
43DOMAIN_RE = re.compile(
44 r"(?:[A-Z0-9](?:[A-Z0-9-]{0,61}[A-Z0-9])?\.)+"
45 r"(?:[A-Z]{2,6}\.?|[A-Z0-9-]{2,}\.?)",
46 re.IGNORECASE,
47)
49PATH_RE = re.compile(r"(^|[ ])(/[a-zA-Z0-9=:?._-]+)+")
51TEMPLATE_RE = re.compile(r"{[a-z_-]+}|@[A-Z_]@", re.IGNORECASE)
53RST_MATCH = re.compile(r"(:[a-z:]+:`[^`]+`|``[^`]+``)")
55SPLIT_RE = re.compile(
56 r"(?:\&(?:nbsp|rsaquo|lt|gt|amp|ldquo|rdquo|times|quot);|"
57 r'[() ,.^`"\'\\/_<>!?;:|{}*^@%#&~=+\r\n✓—‑…\[\]0-9-])+',
58 re.IGNORECASE,
59)
61EMOJI_RE = re.compile(r"[\U00002600-\U000027bf]|[\U0001f000-\U0001fffd]")
63# Docbook tags to ignore
64DB_TAGS = ("screen", "indexterm", "programlisting")
67def replace_format_placeholder(match: re.Match) -> str:
68 return f"x-weblate-{match.start(0)}"
71def strip_format(
72 msg: str, flags: Flags, replacement: str | Callable[[re.Match], str] = ""
73) -> str:
74 """
75 Remove format strings from the strings.
77 These are quite often not changed by translators.
78 """
79 for format_flag, (regex, _is_position_based, _extract_string) in FLAG_RULES.items():
80 if format_flag in flags:
81 return regex.sub("", msg)
83 if "qt-format" in flags:
84 regex = QT_FORMAT_MATCH
85 elif "qt-plural-format" in flags:
86 regex = QT_PLURAL_MATCH
87 elif "ruby-format" in flags:
88 regex = RUBY_FORMAT_MATCH
89 elif "rst-text" in flags:
90 regex = RST_MATCH
91 elif "percent-placeholders" in flags:
92 regex = PERCENT_MATCH
93 elif "bbcode-text" in flags:
94 regex = BBCODE_MATCH
95 else:
96 return msg
97 return regex.sub(replacement, msg)
100def strip_string(msg: str) -> str:
101 """Strip (usually) untranslated parts from the string."""
102 # Strip HTML markup
103 stripped = strip_tags(msg)
105 # Remove emojis
106 stripped = EMOJI_RE.sub(" ", stripped)
108 # Remove email addresses
109 stripped = EMAIL_RE.sub("", stripped)
111 # Strip full URLs
112 stripped = URL_RE.sub("", stripped)
114 # Strip hash tags / IRC channels
115 stripped = HASH_RE.sub("", stripped)
117 # Strip domain names/URLs
118 stripped = DOMAIN_RE.sub("", stripped)
120 # Strip file/URL paths
121 stripped = PATH_RE.sub("", stripped)
123 # Strip template markup
124 return TEMPLATE_RE.sub("", stripped)
127def test_word(word, extra_ignore):
128 """Test whether word should be ignored."""
129 return (
130 len(word) <= 2
131 or word in IGNORE_WORDS
132 or word in LANGUAGES
133 or word in extra_ignore
134 )
137def strip_placeholders(msg: str, unit: Unit) -> str:
138 return re.sub(
139 "|".join(
140 re.escape(param) if isinstance(param, str) else param.pattern
141 for param in unit.all_flags.get_value("placeholders")
142 ),
143 "",
144 msg,
145 )
148class SameCheck(TargetCheck):
149 """Check for untranslated entries."""
151 check_id = "same"
152 name = gettext_lazy("Unchanged translation")
153 description = gettext_lazy("Source and translation are identical.")
155 def should_ignore(self, source: str, unit: Unit) -> bool:
156 """Check whether given unit should be ignored."""
157 from weblate.checks.flags import TYPED_FLAGS
158 from weblate.glossary.models import get_glossary_terms
160 # Ignore some docbook tags
161 if unit.note.startswith("Tag: ") and unit.note[5:] in DB_TAGS:
162 return True
164 stripped = source
165 flags = unit.all_flags
167 # Strip format strings
168 stripped = strip_format(stripped, flags)
170 # Strip placeholder strings
171 if "placeholders" in TYPED_FLAGS and "placeholders" in flags:
172 stripped = strip_placeholders(stripped, unit)
174 if "strict-same" in flags:
175 return not stripped
177 # Ignore name of the project
178 extra_ignore = set(
179 unit.translation.component.project.name.lower().split()
180 + unit.translation.component.name.lower().split()
181 )
183 # Lower case source
184 lower_source = source.lower()
186 # Check special things like 1:4 1/2 or copyright
187 if (
188 len(source.strip("0123456789:/,.")) <= 1
189 or "(c) copyright" in lower_source
190 or "©" in source
191 ):
192 return True
194 # Strip glossary terms
195 if "check-glossary" in flags:
196 # Extract untranslatable terms
197 terms = [
198 re.escape(term.source)
199 for term in get_glossary_terms(unit, include_variants=False)
200 if "read-only" in term.all_flags
201 ]
202 if terms:
203 stripped = re.sub("|".join(terms), "", source, flags=re.IGNORECASE)
205 # Strip typically untranslatable parts
206 stripped = strip_string(stripped)
208 # Ignore strings which don't contain any string to translate
209 # or just single letter (usually unit or something like that)
210 # or are whole uppercase (abbreviations)
211 if len(stripped) <= 1 or stripped.isupper():
212 return True
213 # Check if we have any word which is not in exceptions list
214 # (words which are often same in foreign language)
215 for word in SPLIT_RE.split(stripped.lower()):
216 if not test_word(word, extra_ignore):
217 return False
219 return True
221 def should_skip(self, unit: Unit) -> bool:
222 # Skip read-only units and ignored check
223 if unit.readonly or super().should_skip(unit):
224 return True
226 source_language = unit.translation.component.source_language.base_code
228 return (
229 # Ignore the check for source language,
230 unit.translation.language.is_base({source_language})
231 # English variants will have most things untranslated
232 # Interlingua is also quite often similar to English
233 or (
234 source_language == "en"
235 and unit.translation.language.is_base({"en", "ia"})
236 )
237 )
239 def check_single(self, source: str, target: str, unit: Unit):
240 # One letter things are usually labels or decimal/thousand separators
241 if len(source) <= 1 and len(target) <= 1:
242 return False
244 # Check for ignoring
245 if self.should_ignore(source, unit):
246 return False
248 return source == target