Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/chars.py: 34%

312 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7import re 

8import unicodedata 

9from typing import TYPE_CHECKING, ClassVar 

10 

11from django.utils.translation import gettext_lazy 

12 

13from weblate.checks.base import CountingCheck, TargetCheck, TargetCheckParametrized 

14from weblate.checks.markup import strip_entities 

15from weblate.checks.parser import single_value_flag 

16from weblate.checks.same import strip_format 

17 

18if TYPE_CHECKING: 18 ↛ 19line 18 didn't jump to line 19 because the condition on line 18 was never true

19 from collections.abc import Iterable 

20 

21 from weblate.trans.models import Unit 

22 

23 from .base import FixupType 

24 

25FRENCH_PUNCTUATION_NBSP = {":"} 

26FRENCH_PUNCTUATION_NNBSP = {";", "?", "!"} 

27FRENCH_PUNCTUATION = FRENCH_PUNCTUATION_NBSP.union(FRENCH_PUNCTUATION_NNBSP) 

28FRENCH_PUNCTUATION_SPACING = {"Zs", "Ps", "Pe"} 

29FRENCH_PUNCTUATION_FIXUP_RE_NBSP = "([ \u2009\u202f])([{}])".format( 

30 "".join(FRENCH_PUNCTUATION_NBSP) 

31) 

32FRENCH_PUNCTUATION_FIXUP_RE_NNBSP = "([ \u00a0\u2009])([{}])".format( 

33 "".join(FRENCH_PUNCTUATION_NNBSP) 

34) 

35FRENCH_PUNCTUATION_MISSING_RE_NBSP = "([^\u00a0])([{}])".format( 

36 "".join(FRENCH_PUNCTUATION_NBSP) 

37) 

38FRENCH_PUNCTUATION_MISSING_RE_NNBSP = "([^\u202f])([{}])".format( 

39 "".join(FRENCH_PUNCTUATION_NNBSP) 

40) 

41MY_QUESTION_MARK = "\u1038\u104b" 

42INTERROBANGS = ("?!", "!?", "?!", "!?", "⁈", "⁉") 

43 

44 

45class BeginNewlineCheck(TargetCheck): 

46 """Check for newlines at beginning.""" 

47 

48 check_id = "begin_newline" 

49 name = gettext_lazy("Starting newline") 

50 description = gettext_lazy( 

51 "Source and translation do not both start with a newline." 

52 ) 

53 

54 def check_single(self, source: str, target: str, unit: Unit): 

55 return self.check_chars(source, target, 0, {"\n"}) 

56 

57 

58class EndNewlineCheck(TargetCheck): 

59 """Check for newlines at end.""" 

60 

61 check_id = "end_newline" 

62 name = gettext_lazy("Trailing newline") 

63 description = gettext_lazy("Source and translation do not both end with a newline.") 

64 

65 def check_single(self, source: str, target: str, unit: Unit): 

66 return self.check_chars(source, target, -1, {"\n"}) 

67 

68 

69class BeginSpaceCheck(TargetCheck): 

70 """Whitespace check, starting whitespace usually is important for UI.""" 

71 

72 check_id = "begin_space" 

73 name = gettext_lazy("Starting spaces") 

74 description = gettext_lazy( 

75 "Source and translation do not both start with same number of spaces." 

76 ) 

77 

78 def check_single(self, source: str, target: str, unit: Unit): 

79 # One letter things are usually decimal/thousand separators 

80 if len(source) <= 1 and len(target) <= 1: 

81 return False 

82 

83 stripped_target = target.lstrip(" ") 

84 stripped_source = source.lstrip(" ") 

85 

86 # String translated to spaces only 

87 if not stripped_target: 

88 return False 

89 

90 # Count space chars in source and target 

91 source_space = len(source) - len(stripped_source) 

92 target_space = len(target) - len(stripped_target) 

93 

94 # Compare numbers 

95 return source_space != target_space 

96 

97 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None: 

98 source = unit.source_string 

99 stripped_source = source.lstrip(" ") 

100 spaces = len(source) - len(stripped_source) 

101 replacement = source[:spaces] if spaces else "" 

102 return [("regex", "^ *", replacement, "u")] 

103 

104 

105class KabyleCharactersCheck(TargetCheck): 

106 """Flag and suggest standard Kabyle characters instead of visually similar but incorrect ones.""" 

107 

108 check_id = "kabyle-characters" 

109 name = gettext_lazy("Non‑standard characters in Kabyle") 

110 description = gettext_lazy( 

111 "Use standardized Latin Kabyle characters (e.g. ɣ instead of Greek γ; ɛ instead of ε)." 

112 ) 

113 

114 confusable_to_standard: ClassVar[dict[str, str]] = { 

115 "\u03b3": "\u0263", 

116 "\u0393": "\u0194", 

117 "\u03b5": "\u025b", 

118 "\u0395": "\u0190", 

119 "\u011f": "\u01e7", 

120 "\u011e": "\u01e6", 

121 } 

122 

123 def should_skip(self, unit: Unit) -> bool: 

124 # Only run on Kabyle (covers 'kab' plus any variants) 

125 if not unit.translation.language.is_base({"kab"}): 

126 return True 

127 return super().should_skip(unit) 

128 

129 def check_single(self, source: str, target: str, unit: Unit) -> bool: 

130 # by now we know it's Kabyle, so just look for confusables 

131 return any(char in target for char in self.confusable_to_standard) 

132 

133 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None: 

134 return [ 

135 ("regex", re.escape(confusable), standard, "gu") 

136 for confusable, standard in self.confusable_to_standard.items() 

137 ] 

138 

139 

140class EndSpaceCheck(TargetCheck): 

141 """Whitespace check.""" 

142 

143 check_id = "end_space" 

144 name = gettext_lazy("Trailing space") 

145 description = gettext_lazy("Source and translation do not both end with a space.") 

146 

147 def check_single(self, source: str, target: str, unit: Unit): 

148 # One letter things are usually decimal/thousand separators 

149 if len(source) <= 1 and len(target) <= 1: 

150 return False 

151 if not source or not target: 

152 return False 

153 

154 stripped_target = target.rstrip(" ") 

155 stripped_source = source.rstrip(" ") 

156 

157 # String translated to spaces only 

158 if not stripped_target: 

159 return False 

160 

161 # Count space chars in source and target 

162 source_space = len(source) - len(stripped_source) 

163 target_space = len(target) - len(stripped_target) 

164 

165 # Compare numbers 

166 return source_space != target_space 

167 

168 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None: 

169 source = unit.source_string 

170 stripped_source = source.rstrip(" ") 

171 spaces = len(source) - len(stripped_source) 

172 replacement = source[-spaces:] if spaces else "" 

173 return [("regex", " *$", replacement, "u")] 

174 

175 

176class DoubleSpaceCheck(TargetCheck): 

177 """Doublespace check.""" 

178 

179 check_id = "double_space" 

180 name = gettext_lazy("Double space") 

181 description = gettext_lazy("Translation contains double space.") 

182 

183 def check_single(self, source: str, target: str, unit: Unit): 

184 # One letter things are usually decimal/thousand separators 

185 if len(source) <= 1 and len(target) <= 1: 

186 return False 

187 if not source or not target: 

188 return False 

189 if " " in source: 

190 return False 

191 # Check if target contains double space 

192 return " " in target 

193 

194 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None: 

195 return [("regex", " {2,}", " ", "u")] 

196 

197 

198class EndStopCheck(TargetCheck): 

199 """Check for final stop.""" 

200 

201 check_id = "end_stop" 

202 name = gettext_lazy("Mismatched full stop") 

203 description = gettext_lazy( 

204 "Source and translation do not both end with a full stop." 

205 ) 

206 

207 def _check_my(self, source: str, target: str): 

208 if target.endswith(MY_QUESTION_MARK): 

209 # Laeave this on the question mark check 

210 return False 

211 return self.check_chars(source, target, -1, {".", "။"}) 

212 

213 def should_skip(self, unit: Unit) -> bool: 

214 # Thai and Lojban does not have a full stop 

215 if unit.translation.language.is_base({"th", "jbo"}): 

216 return True 

217 return super().should_skip(unit) 

218 

219 def check_single(self, source: str, target: str, unit: Unit): 

220 if len(source) <= 4: 

221 # Might need to use shortcut in translation 

222 return False 

223 if not target: 

224 return False 

225 # Allow ... to be translated into ellipsis 

226 if source.endswith("...") and target[-1] == "…": 

227 return False 

228 if unit.translation.language.is_cjk() and source[-1] in {":", ";"}: 

229 # Japanese sentence might need to end with full stop 

230 # in case it's used before list. 

231 return self.check_chars(source, target, -1, {";", ":", ":", ".", "。"}) 

232 if unit.translation.language.is_base({"hy"}): 

233 return self.check_chars( 

234 source, 

235 target, 

236 -1, 

237 {".", "。", "।", "۔", "։", "·", "෴", "។", ":", "՝", "?", "!", "`"}, 

238 ) 

239 if unit.translation.language.is_base({"hi", "bn", "or"}): 

240 # Using | instead of । is not typographically correct, but 

241 # seems to be quite usual. \u0964 is correct, but \u09F7 

242 # is also sometimes used instead in some popular editors. 

243 return self.check_chars(source, target, -1, {".", "\u0964", "\u09f7", "|"}) 

244 if unit.translation.language.is_base({"sat"}): 

245 # Santali uses "᱾" as full stop 

246 return self.check_chars(source, target, -1, {".", "᱾"}) 

247 if unit.translation.language.is_base({"my"}): 

248 return self._check_my(source, target) 

249 return self.check_chars( 

250 source, target, -1, {".", "。", "।", "۔", "։", "·", "෴", "។", "።"} 

251 ) 

252 

253 

254class EndColonCheck(TargetCheck): 

255 """Check for final colon.""" 

256 

257 check_id = "end_colon" 

258 name = gettext_lazy("Mismatched colon") 

259 description = gettext_lazy("Source and translation do not both end with a colon.") 

260 

261 def should_skip(self, unit: Unit) -> bool: 

262 # Thai and Lojban does not have a colon 

263 if unit.translation.language.is_base({"th", "jbo"}): 

264 return True 

265 return super().should_skip(unit) 

266 

267 def _check_hy(self, source: str, target: str): 

268 if source[-1] == ":": 

269 return self.check_chars(source, target, -1, {":", "՝", "`"}) 

270 return False 

271 

272 def _check_ja(self, source: str, target: str): 

273 # Japanese sentence might need to end with full stop 

274 # in case it's used before list. 

275 if source[-1] in {":", ";"}: 

276 return self.check_chars(source, target, -1, {";", ":", ":", ".", "。"}) 

277 return False 

278 

279 def check_single(self, source: str, target: str, unit: Unit): 

280 if not source or not target: 

281 return False 

282 if unit.translation.language.is_base({"hy"}): 

283 return self._check_hy(source, target) 

284 if unit.translation.language.is_cjk(): 

285 return self._check_ja(source, target) 

286 return self.check_chars(source, target, -1, {":", ":", "៖"}) 

287 

288 

289class EndQuestionCheck(TargetCheck): 

290 """Check for final question mark.""" 

291 

292 check_id = "end_question" 

293 name = gettext_lazy("Mismatched question mark") 

294 description = gettext_lazy( 

295 "Source and translation do not both end with a question mark." 

296 ) 

297 question_el = ("?", ";", ";") 

298 

299 def should_skip(self, unit: Unit) -> bool: 

300 # Thai and Lojban does not have a question mark 

301 if unit.translation.language.is_base({"th", "jbo"}): 

302 return True 

303 return super().should_skip(unit) 

304 

305 def _check_hy(self, source: str, target: str): 

306 if source[-1] == "?": 

307 return self.check_chars(source, target, -1, {"?", "՞", "։"}) 

308 return False 

309 

310 def _check_el(self, source: str, target: str): 

311 if source[-1] != "?": 

312 return False 

313 return target[-1] not in self.question_el 

314 

315 def _check_my(self, source: str, target: str): 

316 return source.endswith("?") != target.endswith(MY_QUESTION_MARK) 

317 

318 def check_single(self, source: str, target: str, unit: Unit): 

319 if not source or not target: 

320 return False 

321 if source.endswith(INTERROBANGS) or target.endswith(INTERROBANGS): 

322 return False 

323 if unit.translation.language.is_base({"hy"}): 

324 return self._check_hy(source, target) 

325 if unit.translation.language.is_base({"el"}): 

326 return self._check_el(source, target) 

327 if unit.translation.language.is_base({"my"}): 

328 return self._check_my(source, target) 

329 

330 return self.check_chars( 

331 source, target, -1, {"?", "՞", "؟", "⸮", "?", "፧", "꘏", "⳺"} 

332 ) 

333 

334 

335class EndExclamationCheck(TargetCheck): 

336 """Check for final exclamation mark.""" 

337 

338 check_id = "end_exclamation" 

339 name = gettext_lazy("Mismatched exclamation mark") 

340 description = gettext_lazy( 

341 "Source and translation do not both end with an exclamation mark." 

342 ) 

343 

344 def should_skip(self, unit: Unit) -> bool: 

345 # Thai and Lojban and Armenian does not have an exclamation mark 

346 if unit.translation.language.is_base({"hy", "th", "jbo"}): 

347 return True 

348 return super().should_skip(unit) 

349 

350 def check_single(self, source: str, target: str, unit: Unit): 

351 if not source or not target: 

352 return False 

353 if source.endswith(INTERROBANGS) or target.endswith(INTERROBANGS): 

354 return False 

355 if ( 

356 unit.translation.language.is_base({"eu"}) 

357 and source[-1] == "!" 

358 and "¡" in target 

359 and "!" in target 

360 ): 

361 return False 

362 if unit.translation.language.is_base({"my"}): 

363 return self.check_chars(source, target, -1, {"!", "႟"}) 

364 if source.endswith("Texy!") or target.endswith("Texy!"): 

365 return False 

366 return self.check_chars(source, target, -1, {"!", "!", "՜", "᥄", "႟", "߹"}) 

367 

368 

369class EndInterrobangCheck(TargetCheck): 

370 """Check for final interrobang expression.""" 

371 

372 check_id = "end_interrobang" 

373 name = gettext_lazy("Mismatched interrobang") 

374 description = gettext_lazy( 

375 "Source and translation do not both end with an interrobang expression." 

376 ) 

377 

378 def check_single(self, source: str, target: str, unit: Unit): 

379 if not source or not target: 

380 return False 

381 

382 return source.endswith(INTERROBANGS) != target.endswith(INTERROBANGS) 

383 

384 

385class EndEllipsisCheck(TargetCheck): 

386 """Check for ellipsis at the end of string.""" 

387 

388 check_id = "end_ellipsis" 

389 name = gettext_lazy("Mismatched ellipsis") 

390 description = gettext_lazy( 

391 "Source and translation do not both end with an ellipsis." 

392 ) 

393 

394 def should_skip(self, unit: Unit) -> bool: 

395 # Thai and Lojban does not have a ellipsis 

396 if unit.translation.language.is_base({"th", "jbo"}): 

397 return True 

398 return super().should_skip(unit) 

399 

400 def check_single(self, source: str, target: str, unit: Unit): 

401 if not target: 

402 return False 

403 # Allow ... to be translated into ellipsis 

404 if source.endswith("...") and target[-1] == "…": 

405 return False 

406 return self.check_chars(source, target, -1, {"…"}) 

407 

408 

409class EscapedNewlineCountingCheck(CountingCheck): 

410 r"""Check whether there is same amount of escaped \n strings.""" 

411 

412 string = "\\n" 

413 check_id = "escaped_newline" 

414 name = gettext_lazy("Mismatched \\n") 

415 description = gettext_lazy( 

416 "Number of \\n literals in translation does not match source." 

417 ) 

418 

419 ignore_re = re.compile(r"[A-Z]:\\\\[^\\ ]+(\\[^\\ ]+)+") 

420 

421 def check_single(self, source: str, target: str, unit: Unit): 

422 if not target or not source: 

423 return False 

424 

425 target = self.ignore_re.sub("", target) 

426 source = self.ignore_re.sub("", source) 

427 return super().check_single(source, target, unit) 

428 

429 

430class NewLineCountCheck(CountingCheck): 

431 """Check whether there is same amount of new lines.""" 

432 

433 string = "\n" 

434 check_id = "newline-count" 

435 name = gettext_lazy("Mismatching line breaks") 

436 description = gettext_lazy( 

437 "Number of new lines in translation does not match source." 

438 ) 

439 

440 

441class ZeroWidthSpaceCheck(TargetCheck): 

442 """Check for zero width space char (<U+200B>).""" 

443 

444 check_id = "zero-width-space" 

445 name = gettext_lazy("Zero-width space") 

446 description = gettext_lazy("Translation contains extra zero-width space character.") 

447 

448 def check_single(self, source: str, target: str, unit: Unit): 

449 if unit.translation.language.is_base({"km"}): 

450 return False 

451 if "\u200b" in source: 

452 return False 

453 return "\u200b" in target 

454 

455 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None: 

456 return [("regex", "\u200b", "", "gu")] 

457 

458 

459class MaxLengthCheck(TargetCheckParametrized): 

460 """Check for maximum length of translation.""" 

461 

462 check_id = "max-length" 

463 name = gettext_lazy("Maximum length of translation") 

464 description = gettext_lazy("Translation should not exceed given length.") 

465 default_disabled = True 

466 

467 param_type = single_value_flag(int) 

468 

469 def check_target_params( 

470 self, sources: list[str], targets: list[str], unit: Unit, value 

471 ): 

472 replace = self.get_replacement_function(unit) 

473 return any(len(replace(target)) > value for target in targets) 

474 

475 

476class EndSemicolonCheck(TargetCheck): 

477 """Check for semicolon at end.""" 

478 

479 check_id = "end_semicolon" 

480 name = gettext_lazy("Mismatched semicolon") 

481 description = gettext_lazy( 

482 "Source and translation do not both end with a semicolon." 

483 ) 

484 

485 def check_single(self, source: str, target: str, unit: Unit): 

486 if unit.translation.language.is_base({"el"}) and source and source[-1] == "?": 

487 # Complement to question mark check 

488 return False 

489 return self.check_chars( 

490 strip_entities(source), strip_entities(target), -1, {";"} 

491 ) 

492 

493 

494class KashidaCheck(TargetCheck): 

495 check_id = "kashida" 

496 name = gettext_lazy("Kashida letter used") 

497 description = gettext_lazy("The decorative kashida letters should not be used.") 

498 

499 kashida_regex = ( 

500 # Allow kashida after certain letters 

501 "(?<![\u0628\u0643\u0644])" 

502 # List of kashida letters to check 

503 "[\u0640\ufcf2\ufcf3\ufcf4\ufe71\ufe77\ufe79\ufe7b\ufe7d\ufe7f]" 

504 ) 

505 kashida_re = re.compile(kashida_regex) 

506 

507 def check_single(self, source: str, target: str, unit: Unit): 

508 return self.kashida_re.search(target) 

509 

510 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None: 

511 return [("regex", self.kashida_regex, "", "gu")] 

512 

513 

514class PunctuationSpacingCheck(TargetCheck): 

515 check_id = "punctuation_spacing" 

516 name = gettext_lazy("Punctuation spacing") 

517 description = gettext_lazy( 

518 "Missing non breakable space before double punctuation sign." 

519 ) 

520 

521 def should_skip(self, unit: Unit) -> bool: 

522 if ( 

523 not unit.translation.language.is_base({"fr"}) 

524 or unit.translation.language.code == "fr_CA" 

525 ): 

526 return True 

527 return super().should_skip(unit) 

528 

529 def check_single(self, source: str, target: str, unit: Unit) -> bool: 

530 # Remove possible markup 

531 target = strip_format(target, unit.all_flags) 

532 # Remove XML/HTML entities to simplify parsing 

533 target = strip_entities(target) 

534 

535 whitespace = {" ", "\u00a0", "\u202f", "\u2009"} 

536 

537 total = len(target) 

538 for i, char in enumerate(target): 

539 if char in FRENCH_PUNCTUATION: 

540 if i == 0: 

541 # Trigger if punctionation at beginning of the string 

542 return True 

543 if ( 

544 i + 1 < total 

545 and unicodedata.category(target[i + 1]) 

546 not in FRENCH_PUNCTUATION_SPACING 

547 ): 

548 # Ignore when not followed by space or open/close bracket 

549 continue 

550 prev_char = target[i - 1] 

551 if prev_char not in whitespace and prev_char not in FRENCH_PUNCTUATION: 

552 return True 

553 return False 

554 

555 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None: 

556 return [ 

557 # First fix possibly wrong whitespace 

558 ( 

559 "regex", 

560 FRENCH_PUNCTUATION_FIXUP_RE_NBSP, 

561 "\u00a0$2", 

562 "gu", 

563 ), 

564 ( 

565 "regex", 

566 FRENCH_PUNCTUATION_FIXUP_RE_NNBSP, 

567 "\u202f$2", 

568 "gu", 

569 ), 

570 # Then add missing ones 

571 ( 

572 "regex", 

573 FRENCH_PUNCTUATION_MISSING_RE_NBSP, 

574 "$1\u00a0$2", 

575 "gu", 

576 ), 

577 ( 

578 "regex", 

579 FRENCH_PUNCTUATION_MISSING_RE_NNBSP, 

580 "$1\u202f$2", 

581 "gu", 

582 ), 

583 ]