Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/markup.py: 29%

381 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7import re 

8from collections import Counter, defaultdict 

9from functools import cache, lru_cache 

10from itertools import chain 

11from types import SimpleNamespace 

12from typing import TYPE_CHECKING, cast 

13 

14from django.core.exceptions import ValidationError 

15from django.core.validators import URLValidator 

16from django.utils.functional import cached_property 

17from django.utils.html import format_html_join 

18from django.utils.safestring import mark_safe 

19from django.utils.translation import gettext, gettext_lazy 

20from docutils import utils 

21from docutils.core import Publisher 

22from docutils.nodes import ( 

23 Element, 

24 emphasis, 

25 footnote_reference, 

26 literal, 

27 problematic, 

28 reference, 

29 strong, 

30 substitution_reference, 

31) 

32from docutils.parsers.rst import Parser, languages 

33from docutils.parsers.rst.states import Inliner 

34from docutils.readers.standalone import Reader 

35from docutils.writers.null import Writer 

36 

37from weblate.checks.base import TargetCheck 

38from weblate.utils.html import ( 

39 MD_BROKEN_LINK, 

40 MD_LINK, 

41 MD_REFLINK, 

42 MD_SYNTAX, 

43 MD_SYNTAX_GROUPS, 

44 HTMLSanitizer, 

45) 

46from weblate.utils.xml import parse_xml 

47 

48if TYPE_CHECKING: 48 ↛ 49line 48 didn't jump to line 49 because the condition on line 48 was never true

49 from collections.abc import Iterable 

50 

51 from django_stubs_ext import StrOrPromise 

52 from docutils.nodes import ( 

53 system_message, 

54 ) 

55 from lxml.etree import _Element 

56 

57 from weblate.checks.base import MissingExtraDict 

58 from weblate.trans.models import Unit 

59 

60 from .base import FixupType 

61 from .models import Check 

62 

63BBCODE_MATCH = re.compile( 

64 r"(?P<start>\[(?P<tag>[^]]+)(@[^]]*)?\])(.*?)(?P<end>\[\/(?P=tag)\])", re.MULTILINE 

65) 

66 

67 

68XML_MATCH = re.compile(r"<[^>]+>") 

69XML_ENTITY_MATCH = re.compile( 

70 r""" 

71 # Initial & 

72 \& 

73 ( 

74 # CharRef 

75 \x23[0-9]+ 

76 | 

77 # CharRef 

78 \x23x[0-9a-fA-F]+ 

79 | 

80 # EntityRef 

81 # NameStartChar 

82 [:A-Z_a-z\xC0-\xD6\xD8-\xF6\xF8-\u02FF\u0370-\u037D\u037F-\u1FFF\u200C-\u200D\u2070-\u218F\u2C00-\u2FEF\u3001-\uD7FF\uF900-\uFDCF\uFDF0-\uFFFD\u10000-\uEFFFF] 

83 # NameChar 

84 [0-9\xB7.:A-Z_a-z\xC0-\xD6\xD8-\xF6\xF8-\u02FF\u0370-\u037D\u037F-\u1FFF\u200C-\u200D\u2070-\u218F\u2C00-\u2FEF\u3001-\uD7FF\uF900-\uFDCF\uFDF0-\uFFFD\u10000-\uEFFFF\u0300-\u036F\u203F-\u2040-]* 

85 ) 

86 # Closing ; 

87 ; 

88 """, 

89 re.VERBOSE, 

90) 

91 

92# Extracted from Sphinx sphinx/util/docutils.py 

93RST_EXPLICIT_TITLE_RE = re.compile(r"^(.+?)\s*(?<!\x00)<(.*?)>$", re.DOTALL) 

94 

95# These should be present in translation if present in source, but might be translated 

96RST_TRANSLATABLE = { 

97 "guilabel", 

98 "file", 

99 "code", 

100 "math", 

101 "eq", 

102 "abbr", 

103 "dfn", 

104 "menuselection", 

105 "sub", 

106 "sup", 

107 "kbd", 

108 "index", 

109 "samp", 

110} 

111 

112RST_ROLE_RE = [ 

113 re.compile(r"""Unknown interpreted text role "([^"]*)"\."""), 

114 re.compile(r"""Interpreted text role "([^"]*)" not implemented\."""), 

115] 

116 

117 

118def strip_entities(text): 

119 """Strip all HTML entities (we don't care about them).""" 

120 return XML_ENTITY_MATCH.sub(" ", text) 

121 

122 

123class BBCodeCheck(TargetCheck): 

124 """Check for matching bbcode tags.""" 

125 

126 check_id = "bbcode" 

127 name = gettext_lazy("BBCode markup") 

128 description = gettext_lazy("BBCode in translation does not match source.") 

129 default_disabled = True 

130 

131 def __init__(self) -> None: 

132 super().__init__() 

133 self.enable_string = "bbcode-text" 

134 

135 def check_single(self, source: str, target: str, unit: Unit): 

136 # Parse source 

137 src_match = BBCODE_MATCH.findall(source) 

138 

139 # Parse target 

140 tgt_match = BBCODE_MATCH.findall(target) 

141 if len(src_match) != len(tgt_match): 

142 return True 

143 

144 src_tags = {x[1] for x in src_match} 

145 tgt_tags = {x[1] for x in tgt_match} 

146 

147 return src_tags != tgt_tags 

148 

149 def check_highlight(self, source: str, unit: Unit): 

150 if self.should_skip(unit): 150 ↛ 152line 150 didn't jump to line 152 because the condition on line 150 was always true

151 return 

152 for match in BBCODE_MATCH.finditer(source): 

153 for tag in ("start", "end"): 

154 yield match.start(tag), match.end(tag), match.group(tag) 

155 

156 

157class BaseXMLCheck(TargetCheck): 

158 def detect_xml_wrapping(self, text: str) -> tuple[_Element, bool]: 

159 """Detect whether wrapping is desired.""" 

160 try: 

161 return self.parse_xml(text, True), True 

162 except SyntaxError: 

163 return self.parse_xml(text, False), False 

164 

165 def can_parse_xml(self, text: str) -> bool: 

166 try: 

167 self.detect_xml_wrapping(text) 

168 except SyntaxError: 

169 return False 

170 return True 

171 

172 def parse_xml(self, text: str, wrap: bool) -> _Element: 

173 """Parse XML.""" 

174 text = strip_entities(text) 

175 if wrap: 

176 text = f"<weblate>{text}</weblate>" 

177 return parse_xml(text.encode() if "encoding" in text else text) 

178 

179 def should_skip(self, unit: Unit) -> bool: 

180 if super().should_skip(unit): 180 ↛ 181line 180 didn't jump to line 181 because the condition on line 180 was never true

181 return True 

182 

183 flags = unit.all_flags 

184 

185 if "safe-html" in flags: 185 ↛ 186line 185 didn't jump to line 186 because the condition on line 185 was never true

186 return True 

187 

188 if "xml-text" in flags: 188 ↛ 189line 188 didn't jump to line 189 because the condition on line 188 was never true

189 return False 

190 

191 sources = unit.get_source_plurals() 

192 

193 # Quick check if source looks like XML. 

194 if all( 194 ↛ 200line 194 didn't jump to line 200 because the condition on line 194 was always true

195 "<" not in source or not XML_MATCH.findall(source) for source in sources 

196 ): 

197 return True 

198 

199 # Actually verify XML parsing 

200 return not all(self.can_parse_xml(source) for source in sources) 

201 

202 def check_single(self, source: str, target: str, unit: Unit) -> bool: 

203 """Check for single phrase, not dealing with plurals.""" 

204 raise NotImplementedError 

205 

206 

207class XMLValidityCheck(BaseXMLCheck): 

208 """Check whether XML in target is valid.""" 

209 

210 check_id = "xml-invalid" 

211 name = gettext_lazy("XML syntax") 

212 description = gettext_lazy("The translation is not valid XML.") 

213 

214 def check_single(self, source: str, target: str, unit: Unit) -> bool: 

215 # Check if source is XML 

216 try: 

217 wrap = self.detect_xml_wrapping(source)[1] 

218 except SyntaxError: 

219 # Source is not valid XML, we give up 

220 return False 

221 

222 # Check target 

223 try: 

224 self.parse_xml(target, wrap) 

225 except SyntaxError: 

226 # Target is not valid XML 

227 return True 

228 

229 return False 

230 

231 

232class XMLTagsCheck(BaseXMLCheck): 

233 """Check whether XML in target matches source.""" 

234 

235 check_id = "xml-tags" 

236 name = gettext_lazy("XML markup") 

237 description = gettext_lazy("XML tags in translation do not match source.") 

238 

239 def check_single(self, source: str, target: str, unit: Unit): 

240 # Check if source is XML 

241 try: 

242 source_tree, wrap = self.detect_xml_wrapping(source) 

243 source_tags = [(x.tag, x.keys()) for x in source_tree.iter()] 

244 except SyntaxError: 

245 # Source is not valid XML, we give up 

246 return False 

247 

248 # Check target 

249 try: 

250 target_tree = self.parse_xml(target, wrap) 

251 target_tags = [(x.tag, x.keys()) for x in target_tree.iter()] 

252 except SyntaxError: 

253 # Target is not valid XML 

254 return False 

255 

256 # Compare tags 

257 return source_tags != target_tags 

258 

259 def check_highlight(self, source: str, unit: Unit): 

260 if self.should_skip(unit): 260 ↛ 262line 260 didn't jump to line 262 because the condition on line 260 was always true

261 return [] 

262 if not self.can_parse_xml(source): 

263 return [] 

264 # Include XML markup 

265 ret = [ 

266 (match.start(), match.end(), match.group()) 

267 for match in XML_MATCH.finditer(source) 

268 ] 

269 # Add XML entities 

270 skipranges = [x[:2] for x in ret] 

271 skipranges.append((len(source), len(source))) 

272 offset = 0 

273 for match in XML_ENTITY_MATCH.finditer(source): 

274 start = match.start() 

275 end = match.end() 

276 while skipranges[offset][1] < end: 

277 offset += 1 

278 # Avoid including entities inside markup 

279 if start > skipranges[offset][0] and end < skipranges[offset][1]: 

280 continue 

281 ret.append((start, end, match.group())) 

282 return ret 

283 

284 

285class MarkdownBaseCheck(TargetCheck): 

286 default_disabled = True 

287 

288 def __init__(self) -> None: 

289 super().__init__() 

290 self.enable_string = "md-text" 

291 

292 

293class MarkdownRefLinkCheck(MarkdownBaseCheck): 

294 check_id = "md-reflink" 

295 name = gettext_lazy("Markdown references") 

296 description = gettext_lazy("Markdown link references do not match source.") 

297 

298 def check_single(self, source: str, target: str, unit: Unit): 

299 src_match = MD_REFLINK.findall(source) 

300 if not src_match: 

301 return False 

302 tgt_match = MD_REFLINK.findall(target) 

303 

304 src_tags = {x[1] for x in src_match} 

305 tgt_tags = {x[1] for x in tgt_match} 

306 

307 return src_tags != tgt_tags 

308 

309 

310class MarkdownLinkCheck(MarkdownBaseCheck): 

311 check_id = "md-link" 

312 name = gettext_lazy("Markdown links") 

313 description = gettext_lazy("Markdown links do not match source.") 

314 

315 def check_single(self, source: str, target: str, unit: Unit): 

316 src_match = MD_LINK.findall(source) 

317 if not src_match: 

318 return False 

319 tgt_match = MD_LINK.findall(target) 

320 

321 # Check number of links 

322 if len(src_match) != len(tgt_match): 

323 return True 

324 

325 # We don't check actual remote link targets as those might 

326 # be localized as well (consider links to Wikipedia). 

327 # Instead we check only relative links and templated ones. 

328 link_start = (".", "#", "{") 

329 tgt_anchors = {x[2] for x in tgt_match if x[2] and x[2][0] in link_start} 

330 src_anchors = {x[2] for x in src_match if x[2] and x[2][0] in link_start} 

331 return tgt_anchors != src_anchors 

332 

333 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None: 

334 if MD_BROKEN_LINK.findall(unit.target): 

335 return [("regex", MD_BROKEN_LINK.pattern, "](", "u")] 

336 return None 

337 

338 

339class MarkdownSyntaxCheck(MarkdownBaseCheck): 

340 check_id = "md-syntax" 

341 name = gettext_lazy("Markdown syntax") 

342 description = gettext_lazy("Markdown syntax does not match source.") 

343 

344 @staticmethod 

345 def extract_match(match): 

346 for i in range(6): 

347 if match[i]: 

348 return match[i] 

349 return None 

350 

351 def check_single(self, source: str, target: str, unit: Unit): 

352 src_tags = {self.extract_match(x) for x in MD_SYNTAX.findall(source)} 

353 tgt_tags = {self.extract_match(x) for x in MD_SYNTAX.findall(target)} 

354 

355 return src_tags != tgt_tags 

356 

357 def check_highlight(self, source: str, unit: Unit): 

358 if self.should_skip(unit): 358 ↛ 360line 358 didn't jump to line 360 because the condition on line 358 was always true

359 return 

360 for match in MD_SYNTAX.finditer(source): 

361 value = "" 

362 for i in range(MD_SYNTAX_GROUPS): 

363 value = match.group(i + 1) 

364 if value: 

365 break 

366 start = match.start() 

367 end = match.end() 

368 yield (start, start + len(value), value) 

369 yield ((end - len(value), end, value if value != "<" else ">")) 

370 

371 

372class URLCheck(TargetCheck): 

373 check_id = "url" 

374 name = gettext_lazy("URL") 

375 description = gettext_lazy("The translation does not contain an URL.") 

376 default_disabled = True 

377 

378 @cached_property 

379 def validator(self): 

380 return URLValidator() 

381 

382 def check_single(self, source: str, target: str, unit: Unit) -> bool: 

383 if not source: 

384 return False 

385 try: 

386 self.validator(target) # pylint: disable=too-many-function-args 

387 except ValidationError: 

388 return True 

389 return False 

390 

391 

392class SafeHTMLCheck(TargetCheck): 

393 check_id = "safe-html" 

394 name = gettext_lazy("Unsafe HTML") 

395 description = gettext_lazy("The translation uses unsafe HTML markup.") 

396 default_disabled = True 

397 

398 def check_single(self, source: str, target: str, unit: Unit): 

399 # Strip MarkDown links 

400 if "md-text" in unit.all_flags: 

401 target = MD_LINK.sub("", target) 

402 

403 sanitizer = HTMLSanitizer() 

404 cleaned_target = sanitizer.clean(target, source, unit.all_flags) 

405 

406 return cleaned_target != target 

407 

408 

409class RSTBaseCheck(TargetCheck): 

410 default_disabled = True 

411 

412 def __init__(self) -> None: 

413 super().__init__() 

414 self.enable_string = "rst-text" 

415 

416 

417@lru_cache(maxsize=512) 

418def extract_rst_references(text: str) -> tuple[dict[str, str], Counter, list[str]]: 

419 memo = SimpleNamespace() 

420 publisher = get_rst_publisher() 

421 document = utils.new_document("", publisher.settings) 

422 memo.reporter = document.reporter 

423 memo.document = document 

424 memo.language = languages.get_language( 

425 document.settings.language_code, document.reporter 

426 ) 

427 inliner = Inliner() 

428 inliner.init_customizations(document.settings) 

429 nodes, system_messages = inliner.parse(text, 0, memo, document) 

430 

431 message_ids = { 

432 message["ids"][0]: Element.astext(message) 

433 for message in system_messages 

434 if message["ids"] 

435 } 

436 result: list[tuple[str, str]] = [] 

437 alltags: list[str] = [] 

438 

439 for node in nodes: 

440 if isinstance(node, problematic) and "refid" in node.attributes: 

441 for rst_role_re in RST_ROLE_RE: 

442 if match := rst_role_re.match(message_ids[node["refid"]]): 

443 alltags.append(node.rawsource) 

444 role = match.group(1) 

445 if role in RST_TRANSLATABLE: 

446 name = f":{role}:" 

447 elif matched := RST_EXPLICIT_TITLE_RE.match( 

448 node.rawsource[len(role) + 3 : -1] 

449 ): 

450 # Exclude title for checking translatable roles 

451 name = f":{role}:`{matched.group(2)}`" 

452 else: 

453 name = node.rawsource 

454 

455 result.append((name, node.rawsource)) 

456 break 

457 elif isinstance(node, (footnote_reference, substitution_reference)): 

458 result.append((node.rawsource, node.rawsource)) 

459 alltags.append(node.rawsource) 

460 elif isinstance(node, reference): 

461 # Ignore the content as it might be localized, just differentiate 

462 # references with a link and without 

463 refuri = node.get("refuri") 

464 name = "`... <...>`_" if refuri else "`...`_" 

465 result.append((name, node.rawsource)) 

466 if refuri: 

467 alltags.append(f"<{refuri}>") 

468 elif isinstance(node, literal): 

469 result.append(("``...``", node.rawsource)) 

470 elif isinstance(node, emphasis): 

471 result.append(("*...*", node.rawsource)) 

472 elif isinstance(node, strong): 

473 result.append(("**...**", node.rawsource)) 

474 

475 return dict(result), Counter(item[0] for item in result), alltags 

476 

477 

478class RSTReferencesCheck(RSTBaseCheck): 

479 check_id = "rst-references" 

480 name = gettext_lazy("Inconsistent reStructuredText") 

481 description = gettext_lazy( 

482 "Inconsistent reStructuredText markup in the translated message." 

483 ) 

484 

485 def get_missing_text(self, values: Iterable[str]) -> StrOrPromise: 

486 return self.get_values_text( 

487 gettext("The following reStructuredText markup is missing: {}"), values 

488 ) 

489 

490 def get_extra_text(self, values: Iterable[str]) -> StrOrPromise: 

491 return self.get_values_text( 

492 gettext("The following reStructuredText markup is extra: {}"), values 

493 ) 

494 

495 def check_single( 

496 self, source: str, target: str, unit: Unit 

497 ) -> bool | MissingExtraDict: 

498 src_references, src_set, _alltags = extract_rst_references(source) 

499 tgt_references, tgt_set, _alltags = extract_rst_references(target) 

500 

501 missing = src_set - tgt_set 

502 extra = tgt_set - src_set 

503 

504 if missing or extra: 

505 return { 

506 "missing": list( 

507 chain.from_iterable( 

508 [src_references[item]] if src_set[item] == 1 else [item] * count 

509 for item, count in missing.items() 

510 ) 

511 ), 

512 "extra": list( 

513 chain.from_iterable( 

514 [tgt_references[item]] if tgt_set[item] == 1 else [item] * count 

515 for item, count in extra.items() 

516 ) 

517 ), 

518 "errors": [], 

519 } 

520 return False 

521 

522 def get_description(self, check_obj: Check) -> StrOrPromise: 

523 unit = check_obj.unit 

524 

525 errors: list[StrOrPromise] = [] 

526 results: MissingExtraDict = cast("MissingExtraDict", defaultdict(list)) 

527 

528 # Merge plurals 

529 for result in self.check_target_generator( 

530 unit.get_source_plurals(), unit.get_target_plurals(), unit 

531 ): 

532 if isinstance(result, dict): 

533 for key, value in result.items(): 

534 results[key].extend(value) 

535 if results: 

536 errors.extend(self.format_result(results)) 

537 if errors: 

538 return format_html_join( 

539 mark_safe("<br />"), 

540 "{}", 

541 ((error,) for error in errors), 

542 ) 

543 return super().get_description(check_obj) 

544 

545 def check_highlight(self, source: str, unit: Unit): 

546 if self.should_skip(unit): 546 ↛ 548line 546 didn't jump to line 548 because the condition on line 546 was always true

547 return 

548 _references, _counter, alltags = extract_rst_references(source) 

549 if not alltags: 

550 return 

551 match_exp = "|".join(re.escape(tag) for tag in alltags) 

552 for match in re.finditer(match_exp, source): 

553 yield match.start(0), match.end(0), match.group(0) 

554 

555 

556@cache 

557def get_rst_publisher() -> Publisher: 

558 parser = Parser() 

559 reader: Reader = Reader(parser) 

560 writer = Writer() 

561 publisher = Publisher(settings=None, reader=reader, parser=parser, writer=writer) 

562 publisher.get_settings( 

563 # Never halt parsing with an exception 

564 halt_level=5, 

565 # Disable warnings 

566 warning_stream=False, 

567 # Do not allow file insertion 

568 file_insertion_enabled=False, 

569 # Following are needed in case django.contrib.admindocs is imported 

570 # and registers own rst tags 

571 default_reference_context="", 

572 link_base="", 

573 ) 

574 return publisher 

575 

576 

577@lru_cache(maxsize=512) 

578def validate_rst_snippet( 

579 snippet: str, source_tags: tuple[str] | None = None 

580) -> tuple[list[str], list[str]]: 

581 publisher = get_rst_publisher() 

582 document = utils.new_document("", publisher.settings) 

583 

584 errors: list[str] = [] 

585 roles: list[str] = [] 

586 

587 def error_collector(data: system_message) -> None: 

588 """Save the error.""" 

589 message = Element.astext(data) 

590 if message.startswith("Unknown target name:") and "`" not in message: 

591 # Translating targets is okay, just catch obvious errors 

592 return 

593 if message.startswith( 

594 ( 

595 # Duplicates Unknown interpreted in our case 

596 "No role entry", 

597 # Can not work on snippets 

598 "Too many autonumbered footnote", 

599 # Can not work on snippets 

600 "Enumerated list start value not ordinal", 

601 # Substitutions are typically defined at the document level 

602 "Undefined substitution referenced", 

603 ) 

604 ): 

605 return 

606 for rst_role_re in RST_ROLE_RE: 

607 if match := rst_role_re.match(message): 

608 role = match.group(1) 

609 roles.append(role) 

610 if source_tags is not None and role in source_tags: 

611 # Skip if the role was found in the source 

612 return 

613 errors.append(message) 

614 

615 document.reporter.attach_observer(error_collector) 

616 cast("Parser", publisher.reader.parser).parse(snippet, document) 

617 transformer = document.transformer 

618 transformer.populate_from_components( 

619 ( 

620 publisher.reader, 

621 cast("Parser", publisher.reader.parser), 

622 publisher.writer, 

623 ) 

624 ) 

625 while transformer.transforms: 

626 if not transformer.sorted: 

627 # Unsorted initially, and whenever a transform is added. 

628 transformer.transforms.sort() 

629 transformer.transforms.reverse() 

630 transformer.sorted = True 

631 priority, transform_class, pending, kwargs = transformer.transforms.pop() 

632 transform = transform_class(transformer.document, startnode=pending) 

633 transform.apply(**kwargs) 

634 transformer.applied.append((priority, transform_class, pending, kwargs)) 

635 return errors, roles 

636 

637 

638class RSTSyntaxCheck(RSTBaseCheck): 

639 check_id = "rst-syntax" 

640 name = gettext_lazy("reStructuredText syntax error") 

641 description = gettext_lazy("reStructuredText syntax error in the translation.") 

642 

643 def check_single( 

644 self, source: str, target: str, unit: Unit 

645 ) -> bool | MissingExtraDict: 

646 _errors, source_roles = validate_rst_snippet(source) 

647 errors, _target_roles = validate_rst_snippet(target, tuple(source_roles)) 

648 

649 if errors: 

650 return {"errors": errors} 

651 return False 

652 

653 def get_description(self, check_obj: Check) -> StrOrPromise: 

654 unit = check_obj.unit 

655 

656 errors: list[StrOrPromise] = [] 

657 results: MissingExtraDict = cast("MissingExtraDict", defaultdict(list)) 

658 

659 # Merge plurals 

660 for result in self.check_target_generator( 

661 unit.get_source_plurals(), unit.get_target_plurals(), unit 

662 ): 

663 if isinstance(result, dict): 

664 for key, value in result.items(): 

665 results[key].extend(value) 

666 if results: 

667 errors.extend(self.format_result(results)) 

668 if errors: 

669 return format_html_join( 

670 mark_safe("<br />"), 

671 "{}", 

672 ((error,) for error in errors), 

673 ) 

674 return super().get_description(check_obj)