Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/markup.py: 29%
381 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5from __future__ import annotations
7import re
8from collections import Counter, defaultdict
9from functools import cache, lru_cache
10from itertools import chain
11from types import SimpleNamespace
12from typing import TYPE_CHECKING, cast
14from django.core.exceptions import ValidationError
15from django.core.validators import URLValidator
16from django.utils.functional import cached_property
17from django.utils.html import format_html_join
18from django.utils.safestring import mark_safe
19from django.utils.translation import gettext, gettext_lazy
20from docutils import utils
21from docutils.core import Publisher
22from docutils.nodes import (
23 Element,
24 emphasis,
25 footnote_reference,
26 literal,
27 problematic,
28 reference,
29 strong,
30 substitution_reference,
31)
32from docutils.parsers.rst import Parser, languages
33from docutils.parsers.rst.states import Inliner
34from docutils.readers.standalone import Reader
35from docutils.writers.null import Writer
37from weblate.checks.base import TargetCheck
38from weblate.utils.html import (
39 MD_BROKEN_LINK,
40 MD_LINK,
41 MD_REFLINK,
42 MD_SYNTAX,
43 MD_SYNTAX_GROUPS,
44 HTMLSanitizer,
45)
46from weblate.utils.xml import parse_xml
48if TYPE_CHECKING: 48 ↛ 49line 48 didn't jump to line 49 because the condition on line 48 was never true
49 from collections.abc import Iterable
51 from django_stubs_ext import StrOrPromise
52 from docutils.nodes import (
53 system_message,
54 )
55 from lxml.etree import _Element
57 from weblate.checks.base import MissingExtraDict
58 from weblate.trans.models import Unit
60 from .base import FixupType
61 from .models import Check
63BBCODE_MATCH = re.compile(
64 r"(?P<start>\[(?P<tag>[^]]+)(@[^]]*)?\])(.*?)(?P<end>\[\/(?P=tag)\])", re.MULTILINE
65)
68XML_MATCH = re.compile(r"<[^>]+>")
69XML_ENTITY_MATCH = re.compile(
70 r"""
71 # Initial &
72 \&
73 (
74 # CharRef
75 \x23[0-9]+
76 |
77 # CharRef
78 \x23x[0-9a-fA-F]+
79 |
80 # EntityRef
81 # NameStartChar
82 [:A-Z_a-z\xC0-\xD6\xD8-\xF6\xF8-\u02FF\u0370-\u037D\u037F-\u1FFF\u200C-\u200D\u2070-\u218F\u2C00-\u2FEF\u3001-\uD7FF\uF900-\uFDCF\uFDF0-\uFFFD\u10000-\uEFFFF]
83 # NameChar
84 [0-9\xB7.:A-Z_a-z\xC0-\xD6\xD8-\xF6\xF8-\u02FF\u0370-\u037D\u037F-\u1FFF\u200C-\u200D\u2070-\u218F\u2C00-\u2FEF\u3001-\uD7FF\uF900-\uFDCF\uFDF0-\uFFFD\u10000-\uEFFFF\u0300-\u036F\u203F-\u2040-]*
85 )
86 # Closing ;
87 ;
88 """,
89 re.VERBOSE,
90)
92# Extracted from Sphinx sphinx/util/docutils.py
93RST_EXPLICIT_TITLE_RE = re.compile(r"^(.+?)\s*(?<!\x00)<(.*?)>$", re.DOTALL)
95# These should be present in translation if present in source, but might be translated
96RST_TRANSLATABLE = {
97 "guilabel",
98 "file",
99 "code",
100 "math",
101 "eq",
102 "abbr",
103 "dfn",
104 "menuselection",
105 "sub",
106 "sup",
107 "kbd",
108 "index",
109 "samp",
110}
112RST_ROLE_RE = [
113 re.compile(r"""Unknown interpreted text role "([^"]*)"\."""),
114 re.compile(r"""Interpreted text role "([^"]*)" not implemented\."""),
115]
118def strip_entities(text):
119 """Strip all HTML entities (we don't care about them)."""
120 return XML_ENTITY_MATCH.sub(" ", text)
123class BBCodeCheck(TargetCheck):
124 """Check for matching bbcode tags."""
126 check_id = "bbcode"
127 name = gettext_lazy("BBCode markup")
128 description = gettext_lazy("BBCode in translation does not match source.")
129 default_disabled = True
131 def __init__(self) -> None:
132 super().__init__()
133 self.enable_string = "bbcode-text"
135 def check_single(self, source: str, target: str, unit: Unit):
136 # Parse source
137 src_match = BBCODE_MATCH.findall(source)
139 # Parse target
140 tgt_match = BBCODE_MATCH.findall(target)
141 if len(src_match) != len(tgt_match):
142 return True
144 src_tags = {x[1] for x in src_match}
145 tgt_tags = {x[1] for x in tgt_match}
147 return src_tags != tgt_tags
149 def check_highlight(self, source: str, unit: Unit):
150 if self.should_skip(unit): 150 ↛ 152line 150 didn't jump to line 152 because the condition on line 150 was always true
151 return
152 for match in BBCODE_MATCH.finditer(source):
153 for tag in ("start", "end"):
154 yield match.start(tag), match.end(tag), match.group(tag)
157class BaseXMLCheck(TargetCheck):
158 def detect_xml_wrapping(self, text: str) -> tuple[_Element, bool]:
159 """Detect whether wrapping is desired."""
160 try:
161 return self.parse_xml(text, True), True
162 except SyntaxError:
163 return self.parse_xml(text, False), False
165 def can_parse_xml(self, text: str) -> bool:
166 try:
167 self.detect_xml_wrapping(text)
168 except SyntaxError:
169 return False
170 return True
172 def parse_xml(self, text: str, wrap: bool) -> _Element:
173 """Parse XML."""
174 text = strip_entities(text)
175 if wrap:
176 text = f"<weblate>{text}</weblate>"
177 return parse_xml(text.encode() if "encoding" in text else text)
179 def should_skip(self, unit: Unit) -> bool:
180 if super().should_skip(unit): 180 ↛ 181line 180 didn't jump to line 181 because the condition on line 180 was never true
181 return True
183 flags = unit.all_flags
185 if "safe-html" in flags: 185 ↛ 186line 185 didn't jump to line 186 because the condition on line 185 was never true
186 return True
188 if "xml-text" in flags: 188 ↛ 189line 188 didn't jump to line 189 because the condition on line 188 was never true
189 return False
191 sources = unit.get_source_plurals()
193 # Quick check if source looks like XML.
194 if all( 194 ↛ 200line 194 didn't jump to line 200 because the condition on line 194 was always true
195 "<" not in source or not XML_MATCH.findall(source) for source in sources
196 ):
197 return True
199 # Actually verify XML parsing
200 return not all(self.can_parse_xml(source) for source in sources)
202 def check_single(self, source: str, target: str, unit: Unit) -> bool:
203 """Check for single phrase, not dealing with plurals."""
204 raise NotImplementedError
207class XMLValidityCheck(BaseXMLCheck):
208 """Check whether XML in target is valid."""
210 check_id = "xml-invalid"
211 name = gettext_lazy("XML syntax")
212 description = gettext_lazy("The translation is not valid XML.")
214 def check_single(self, source: str, target: str, unit: Unit) -> bool:
215 # Check if source is XML
216 try:
217 wrap = self.detect_xml_wrapping(source)[1]
218 except SyntaxError:
219 # Source is not valid XML, we give up
220 return False
222 # Check target
223 try:
224 self.parse_xml(target, wrap)
225 except SyntaxError:
226 # Target is not valid XML
227 return True
229 return False
232class XMLTagsCheck(BaseXMLCheck):
233 """Check whether XML in target matches source."""
235 check_id = "xml-tags"
236 name = gettext_lazy("XML markup")
237 description = gettext_lazy("XML tags in translation do not match source.")
239 def check_single(self, source: str, target: str, unit: Unit):
240 # Check if source is XML
241 try:
242 source_tree, wrap = self.detect_xml_wrapping(source)
243 source_tags = [(x.tag, x.keys()) for x in source_tree.iter()]
244 except SyntaxError:
245 # Source is not valid XML, we give up
246 return False
248 # Check target
249 try:
250 target_tree = self.parse_xml(target, wrap)
251 target_tags = [(x.tag, x.keys()) for x in target_tree.iter()]
252 except SyntaxError:
253 # Target is not valid XML
254 return False
256 # Compare tags
257 return source_tags != target_tags
259 def check_highlight(self, source: str, unit: Unit):
260 if self.should_skip(unit): 260 ↛ 262line 260 didn't jump to line 262 because the condition on line 260 was always true
261 return []
262 if not self.can_parse_xml(source):
263 return []
264 # Include XML markup
265 ret = [
266 (match.start(), match.end(), match.group())
267 for match in XML_MATCH.finditer(source)
268 ]
269 # Add XML entities
270 skipranges = [x[:2] for x in ret]
271 skipranges.append((len(source), len(source)))
272 offset = 0
273 for match in XML_ENTITY_MATCH.finditer(source):
274 start = match.start()
275 end = match.end()
276 while skipranges[offset][1] < end:
277 offset += 1
278 # Avoid including entities inside markup
279 if start > skipranges[offset][0] and end < skipranges[offset][1]:
280 continue
281 ret.append((start, end, match.group()))
282 return ret
285class MarkdownBaseCheck(TargetCheck):
286 default_disabled = True
288 def __init__(self) -> None:
289 super().__init__()
290 self.enable_string = "md-text"
293class MarkdownRefLinkCheck(MarkdownBaseCheck):
294 check_id = "md-reflink"
295 name = gettext_lazy("Markdown references")
296 description = gettext_lazy("Markdown link references do not match source.")
298 def check_single(self, source: str, target: str, unit: Unit):
299 src_match = MD_REFLINK.findall(source)
300 if not src_match:
301 return False
302 tgt_match = MD_REFLINK.findall(target)
304 src_tags = {x[1] for x in src_match}
305 tgt_tags = {x[1] for x in tgt_match}
307 return src_tags != tgt_tags
310class MarkdownLinkCheck(MarkdownBaseCheck):
311 check_id = "md-link"
312 name = gettext_lazy("Markdown links")
313 description = gettext_lazy("Markdown links do not match source.")
315 def check_single(self, source: str, target: str, unit: Unit):
316 src_match = MD_LINK.findall(source)
317 if not src_match:
318 return False
319 tgt_match = MD_LINK.findall(target)
321 # Check number of links
322 if len(src_match) != len(tgt_match):
323 return True
325 # We don't check actual remote link targets as those might
326 # be localized as well (consider links to Wikipedia).
327 # Instead we check only relative links and templated ones.
328 link_start = (".", "#", "{")
329 tgt_anchors = {x[2] for x in tgt_match if x[2] and x[2][0] in link_start}
330 src_anchors = {x[2] for x in src_match if x[2] and x[2][0] in link_start}
331 return tgt_anchors != src_anchors
333 def get_fixup(self, unit: Unit) -> Iterable[FixupType] | None:
334 if MD_BROKEN_LINK.findall(unit.target):
335 return [("regex", MD_BROKEN_LINK.pattern, "](", "u")]
336 return None
339class MarkdownSyntaxCheck(MarkdownBaseCheck):
340 check_id = "md-syntax"
341 name = gettext_lazy("Markdown syntax")
342 description = gettext_lazy("Markdown syntax does not match source.")
344 @staticmethod
345 def extract_match(match):
346 for i in range(6):
347 if match[i]:
348 return match[i]
349 return None
351 def check_single(self, source: str, target: str, unit: Unit):
352 src_tags = {self.extract_match(x) for x in MD_SYNTAX.findall(source)}
353 tgt_tags = {self.extract_match(x) for x in MD_SYNTAX.findall(target)}
355 return src_tags != tgt_tags
357 def check_highlight(self, source: str, unit: Unit):
358 if self.should_skip(unit): 358 ↛ 360line 358 didn't jump to line 360 because the condition on line 358 was always true
359 return
360 for match in MD_SYNTAX.finditer(source):
361 value = ""
362 for i in range(MD_SYNTAX_GROUPS):
363 value = match.group(i + 1)
364 if value:
365 break
366 start = match.start()
367 end = match.end()
368 yield (start, start + len(value), value)
369 yield ((end - len(value), end, value if value != "<" else ">"))
372class URLCheck(TargetCheck):
373 check_id = "url"
374 name = gettext_lazy("URL")
375 description = gettext_lazy("The translation does not contain an URL.")
376 default_disabled = True
378 @cached_property
379 def validator(self):
380 return URLValidator()
382 def check_single(self, source: str, target: str, unit: Unit) -> bool:
383 if not source:
384 return False
385 try:
386 self.validator(target) # pylint: disable=too-many-function-args
387 except ValidationError:
388 return True
389 return False
392class SafeHTMLCheck(TargetCheck):
393 check_id = "safe-html"
394 name = gettext_lazy("Unsafe HTML")
395 description = gettext_lazy("The translation uses unsafe HTML markup.")
396 default_disabled = True
398 def check_single(self, source: str, target: str, unit: Unit):
399 # Strip MarkDown links
400 if "md-text" in unit.all_flags:
401 target = MD_LINK.sub("", target)
403 sanitizer = HTMLSanitizer()
404 cleaned_target = sanitizer.clean(target, source, unit.all_flags)
406 return cleaned_target != target
409class RSTBaseCheck(TargetCheck):
410 default_disabled = True
412 def __init__(self) -> None:
413 super().__init__()
414 self.enable_string = "rst-text"
417@lru_cache(maxsize=512)
418def extract_rst_references(text: str) -> tuple[dict[str, str], Counter, list[str]]:
419 memo = SimpleNamespace()
420 publisher = get_rst_publisher()
421 document = utils.new_document("", publisher.settings)
422 memo.reporter = document.reporter
423 memo.document = document
424 memo.language = languages.get_language(
425 document.settings.language_code, document.reporter
426 )
427 inliner = Inliner()
428 inliner.init_customizations(document.settings)
429 nodes, system_messages = inliner.parse(text, 0, memo, document)
431 message_ids = {
432 message["ids"][0]: Element.astext(message)
433 for message in system_messages
434 if message["ids"]
435 }
436 result: list[tuple[str, str]] = []
437 alltags: list[str] = []
439 for node in nodes:
440 if isinstance(node, problematic) and "refid" in node.attributes:
441 for rst_role_re in RST_ROLE_RE:
442 if match := rst_role_re.match(message_ids[node["refid"]]):
443 alltags.append(node.rawsource)
444 role = match.group(1)
445 if role in RST_TRANSLATABLE:
446 name = f":{role}:"
447 elif matched := RST_EXPLICIT_TITLE_RE.match(
448 node.rawsource[len(role) + 3 : -1]
449 ):
450 # Exclude title for checking translatable roles
451 name = f":{role}:`{matched.group(2)}`"
452 else:
453 name = node.rawsource
455 result.append((name, node.rawsource))
456 break
457 elif isinstance(node, (footnote_reference, substitution_reference)):
458 result.append((node.rawsource, node.rawsource))
459 alltags.append(node.rawsource)
460 elif isinstance(node, reference):
461 # Ignore the content as it might be localized, just differentiate
462 # references with a link and without
463 refuri = node.get("refuri")
464 name = "`... <...>`_" if refuri else "`...`_"
465 result.append((name, node.rawsource))
466 if refuri:
467 alltags.append(f"<{refuri}>")
468 elif isinstance(node, literal):
469 result.append(("``...``", node.rawsource))
470 elif isinstance(node, emphasis):
471 result.append(("*...*", node.rawsource))
472 elif isinstance(node, strong):
473 result.append(("**...**", node.rawsource))
475 return dict(result), Counter(item[0] for item in result), alltags
478class RSTReferencesCheck(RSTBaseCheck):
479 check_id = "rst-references"
480 name = gettext_lazy("Inconsistent reStructuredText")
481 description = gettext_lazy(
482 "Inconsistent reStructuredText markup in the translated message."
483 )
485 def get_missing_text(self, values: Iterable[str]) -> StrOrPromise:
486 return self.get_values_text(
487 gettext("The following reStructuredText markup is missing: {}"), values
488 )
490 def get_extra_text(self, values: Iterable[str]) -> StrOrPromise:
491 return self.get_values_text(
492 gettext("The following reStructuredText markup is extra: {}"), values
493 )
495 def check_single(
496 self, source: str, target: str, unit: Unit
497 ) -> bool | MissingExtraDict:
498 src_references, src_set, _alltags = extract_rst_references(source)
499 tgt_references, tgt_set, _alltags = extract_rst_references(target)
501 missing = src_set - tgt_set
502 extra = tgt_set - src_set
504 if missing or extra:
505 return {
506 "missing": list(
507 chain.from_iterable(
508 [src_references[item]] if src_set[item] == 1 else [item] * count
509 for item, count in missing.items()
510 )
511 ),
512 "extra": list(
513 chain.from_iterable(
514 [tgt_references[item]] if tgt_set[item] == 1 else [item] * count
515 for item, count in extra.items()
516 )
517 ),
518 "errors": [],
519 }
520 return False
522 def get_description(self, check_obj: Check) -> StrOrPromise:
523 unit = check_obj.unit
525 errors: list[StrOrPromise] = []
526 results: MissingExtraDict = cast("MissingExtraDict", defaultdict(list))
528 # Merge plurals
529 for result in self.check_target_generator(
530 unit.get_source_plurals(), unit.get_target_plurals(), unit
531 ):
532 if isinstance(result, dict):
533 for key, value in result.items():
534 results[key].extend(value)
535 if results:
536 errors.extend(self.format_result(results))
537 if errors:
538 return format_html_join(
539 mark_safe("<br />"),
540 "{}",
541 ((error,) for error in errors),
542 )
543 return super().get_description(check_obj)
545 def check_highlight(self, source: str, unit: Unit):
546 if self.should_skip(unit): 546 ↛ 548line 546 didn't jump to line 548 because the condition on line 546 was always true
547 return
548 _references, _counter, alltags = extract_rst_references(source)
549 if not alltags:
550 return
551 match_exp = "|".join(re.escape(tag) for tag in alltags)
552 for match in re.finditer(match_exp, source):
553 yield match.start(0), match.end(0), match.group(0)
556@cache
557def get_rst_publisher() -> Publisher:
558 parser = Parser()
559 reader: Reader = Reader(parser)
560 writer = Writer()
561 publisher = Publisher(settings=None, reader=reader, parser=parser, writer=writer)
562 publisher.get_settings(
563 # Never halt parsing with an exception
564 halt_level=5,
565 # Disable warnings
566 warning_stream=False,
567 # Do not allow file insertion
568 file_insertion_enabled=False,
569 # Following are needed in case django.contrib.admindocs is imported
570 # and registers own rst tags
571 default_reference_context="",
572 link_base="",
573 )
574 return publisher
577@lru_cache(maxsize=512)
578def validate_rst_snippet(
579 snippet: str, source_tags: tuple[str] | None = None
580) -> tuple[list[str], list[str]]:
581 publisher = get_rst_publisher()
582 document = utils.new_document("", publisher.settings)
584 errors: list[str] = []
585 roles: list[str] = []
587 def error_collector(data: system_message) -> None:
588 """Save the error."""
589 message = Element.astext(data)
590 if message.startswith("Unknown target name:") and "`" not in message:
591 # Translating targets is okay, just catch obvious errors
592 return
593 if message.startswith(
594 (
595 # Duplicates Unknown interpreted in our case
596 "No role entry",
597 # Can not work on snippets
598 "Too many autonumbered footnote",
599 # Can not work on snippets
600 "Enumerated list start value not ordinal",
601 # Substitutions are typically defined at the document level
602 "Undefined substitution referenced",
603 )
604 ):
605 return
606 for rst_role_re in RST_ROLE_RE:
607 if match := rst_role_re.match(message):
608 role = match.group(1)
609 roles.append(role)
610 if source_tags is not None and role in source_tags:
611 # Skip if the role was found in the source
612 return
613 errors.append(message)
615 document.reporter.attach_observer(error_collector)
616 cast("Parser", publisher.reader.parser).parse(snippet, document)
617 transformer = document.transformer
618 transformer.populate_from_components(
619 (
620 publisher.reader,
621 cast("Parser", publisher.reader.parser),
622 publisher.writer,
623 )
624 )
625 while transformer.transforms:
626 if not transformer.sorted:
627 # Unsorted initially, and whenever a transform is added.
628 transformer.transforms.sort()
629 transformer.transforms.reverse()
630 transformer.sorted = True
631 priority, transform_class, pending, kwargs = transformer.transforms.pop()
632 transform = transform_class(transformer.document, startnode=pending)
633 transform.apply(**kwargs)
634 transformer.applied.append((priority, transform_class, pending, kwargs))
635 return errors, roles
638class RSTSyntaxCheck(RSTBaseCheck):
639 check_id = "rst-syntax"
640 name = gettext_lazy("reStructuredText syntax error")
641 description = gettext_lazy("reStructuredText syntax error in the translation.")
643 def check_single(
644 self, source: str, target: str, unit: Unit
645 ) -> bool | MissingExtraDict:
646 _errors, source_roles = validate_rst_snippet(source)
647 errors, _target_roles = validate_rst_snippet(target, tuple(source_roles))
649 if errors:
650 return {"errors": errors}
651 return False
653 def get_description(self, check_obj: Check) -> StrOrPromise:
654 unit = check_obj.unit
656 errors: list[StrOrPromise] = []
657 results: MissingExtraDict = cast("MissingExtraDict", defaultdict(list))
659 # Merge plurals
660 for result in self.check_target_generator(
661 unit.get_source_plurals(), unit.get_target_plurals(), unit
662 ):
663 if isinstance(result, dict):
664 for key, value in result.items():
665 results[key].extend(value)
666 if results:
667 errors.extend(self.format_result(results))
668 if errors:
669 return format_html_join(
670 mark_safe("<br />"),
671 "{}",
672 ((error,) for error in errors),
673 )
674 return super().get_description(check_obj)