Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/fluent/inner_html.py: 24%
483 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Henry Wilkes <henry@torproject.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5from __future__ import annotations
7import html
8import re
9from typing import TYPE_CHECKING
11from django.utils.translation import gettext, gettext_lazy
13from weblate.checks.base import SourceCheck, TargetCheck
14from weblate.checks.fluent.utils import (
15 FluentPatterns,
16 FluentUnitConverter,
17 format_html_code,
18 format_html_error_list,
19 translation_from_check,
20 variant_name,
21)
22from weblate.utils.html import format_html_join_comma, list_to_tuples
24if TYPE_CHECKING: 24 ↛ 25line 24 didn't jump to line 25 because the condition on line 24 was never true
25 from collections.abc import Iterable, Iterator
27 from django.utils.safestring import SafeString
28 from django_stubs_ext import StrOrPromise
29 from translate.storage.fluent import FluentSelectorBranch
31 from weblate.checks.fluent.utils import CheckModel, HighlightsType, TransUnitModel
33# Standard html elements that do not have content or end tags.
34_VOID_ELEMENTS = [
35 "area",
36 "base",
37 "br",
38 "col",
39 "embed",
40 "hr",
41 "img",
42 "input",
43 "link",
44 "meta",
45 "source",
46 "track",
47 "wbr",
48]
51class _HTMLNode:
52 """Represents a found HTML node."""
54 def __init__(self, tag: str, parent: _HTMLNode | None) -> None:
55 self.parent = parent
56 self.children: list[_HTMLNode] = []
57 self.tag = tag
58 # Keep attributes as a list to keep duplicates.
59 self.attributes: dict[str, str] = {}
60 if parent:
61 parent.children.append(self)
63 def descendants(self) -> Iterator[_HTMLNode]:
64 """All descendant nodes, not including itself."""
65 for child in self.children:
66 yield child
67 yield from child.descendants()
69 def matches(self, other: _HTMLNode) -> bool:
70 """Whether the two nodes match."""
71 if self.tag != other.tag:
72 return False
73 if self.attributes != other.attributes:
74 return False
75 # Two nodes are only match if their ancestors match.
76 if self.parent is None:
77 return other.parent is None
78 if other.parent is None:
79 return False
80 return self.parent.matches(other.parent)
82 def tags(self) -> tuple[str, str]:
83 """Get the start and end tags for this node."""
84 start = f"<{self.tag}"
85 for attr, val in self.attributes.items():
86 if '"' in val:
87 start += f" {attr}='{val}'"
88 else:
89 start += f' {attr}="{val}"'
90 if self.tag.lower() in _VOID_ELEMENTS:
91 start += "/>"
92 return (start, "")
93 start += ">"
94 return (start, f"</{self.tag}>")
96 def present(self) -> str:
97 """Present this node and all of its ancestors for the user."""
98 start = ""
99 end = ""
101 serialized, end = self.tags()
102 if end:
103 serialized += "…" + end
105 node = self.parent
106 while node and node.parent:
107 # Also show the parent elements, minus the root.
108 start, end = node.tags()
109 serialized = start + serialized + end
110 node = node.parent
112 return serialized
115class _HTMLParseError(BaseException):
116 """Generic error class for our internal parsing errors."""
118 def description(self) -> SafeString:
119 raise NotImplementedError
122class _HTMLFluentReferenceTagError(_HTMLParseError):
123 def __init__(self, sequence: str) -> None:
124 self.sequence = sequence
126 def description(self) -> SafeString:
127 return format_html_code(
128 gettext(
129 "The Fluent reference in {sequence} may expand into a HTML "
130 "tag. Maybe use {suggestion}."
131 ),
132 sequence=self.sequence,
133 suggestion=self.sequence.replace("<", "<", 1),
134 )
137class _HTMLFluentReferenceCharacterReferenceError(_HTMLParseError):
138 def __init__(self, sequence: str) -> None:
139 self.sequence = sequence
141 def description(self) -> SafeString:
142 return format_html_code(
143 gettext(
144 "The Fluent reference in {sequence} may expand into a HTML "
145 "character reference. Maybe use {suggestion}."
146 ),
147 sequence=self.sequence,
148 suggestion=self.sequence.replace("&", "&", 1),
149 )
152class _HTMLInvalidTagSequenceError(_HTMLParseError):
153 """
154 Base class for parsing errors in a tag-like sequence.
156 We assume the user may not have wanted to create a HTML tag, so we will show
157 a suggestion on how to avoid it.
158 """
160 def __init__(self, sequence: str) -> None:
161 self.sequence = sequence
163 @property
164 def suggestion(self) -> str:
165 return self.sequence.replace("<", "<", 1)
168class _HTMLInvalidEndTagError(_HTMLInvalidTagSequenceError):
169 def description(self) -> SafeString:
170 return format_html_code(
171 gettext(
172 "The sequence {sequence} begins with a HTML closing tag, "
173 "but the name or syntax is not valid. "
174 "If you do not want a closing tag, use {suggestion}."
175 ),
176 sequence=self.sequence,
177 suggestion=self.suggestion,
178 )
181class _HTMLInvalidStartTagNameError(_HTMLInvalidTagSequenceError):
182 def description(self) -> SafeString:
183 return format_html_code(
184 gettext(
185 "The sequence {sequence} begins with a HTML tag, "
186 "but the name is not valid. "
187 "If you do not want to begin a HTML tag, use {suggestion}."
188 ),
189 sequence=self.sequence,
190 suggestion=self.suggestion,
191 )
194class _HTMLStartTagNotClosedError(_HTMLInvalidTagSequenceError):
195 def description(self) -> SafeString:
196 return format_html_code(
197 gettext(
198 "The sequence {sequence} begins with a HTML tag, "
199 "but the tag is never closed by {close}. "
200 "If you do not want to begin a HTML tag, use {suggestion}."
201 ),
202 sequence=self.sequence,
203 close=">",
204 suggestion=self.suggestion,
205 )
208class _HTMLTagTypeNotAllowedError(_HTMLInvalidTagSequenceError):
209 def description(self) -> SafeString:
210 return format_html_code(
211 gettext(
212 "Fluent inner HTML should not include {sequence}. "
213 "Maybe use {suggestion}."
214 ),
215 sequence=self.sequence,
216 suggestion=self.suggestion,
217 )
220class _HTMLUnexpectedAttributeError(_HTMLInvalidTagSequenceError):
221 def __init__(self, tag: str, attribute: str) -> None:
222 super().__init__(tag)
223 self.attribute = attribute
225 def description(self) -> SafeString:
226 return format_html_code(
227 gettext(
228 "The sequence {sequence} begins with a HTML tag, "
229 "but the sequence {attribute} is not a valid attribute "
230 "with a value. "
231 "If you do not want to begin a HTML tag, use {suggestion}."
232 ),
233 sequence=self.sequence,
234 attribute=self.attribute,
235 suggestion=self.suggestion,
236 )
239class _HTMLInvalidAttributeNameError(_HTMLParseError):
240 def __init__(self, tag: str, name: str) -> None:
241 self.tag = tag
242 self.name = name
244 def description(self) -> SafeString:
245 return format_html_code(
246 gettext("The HTML attribute name {name} for the {tag} tag is not valid."),
247 name=self.name,
248 tag=f"<{self.tag}>",
249 )
252class _HTMLInvalidAttributeValueError(_HTMLParseError):
253 def __init__(self, tag: str, name: str, value: str) -> None:
254 self.tag = tag
255 self.name = name
256 self.value = value
258 def description(self) -> SafeString:
259 return format_html_code(
260 gettext(
261 "The HTML {name} attribute value {value} for the {tag} "
262 "tag is not a valid quoted value."
263 ),
264 name=self.name,
265 value=self.value,
266 tag=f"<{self.tag}>",
267 )
270class _HTMLDuplicateAttributeNameError(_HTMLParseError):
271 def __init__(self, tag: str, name: str) -> None:
272 self.tag = tag
273 self.name = name
275 def description(self) -> SafeString:
276 return format_html_code(
277 gettext("The HTML {name} attribute appears twice for the {tag} tag."),
278 name=self.name,
279 tag=f"<{self.tag}>",
280 )
283class _HTMLUnmatchedEndTagError(_HTMLParseError):
284 def __init__(self, tag: str) -> None:
285 self.tag = tag
287 def description(self) -> SafeString:
288 return format_html_code(
289 gettext("Unmatched HTML end tag: {tag}."), tag=f"</{self.tag}>"
290 )
293class _HTMLVoidEndTagError(_HTMLParseError):
294 def __init__(self, tag: str) -> None:
295 self.tag = tag
297 def description(self) -> SafeString:
298 return format_html_code(
299 gettext("The HTML {name} void element should not have an end tag: {end}."),
300 name=self.tag,
301 end=f"</{self.tag}>",
302 )
305class _HTMLNonVoidSelfClosedError(_HTMLParseError):
306 def __init__(self, tag: str) -> None:
307 self.tag = tag
309 def description(self) -> SafeString:
310 return format_html_code(
311 gettext(
312 "The HTML {name} element is not a known void element, so "
313 "should not have a self-closing tag: {self_close}."
314 ),
315 name=self.tag,
316 self_close=f"<{self.tag}/>",
317 )
320class _HTMLMissingEndTagError(_HTMLParseError):
321 def __init__(self, tag: str) -> None:
322 self.tag = tag
324 def description(self) -> SafeString:
325 return format_html_code(
326 gettext("The HTML {tag} tag is missing a matching end tag: {end}."),
327 tag=f"<{self.tag}>",
328 end=f"</{self.tag}>",
329 )
332class _HTMLUnexpectedCharacterReferenceError(_HTMLParseError):
333 def __init__(self, sequence: str) -> None:
334 self.sequence = sequence
336 def description(self) -> SafeString:
337 # We primarily assume the user did not intend to create a character
338 # reference.
339 suggestion = self.sequence.replace("&", "&", 1)
340 return format_html_code(
341 gettext(
342 "The sequence {sequence} will begin a HTML character "
343 "reference, but does not end with {semicolon}. "
344 "If you do not want a character reference, use {suggestion}."
345 ),
346 sequence=self.sequence,
347 semicolon=";",
348 suggestion=suggestion,
349 )
352class _HTMLInvalidCharacterReferenceError(_HTMLParseError):
353 def __init__(self, sequence: str) -> None:
354 self.sequence = sequence
356 def description(self) -> SafeString:
357 # We primarily assume the user did not intend to create a character
358 # reference.
359 return format_html_code(
360 gettext("The sequence {sequence} is not a valid HTML character reference."),
361 sequence=self.sequence,
362 )
365class _HTMLSourcePosition:
366 """Keeps track of some source string and a position in that string."""
368 def __init__(self, serialized: str) -> None:
369 self.pos = 0
370 self.source = ""
371 self._start_end_literals: list[tuple[int, int]] = []
373 # Since FluentPatternVariants correspond to flat Patterns, we
374 # don't expect any SelectExpressions, but it might contain other
375 # "flat" references. E.g.
376 # contact { $person } at { 123 }
377 #
378 # We need this to be valid HTML *after* expanding the literals.
379 for _pos, text, literal in FluentPatterns.split_literal_expressions(serialized):
380 self.source += text
381 if literal:
382 start = len(self.source)
383 self._start_end_literals.append((start, start + len(literal)))
384 self.source += literal
386 def reset(self) -> None:
387 """Reset the position back to the start."""
388 self.pos = 0
390 def at_literal(self) -> bool:
391 """Whether the current position is within a Fluent literal or not."""
392 return any(
393 self.pos >= start and self.pos < end
394 for start, end in self._start_end_literals
395 )
397 def match(self, regex: re.Pattern) -> re.Match | None:
398 """
399 Try and match with the given regular expression.
401 The match is performed at the current source position.
403 If the expressions match, the source position will automatically move
404 forward to the end of the match.
405 """
406 match = regex.match(self.source, self.pos)
407 if not match:
408 return None
409 self.pos = match.end()
410 return match
412 def get(self, regex: re.Pattern) -> str:
413 """
414 Get the matched text for the given regular expression.
416 The text is fetched from the current source position, without changing
417 its value.
419 If nothing is matched, then an empty string is returned.
420 """
421 match = regex.match(self.source, self.pos)
422 if not match:
423 return ""
424 return match.group()
426 def peak_matches(self, regex: re.Pattern) -> bool:
427 """
428 Peak whether the we match with the given regular expression.
430 The match is performed at the current source position.
432 If there is a match, this will return True without changing the source
433 position. Otherwise it will return False.
434 """
435 return regex.match(self.source, self.pos) is not None
438class _CountedNodes:
439 """
440 Tracks the number of matching _HTMLNodes in a list.
442 This will collect matching nodes together so they can be counted and
443 compared, without caring about the order.
444 """
446 def __init__(self, node_list: Iterable[_HTMLNode]) -> None:
447 # Each entry in matching_nodes is a list of nodes that all match. The
448 # length of this list is the count.
449 self.matching_nodes: list[list[_HTMLNode]] = []
450 for node in node_list:
451 add = True
452 for other_nodes in self.matching_nodes:
453 if other_nodes[0].matches(node):
454 add = False
455 other_nodes.append(node)
456 break
457 if add:
458 self.matching_nodes.append([node])
460 def matches(self, other: _CountedNodes) -> bool:
461 """
462 Whether the two instances match.
464 This will match if the two instances have matching _HTMLNodes with the
465 same count in both.
466 """
467 if len(self.matching_nodes) != len(other.matching_nodes):
468 return False
469 for nodes in self.matching_nodes:
470 has_match = False
471 for _index, other_nodes in enumerate(other.matching_nodes):
472 if nodes[0].matches(other_nodes[0]):
473 if len(nodes) != len(other_nodes):
474 return False
475 has_match = True
476 break
477 if not has_match:
478 return False
479 # Both are the same length and each node in self was matched, so there
480 # shouldn't be any nodes in other that were not matched once.
481 return True
483 def count(self, node: _HTMLNode) -> int:
484 """Count how many nodes match the given node."""
485 for other_nodes in self.matching_nodes:
486 if node.matches(other_nodes[0]):
487 return len(other_nodes)
488 return 0
491class _VariantNodes:
492 """Represents a variant string and the _HTMLNodes it contains."""
494 def __init__(
495 self,
496 path: list[FluentSelectorBranch],
497 root: _HTMLNode,
498 ) -> None:
499 self.path = path
500 self.root = root
501 self._nodes: list[_HTMLNode] | None = None
502 self._counted_nodes: _CountedNodes | None = None
504 @property
505 def nodes(self) -> list[_HTMLNode]:
506 """The nodes found in this variant."""
507 if self._nodes is None:
508 self._nodes = list(self.root.descendants())
509 return self._nodes
511 @property
512 def counted_nodes(self) -> _CountedNodes:
513 """The nodes found in this variant, grouped with matching nodes."""
514 if self._counted_nodes is None:
515 self._counted_nodes = _CountedNodes(self.nodes)
516 return self._counted_nodes
518 def name(self) -> str:
519 """Generate name for this variant."""
520 return variant_name(self.path)
523class _FluentInnerHTMLCheck:
524 """
525 Check that we have valid inner HTML.
527 This will check that the given translation unit's source or target has inner
528 HTML that will not lead to a loss of content when parsed.
529 """
531 # Pattern to search for the next tag.
532 _NEXT_TAG_REGEX = re.compile(r"[^<]*<")
533 # Pattern that starts an end-tag.
534 _OPEN_END_TAG_REGEX = re.compile(r"\/")
536 # Do not allow processing instructions, or comments or CDATA or DOCTYPE.
537 _TAG_NOT_ALLOWED_FIRST_CHAR_REGEX = re.compile(r"[!?]")
539 # First character of tag must be ASCII alpha, as per HTML spec. Unlike the
540 # spec, we also restrict the other characters to be ASCII alphanumeric or
541 # "-".
542 _TAG_FIRST_CHAR = r"[a-zA-Z]"
543 _TAG_FIRST_CHAR_REGEX = re.compile(_TAG_FIRST_CHAR)
544 _TAG_PATTERN = _TAG_FIRST_CHAR + r"[a-zA-Z0-9-]*"
546 # Only allow a limited set of "blank" characters within a tag.
547 _BLANK_CHAR = r"[ \t\n]"
548 # Match either the closing of a start-tag, or some whitespace, or the end of
549 # the string.
550 _CLOSE_BLANK_OR_END_PATTERN = (
551 r"("
552 + (_BLANK_CHAR + r"*(?P<close>\/?>)") # Blanks followed by ">" or "/>",
553 + r"|" # or
554 + (_BLANK_CHAR + r"*(?P<eof>$)") # end of the string,
555 + r"|" # or
556 + (_BLANK_CHAR + r"+") # a required blank.
557 # NOTE: The <eof> group will match before the blank group. E.g. " x"
558 # will match blank, but " " will match <eof>.
559 + r")"
560 )
561 _START_TAG_REGEX = re.compile(
562 r"(?P<tag>" + _TAG_PATTERN + r")" + _CLOSE_BLANK_OR_END_PATTERN
563 )
564 # Only allow a limited set of characters for attribute names, which should
565 # cover HTML attributes and "data-" attributes.
566 # Also require ending with an "=".
567 _ATTRIBUTE_NAME_PATTERN = r"(?P<name>[a-zA-Z][a-zA-Z0-9_.:-]*)="
568 # Only allow quoted values.
569 # NOTE: We do not use a raw string since we want the '\"' to become a
570 # literal '"'.
571 _ATTRIBUTE_VALUE_PATTERN = "('(?P<value1>[^']*)'|\"(?P<value2>[^\"]*)\")"
572 _ATTRIBUTE_NAME_REGEX = re.compile(_ATTRIBUTE_NAME_PATTERN)
573 _ATTRIBUTE_VALUE_REGEX = re.compile(
574 _ATTRIBUTE_VALUE_PATTERN
575 # Followed by a blank or the closing character.
576 + _CLOSE_BLANK_OR_END_PATTERN
577 )
578 _END_TAG_REGEX = re.compile(r"(?P<tag>" + _TAG_PATTERN + r")" + _BLANK_CHAR + r"*>")
580 # Pattern used to pull text within a suspected tag up until the next
581 # attribute or the tag closes.
582 _NON_BLANK_OR_CLOSE_REGEX = re.compile(r"[^ \t\n>]*")
584 _FLUENT_REF_FIRST_CHAR_REGEX = re.compile(r"\{")
585 # This pattern will match basic fluent references, but will break down for
586 # references that contain sub-placeables or "}" literals. However, we only
587 # use this for warning messages, so it is good enough.
588 # Similarly we make the last "}" optional, even though this would indicate
589 # invalid Fluent syntax, but we want a guaranteed match if we already match
590 _FLUENT_REF_REGEX = re.compile(r"\{[^}]*\}?")
592 # For highlighting tags.
593 ALL_TAGS_REGEX = re.compile(
594 # Start tag.
595 r"<"
596 + _TAG_PATTERN
597 # Attributes.
598 + r"("
599 + (_BLANK_CHAR + r"+" + _ATTRIBUTE_NAME_PATTERN + _ATTRIBUTE_VALUE_PATTERN)
600 + r")*"
601 # Close start tag.
602 + (_BLANK_CHAR + r"*\/?>")
603 + r"|"
604 # End tag.
605 + (r"<\/" + _TAG_PATTERN + _BLANK_CHAR + r"*>")
606 )
608 @classmethod
609 def _non_blank_or_close(cls, source: _HTMLSourcePosition) -> str:
610 """
611 Get all the non-blank and non ">" characters found at the start.
613 This is used to grab some joined sequence of characters within a HTML
614 tag to show back to the user.
615 """
616 return source.get(cls._NON_BLANK_OR_CLOSE_REGEX)
618 @classmethod
619 def _parse_end_tag(
620 cls, source: _HTMLSourcePosition, open_nodes: list[_HTMLNode]
621 ) -> None:
622 """Parse an end tag, starting after the "</"."""
623 end_tag_match = source.match(cls._END_TAG_REGEX)
624 if not end_tag_match:
625 # May correspond to using a non-ASCII alphanumeric value in the tag
626 # name, which whilst technically allowed for HTML, is not allowed by
627 # this check.
628 #
629 # Otherwise, this is expected to correspond to some HTML parsing
630 # error, like:
631 # + invalid-first-character-of-tag-name
632 # + eof-in-tag
633 # + end-tag-with-attributes
634 # + missing-end-tag-name
635 #
636 # In which cases some content will be *lost* when parsed as HTML by
637 # being commented out or ignored.
638 #
639 # Some parsing errors, like eof-before-tag-name, will not lead to a
640 # loss of content, but aren't allowed here for consistency.
641 raise _HTMLInvalidEndTagError("</" + cls._non_blank_or_close(source))
643 tag = end_tag_match.group("tag")
645 if tag.lower() in _VOID_ELEMENTS:
646 raise _HTMLVoidEndTagError(tag)
648 node = open_nodes.pop()
649 # Never close our dummy "root" element. So raise error if open_nodes is
650 # now empty.
651 if not open_nodes or node.tag != tag:
652 raise _HTMLUnmatchedEndTagError(tag)
654 @classmethod
655 def _parse_attribute(
656 cls,
657 source: _HTMLSourcePosition,
658 node: _HTMLNode,
659 ) -> re.Match:
660 """Parse a start tag attribute, starting at the attribute name."""
661 name_match = source.match(cls._ATTRIBUTE_NAME_REGEX)
662 if not name_match:
663 non_blank = cls._non_blank_or_close(source)
664 if "=" not in non_blank:
665 # Doesn't look like an attribute with a value.
666 raise _HTMLUnexpectedAttributeError("<" + node.tag, non_blank)
667 raise _HTMLInvalidAttributeNameError(
668 node.tag,
669 non_blank[: non_blank.index("=")],
670 )
672 name = name_match.group("name")
673 if name in node.attributes:
674 raise _HTMLDuplicateAttributeNameError(node.tag, name)
675 value_match = source.match(cls._ATTRIBUTE_VALUE_REGEX)
676 if not value_match:
677 raise _HTMLInvalidAttributeValueError(
678 node.tag, name, cls._non_blank_or_close(source)
679 )
681 value = value_match.group("value1")
682 if value is None:
683 value = value_match.group("value2")
685 node.attributes[name] = value
687 return value_match
689 @classmethod
690 def _parse_start_tag(
691 cls, source: _HTMLSourcePosition, open_nodes: list[_HTMLNode]
692 ) -> None:
693 """Parse a start tag, starting after the opening "<"."""
694 if (
695 # If we are pointing to a literal character, then this "{" should
696 # not be part of a fluent reference, but be a literal "{" instead.
697 not source.at_literal()
698 and source.peak_matches(cls._FLUENT_REF_FIRST_CHAR_REGEX)
699 ):
700 # Starts a Fluent reference. E.g.
701 # contact <{ $name }
702 # If the "$name" starts with an ASCII alpha it could expand to
703 # a tag.
704 #
705 # NOTE: This won't capture all cases where the reference might
706 # expand into some HTML, but generally we expect the fluent
707 # application to ensure their references are sanitized for
708 # inner HTML. But this seems like a case where a sanitized value
709 # might become unintentionally bad.
710 #
711 # NOTE: If the fluent reference appears elsewhere within the tag
712 # it will be invalid (not a valid tag name or attribute name) except
713 # within a quoted attribute value. A quoted attribute value can
714 # still cause problems when it is expanded with the value. E.g. if
715 # the value closes the quotes and injects parts, but we are not
716 # trying to protect against this.
717 raise _HTMLFluentReferenceTagError("<" + source.get(cls._FLUENT_REF_REGEX))
719 tag_not_allowed_match = source.match(cls._TAG_NOT_ALLOWED_FIRST_CHAR_REGEX)
720 if tag_not_allowed_match:
721 raise _HTMLTagTypeNotAllowedError("<" + tag_not_allowed_match.group())
723 if not source.peak_matches(cls._TAG_FIRST_CHAR_REGEX):
724 # Corresponds to the HTML parsing errors
725 # + invalid-first-character-of-tag-name, and
726 # + eof-before-tag-name
727 # for a start tag, so the "<" will be treated as text content.
728 # So can keep this "<" character and move on.
729 return
731 tag_match = source.match(cls._START_TAG_REGEX)
732 if not tag_match:
733 raise _HTMLInvalidStartTagNameError("<" + cls._non_blank_or_close(source))
735 tag = tag_match.group("tag")
737 node = _HTMLNode(tag, open_nodes[-1])
739 end_match = tag_match
740 while True:
741 closing_part = end_match.group("close")
742 if closing_part is not None:
743 # Void element has no content, so is not added to the list
744 # of open_nodes. Being self-closing is optional.
745 if tag.lower() not in _VOID_ELEMENTS:
746 # Non-void elements should not be self-closing.
747 if closing_part == "/>":
748 raise _HTMLNonVoidSelfClosedError(tag)
749 open_nodes.append(node)
750 return
752 if end_match.group("eof") is not None:
753 # Corresponds to HTML parsing error eof-in-tag.
754 # We have some blank, but then reach the end of the string
755 # before the tag is closed.
756 raise _HTMLStartTagNotClosedError("<" + tag)
758 end_match = cls._parse_attribute(source, node)
760 # Check for:
761 # + valid character refs (like "<", "½", "P" or "€"), or
762 # + character refs that may expand in some way, but likely were not intended
763 # by the user because they are missing the ";" (like "ðical"), or
764 # + sequences that look like character refs that the user wants to expand
765 # (because they are alphanumeric and end with ";"), but may not actually
766 # expand as expected.
767 _CHARACTER_REFS_REGEX = re.compile(r"[^&]*(?P<ref>&#?[a-zA-Z0-9]*;?)")
769 @classmethod
770 def _check_character_refs(cls, source: _HTMLSourcePosition) -> None:
771 """
772 Check that the character references found in the source.
774 A character reference must be either intentional (with a semicolon) and
775 valid, or unintentional but will not expand into a reference under the
776 HTML spec, so are safe to keep in HTML.
777 """
778 # NOTE: The HTML5 spec (13.2.5.73 Named character reference state), for
779 # historical reasons, will treat character references that do not end
780 # with a ";", but are followed by an alphanumeric or "=", differently
781 # depending on whether we are in text content or an attribute value.
782 # E.g. "<x" will become "<x" for text content but will remain as
783 # "<x" for an attribute value.
784 # Even though "<x" will not expand for an attribute, we still want to
785 # raise an error against it for consistency. So we treat this the same
786 # as text content.
787 while True:
788 # Will always match at least "&" if it exists.
789 ref_match = source.match(cls._CHARACTER_REFS_REGEX)
790 if not ref_match:
791 return
793 sequence = ref_match.group("ref")
794 closes_with_semicolon = sequence[-1] == ";"
796 if (
797 not closes_with_semicolon
798 # If we are pointing to a literal character, then this "{"
799 # should not be part of a fluent reference, but be a literal "{"
800 # instead, which will break the HTML character reference.
801 and not source.at_literal()
802 # Next character is "{".
803 and source.peak_matches(cls._FLUENT_REF_FIRST_CHAR_REGEX)
804 ):
805 # Could expand to a character reference when the fluent
806 # reference is substituted. E.g.
807 # &l{ $var }
808 # or
809 # &#{ $var }
810 raise _HTMLFluentReferenceCharacterReferenceError(
811 sequence + source.get(cls._FLUENT_REF_REGEX)
812 )
814 expanded = html.unescape(sequence)
815 if not closes_with_semicolon and expanded != sequence:
816 # Roughly corresponds to the parse error
817 # missing-semicolon-after-character-reference
818 # We treat this as an unintended expansion on the user's part.
819 raise _HTMLUnexpectedCharacterReferenceError(sequence)
820 if closes_with_semicolon and expanded[-1] == ";":
821 # Failed to expand at all (e.g. for the parse error
822 # unknown-named-character-reference), or expanded earlier than
823 # the intended ";".
824 #
825 # We treat this as the user trying to do an expansion that
826 # fails.
827 raise _HTMLInvalidCharacterReferenceError(sequence)
829 @classmethod
830 def _parse_basic_inner_html(cls, serialized: str) -> _HTMLNode:
831 """
832 Parse the given source as basic inner HTML.
834 Returns the list of all nodes found.
835 """
836 source = _HTMLSourcePosition(serialized)
838 cls._check_character_refs(source)
839 source.reset()
841 open_nodes = [_HTMLNode("root", None)]
842 while True:
843 # We start in the HTML "data state", this can end with:
844 # + "&" which enters the "character reference state", we check this
845 # in _check_character_refs.
846 # + "<" which enters the "tag open state".
847 #
848 # NOTE: This skips past any ">" characters. We expect these
849 # characters to be handled ok for innerHTML.
850 open_tag_match = source.match(cls._NEXT_TAG_REGEX)
851 if open_tag_match is None:
852 break
854 if source.match(cls._OPEN_END_TAG_REGEX):
855 cls._parse_end_tag(source, open_nodes)
856 else:
857 cls._parse_start_tag(source, open_nodes)
859 if len(open_nodes) > 1:
860 raise _HTMLMissingEndTagError(open_nodes[1].tag)
862 return open_nodes[0]
864 @classmethod
865 def get_fluent_inner_html(
866 cls,
867 unit: TransUnitModel,
868 unit_source: str,
869 ) -> list[_VariantNodes] | None:
870 """
871 Get the list of HTML nodes found in each variant.
873 Returns None if there is a syntax error or no value.
874 """
875 unit_parts = FluentUnitConverter(unit, unit_source).to_fluent_parts()
876 if unit_parts is None:
877 return None
878 for part in unit_parts:
879 if not part.name:
880 # We only want to process the Fluent value part as HTML since
881 # the attributes will not likely become inner HTML.
882 return [
883 _VariantNodes(
884 path,
885 cls._parse_basic_inner_html(
886 part.top_branch.to_flat_string(path)
887 ),
888 )
889 for path in part.top_branch.branch_paths()
890 ]
891 # No value part.
892 return None
895class FluentSourceInnerHTMLCheck(_FluentInnerHTMLCheck, SourceCheck):
896 """
897 Check that the source value works as inner HTML.
899 Fluent is often used in contexts where the value for a Message (or Term) is
900 meant to be used directly as ``.innerHTML`` (rather than ``.textContent``)
901 for some HTML element. For example, when using the Fluent DOM package.
903 The aim of this check is to predict how the value will be parsed as inner
904 HTML, assuming a HTML5 conforming parser, to catch cases where there would
905 be some "unintended" loss of the string, without being too strict about
906 technical parsing errors that do *not* lead to a loss of the string.
908 This check is applied to the value of Fluent Messages or Terms, but not
909 their Attributes. For Messages, the Fluent Attributes are often just HTML
910 attribute values, so can be arbitrary strings. For Terms, the Fluent
911 Attributes are often language properties that can only be referenced in the
912 selectors of Fluent Select Expressions.
914 Generally, most Fluent values are not expected to contain any HTML markup.
915 Therefore, this check does not expect or want translators and developers to
916 have to care about strictly avoiding *any* technical HTML5 parsing errors
917 (let alone XHTML parsing errors). Instead, this check will just want to warn
918 them when they may have unintentionally opened a HTML tag or inserted a
919 character reference.
921 Moreover, for the Fluent values that intentionally contain HTML tags or
922 character references, this check will verify some "good practices", such as
923 matching closing and ending tags, valid character references, and quoted
924 attribute values. In addition, whilst the HTML5 specification technically
925 allows for quite arbitrary tag and attribute names, this check will restrain
926 them to some basic ASCII values that should cover the standard HTML5 element
927 tags and attributes, as well as allow *some* custom element or attribute
928 names. This is partially to ensure that the user is using HTML
929 intentionally.
931 NOTE: This check will *not* ensure the inner HTML is safe or sanitized, and
932 is not meant to protect against malicious attempts to alter the inner HTML.
933 Moreover, it should be remembered that Fluent variables and references may
934 expand to arbitrary strings, so could expand to arbitrary HTML unless they
935 are escaped. As an exception, a ``<`` or ``&`` character before a Fluent
936 reference will trigger this check since even an escaped value could lead to
937 unexpected results.
939 NOTE: The Fluent DOM package has further limitations, such as allowed tags
940 and attributes, which this check will not enforce.
941 """
943 check_id = "fluent-source-inner-html"
944 name = gettext_lazy("Fluent source inner HTML")
945 description = gettext_lazy("Fluent source should be valid inner HTML.")
946 default_disabled = True
948 def check_source_unit(self, sources: list[str], unit: TransUnitModel) -> bool:
949 try:
950 self.get_fluent_inner_html(unit, sources[0])
951 except _HTMLParseError:
952 return True
953 return False
955 def get_description(self, check_model: CheckModel) -> StrOrPromise:
956 unit, source, _target = translation_from_check(check_model)
957 try:
958 self.get_fluent_inner_html(unit, source)
959 except _HTMLParseError as err:
960 return err.description()
961 return super().get_description(check_model)
964class _VariantNodesDifference:
965 """
966 The difference between the nodes found in the source and target.
968 Each variant in the source will be compared against each variant in the
969 target to see if they have a matching set of nodes with the same number or
970 appearances, but not necessarily in the same order.
972 If there is any source variant that does not have at least one match in the
973 target, it will be flagged as a missing variant. Similarly, if there is any
974 target variant with no matching source variant, it will be flagged as an
975 extra variant.
976 """
978 def __init__(
979 self,
980 source_variant_nodes: list[_VariantNodes],
981 target_variant_nodes: list[_VariantNodes],
982 ) -> None:
983 self._source_variants = source_variant_nodes
984 self._target_variants = target_variant_nodes
986 self._missing_variants = [
987 variant
988 for variant in self._source_variants
989 if not self._has_match(variant, self._target_variants)
990 ]
991 self._extra_variants = [
992 variant
993 for variant in self._target_variants
994 if not self._has_match(variant, self._source_variants)
995 ]
997 @staticmethod
998 def _has_match(
999 variant: _VariantNodes,
1000 search_list: list[_VariantNodes],
1001 ) -> bool:
1002 return any(
1003 variant.counted_nodes.matches(other.counted_nodes) for other in search_list
1004 )
1006 def __bool__(self) -> bool:
1007 return bool(self._missing_variants or self._extra_variants)
1009 @staticmethod
1010 def _missing_element_message(
1011 tag: str,
1012 variants: str,
1013 ) -> SafeString:
1014 if not variants:
1015 return format_html_code(
1016 gettext("Fluent value is missing a HTML {tag} tag."),
1017 tag=tag,
1018 )
1019 return format_html_code(
1020 gettext(
1021 "Fluent value is missing a HTML {tag} tag "
1022 "for the following variants: {variant_list}."
1023 ),
1024 tag=tag,
1025 variant_list=variants,
1026 )
1028 @staticmethod
1029 def _extra_element_message(
1030 tag: str,
1031 variants: str,
1032 ) -> SafeString:
1033 if not variants:
1034 return format_html_code(
1035 gettext("Fluent value has an unexpected extra HTML {tag} tag."),
1036 tag=tag,
1037 )
1038 return format_html_code(
1039 gettext(
1040 "Fluent value has an unexpected extra HTML {tag} tag "
1041 "for the following variants: {variant_list}."
1042 ),
1043 tag=tag,
1044 variant_list=variants,
1045 )
1047 @staticmethod
1048 def _present_variant_list(
1049 variant_list: list[_VariantNodes] | None,
1050 ) -> str:
1051 if not variant_list:
1052 return ""
1053 return format_html_join_comma(
1054 "{}", list_to_tuples(variant.name() for variant in variant_list)
1055 )
1057 def _unique_target_nodes(self) -> Iterator[_HTMLNode]:
1058 unique_nodes: list[_HTMLNode] = []
1059 for variant in self._target_variants:
1060 for node in variant.nodes:
1061 add = True
1062 for other in unique_nodes:
1063 if other.matches(node):
1064 add = False
1065 break
1066 if add:
1067 unique_nodes.append(node)
1068 yield node
1070 def _errors_relative_to(
1071 self,
1072 source_counted_nodes: _CountedNodes,
1073 ) -> Iterator[SafeString]:
1074 for nodes in source_counted_nodes.matching_nodes:
1075 count = len(nodes)
1076 variants_missing_node = []
1077 all_variants = True
1078 for variant in self._target_variants:
1079 if variant.counted_nodes.count(nodes[0]) < count:
1080 variants_missing_node.append(variant)
1081 else:
1082 all_variants = False
1083 if not variants_missing_node:
1084 continue
1085 yield self._missing_element_message(
1086 nodes[0].present(),
1087 self._present_variant_list(
1088 None if all_variants else variants_missing_node
1089 ),
1090 )
1092 for node in self._unique_target_nodes():
1093 count = source_counted_nodes.count(node)
1094 variants_extra_node = []
1095 all_variants = True
1096 for variant in self._target_variants:
1097 if variant.counted_nodes.count(node) > count:
1098 variants_extra_node.append(variant)
1099 else:
1100 all_variants = False
1101 if not variants_extra_node:
1102 continue
1103 yield self._extra_element_message(
1104 node.present(),
1105 self._present_variant_list(
1106 None if all_variants else variants_extra_node
1107 ),
1108 )
1110 def _missing_variants_message(
1111 self,
1112 variants: list[_VariantNodes],
1113 ) -> SafeString:
1114 # NOTE: variants should all have names since the source contains at
1115 # least two variants in order to reach this step.
1116 variant_list = self._present_variant_list(variants)
1117 return format_html_code(
1118 gettext(
1119 "The following variants in the original Fluent value do not "
1120 "have at least one matching variant in the translation with "
1121 "the same set of HTML elements: {variant_list}."
1122 ),
1123 variant_list=variant_list,
1124 )
1126 def _extra_variants_message(
1127 self,
1128 variants: list[_VariantNodes] | None,
1129 ) -> SafeString:
1130 variant_list = self._present_variant_list(variants)
1131 if not variant_list:
1132 return format_html_code(
1133 gettext(
1134 "The translated Fluent value does not "
1135 "have a matching variant in the original with the same "
1136 "set of HTML elements."
1137 ),
1138 )
1139 return format_html_code(
1140 gettext(
1141 "The following variants in the translated Fluent "
1142 "value do not have a matching variant in "
1143 "the original with the same set of HTML elements: "
1144 "{variant_list}."
1145 ),
1146 variant_list=variant_list,
1147 )
1149 def _errors_for_unmatched_variants(
1150 self,
1151 ) -> Iterator[SafeString]:
1152 if self._missing_variants:
1153 yield self._missing_variants_message(self._missing_variants)
1154 if self._extra_variants:
1155 # Don't want to print a list of variants if we only have one in the
1156 # original.
1157 have_target_variants = len(self._target_variants) > 1
1158 yield self._extra_variants_message(
1159 self._extra_variants if have_target_variants else None
1160 )
1162 def description(self) -> SafeString:
1163 """Generate a description of the differences between the source and target."""
1164 # We want to be able to compare each target variant against some common
1165 # set of expected nodes. This allows us to determine which specific
1166 # nodes are missing or extra.
1167 # This is only possible if each variant in the source has the same set
1168 # of nodes to allow for this one-to-one comparison. But we expect this
1169 # will happen in most cases.
1170 common_nodes: _CountedNodes | None = None
1171 for variant in self._source_variants:
1172 if common_nodes is None:
1173 common_nodes = variant.counted_nodes
1174 elif not common_nodes.matches(variant.counted_nodes):
1175 common_nodes = None
1176 break
1178 if common_nodes is not None:
1179 return format_html_error_list(self._errors_relative_to(common_nodes))
1180 # The source contains multiple variants with different node counts.
1181 return format_html_error_list(self._errors_for_unmatched_variants())
1184class FluentTargetInnerHTMLCheck(_FluentInnerHTMLCheck, TargetCheck):
1185 """
1186 Check that the target value has the same HTML nodes as the source.
1188 This check will verify that the translated value of a Message or Term
1189 contains the same HTML elements as the source value.
1191 First, if the source value fails the `check-fluent-source-inner-html` check,
1192 then this check will do nothing. Otherwise, the translated value will also
1193 be checked under the same conditions.
1195 Second, the HTML elements found in the translated value will be compared
1196 against the HTML elements found in the source value. Two elements will match
1197 if they share the exact same tag name, the exact same attributes and values,
1198 and all their ancestors match in the same way. This check will ensure that
1199 all the elements in the source appear somewhere in the translation, with the
1200 same *number* of appearances, and with no additional elements added. When
1201 there are multiple elements in the value, they need not appear in the same
1202 order in the translation value.
1204 When the source or translation contains Fluent Select Expressions, then each
1205 possible variant in the source must be matched with at least one variant in
1206 the translation with the same HTML elements, and vice versa.
1208 When using Fluent in combination with the Fluent DOM package, this check
1209 will ensure that the translation also includes any required
1210 ``data-l10n-name`` elements that appear in the source, or any of the allowed
1211 inline elements like ``<br>``.
1212 """
1214 # E.g. if the source is
1215 #
1216 # m = You <em>must</em> visit <a data-l10n-name="link">my homepage</a>
1217 #
1218 # Then we would expect the translation to include the <em> element and the
1219 # <a> element *including* the same "data-l10n-name" attribute and value.
1221 check_id = "fluent-target-inner-html"
1222 name = gettext_lazy("Fluent translation inner HTML")
1223 description = gettext_lazy("Fluent target should be valid inner HTML that matches.")
1224 default_disabled = True
1226 @classmethod
1227 def _compare_inner_html(
1228 cls, unit: TransUnitModel, source: str, target: str
1229 ) -> _VariantNodesDifference | None:
1230 # May raise a _HTMLParseError.
1231 target_variant_nodes = cls.get_fluent_inner_html(unit, target)
1233 try:
1234 source_variant_nodes = cls.get_fluent_inner_html(unit, source)
1235 except _HTMLParseError:
1236 # If the source is invalid, we do not expect the target to match.
1237 return None
1239 if source_variant_nodes is None:
1240 # Invalid syntax or no value, leave this to the syntax check and
1241 # part check.
1242 return None
1244 if target_variant_nodes is None:
1245 # Invalid syntax or no value in target. Leave this to the syntax
1246 # check and part check.
1247 return None
1249 # Compare every variant in the target against every variant in the
1250 # source. Each variant's list of nodes should match at least one other
1251 # variant's list of nodes.
1252 return _VariantNodesDifference(
1253 source_variant_nodes,
1254 target_variant_nodes,
1255 )
1257 def check_single(
1258 self,
1259 source: str,
1260 target: str,
1261 unit: TransUnitModel,
1262 ) -> bool:
1263 try:
1264 difference = self._compare_inner_html(unit, source, target)
1265 except _HTMLParseError:
1266 return True
1267 return bool(difference)
1269 def get_description(self, check_model: CheckModel) -> StrOrPromise:
1270 unit, source, target = translation_from_check(check_model)
1271 try:
1272 difference = self._compare_inner_html(unit, source, target)
1273 except _HTMLParseError as err:
1274 return err.description()
1276 if not difference:
1277 return super().get_description(check_model)
1279 return difference.description()
1281 def check_highlight(
1282 self,
1283 source: str,
1284 unit: TransUnitModel,
1285 ) -> HighlightsType:
1286 if self.should_skip(unit): 1286 ↛ 1291line 1286 didn't jump to line 1291 because the condition on line 1286 was always true
1287 return []
1289 # We simply highlight all HTML tags that are valid tags according to our
1290 # parser, regardless of whether it matches a tag found in the source.
1291 return [
1292 (match.start(), match.end(), match.group())
1293 for match in self.ALL_TAGS_REGEX.finditer(source)
1294 ]