Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/fluent/inner_html.py: 24%

483 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Henry Wilkes <henry@torproject.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7import html 

8import re 

9from typing import TYPE_CHECKING 

10 

11from django.utils.translation import gettext, gettext_lazy 

12 

13from weblate.checks.base import SourceCheck, TargetCheck 

14from weblate.checks.fluent.utils import ( 

15 FluentPatterns, 

16 FluentUnitConverter, 

17 format_html_code, 

18 format_html_error_list, 

19 translation_from_check, 

20 variant_name, 

21) 

22from weblate.utils.html import format_html_join_comma, list_to_tuples 

23 

24if TYPE_CHECKING: 24 ↛ 25line 24 didn't jump to line 25 because the condition on line 24 was never true

25 from collections.abc import Iterable, Iterator 

26 

27 from django.utils.safestring import SafeString 

28 from django_stubs_ext import StrOrPromise 

29 from translate.storage.fluent import FluentSelectorBranch 

30 

31 from weblate.checks.fluent.utils import CheckModel, HighlightsType, TransUnitModel 

32 

33# Standard html elements that do not have content or end tags. 

34_VOID_ELEMENTS = [ 

35 "area", 

36 "base", 

37 "br", 

38 "col", 

39 "embed", 

40 "hr", 

41 "img", 

42 "input", 

43 "link", 

44 "meta", 

45 "source", 

46 "track", 

47 "wbr", 

48] 

49 

50 

51class _HTMLNode: 

52 """Represents a found HTML node.""" 

53 

54 def __init__(self, tag: str, parent: _HTMLNode | None) -> None: 

55 self.parent = parent 

56 self.children: list[_HTMLNode] = [] 

57 self.tag = tag 

58 # Keep attributes as a list to keep duplicates. 

59 self.attributes: dict[str, str] = {} 

60 if parent: 

61 parent.children.append(self) 

62 

63 def descendants(self) -> Iterator[_HTMLNode]: 

64 """All descendant nodes, not including itself.""" 

65 for child in self.children: 

66 yield child 

67 yield from child.descendants() 

68 

69 def matches(self, other: _HTMLNode) -> bool: 

70 """Whether the two nodes match.""" 

71 if self.tag != other.tag: 

72 return False 

73 if self.attributes != other.attributes: 

74 return False 

75 # Two nodes are only match if their ancestors match. 

76 if self.parent is None: 

77 return other.parent is None 

78 if other.parent is None: 

79 return False 

80 return self.parent.matches(other.parent) 

81 

82 def tags(self) -> tuple[str, str]: 

83 """Get the start and end tags for this node.""" 

84 start = f"<{self.tag}" 

85 for attr, val in self.attributes.items(): 

86 if '"' in val: 

87 start += f" {attr}='{val}'" 

88 else: 

89 start += f' {attr}="{val}"' 

90 if self.tag.lower() in _VOID_ELEMENTS: 

91 start += "/>" 

92 return (start, "") 

93 start += ">" 

94 return (start, f"</{self.tag}>") 

95 

96 def present(self) -> str: 

97 """Present this node and all of its ancestors for the user.""" 

98 start = "" 

99 end = "" 

100 

101 serialized, end = self.tags() 

102 if end: 

103 serialized += "…" + end 

104 

105 node = self.parent 

106 while node and node.parent: 

107 # Also show the parent elements, minus the root. 

108 start, end = node.tags() 

109 serialized = start + serialized + end 

110 node = node.parent 

111 

112 return serialized 

113 

114 

115class _HTMLParseError(BaseException): 

116 """Generic error class for our internal parsing errors.""" 

117 

118 def description(self) -> SafeString: 

119 raise NotImplementedError 

120 

121 

122class _HTMLFluentReferenceTagError(_HTMLParseError): 

123 def __init__(self, sequence: str) -> None: 

124 self.sequence = sequence 

125 

126 def description(self) -> SafeString: 

127 return format_html_code( 

128 gettext( 

129 "The Fluent reference in {sequence} may expand into a HTML " 

130 "tag. Maybe use {suggestion}." 

131 ), 

132 sequence=self.sequence, 

133 suggestion=self.sequence.replace("<", "&lt;", 1), 

134 ) 

135 

136 

137class _HTMLFluentReferenceCharacterReferenceError(_HTMLParseError): 

138 def __init__(self, sequence: str) -> None: 

139 self.sequence = sequence 

140 

141 def description(self) -> SafeString: 

142 return format_html_code( 

143 gettext( 

144 "The Fluent reference in {sequence} may expand into a HTML " 

145 "character reference. Maybe use {suggestion}." 

146 ), 

147 sequence=self.sequence, 

148 suggestion=self.sequence.replace("&", "&amp;", 1), 

149 ) 

150 

151 

152class _HTMLInvalidTagSequenceError(_HTMLParseError): 

153 """ 

154 Base class for parsing errors in a tag-like sequence. 

155 

156 We assume the user may not have wanted to create a HTML tag, so we will show 

157 a suggestion on how to avoid it. 

158 """ 

159 

160 def __init__(self, sequence: str) -> None: 

161 self.sequence = sequence 

162 

163 @property 

164 def suggestion(self) -> str: 

165 return self.sequence.replace("<", "&lt;", 1) 

166 

167 

168class _HTMLInvalidEndTagError(_HTMLInvalidTagSequenceError): 

169 def description(self) -> SafeString: 

170 return format_html_code( 

171 gettext( 

172 "The sequence {sequence} begins with a HTML closing tag, " 

173 "but the name or syntax is not valid. " 

174 "If you do not want a closing tag, use {suggestion}." 

175 ), 

176 sequence=self.sequence, 

177 suggestion=self.suggestion, 

178 ) 

179 

180 

181class _HTMLInvalidStartTagNameError(_HTMLInvalidTagSequenceError): 

182 def description(self) -> SafeString: 

183 return format_html_code( 

184 gettext( 

185 "The sequence {sequence} begins with a HTML tag, " 

186 "but the name is not valid. " 

187 "If you do not want to begin a HTML tag, use {suggestion}." 

188 ), 

189 sequence=self.sequence, 

190 suggestion=self.suggestion, 

191 ) 

192 

193 

194class _HTMLStartTagNotClosedError(_HTMLInvalidTagSequenceError): 

195 def description(self) -> SafeString: 

196 return format_html_code( 

197 gettext( 

198 "The sequence {sequence} begins with a HTML tag, " 

199 "but the tag is never closed by {close}. " 

200 "If you do not want to begin a HTML tag, use {suggestion}." 

201 ), 

202 sequence=self.sequence, 

203 close=">", 

204 suggestion=self.suggestion, 

205 ) 

206 

207 

208class _HTMLTagTypeNotAllowedError(_HTMLInvalidTagSequenceError): 

209 def description(self) -> SafeString: 

210 return format_html_code( 

211 gettext( 

212 "Fluent inner HTML should not include {sequence}. " 

213 "Maybe use {suggestion}." 

214 ), 

215 sequence=self.sequence, 

216 suggestion=self.suggestion, 

217 ) 

218 

219 

220class _HTMLUnexpectedAttributeError(_HTMLInvalidTagSequenceError): 

221 def __init__(self, tag: str, attribute: str) -> None: 

222 super().__init__(tag) 

223 self.attribute = attribute 

224 

225 def description(self) -> SafeString: 

226 return format_html_code( 

227 gettext( 

228 "The sequence {sequence} begins with a HTML tag, " 

229 "but the sequence {attribute} is not a valid attribute " 

230 "with a value. " 

231 "If you do not want to begin a HTML tag, use {suggestion}." 

232 ), 

233 sequence=self.sequence, 

234 attribute=self.attribute, 

235 suggestion=self.suggestion, 

236 ) 

237 

238 

239class _HTMLInvalidAttributeNameError(_HTMLParseError): 

240 def __init__(self, tag: str, name: str) -> None: 

241 self.tag = tag 

242 self.name = name 

243 

244 def description(self) -> SafeString: 

245 return format_html_code( 

246 gettext("The HTML attribute name {name} for the {tag} tag is not valid."), 

247 name=self.name, 

248 tag=f"<{self.tag}>", 

249 ) 

250 

251 

252class _HTMLInvalidAttributeValueError(_HTMLParseError): 

253 def __init__(self, tag: str, name: str, value: str) -> None: 

254 self.tag = tag 

255 self.name = name 

256 self.value = value 

257 

258 def description(self) -> SafeString: 

259 return format_html_code( 

260 gettext( 

261 "The HTML {name} attribute value {value} for the {tag} " 

262 "tag is not a valid quoted value." 

263 ), 

264 name=self.name, 

265 value=self.value, 

266 tag=f"<{self.tag}>", 

267 ) 

268 

269 

270class _HTMLDuplicateAttributeNameError(_HTMLParseError): 

271 def __init__(self, tag: str, name: str) -> None: 

272 self.tag = tag 

273 self.name = name 

274 

275 def description(self) -> SafeString: 

276 return format_html_code( 

277 gettext("The HTML {name} attribute appears twice for the {tag} tag."), 

278 name=self.name, 

279 tag=f"<{self.tag}>", 

280 ) 

281 

282 

283class _HTMLUnmatchedEndTagError(_HTMLParseError): 

284 def __init__(self, tag: str) -> None: 

285 self.tag = tag 

286 

287 def description(self) -> SafeString: 

288 return format_html_code( 

289 gettext("Unmatched HTML end tag: {tag}."), tag=f"</{self.tag}>" 

290 ) 

291 

292 

293class _HTMLVoidEndTagError(_HTMLParseError): 

294 def __init__(self, tag: str) -> None: 

295 self.tag = tag 

296 

297 def description(self) -> SafeString: 

298 return format_html_code( 

299 gettext("The HTML {name} void element should not have an end tag: {end}."), 

300 name=self.tag, 

301 end=f"</{self.tag}>", 

302 ) 

303 

304 

305class _HTMLNonVoidSelfClosedError(_HTMLParseError): 

306 def __init__(self, tag: str) -> None: 

307 self.tag = tag 

308 

309 def description(self) -> SafeString: 

310 return format_html_code( 

311 gettext( 

312 "The HTML {name} element is not a known void element, so " 

313 "should not have a self-closing tag: {self_close}." 

314 ), 

315 name=self.tag, 

316 self_close=f"<{self.tag}/>", 

317 ) 

318 

319 

320class _HTMLMissingEndTagError(_HTMLParseError): 

321 def __init__(self, tag: str) -> None: 

322 self.tag = tag 

323 

324 def description(self) -> SafeString: 

325 return format_html_code( 

326 gettext("The HTML {tag} tag is missing a matching end tag: {end}."), 

327 tag=f"<{self.tag}>", 

328 end=f"</{self.tag}>", 

329 ) 

330 

331 

332class _HTMLUnexpectedCharacterReferenceError(_HTMLParseError): 

333 def __init__(self, sequence: str) -> None: 

334 self.sequence = sequence 

335 

336 def description(self) -> SafeString: 

337 # We primarily assume the user did not intend to create a character 

338 # reference. 

339 suggestion = self.sequence.replace("&", "&amp;", 1) 

340 return format_html_code( 

341 gettext( 

342 "The sequence {sequence} will begin a HTML character " 

343 "reference, but does not end with {semicolon}. " 

344 "If you do not want a character reference, use {suggestion}." 

345 ), 

346 sequence=self.sequence, 

347 semicolon=";", 

348 suggestion=suggestion, 

349 ) 

350 

351 

352class _HTMLInvalidCharacterReferenceError(_HTMLParseError): 

353 def __init__(self, sequence: str) -> None: 

354 self.sequence = sequence 

355 

356 def description(self) -> SafeString: 

357 # We primarily assume the user did not intend to create a character 

358 # reference. 

359 return format_html_code( 

360 gettext("The sequence {sequence} is not a valid HTML character reference."), 

361 sequence=self.sequence, 

362 ) 

363 

364 

365class _HTMLSourcePosition: 

366 """Keeps track of some source string and a position in that string.""" 

367 

368 def __init__(self, serialized: str) -> None: 

369 self.pos = 0 

370 self.source = "" 

371 self._start_end_literals: list[tuple[int, int]] = [] 

372 

373 # Since FluentPatternVariants correspond to flat Patterns, we 

374 # don't expect any SelectExpressions, but it might contain other 

375 # "flat" references. E.g. 

376 # contact { $person } at { 123 } 

377 # 

378 # We need this to be valid HTML *after* expanding the literals. 

379 for _pos, text, literal in FluentPatterns.split_literal_expressions(serialized): 

380 self.source += text 

381 if literal: 

382 start = len(self.source) 

383 self._start_end_literals.append((start, start + len(literal))) 

384 self.source += literal 

385 

386 def reset(self) -> None: 

387 """Reset the position back to the start.""" 

388 self.pos = 0 

389 

390 def at_literal(self) -> bool: 

391 """Whether the current position is within a Fluent literal or not.""" 

392 return any( 

393 self.pos >= start and self.pos < end 

394 for start, end in self._start_end_literals 

395 ) 

396 

397 def match(self, regex: re.Pattern) -> re.Match | None: 

398 """ 

399 Try and match with the given regular expression. 

400 

401 The match is performed at the current source position. 

402 

403 If the expressions match, the source position will automatically move 

404 forward to the end of the match. 

405 """ 

406 match = regex.match(self.source, self.pos) 

407 if not match: 

408 return None 

409 self.pos = match.end() 

410 return match 

411 

412 def get(self, regex: re.Pattern) -> str: 

413 """ 

414 Get the matched text for the given regular expression. 

415 

416 The text is fetched from the current source position, without changing 

417 its value. 

418 

419 If nothing is matched, then an empty string is returned. 

420 """ 

421 match = regex.match(self.source, self.pos) 

422 if not match: 

423 return "" 

424 return match.group() 

425 

426 def peak_matches(self, regex: re.Pattern) -> bool: 

427 """ 

428 Peak whether the we match with the given regular expression. 

429 

430 The match is performed at the current source position. 

431 

432 If there is a match, this will return True without changing the source 

433 position. Otherwise it will return False. 

434 """ 

435 return regex.match(self.source, self.pos) is not None 

436 

437 

438class _CountedNodes: 

439 """ 

440 Tracks the number of matching _HTMLNodes in a list. 

441 

442 This will collect matching nodes together so they can be counted and 

443 compared, without caring about the order. 

444 """ 

445 

446 def __init__(self, node_list: Iterable[_HTMLNode]) -> None: 

447 # Each entry in matching_nodes is a list of nodes that all match. The 

448 # length of this list is the count. 

449 self.matching_nodes: list[list[_HTMLNode]] = [] 

450 for node in node_list: 

451 add = True 

452 for other_nodes in self.matching_nodes: 

453 if other_nodes[0].matches(node): 

454 add = False 

455 other_nodes.append(node) 

456 break 

457 if add: 

458 self.matching_nodes.append([node]) 

459 

460 def matches(self, other: _CountedNodes) -> bool: 

461 """ 

462 Whether the two instances match. 

463 

464 This will match if the two instances have matching _HTMLNodes with the 

465 same count in both. 

466 """ 

467 if len(self.matching_nodes) != len(other.matching_nodes): 

468 return False 

469 for nodes in self.matching_nodes: 

470 has_match = False 

471 for _index, other_nodes in enumerate(other.matching_nodes): 

472 if nodes[0].matches(other_nodes[0]): 

473 if len(nodes) != len(other_nodes): 

474 return False 

475 has_match = True 

476 break 

477 if not has_match: 

478 return False 

479 # Both are the same length and each node in self was matched, so there 

480 # shouldn't be any nodes in other that were not matched once. 

481 return True 

482 

483 def count(self, node: _HTMLNode) -> int: 

484 """Count how many nodes match the given node.""" 

485 for other_nodes in self.matching_nodes: 

486 if node.matches(other_nodes[0]): 

487 return len(other_nodes) 

488 return 0 

489 

490 

491class _VariantNodes: 

492 """Represents a variant string and the _HTMLNodes it contains.""" 

493 

494 def __init__( 

495 self, 

496 path: list[FluentSelectorBranch], 

497 root: _HTMLNode, 

498 ) -> None: 

499 self.path = path 

500 self.root = root 

501 self._nodes: list[_HTMLNode] | None = None 

502 self._counted_nodes: _CountedNodes | None = None 

503 

504 @property 

505 def nodes(self) -> list[_HTMLNode]: 

506 """The nodes found in this variant.""" 

507 if self._nodes is None: 

508 self._nodes = list(self.root.descendants()) 

509 return self._nodes 

510 

511 @property 

512 def counted_nodes(self) -> _CountedNodes: 

513 """The nodes found in this variant, grouped with matching nodes.""" 

514 if self._counted_nodes is None: 

515 self._counted_nodes = _CountedNodes(self.nodes) 

516 return self._counted_nodes 

517 

518 def name(self) -> str: 

519 """Generate name for this variant.""" 

520 return variant_name(self.path) 

521 

522 

523class _FluentInnerHTMLCheck: 

524 """ 

525 Check that we have valid inner HTML. 

526 

527 This will check that the given translation unit's source or target has inner 

528 HTML that will not lead to a loss of content when parsed. 

529 """ 

530 

531 # Pattern to search for the next tag. 

532 _NEXT_TAG_REGEX = re.compile(r"[^<]*<") 

533 # Pattern that starts an end-tag. 

534 _OPEN_END_TAG_REGEX = re.compile(r"\/") 

535 

536 # Do not allow processing instructions, or comments or CDATA or DOCTYPE. 

537 _TAG_NOT_ALLOWED_FIRST_CHAR_REGEX = re.compile(r"[!?]") 

538 

539 # First character of tag must be ASCII alpha, as per HTML spec. Unlike the 

540 # spec, we also restrict the other characters to be ASCII alphanumeric or 

541 # "-". 

542 _TAG_FIRST_CHAR = r"[a-zA-Z]" 

543 _TAG_FIRST_CHAR_REGEX = re.compile(_TAG_FIRST_CHAR) 

544 _TAG_PATTERN = _TAG_FIRST_CHAR + r"[a-zA-Z0-9-]*" 

545 

546 # Only allow a limited set of "blank" characters within a tag. 

547 _BLANK_CHAR = r"[ \t\n]" 

548 # Match either the closing of a start-tag, or some whitespace, or the end of 

549 # the string. 

550 _CLOSE_BLANK_OR_END_PATTERN = ( 

551 r"(" 

552 + (_BLANK_CHAR + r"*(?P<close>\/?>)") # Blanks followed by ">" or "/>", 

553 + r"|" # or 

554 + (_BLANK_CHAR + r"*(?P<eof>$)") # end of the string, 

555 + r"|" # or 

556 + (_BLANK_CHAR + r"+") # a required blank. 

557 # NOTE: The <eof> group will match before the blank group. E.g. " x" 

558 # will match blank, but " " will match <eof>. 

559 + r")" 

560 ) 

561 _START_TAG_REGEX = re.compile( 

562 r"(?P<tag>" + _TAG_PATTERN + r")" + _CLOSE_BLANK_OR_END_PATTERN 

563 ) 

564 # Only allow a limited set of characters for attribute names, which should 

565 # cover HTML attributes and "data-" attributes. 

566 # Also require ending with an "=". 

567 _ATTRIBUTE_NAME_PATTERN = r"(?P<name>[a-zA-Z][a-zA-Z0-9_.:-]*)=" 

568 # Only allow quoted values. 

569 # NOTE: We do not use a raw string since we want the '\"' to become a 

570 # literal '"'. 

571 _ATTRIBUTE_VALUE_PATTERN = "('(?P<value1>[^']*)'|\"(?P<value2>[^\"]*)\")" 

572 _ATTRIBUTE_NAME_REGEX = re.compile(_ATTRIBUTE_NAME_PATTERN) 

573 _ATTRIBUTE_VALUE_REGEX = re.compile( 

574 _ATTRIBUTE_VALUE_PATTERN 

575 # Followed by a blank or the closing character. 

576 + _CLOSE_BLANK_OR_END_PATTERN 

577 ) 

578 _END_TAG_REGEX = re.compile(r"(?P<tag>" + _TAG_PATTERN + r")" + _BLANK_CHAR + r"*>") 

579 

580 # Pattern used to pull text within a suspected tag up until the next 

581 # attribute or the tag closes. 

582 _NON_BLANK_OR_CLOSE_REGEX = re.compile(r"[^ \t\n>]*") 

583 

584 _FLUENT_REF_FIRST_CHAR_REGEX = re.compile(r"\{") 

585 # This pattern will match basic fluent references, but will break down for 

586 # references that contain sub-placeables or "}" literals. However, we only 

587 # use this for warning messages, so it is good enough. 

588 # Similarly we make the last "}" optional, even though this would indicate 

589 # invalid Fluent syntax, but we want a guaranteed match if we already match 

590 _FLUENT_REF_REGEX = re.compile(r"\{[^}]*\}?") 

591 

592 # For highlighting tags. 

593 ALL_TAGS_REGEX = re.compile( 

594 # Start tag. 

595 r"<" 

596 + _TAG_PATTERN 

597 # Attributes. 

598 + r"(" 

599 + (_BLANK_CHAR + r"+" + _ATTRIBUTE_NAME_PATTERN + _ATTRIBUTE_VALUE_PATTERN) 

600 + r")*" 

601 # Close start tag. 

602 + (_BLANK_CHAR + r"*\/?>") 

603 + r"|" 

604 # End tag. 

605 + (r"<\/" + _TAG_PATTERN + _BLANK_CHAR + r"*>") 

606 ) 

607 

608 @classmethod 

609 def _non_blank_or_close(cls, source: _HTMLSourcePosition) -> str: 

610 """ 

611 Get all the non-blank and non ">" characters found at the start. 

612 

613 This is used to grab some joined sequence of characters within a HTML 

614 tag to show back to the user. 

615 """ 

616 return source.get(cls._NON_BLANK_OR_CLOSE_REGEX) 

617 

618 @classmethod 

619 def _parse_end_tag( 

620 cls, source: _HTMLSourcePosition, open_nodes: list[_HTMLNode] 

621 ) -> None: 

622 """Parse an end tag, starting after the "</".""" 

623 end_tag_match = source.match(cls._END_TAG_REGEX) 

624 if not end_tag_match: 

625 # May correspond to using a non-ASCII alphanumeric value in the tag 

626 # name, which whilst technically allowed for HTML, is not allowed by 

627 # this check. 

628 # 

629 # Otherwise, this is expected to correspond to some HTML parsing 

630 # error, like: 

631 # + invalid-first-character-of-tag-name 

632 # + eof-in-tag 

633 # + end-tag-with-attributes 

634 # + missing-end-tag-name 

635 # 

636 # In which cases some content will be *lost* when parsed as HTML by 

637 # being commented out or ignored. 

638 # 

639 # Some parsing errors, like eof-before-tag-name, will not lead to a 

640 # loss of content, but aren't allowed here for consistency. 

641 raise _HTMLInvalidEndTagError("</" + cls._non_blank_or_close(source)) 

642 

643 tag = end_tag_match.group("tag") 

644 

645 if tag.lower() in _VOID_ELEMENTS: 

646 raise _HTMLVoidEndTagError(tag) 

647 

648 node = open_nodes.pop() 

649 # Never close our dummy "root" element. So raise error if open_nodes is 

650 # now empty. 

651 if not open_nodes or node.tag != tag: 

652 raise _HTMLUnmatchedEndTagError(tag) 

653 

654 @classmethod 

655 def _parse_attribute( 

656 cls, 

657 source: _HTMLSourcePosition, 

658 node: _HTMLNode, 

659 ) -> re.Match: 

660 """Parse a start tag attribute, starting at the attribute name.""" 

661 name_match = source.match(cls._ATTRIBUTE_NAME_REGEX) 

662 if not name_match: 

663 non_blank = cls._non_blank_or_close(source) 

664 if "=" not in non_blank: 

665 # Doesn't look like an attribute with a value. 

666 raise _HTMLUnexpectedAttributeError("<" + node.tag, non_blank) 

667 raise _HTMLInvalidAttributeNameError( 

668 node.tag, 

669 non_blank[: non_blank.index("=")], 

670 ) 

671 

672 name = name_match.group("name") 

673 if name in node.attributes: 

674 raise _HTMLDuplicateAttributeNameError(node.tag, name) 

675 value_match = source.match(cls._ATTRIBUTE_VALUE_REGEX) 

676 if not value_match: 

677 raise _HTMLInvalidAttributeValueError( 

678 node.tag, name, cls._non_blank_or_close(source) 

679 ) 

680 

681 value = value_match.group("value1") 

682 if value is None: 

683 value = value_match.group("value2") 

684 

685 node.attributes[name] = value 

686 

687 return value_match 

688 

689 @classmethod 

690 def _parse_start_tag( 

691 cls, source: _HTMLSourcePosition, open_nodes: list[_HTMLNode] 

692 ) -> None: 

693 """Parse a start tag, starting after the opening "<".""" 

694 if ( 

695 # If we are pointing to a literal character, then this "{" should 

696 # not be part of a fluent reference, but be a literal "{" instead. 

697 not source.at_literal() 

698 and source.peak_matches(cls._FLUENT_REF_FIRST_CHAR_REGEX) 

699 ): 

700 # Starts a Fluent reference. E.g. 

701 # contact <{ $name } 

702 # If the "$name" starts with an ASCII alpha it could expand to 

703 # a tag. 

704 # 

705 # NOTE: This won't capture all cases where the reference might 

706 # expand into some HTML, but generally we expect the fluent 

707 # application to ensure their references are sanitized for 

708 # inner HTML. But this seems like a case where a sanitized value 

709 # might become unintentionally bad. 

710 # 

711 # NOTE: If the fluent reference appears elsewhere within the tag 

712 # it will be invalid (not a valid tag name or attribute name) except 

713 # within a quoted attribute value. A quoted attribute value can 

714 # still cause problems when it is expanded with the value. E.g. if 

715 # the value closes the quotes and injects parts, but we are not 

716 # trying to protect against this. 

717 raise _HTMLFluentReferenceTagError("<" + source.get(cls._FLUENT_REF_REGEX)) 

718 

719 tag_not_allowed_match = source.match(cls._TAG_NOT_ALLOWED_FIRST_CHAR_REGEX) 

720 if tag_not_allowed_match: 

721 raise _HTMLTagTypeNotAllowedError("<" + tag_not_allowed_match.group()) 

722 

723 if not source.peak_matches(cls._TAG_FIRST_CHAR_REGEX): 

724 # Corresponds to the HTML parsing errors 

725 # + invalid-first-character-of-tag-name, and 

726 # + eof-before-tag-name 

727 # for a start tag, so the "<" will be treated as text content. 

728 # So can keep this "<" character and move on. 

729 return 

730 

731 tag_match = source.match(cls._START_TAG_REGEX) 

732 if not tag_match: 

733 raise _HTMLInvalidStartTagNameError("<" + cls._non_blank_or_close(source)) 

734 

735 tag = tag_match.group("tag") 

736 

737 node = _HTMLNode(tag, open_nodes[-1]) 

738 

739 end_match = tag_match 

740 while True: 

741 closing_part = end_match.group("close") 

742 if closing_part is not None: 

743 # Void element has no content, so is not added to the list 

744 # of open_nodes. Being self-closing is optional. 

745 if tag.lower() not in _VOID_ELEMENTS: 

746 # Non-void elements should not be self-closing. 

747 if closing_part == "/>": 

748 raise _HTMLNonVoidSelfClosedError(tag) 

749 open_nodes.append(node) 

750 return 

751 

752 if end_match.group("eof") is not None: 

753 # Corresponds to HTML parsing error eof-in-tag. 

754 # We have some blank, but then reach the end of the string 

755 # before the tag is closed. 

756 raise _HTMLStartTagNotClosedError("<" + tag) 

757 

758 end_match = cls._parse_attribute(source, node) 

759 

760 # Check for: 

761 # + valid character refs (like "&lt;", "&frac12;", "&#80;" or "&#x80"), or 

762 # + character refs that may expand in some way, but likely were not intended 

763 # by the user because they are missing the ";" (like "&ethical"), or 

764 # + sequences that look like character refs that the user wants to expand 

765 # (because they are alphanumeric and end with ";"), but may not actually 

766 # expand as expected. 

767 _CHARACTER_REFS_REGEX = re.compile(r"[^&]*(?P<ref>&#?[a-zA-Z0-9]*;?)") 

768 

769 @classmethod 

770 def _check_character_refs(cls, source: _HTMLSourcePosition) -> None: 

771 """ 

772 Check that the character references found in the source. 

773 

774 A character reference must be either intentional (with a semicolon) and 

775 valid, or unintentional but will not expand into a reference under the 

776 HTML spec, so are safe to keep in HTML. 

777 """ 

778 # NOTE: The HTML5 spec (13.2.5.73 Named character reference state), for 

779 # historical reasons, will treat character references that do not end 

780 # with a ";", but are followed by an alphanumeric or "=", differently 

781 # depending on whether we are in text content or an attribute value. 

782 # E.g. "&ltx" will become "<x" for text content but will remain as 

783 # "&ltx" for an attribute value. 

784 # Even though "&ltx" will not expand for an attribute, we still want to 

785 # raise an error against it for consistency. So we treat this the same 

786 # as text content. 

787 while True: 

788 # Will always match at least "&" if it exists. 

789 ref_match = source.match(cls._CHARACTER_REFS_REGEX) 

790 if not ref_match: 

791 return 

792 

793 sequence = ref_match.group("ref") 

794 closes_with_semicolon = sequence[-1] == ";" 

795 

796 if ( 

797 not closes_with_semicolon 

798 # If we are pointing to a literal character, then this "{" 

799 # should not be part of a fluent reference, but be a literal "{" 

800 # instead, which will break the HTML character reference. 

801 and not source.at_literal() 

802 # Next character is "{". 

803 and source.peak_matches(cls._FLUENT_REF_FIRST_CHAR_REGEX) 

804 ): 

805 # Could expand to a character reference when the fluent 

806 # reference is substituted. E.g. 

807 # &l{ $var } 

808 # or 

809 # &#{ $var } 

810 raise _HTMLFluentReferenceCharacterReferenceError( 

811 sequence + source.get(cls._FLUENT_REF_REGEX) 

812 ) 

813 

814 expanded = html.unescape(sequence) 

815 if not closes_with_semicolon and expanded != sequence: 

816 # Roughly corresponds to the parse error 

817 # missing-semicolon-after-character-reference 

818 # We treat this as an unintended expansion on the user's part. 

819 raise _HTMLUnexpectedCharacterReferenceError(sequence) 

820 if closes_with_semicolon and expanded[-1] == ";": 

821 # Failed to expand at all (e.g. for the parse error 

822 # unknown-named-character-reference), or expanded earlier than 

823 # the intended ";". 

824 # 

825 # We treat this as the user trying to do an expansion that 

826 # fails. 

827 raise _HTMLInvalidCharacterReferenceError(sequence) 

828 

829 @classmethod 

830 def _parse_basic_inner_html(cls, serialized: str) -> _HTMLNode: 

831 """ 

832 Parse the given source as basic inner HTML. 

833 

834 Returns the list of all nodes found. 

835 """ 

836 source = _HTMLSourcePosition(serialized) 

837 

838 cls._check_character_refs(source) 

839 source.reset() 

840 

841 open_nodes = [_HTMLNode("root", None)] 

842 while True: 

843 # We start in the HTML "data state", this can end with: 

844 # + "&" which enters the "character reference state", we check this 

845 # in _check_character_refs. 

846 # + "<" which enters the "tag open state". 

847 # 

848 # NOTE: This skips past any ">" characters. We expect these 

849 # characters to be handled ok for innerHTML. 

850 open_tag_match = source.match(cls._NEXT_TAG_REGEX) 

851 if open_tag_match is None: 

852 break 

853 

854 if source.match(cls._OPEN_END_TAG_REGEX): 

855 cls._parse_end_tag(source, open_nodes) 

856 else: 

857 cls._parse_start_tag(source, open_nodes) 

858 

859 if len(open_nodes) > 1: 

860 raise _HTMLMissingEndTagError(open_nodes[1].tag) 

861 

862 return open_nodes[0] 

863 

864 @classmethod 

865 def get_fluent_inner_html( 

866 cls, 

867 unit: TransUnitModel, 

868 unit_source: str, 

869 ) -> list[_VariantNodes] | None: 

870 """ 

871 Get the list of HTML nodes found in each variant. 

872 

873 Returns None if there is a syntax error or no value. 

874 """ 

875 unit_parts = FluentUnitConverter(unit, unit_source).to_fluent_parts() 

876 if unit_parts is None: 

877 return None 

878 for part in unit_parts: 

879 if not part.name: 

880 # We only want to process the Fluent value part as HTML since 

881 # the attributes will not likely become inner HTML. 

882 return [ 

883 _VariantNodes( 

884 path, 

885 cls._parse_basic_inner_html( 

886 part.top_branch.to_flat_string(path) 

887 ), 

888 ) 

889 for path in part.top_branch.branch_paths() 

890 ] 

891 # No value part. 

892 return None 

893 

894 

895class FluentSourceInnerHTMLCheck(_FluentInnerHTMLCheck, SourceCheck): 

896 """ 

897 Check that the source value works as inner HTML. 

898 

899 Fluent is often used in contexts where the value for a Message (or Term) is 

900 meant to be used directly as ``.innerHTML`` (rather than ``.textContent``) 

901 for some HTML element. For example, when using the Fluent DOM package. 

902 

903 The aim of this check is to predict how the value will be parsed as inner 

904 HTML, assuming a HTML5 conforming parser, to catch cases where there would 

905 be some "unintended" loss of the string, without being too strict about 

906 technical parsing errors that do *not* lead to a loss of the string. 

907 

908 This check is applied to the value of Fluent Messages or Terms, but not 

909 their Attributes. For Messages, the Fluent Attributes are often just HTML 

910 attribute values, so can be arbitrary strings. For Terms, the Fluent 

911 Attributes are often language properties that can only be referenced in the 

912 selectors of Fluent Select Expressions. 

913 

914 Generally, most Fluent values are not expected to contain any HTML markup. 

915 Therefore, this check does not expect or want translators and developers to 

916 have to care about strictly avoiding *any* technical HTML5 parsing errors 

917 (let alone XHTML parsing errors). Instead, this check will just want to warn 

918 them when they may have unintentionally opened a HTML tag or inserted a 

919 character reference. 

920 

921 Moreover, for the Fluent values that intentionally contain HTML tags or 

922 character references, this check will verify some "good practices", such as 

923 matching closing and ending tags, valid character references, and quoted 

924 attribute values. In addition, whilst the HTML5 specification technically 

925 allows for quite arbitrary tag and attribute names, this check will restrain 

926 them to some basic ASCII values that should cover the standard HTML5 element 

927 tags and attributes, as well as allow *some* custom element or attribute 

928 names. This is partially to ensure that the user is using HTML 

929 intentionally. 

930 

931 NOTE: This check will *not* ensure the inner HTML is safe or sanitized, and 

932 is not meant to protect against malicious attempts to alter the inner HTML. 

933 Moreover, it should be remembered that Fluent variables and references may 

934 expand to arbitrary strings, so could expand to arbitrary HTML unless they 

935 are escaped. As an exception, a ``<`` or ``&`` character before a Fluent 

936 reference will trigger this check since even an escaped value could lead to 

937 unexpected results. 

938 

939 NOTE: The Fluent DOM package has further limitations, such as allowed tags 

940 and attributes, which this check will not enforce. 

941 """ 

942 

943 check_id = "fluent-source-inner-html" 

944 name = gettext_lazy("Fluent source inner HTML") 

945 description = gettext_lazy("Fluent source should be valid inner HTML.") 

946 default_disabled = True 

947 

948 def check_source_unit(self, sources: list[str], unit: TransUnitModel) -> bool: 

949 try: 

950 self.get_fluent_inner_html(unit, sources[0]) 

951 except _HTMLParseError: 

952 return True 

953 return False 

954 

955 def get_description(self, check_model: CheckModel) -> StrOrPromise: 

956 unit, source, _target = translation_from_check(check_model) 

957 try: 

958 self.get_fluent_inner_html(unit, source) 

959 except _HTMLParseError as err: 

960 return err.description() 

961 return super().get_description(check_model) 

962 

963 

964class _VariantNodesDifference: 

965 """ 

966 The difference between the nodes found in the source and target. 

967 

968 Each variant in the source will be compared against each variant in the 

969 target to see if they have a matching set of nodes with the same number or 

970 appearances, but not necessarily in the same order. 

971 

972 If there is any source variant that does not have at least one match in the 

973 target, it will be flagged as a missing variant. Similarly, if there is any 

974 target variant with no matching source variant, it will be flagged as an 

975 extra variant. 

976 """ 

977 

978 def __init__( 

979 self, 

980 source_variant_nodes: list[_VariantNodes], 

981 target_variant_nodes: list[_VariantNodes], 

982 ) -> None: 

983 self._source_variants = source_variant_nodes 

984 self._target_variants = target_variant_nodes 

985 

986 self._missing_variants = [ 

987 variant 

988 for variant in self._source_variants 

989 if not self._has_match(variant, self._target_variants) 

990 ] 

991 self._extra_variants = [ 

992 variant 

993 for variant in self._target_variants 

994 if not self._has_match(variant, self._source_variants) 

995 ] 

996 

997 @staticmethod 

998 def _has_match( 

999 variant: _VariantNodes, 

1000 search_list: list[_VariantNodes], 

1001 ) -> bool: 

1002 return any( 

1003 variant.counted_nodes.matches(other.counted_nodes) for other in search_list 

1004 ) 

1005 

1006 def __bool__(self) -> bool: 

1007 return bool(self._missing_variants or self._extra_variants) 

1008 

1009 @staticmethod 

1010 def _missing_element_message( 

1011 tag: str, 

1012 variants: str, 

1013 ) -> SafeString: 

1014 if not variants: 

1015 return format_html_code( 

1016 gettext("Fluent value is missing a HTML {tag} tag."), 

1017 tag=tag, 

1018 ) 

1019 return format_html_code( 

1020 gettext( 

1021 "Fluent value is missing a HTML {tag} tag " 

1022 "for the following variants: {variant_list}." 

1023 ), 

1024 tag=tag, 

1025 variant_list=variants, 

1026 ) 

1027 

1028 @staticmethod 

1029 def _extra_element_message( 

1030 tag: str, 

1031 variants: str, 

1032 ) -> SafeString: 

1033 if not variants: 

1034 return format_html_code( 

1035 gettext("Fluent value has an unexpected extra HTML {tag} tag."), 

1036 tag=tag, 

1037 ) 

1038 return format_html_code( 

1039 gettext( 

1040 "Fluent value has an unexpected extra HTML {tag} tag " 

1041 "for the following variants: {variant_list}." 

1042 ), 

1043 tag=tag, 

1044 variant_list=variants, 

1045 ) 

1046 

1047 @staticmethod 

1048 def _present_variant_list( 

1049 variant_list: list[_VariantNodes] | None, 

1050 ) -> str: 

1051 if not variant_list: 

1052 return "" 

1053 return format_html_join_comma( 

1054 "{}", list_to_tuples(variant.name() for variant in variant_list) 

1055 ) 

1056 

1057 def _unique_target_nodes(self) -> Iterator[_HTMLNode]: 

1058 unique_nodes: list[_HTMLNode] = [] 

1059 for variant in self._target_variants: 

1060 for node in variant.nodes: 

1061 add = True 

1062 for other in unique_nodes: 

1063 if other.matches(node): 

1064 add = False 

1065 break 

1066 if add: 

1067 unique_nodes.append(node) 

1068 yield node 

1069 

1070 def _errors_relative_to( 

1071 self, 

1072 source_counted_nodes: _CountedNodes, 

1073 ) -> Iterator[SafeString]: 

1074 for nodes in source_counted_nodes.matching_nodes: 

1075 count = len(nodes) 

1076 variants_missing_node = [] 

1077 all_variants = True 

1078 for variant in self._target_variants: 

1079 if variant.counted_nodes.count(nodes[0]) < count: 

1080 variants_missing_node.append(variant) 

1081 else: 

1082 all_variants = False 

1083 if not variants_missing_node: 

1084 continue 

1085 yield self._missing_element_message( 

1086 nodes[0].present(), 

1087 self._present_variant_list( 

1088 None if all_variants else variants_missing_node 

1089 ), 

1090 ) 

1091 

1092 for node in self._unique_target_nodes(): 

1093 count = source_counted_nodes.count(node) 

1094 variants_extra_node = [] 

1095 all_variants = True 

1096 for variant in self._target_variants: 

1097 if variant.counted_nodes.count(node) > count: 

1098 variants_extra_node.append(variant) 

1099 else: 

1100 all_variants = False 

1101 if not variants_extra_node: 

1102 continue 

1103 yield self._extra_element_message( 

1104 node.present(), 

1105 self._present_variant_list( 

1106 None if all_variants else variants_extra_node 

1107 ), 

1108 ) 

1109 

1110 def _missing_variants_message( 

1111 self, 

1112 variants: list[_VariantNodes], 

1113 ) -> SafeString: 

1114 # NOTE: variants should all have names since the source contains at 

1115 # least two variants in order to reach this step. 

1116 variant_list = self._present_variant_list(variants) 

1117 return format_html_code( 

1118 gettext( 

1119 "The following variants in the original Fluent value do not " 

1120 "have at least one matching variant in the translation with " 

1121 "the same set of HTML elements: {variant_list}." 

1122 ), 

1123 variant_list=variant_list, 

1124 ) 

1125 

1126 def _extra_variants_message( 

1127 self, 

1128 variants: list[_VariantNodes] | None, 

1129 ) -> SafeString: 

1130 variant_list = self._present_variant_list(variants) 

1131 if not variant_list: 

1132 return format_html_code( 

1133 gettext( 

1134 "The translated Fluent value does not " 

1135 "have a matching variant in the original with the same " 

1136 "set of HTML elements." 

1137 ), 

1138 ) 

1139 return format_html_code( 

1140 gettext( 

1141 "The following variants in the translated Fluent " 

1142 "value do not have a matching variant in " 

1143 "the original with the same set of HTML elements: " 

1144 "{variant_list}." 

1145 ), 

1146 variant_list=variant_list, 

1147 ) 

1148 

1149 def _errors_for_unmatched_variants( 

1150 self, 

1151 ) -> Iterator[SafeString]: 

1152 if self._missing_variants: 

1153 yield self._missing_variants_message(self._missing_variants) 

1154 if self._extra_variants: 

1155 # Don't want to print a list of variants if we only have one in the 

1156 # original. 

1157 have_target_variants = len(self._target_variants) > 1 

1158 yield self._extra_variants_message( 

1159 self._extra_variants if have_target_variants else None 

1160 ) 

1161 

1162 def description(self) -> SafeString: 

1163 """Generate a description of the differences between the source and target.""" 

1164 # We want to be able to compare each target variant against some common 

1165 # set of expected nodes. This allows us to determine which specific 

1166 # nodes are missing or extra. 

1167 # This is only possible if each variant in the source has the same set 

1168 # of nodes to allow for this one-to-one comparison. But we expect this 

1169 # will happen in most cases. 

1170 common_nodes: _CountedNodes | None = None 

1171 for variant in self._source_variants: 

1172 if common_nodes is None: 

1173 common_nodes = variant.counted_nodes 

1174 elif not common_nodes.matches(variant.counted_nodes): 

1175 common_nodes = None 

1176 break 

1177 

1178 if common_nodes is not None: 

1179 return format_html_error_list(self._errors_relative_to(common_nodes)) 

1180 # The source contains multiple variants with different node counts. 

1181 return format_html_error_list(self._errors_for_unmatched_variants()) 

1182 

1183 

1184class FluentTargetInnerHTMLCheck(_FluentInnerHTMLCheck, TargetCheck): 

1185 """ 

1186 Check that the target value has the same HTML nodes as the source. 

1187 

1188 This check will verify that the translated value of a Message or Term 

1189 contains the same HTML elements as the source value. 

1190 

1191 First, if the source value fails the `check-fluent-source-inner-html` check, 

1192 then this check will do nothing. Otherwise, the translated value will also 

1193 be checked under the same conditions. 

1194 

1195 Second, the HTML elements found in the translated value will be compared 

1196 against the HTML elements found in the source value. Two elements will match 

1197 if they share the exact same tag name, the exact same attributes and values, 

1198 and all their ancestors match in the same way. This check will ensure that 

1199 all the elements in the source appear somewhere in the translation, with the 

1200 same *number* of appearances, and with no additional elements added. When 

1201 there are multiple elements in the value, they need not appear in the same 

1202 order in the translation value. 

1203 

1204 When the source or translation contains Fluent Select Expressions, then each 

1205 possible variant in the source must be matched with at least one variant in 

1206 the translation with the same HTML elements, and vice versa. 

1207 

1208 When using Fluent in combination with the Fluent DOM package, this check 

1209 will ensure that the translation also includes any required 

1210 ``data-l10n-name`` elements that appear in the source, or any of the allowed 

1211 inline elements like ``<br>``. 

1212 """ 

1213 

1214 # E.g. if the source is 

1215 # 

1216 # m = You <em>must</em> visit <a data-l10n-name="link">my homepage</a> 

1217 # 

1218 # Then we would expect the translation to include the <em> element and the 

1219 # <a> element *including* the same "data-l10n-name" attribute and value. 

1220 

1221 check_id = "fluent-target-inner-html" 

1222 name = gettext_lazy("Fluent translation inner HTML") 

1223 description = gettext_lazy("Fluent target should be valid inner HTML that matches.") 

1224 default_disabled = True 

1225 

1226 @classmethod 

1227 def _compare_inner_html( 

1228 cls, unit: TransUnitModel, source: str, target: str 

1229 ) -> _VariantNodesDifference | None: 

1230 # May raise a _HTMLParseError. 

1231 target_variant_nodes = cls.get_fluent_inner_html(unit, target) 

1232 

1233 try: 

1234 source_variant_nodes = cls.get_fluent_inner_html(unit, source) 

1235 except _HTMLParseError: 

1236 # If the source is invalid, we do not expect the target to match. 

1237 return None 

1238 

1239 if source_variant_nodes is None: 

1240 # Invalid syntax or no value, leave this to the syntax check and 

1241 # part check. 

1242 return None 

1243 

1244 if target_variant_nodes is None: 

1245 # Invalid syntax or no value in target. Leave this to the syntax 

1246 # check and part check. 

1247 return None 

1248 

1249 # Compare every variant in the target against every variant in the 

1250 # source. Each variant's list of nodes should match at least one other 

1251 # variant's list of nodes. 

1252 return _VariantNodesDifference( 

1253 source_variant_nodes, 

1254 target_variant_nodes, 

1255 ) 

1256 

1257 def check_single( 

1258 self, 

1259 source: str, 

1260 target: str, 

1261 unit: TransUnitModel, 

1262 ) -> bool: 

1263 try: 

1264 difference = self._compare_inner_html(unit, source, target) 

1265 except _HTMLParseError: 

1266 return True 

1267 return bool(difference) 

1268 

1269 def get_description(self, check_model: CheckModel) -> StrOrPromise: 

1270 unit, source, target = translation_from_check(check_model) 

1271 try: 

1272 difference = self._compare_inner_html(unit, source, target) 

1273 except _HTMLParseError as err: 

1274 return err.description() 

1275 

1276 if not difference: 

1277 return super().get_description(check_model) 

1278 

1279 return difference.description() 

1280 

1281 def check_highlight( 

1282 self, 

1283 source: str, 

1284 unit: TransUnitModel, 

1285 ) -> HighlightsType: 

1286 if self.should_skip(unit): 1286 ↛ 1291line 1286 didn't jump to line 1291 because the condition on line 1286 was always true

1287 return [] 

1288 

1289 # We simply highlight all HTML tags that are valid tags according to our 

1290 # parser, regardless of whether it matches a tag found in the source. 

1291 return [ 

1292 (match.start(), match.end(), match.group()) 

1293 for match in self.ALL_TAGS_REGEX.finditer(source) 

1294 ]