Coverage for app/venv/lib/python3.14/site-packages/weblate/checks/icu.py: 12%

255 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7from collections import defaultdict 

8from typing import TYPE_CHECKING 

9 

10from django.utils.translation import gettext, gettext_lazy 

11from pyicumessageformat import Parser 

12 

13from weblate.checks.base import SourceCheck 

14from weblate.checks.format import BaseFormatCheck 

15from weblate.utils.html import format_html_join_comma, list_to_tuples 

16 

17if TYPE_CHECKING: 17 ↛ 18line 17 didn't jump to line 18 because the condition on line 17 was never true

18 from weblate.trans.models import Unit 

19 

20# Unique value for checking tags. Since types are 

21# always strings, this will never be encountered. 

22TAG_TYPE = -100 

23 

24# These types are to be considered numeric. Numeric placeholders 

25# can be of any numeric type without triggering a warning from 

26# the checker. 

27NUMERIC_TYPES = {"number", "plural", "selectordinal"} 

28 

29# These types have their sub-messages checked to ensure that 

30# sub-message selectors are valid. 

31PLURAL_TYPES = {"plural", "selectordinal"} 

32 

33# ... and these are the valid selectors, along with selectors 

34# for specific values, formatted such as: =0, =1, etc. 

35PLURAL_SELECTORS = {"zero", "one", "two", "few", "many", "other"} 

36 

37 

38# We construct two Parser instances, one for tags and one without. 

39# Both parsers are configured to allow spaces inside formats, to not 

40# require other (which we can do better ourselves), and to be 

41# permissive about what types can have sub-messages. 

42standard_parser = Parser( 

43 { 

44 "include_indices": True, 

45 "loose_submessages": True, 

46 "allow_format_spaces": True, 

47 "require_other": False, 

48 "allow_tags": False, 

49 } 

50) 

51 

52tag_parser = Parser( 

53 { 

54 "include_indices": True, 

55 "loose_submessages": True, 

56 "allow_format_spaces": True, 

57 "require_other": False, 

58 "allow_tags": True, 

59 "strict_tags": False, 

60 "tag_type": TAG_TYPE, 

61 } 

62) 

63 

64 

65strict_tag_parser = Parser( 

66 { 

67 "include_indices": True, 

68 "loose_submessages": True, 

69 "allow_format_spaces": True, 

70 "require_other": False, 

71 "allow_tags": True, 

72 "strict_tags": True, 

73 "tag_type": TAG_TYPE, 

74 } 

75) 

76 

77 

78def parse_icu( 

79 source: str, 

80 allow_tags: bool, 

81 strict_tags: bool, 

82 tag_prefix: str | None = None, 

83 want_tokens=False, 

84) -> tuple[list[str] | None, Exception | None, list[str] | None]: 

85 """Parse an ICU MessageFormat message.""" 

86 ast = None 

87 err: Exception | None = None 

88 tokens: list[str] | None = [] if want_tokens else None 

89 parser = standard_parser 

90 if allow_tags: 

91 parser = strict_tag_parser if strict_tags else tag_parser 

92 

93 parser.options["tag_prefix"] = tag_prefix 

94 

95 try: 

96 ast = parser.parse(source, tokens) 

97 except SyntaxError as e: 

98 err = e 

99 

100 parser.options["tag_prefix"] = None 

101 

102 return ast, err, tokens 

103 

104 

105def check_bad_plural_selector(selector): 

106 if selector in PLURAL_SELECTORS: 

107 return False 

108 return selector[0] != "=" 

109 

110 

111def update_maybe_value(value, old): 

112 """ 

113 Certain placeholder values can have one of four values. 

114 

115 `None`, `True`, `False`, or `0`. 

116 

117 `None` represents a value never set. 

118 `True` or `False` represents a value selection. 

119 `0` represents a set value with conflicting values. 

120 

121 This is useful if there are multiple placeholders with 

122 conflicting type info. 

123 """ 

124 if old is None or old == value: 

125 return value 

126 return 0 

127 

128 

129def extract_highlights(token, source: str): 

130 """Extract all placeholders from an AST selected for highlighting.""" 

131 if isinstance(token, str): 

132 return 

133 

134 if isinstance(token, list): 

135 for tok in token: 

136 yield from extract_highlights(tok, source) 

137 

138 # Sanity check the token. They should always have 

139 # start and end. 

140 if "start" not in token or "end" not in token: 

141 return 

142 

143 start = token["start"] 

144 end = token["end"] 

145 usable = start < len(source) 

146 

147 if token.get("hash"): 

148 usable = False 

149 

150 if "options" in token: 

151 usable = False 

152 for subast in token["options"].values(): 

153 yield from extract_highlights(subast, source) 

154 

155 if "contents" in token: 

156 usable = False 

157 yield from extract_highlights(token["contents"], source) 

158 

159 if usable: 

160 yield (start, end, source[start:end]) 

161 

162 

163def extract_placeholders(token, variables=None): 

164 """Extract all placeholders from an AST and summarize their types.""" 

165 if variables is None: 

166 variables = {} 

167 

168 if isinstance(token, str): 

169 # Skip strings. Those aren't interesting. 

170 return variables 

171 

172 if isinstance(token, list): 

173 # If we have a list, then we have a list of tokens so iterate 

174 # over the entire list. 

175 for tok in token: 

176 extract_placeholders(tok, variables) 

177 

178 return variables 

179 

180 if "name" not in token: 

181 # There should always be a name. This is highly suspicious. 

182 # Should this raise an exception? 

183 return variables 

184 

185 name = token["name"] 

186 ttype = token.get("type") 

187 data = variables.setdefault( 

188 name, 

189 { 

190 "name": name, 

191 "types": set(), 

192 "formats": set(), 

193 "is_number": None, 

194 "is_tag": None, 

195 "is_empty": None, 

196 }, 

197 ) 

198 

199 if ttype: 

200 is_tag = ttype is TAG_TYPE 

201 data["is_tag"] = update_maybe_value(is_tag, data["is_tag"]) 

202 

203 if is_tag: 

204 data["is_empty"] = update_maybe_value( 

205 "contents" not in token or not token["contents"], data["is_empty"] 

206 ) 

207 else: 

208 data["types"].add(ttype) 

209 data["is_number"] = update_maybe_value( 

210 ttype in NUMERIC_TYPES, data["is_number"] 

211 ) 

212 if "format" in token: 

213 data["formats"].add(token["format"]) 

214 

215 elif name == "count": 

216 # Assume count is a numeric type 

217 data["types"].add("number") 

218 data["is_number"] = update_maybe_value(True, data["is_number"]) 

219 

220 if "options" in token: 

221 choices = data.setdefault("choices", set()) 

222 

223 # We need to do three things with options: 

224 for selector, subast in token["options"].items(): 

225 # First, we log the selector for later comparison. 

226 choices.add(selector) 

227 

228 # Second, we ensure the selector is valid if we're working 

229 # with a plural/selectordinal type. 

230 if ttype in PLURAL_TYPES and check_bad_plural_selector(selector): 

231 data.setdefault("bad_plural", set()).add(selector) 

232 

233 # Finally, we process the sub-ast for this option. 

234 extract_placeholders(subast, variables) 

235 

236 # Make sure we process the contents sub-ast if one exists. 

237 if "contents" in token: 

238 extract_placeholders(token["contents"], variables) 

239 

240 return variables 

241 

242 

243class ICUCheckMixin: 

244 def get_flags(self, unit: Unit): 

245 if unit and unit.all_flags.has_value("icu-flags"): 

246 return unit.all_flags.get_value("icu-flags") 

247 return [] 

248 

249 def get_tag_prefix(self, unit: Unit): 

250 if unit and unit.all_flags.has_value("icu-tag-prefix"): 

251 return unit.all_flags.get_value("icu-tag-prefix") 

252 return None 

253 

254 

255class ICUSourceCheck(ICUCheckMixin, SourceCheck): 

256 """Check for ICU MessageFormat syntax.""" 

257 

258 check_id = "icu_message_format_syntax" 

259 name = gettext_lazy("ICU MessageFormat syntax") 

260 description = gettext_lazy("Syntax errors in ICU MessageFormat strings.") 

261 default_disabled = True 

262 

263 def __init__(self) -> None: 

264 super().__init__() 

265 self.enable_string = "icu-message-format" 

266 self.ignore_string = f"ignore-{self.enable_string}" 

267 

268 def check_source_unit(self, sources: list[str], unit: Unit) -> bool: 

269 """Checker for source strings. Only check for syntax issues.""" 

270 if not sources or not sources[0]: 

271 return False 

272 

273 flags = self.get_flags(unit) 

274 strict_tags = "strict-xml" in flags 

275 allow_tags = strict_tags or "xml" in flags 

276 tag_prefix = self.get_tag_prefix(unit) 

277 

278 _ast, src_err, _tokens = parse_icu( 

279 sources[0], allow_tags, strict_tags, tag_prefix 

280 ) 

281 return bool(src_err) 

282 

283 

284class ICUMessageFormatCheck(ICUCheckMixin, BaseFormatCheck): 

285 """Check for ICU MessageFormat string.""" 

286 

287 check_id = "icu_message_format" 

288 name = gettext_lazy("ICU MessageFormat") 

289 description = gettext_lazy( 

290 "Syntax errors and/or placeholder mismatches in ICU MessageFormat strings." 

291 ) 

292 

293 def check_format(self, source: str, target: str, ignore_missing, unit: Unit): 

294 """Checker for ICU MessageFormat strings.""" 

295 if not target or not source: 

296 return False 

297 

298 flags = self.get_flags(unit) 

299 strict_tags = "strict-xml" in flags 

300 allow_tags = strict_tags or "xml" in flags 

301 tag_prefix = self.get_tag_prefix(unit) 

302 

303 result = defaultdict(list) 

304 src_ast, src_err, _tokens = parse_icu( 

305 source, allow_tags, strict_tags, tag_prefix 

306 ) 

307 

308 # Check to see if we're running on a source string only. 

309 # If we are, then we can only run a syntax check on the 

310 # source and be done. 

311 if unit and unit.is_source: 

312 if src_err: 

313 result["syntax"].append(src_err) 

314 return result 

315 return False 

316 

317 tgt_ast, tgt_err, _tokens = parse_icu( 

318 target, allow_tags, strict_tags, tag_prefix 

319 ) 

320 if tgt_err: 

321 result["syntax"].append(tgt_err) 

322 

323 if tgt_err: 

324 return result 

325 if src_err: 

326 # We cannot run any further checks if the source 

327 # string isn't valid, so just accept that the target 

328 # string is valid for now. 

329 return False 

330 

331 # Both strings are valid! Congratulations. Let's extract 

332 # information on all the placeholders in both strings, and 

333 # compare them to see if anything is wrong. 

334 src_vars = extract_placeholders(src_ast) 

335 tgt_vars = extract_placeholders(tgt_ast) 

336 

337 # First, we check all the variables in the target. 

338 for name, data in tgt_vars.items(): 

339 self.check_for_other(result, name, data, flags) 

340 

341 if name in src_vars: 

342 src_data = src_vars[name] 

343 

344 self.check_bad_plural(result, name, data, src_data, flags) 

345 self.check_bad_submessage(result, name, data, src_data, flags) 

346 self.check_wrong_type(result, name, data, src_data, flags) 

347 

348 if allow_tags: 

349 self.check_tags(result, name, data, src_data, flags) 

350 

351 else: 

352 self.check_bad_submessage(result, name, data, None, flags) 

353 

354 # The variable does not exist in the source, 

355 # which suggests a mistake. 

356 if "-extra" not in flags: 

357 result["extra"].append(name) 

358 

359 # We also want to check for variables used in the 

360 # source but not in the target. 

361 self.check_missing(result, src_vars, tgt_vars, flags) 

362 

363 if result: 

364 return result 

365 return False 

366 

367 def check_missing(self, result, src_vars, tgt_vars, flags) -> None: 

368 """Detect any variables in the target not in the source.""" 

369 if "-missing" in flags: 

370 return 

371 

372 missing = [name for name in src_vars if name not in tgt_vars] 

373 

374 if missing: 

375 result["missing"] = missing 

376 

377 def check_for_other(self, result, name, data, flags) -> None: 

378 """Ensure types with sub-messages have other.""" 

379 if "-require_other" in flags: 

380 return 

381 

382 choices = data.get("choices") 

383 if choices and "other" not in choices: 

384 result["no_other"].append(name) 

385 

386 def check_bad_plural(self, result, name, data, src_data, flags) -> None: 

387 """Forward bad plural selectors detected during extraction.""" 

388 if "-plural_selectors" in flags: 

389 return 

390 

391 if "bad_plural" in data: 

392 result["bad_plural"].append([name, data["bad_plural"]]) 

393 

394 def check_bad_submessage(self, result, name, data, src_data, flags) -> None: 

395 """Detect any bad sub-message selectors.""" 

396 if "-submessage_selectors" in flags: 

397 return 

398 

399 bad = set() 

400 

401 # We also want to check individual select choices. 

402 if ( 

403 src_data 

404 and "select" in data["types"] 

405 and "select" in src_data["types"] 

406 and "choices" in data 

407 and "choices" in src_data 

408 ): 

409 choices = data["choices"] 

410 src_choices = src_data["choices"] 

411 

412 for selector in choices: 

413 if selector not in src_choices: 

414 bad.add(selector) 

415 

416 if bad: 

417 result["bad_submessage"].append([name, bad]) 

418 

419 def check_wrong_type(self, result, name, data, src_data, flags) -> None: 

420 """Ensure that types match, when possible.""" 

421 if "-types" in flags: 

422 return 

423 

424 # If we're dealing with a number, we want to use 

425 # special number logic, since numbers work with 

426 # multiple types. 

427 if isinstance(src_data["is_number"], bool) and src_data["is_number"]: 

428 if src_data["is_number"] != data["is_number"]: 

429 result["wrong_type"].append(name) 

430 

431 else: 

432 for ttype in data["types"]: 

433 if ttype not in src_data["types"]: 

434 result["wrong_type"].append(name) 

435 break 

436 

437 def check_tags(self, result, name, data, src_data, flags) -> None: 

438 """Correct any erroneous XML tags.""" 

439 if "-tags" in flags: 

440 return 

441 

442 if isinstance(src_data["is_tag"], bool) or data["is_tag"] is not None: 

443 if src_data["is_tag"]: 

444 if not data["is_tag"]: 

445 result["should_be_tag"].append(name) 

446 

447 elif ( 

448 isinstance(src_data["is_empty"], bool) 

449 and src_data["is_empty"] != data["is_empty"] 

450 ): 

451 if src_data["is_empty"]: 

452 result["tag_not_empty"].append(name) 

453 else: 

454 result["tag_empty"].append(name) 

455 

456 elif data["is_tag"]: 

457 result["not_tag"].append(name) 

458 

459 def format_result(self, result): 

460 if result.get("syntax"): 

461 yield gettext("Syntax error: %s") % format_html_join_comma( 

462 "{}", 

463 list_to_tuples(err.msg or "unknown error" for err in result["syntax"]), 

464 ) 

465 

466 if result.get("extra"): 

467 yield gettext( 

468 "One or more unknown placeholders in the translation: %s" 

469 ) % format_html_join_comma("{}", list_to_tuples(result["extra"])) 

470 

471 if result.get("missing"): 

472 yield gettext( 

473 "One or more placeholders missing in the translation: %s" 

474 ) % format_html_join_comma("{}", list_to_tuples(result["missing"])) 

475 

476 if result.get("wrong_type"): 

477 yield gettext( 

478 "One or more placeholder types are incorrect: %s" 

479 ) % format_html_join_comma("{}", list_to_tuples(result["wrong_type"])) 

480 

481 if result.get("no_other"): 

482 yield gettext("Missing other sub-message for: %s") % format_html_join_comma( 

483 "{}", list_to_tuples(result["no_other"]) 

484 ) 

485 

486 if result.get("bad_plural"): 

487 yield gettext( 

488 "Incorrect plural selectors for: %s" 

489 ) % format_html_join_comma( 

490 "{}", (f"{x[0]} ({', '.join(x[1])})" for x in result["bad_plural"]) 

491 ) 

492 

493 if result.get("bad_submessage"): 

494 yield gettext( 

495 "Incorrect sub-message selectors for: %s" 

496 ) % format_html_join_comma( 

497 "{}", (f"{x[0]} ({', '.join(x[1])})" for x in result["bad_submessage"]) 

498 ) 

499 

500 if result.get("should_be_tag"): 

501 yield gettext( 

502 "One or more placeholders should have " 

503 "a corresponding XML tag in the translation: %s" 

504 ) % format_html_join_comma("{}", list_to_tuples(result["should_be_tag"])) 

505 

506 if result.get("not_tag"): 

507 yield gettext( 

508 "One or more placeholders should not be " 

509 "an XML tag in the translation: %s" 

510 ) % format_html_join_comma("{}", list_to_tuples(result["not_tag"])) 

511 

512 if result.get("tag_not_empty"): 

513 yield gettext( 

514 "One or more XML tags has unexpected content in the translation: %s" 

515 ) % format_html_join_comma("{}", list_to_tuples(result["tag_not_empty"])) 

516 

517 if result.get("tag_empty"): 

518 yield gettext( 

519 "One or more XML tags missing content in the translation: %s" 

520 ) % format_html_join_comma("{}", list_to_tuples(result["tag_empty"])) 

521 

522 def check_highlight(self, source: str, unit: Unit): 

523 if self.should_skip(unit): 523 ↛ 526line 523 didn't jump to line 526 because the condition on line 523 was always true

524 return 

525 

526 flags = self.get_flags(unit) 

527 if "-highlight" in flags: 

528 return 

529 

530 strict_tags = "strict-xml" in flags 

531 allow_tags = strict_tags or "xml" in flags 

532 tag_prefix = self.get_tag_prefix(unit) 

533 

534 ast, _err, _tokens = parse_icu( 

535 source, allow_tags, strict_tags, tag_prefix, True 

536 ) 

537 if not ast: 

538 return 

539 

540 yield from extract_highlights(ast, source)