Coverage for app/venv/lib/python3.14/site-packages/weblate/machinery/base.py: 24%

480 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5"""Base code for machine translation services.""" 

6 

7from __future__ import annotations 

8 

9import contextlib 

10import random 

11import re 

12import time 

13from collections import defaultdict 

14from hashlib import md5 

15from html import escape, unescape 

16from itertools import chain 

17from typing import TYPE_CHECKING, ClassVar 

18from urllib.parse import quote 

19 

20from django.core.cache import cache 

21from django.core.exceptions import ValidationError 

22from django.utils.functional import cached_property 

23from django.utils.translation import gettext 

24from requests.exceptions import HTTPError, JSONDecodeError, RequestException 

25 

26from weblate.checks.utils import highlight_string 

27from weblate.lang.models import Language, PluralMapper 

28from weblate.machinery.forms import BaseMachineryForm 

29from weblate.utils.errors import report_error 

30from weblate.utils.hash import calculate_dict_hash, calculate_hash, hash_to_checksum 

31from weblate.utils.requests import request 

32from weblate.utils.similarity import Comparer 

33from weblate.utils.site import get_site_url 

34 

35from .types import ( 

36 SourceLanguageChoices, 

37) 

38 

39if TYPE_CHECKING: 39 ↛ 40line 39 didn't jump to line 40 because the condition on line 39 was never true

40 from collections.abc import Iterable, Iterator 

41 

42 from requests.auth import AuthBase 

43 

44 from weblate.auth.models import User 

45 from weblate.trans.models import Translation, Unit 

46 from weblate.trans.models.unit import UnitQuerySet 

47 

48 from .types import ( 

49 DownloadMultipleTranslations, 

50 DownloadTranslations, 

51 SettingsDict, 

52 TranslationResultDict, 

53 UnitMemoryResultDict, 

54 ) 

55 

56 

57def get_machinery_language(language: Language) -> Language: 

58 if language.code.endswith("_devel"): 

59 return Language.objects.get(code=language.code[:-6]) 

60 return language 

61 

62 

63class MachineTranslationError(Exception): 

64 """Generic Machine translation error.""" 

65 

66 

67class MachineryRateLimitError(MachineTranslationError): 

68 """Raised when rate limiting is detected.""" 

69 

70 

71class UnsupportedLanguageError(MachineTranslationError): 

72 """Raised when language is not supported.""" 

73 

74 

75class GlossaryAlreadyExistsError(MachineTranslationError): 

76 """Raised when glossary creation fails because it already exists.""" 

77 

78 

79class GlossaryDoesNotExistError(MachineTranslationError): 

80 """Raised when glossary deletion fails because it does not exist.""" 

81 

82 

83class BatchMachineTranslation: 

84 """Generic object for machine translation services.""" 

85 

86 name = "MT" 

87 max_score = 100 

88 rank_boost = 0 

89 cache_translations = True 

90 language_map: ClassVar[dict[str, str]] = {} 

91 same_languages = False 

92 do_cleanup = True 

93 # Batch size is currently used in autotranslate 

94 batch_size = 20 

95 accounting_key = "external" 

96 force_uncleanup = False 

97 highlight_syntax = False 

98 glossary_support = False 

99 settings_form: type[BaseMachineryForm] | None = BaseMachineryForm 

100 request_timeout = 5 

101 is_available = True 

102 replacement_start = "[X" 

103 replacement_end = "X]" 

104 # Cache results for 30 days 

105 cache_expiry = 30 * 24 * 3600 

106 

107 @classmethod 

108 def get_rank(cls): 

109 return cls.max_score + cls.rank_boost 

110 

111 def __init__(self, settings: SettingsDict) -> None: 

112 """Create new machine translation object.""" 

113 self.mtid = self.get_identifier() 

114 self.rate_limit_cache = f"{self.mtid}-rate-limit" 

115 self.languages_cache = f"{self.mtid}-languages" 

116 self.comparer = Comparer() 

117 self.supported_languages_error: Exception | None = None 

118 self.supported_languages_error_age: float = 0 

119 self.settings = settings 

120 

121 def delete_cache(self) -> None: 

122 cache.delete_many([self.rate_limit_cache, self.languages_cache]) 

123 

124 def validate_settings(self) -> None: 

125 try: 

126 self.download_languages() 

127 except Exception as error: 

128 raise ValidationError( 

129 gettext("Could not fetch supported languages: %s") % error 

130 ) from error 

131 try: 

132 self.download_multiple_translations("en", "de", [("test", None)], None, 75) 

133 except Exception as error: 

134 raise ValidationError( 

135 gettext("Could not fetch translation: %s") % error 

136 ) from error 

137 

138 @property 

139 def api_base_url(self): 

140 base = self.settings["url"] 

141 if base.endswith("/"): 

142 base = base.rstrip("/") 

143 return base 

144 

145 def get_api_url(self, *parts): 

146 """Generate service URL gracefully handle trailing slashes.""" 

147 return "/".join( 

148 chain([self.api_base_url], (quote(part, b"") for part in parts)) 

149 ) 

150 

151 @classmethod 

152 def get_identifier(cls): 

153 return cls.name.lower().replace(" ", "-") 

154 

155 @classmethod 

156 def get_doc_anchor(cls) -> str: 

157 return f"mt-{cls.get_identifier()}" 

158 

159 def account_usage(self, project, delta: int = 1) -> None: 

160 key = f"machinery-accounting:{self.accounting_key}:{project.id}" 

161 try: 

162 cache.incr(key, delta=delta) 

163 except ValueError: 

164 cache.set(key, delta, 24 * 3600) 

165 

166 def get_headers(self) -> dict[str, str]: 

167 """Add authentication headers to request.""" 

168 return {} 

169 

170 def get_auth(self) -> tuple[str, str] | AuthBase | None: 

171 return None 

172 

173 def check_failure(self, response) -> None: 

174 # Directly raise error as last resort, subclass can prepend this 

175 # with something more clever 

176 try: 

177 response.raise_for_status() 

178 except HTTPError as error: 

179 detail = response.text 

180 try: 

181 payload = response.json() 

182 except JSONDecodeError: 

183 pass 

184 else: 

185 if isinstance(payload, dict) and payload: 

186 if detail_error := payload.get("error"): 

187 if isinstance(detail_error, str): 

188 detail = detail_error 

189 elif isinstance(detail_error, dict): 

190 if "message" in detail_error: 

191 detail = detail_error["message"] 

192 else: 

193 detail = str(detail_error) 

194 else: 

195 detail = str(payload) 

196 

197 if detail: 

198 message = f"{error.args[0]}: {detail[:200]}" 

199 raise HTTPError(message, response=response) from error 

200 raise 

201 

202 def request(self, method, url, skip_auth=False, **kwargs): 

203 """Perform JSON request.""" 

204 # Create custom headers 

205 headers = { 

206 "Referer": get_site_url(), 

207 "Accept": "application/json; charset=utf-8", 

208 } 

209 if "headers" in kwargs: 

210 headers.update(kwargs.pop("headers")) 

211 # Optional authentication 

212 if not skip_auth: 

213 headers.update(self.get_headers()) 

214 

215 # Fire request 

216 response = request( 

217 method, 

218 url, 

219 headers=headers, 

220 timeout=self.request_timeout, 

221 auth=self.get_auth(), 

222 raise_for_status=False, 

223 **kwargs, 

224 ) 

225 

226 self.check_failure(response) 

227 

228 return response 

229 

230 def download_languages(self): 

231 """Download list of supported languages from a service.""" 

232 return [] 

233 

234 def map_language_code(self, code: str) -> str: 

235 """Map language code to service specific.""" 

236 code = code.removesuffix("_devel") 

237 if code in self.language_map: 

238 return self.language_map[code] 

239 return code 

240 

241 def report_error( 

242 self, cause: str, extra_log: str | None = None, message: bool = False 

243 ) -> None: 

244 """Report error situations.""" 

245 report_error( 

246 f"machinery[{self.name}]: {cause}", extra_log=extra_log, message=message 

247 ) 

248 

249 @cached_property 

250 def supported_languages(self): 

251 """Return list of supported languages.""" 

252 # Try using list from cache 

253 languages_cache = cache.get(self.languages_cache) 

254 if languages_cache is not None: 

255 # hiredis-py 3 makes list from set 

256 return set(languages_cache) 

257 

258 if self.is_rate_limited(): 

259 return set() 

260 

261 # Download 

262 try: 

263 languages = set(self.download_languages()) 

264 except Exception as exc: 

265 self.supported_languages_error = exc 

266 self.supported_languages_error_age = time.time() 

267 self.report_error("Could not fetch languages, using defaults") 

268 return set() 

269 

270 # Update cache 

271 cache.set(self.languages_cache, languages, 3600 * 48) 

272 return languages 

273 

274 def is_supported(self, source, language): 

275 """Check whether given language combination is supported.""" 

276 return ( 

277 language in self.supported_languages 

278 and source in self.supported_languages 

279 and source != language 

280 ) 

281 

282 def is_rate_limited(self): 

283 return cache.get(self.rate_limit_cache, False) 

284 

285 def set_rate_limit(self): 

286 return cache.set(self.rate_limit_cache, True, 1800) 

287 

288 def is_rate_limit_error(self, exc) -> bool: 

289 if isinstance(exc, MachineryRateLimitError): 

290 return True 

291 if not isinstance(exc, HTTPError): 

292 return False 

293 # Apply rate limiting for following status codes: 

294 # HTTP 456 Client Error: Quota Exceeded (DeepL) 

295 # HTTP 429 Too Many Requests 

296 # HTTP 401 Unauthorized 

297 # HTTP 403 Forbidden 

298 # HTTP 503 Service Unavailable 

299 return exc.response.status_code in {456, 429, 401, 403, 503} 

300 

301 def get_cache_key( 

302 self, scope: str, *, parts: Iterable[str | int] = (), text: str | None = None 

303 ) -> str: 

304 """ 

305 Cache key for caching translations. 

306 

307 Used to avoid fetching same translations again. 

308 

309 This includes project ID for project scoped entries via 

310 Project.get_machinery_settings. 

311 """ 

312 key = [ 

313 "mt", 

314 self.mtid, 

315 scope, 

316 calculate_dict_hash(self.settings), 

317 *parts, 

318 ] 

319 if text is not None: 

320 key.append(calculate_hash(text)) 

321 

322 return ":".join(str(part) for part in key) 

323 

324 def unescape_text(self, text: str): 

325 """Unescaping of the text with replacements.""" 

326 return text 

327 

328 def escape_text(self, text: str): 

329 """Escaping of the text with replacements.""" 

330 return text 

331 

332 def make_re_placeholder(self, text: str): 

333 """Convert placeholder into a regular expression.""" 

334 # Allow additional space before ] 

335 return re.escape(text[:-1]) + " *" + re.escape(text[-1:]) 

336 

337 def format_replacement( 

338 self, h_start: int, h_end: int, h_text: str, h_kind: Unit | None 

339 ) -> str: 

340 """Generate a single replacement.""" 

341 return f"{self.replacement_start}{h_start}{self.replacement_end}" 

342 

343 def get_highlights( 

344 self, text: str, unit 

345 ) -> Iterable[tuple[int, int, str, Unit | None]]: 

346 for h_start, h_end, h_text in highlight_string( 

347 text, unit, highlight_syntax=self.highlight_syntax 

348 ): 

349 yield h_start, h_end, h_text, None 

350 

351 def cleanup_text(self, text: str, unit: Unit) -> tuple[str, dict[str, str]]: 

352 """Remove placeholder to avoid confusing the machine translation.""" 

353 replacements: dict[str, str] = {} 

354 if not self.do_cleanup: 

355 return text, replacements 

356 

357 parts = [] 

358 start = 0 

359 for h_start, h_end, h_text, h_kind in self.get_highlights(text, unit): 

360 parts.append(self.escape_text(text[start:h_start])) 

361 h_text = self.escape_text(h_text) 

362 placeholder = self.format_replacement(h_start, h_end, h_text, h_kind) 

363 replacements[placeholder] = h_text 

364 parts.append(placeholder) 

365 start = h_end 

366 

367 parts.append(self.escape_text(text[start:])) 

368 

369 return "".join(parts), replacements 

370 

371 def uncleanup_text(self, replacements: dict[str, str], text: str) -> str: 

372 for source, target in replacements.items(): 

373 text = re.sub(self.make_re_placeholder(source), target, text) 

374 return self.unescape_text(text) 

375 

376 def uncleanup_results( 

377 self, replacements: dict[str, str], results: list[TranslationResultDict] 

378 ) -> None: 

379 """Reverts replacements done by cleanup_text.""" 

380 for result in results: 

381 result["text"] = self.uncleanup_text(replacements, result["text"]) 

382 result["source"] = self.uncleanup_text(replacements, result["source"]) 

383 

384 def get_language_possibilities(self, language: Language) -> Iterator[str]: 

385 code = language.code 

386 mapped_code = self.map_language_code(code) 

387 if not mapped_code: 

388 return 

389 yield mapped_code 

390 code = code.replace("-", "_") 

391 while "_" in code: 

392 code = code.rsplit("_", 1)[0] 

393 yield self.map_language_code(code) 

394 

395 def get_languages( 

396 self, source_language: Language, target_language: Language 

397 ) -> tuple[str, str]: 

398 if source_language == target_language and not self.same_languages: 

399 msg = "Same languages" 

400 raise UnsupportedLanguageError(msg) 

401 

402 for source in self.get_language_possibilities(source_language): 

403 for target in self.get_language_possibilities(target_language): 

404 if self.is_supported(source, target): 

405 return source, target 

406 

407 if self.supported_languages_error: 

408 if self.supported_languages_error_age + 3600 > time.time(): 

409 raise MachineTranslationError(repr(self.supported_languages_error)) 

410 self.supported_languages_error = None 

411 self.supported_languages_error_age = 0 

412 

413 msg = "Not supported" 

414 raise UnsupportedLanguageError(msg) 

415 

416 def get_cached( 

417 self, 

418 unit, 

419 source_language, 

420 target_language, 

421 text, 

422 threshold, 

423 replacements, 

424 *extra_parts, 

425 ) -> tuple[str | None, list[TranslationResultDict] | None]: 

426 if not self.cache_translations: 

427 return None, None 

428 cache_key = self.get_cache_key( 

429 "translation", 

430 parts=(source_language, target_language, threshold, *extra_parts), 

431 text=text, 

432 ) 

433 result = cache.get(cache_key) 

434 if result and (replacements or self.force_uncleanup): 

435 self.uncleanup_results(replacements, result) 

436 return cache_key, result 

437 

438 def search(self, unit, text, user: User | None): 

439 """Search for known translations of `text`.""" 

440 translation = unit.translation 

441 try: 

442 source_language, target_language = self.get_languages( 

443 translation.component.source_language, translation.language 

444 ) 

445 except UnsupportedLanguageError: 

446 unit.translation.log_debug( 

447 "machinery failed: not supported language pair: %s - %s", 

448 translation.component.source_language.code, 

449 translation.language.code, 

450 ) 

451 return [] 

452 

453 self.account_usage(translation.component.project) 

454 return self._translate( 

455 source_language, target_language, [(text, unit)], user, threshold=10 

456 )[text] 

457 

458 def get_default_source_language(self, translation: Translation) -> Language: 

459 """Return default source language for the translation.""" 

460 return translation.component.source_language 

461 

462 def get_source_language(self, translation: Translation) -> Language: 

463 selection = self.settings.get("source_language", SourceLanguageChoices.AUTO) 

464 

465 if selection == SourceLanguageChoices.SOURCE: 

466 return translation.component.source_language 

467 

468 if selection == SourceLanguageChoices.SECONDARY: 

469 # Use secondary if configured 

470 if translation.component.secondary_language: 

471 return translation.component.secondary_language 

472 if translation.component.project.secondary_language: 

473 return translation.component.project.secondary_language 

474 

475 return self.get_default_source_language(translation) 

476 

477 def translate( 

478 self, 

479 unit: Unit, 

480 user: User | None = None, 

481 threshold: int = 75, 

482 *, 

483 source_language: Language | None = None, 

484 ): 

485 """Return list of machine translations.""" 

486 translation = unit.translation 

487 if source_language is None: 

488 # Fall back to component source language 

489 source_language = self.get_source_language(translation) 

490 translating_from_source: bool = ( 

491 translation.component.source_language == source_language 

492 ) 

493 

494 try: 

495 mapped_source_language, target_language = self.get_languages( 

496 source_language, translation.language 

497 ) 

498 except UnsupportedLanguageError: 

499 unit.translation.log_debug( 

500 "machinery failed: not supported language pair: %s - %s", 

501 source_language.code, 

502 translation.language.code, 

503 ) 

504 return [] 

505 

506 self.account_usage(translation.component.project) 

507 

508 source_plural = source_language.plural 

509 target_plural = translation.plural 

510 plural_mapper = PluralMapper(source_plural, target_plural) 

511 alternate_units: dict[int, Unit] | None = None 

512 if not translating_from_source: 

513 alternate_units = plural_mapper.get_other_units([unit], source_language) 

514 

515 plural_mapper.map_units([unit], alternate_units) 

516 translations = self._translate( 

517 mapped_source_language, 

518 target_language, 

519 [(text, unit) for text in unit.plural_map], 

520 user, 

521 threshold=threshold, 

522 ) 

523 return [translations[text] for text in unit.plural_map] 

524 

525 def download_multiple_translations( 

526 self, 

527 source_language, 

528 target_language, 

529 sources: list[tuple[str, Unit | None]], 

530 user: User | None = None, 

531 threshold: int = 75, 

532 ) -> DownloadMultipleTranslations: 

533 """ 

534 Download dictionary of a lists of possible translations from a service. 

535 

536 Should return dict with translation text, translation quality, source of 

537 translation, source string. 

538 

539 You can use self.name as source of translation, if you can not give 

540 better hint and text parameter as source string if you do no fuzzy 

541 matching. 

542 """ 

543 raise NotImplementedError 

544 

545 def _translate( 

546 self, 

547 source_language, 

548 target_language, 

549 sources: list[tuple[str, Unit]], 

550 user=None, 

551 threshold: int = 75, 

552 ) -> DownloadMultipleTranslations: 

553 output: DownloadMultipleTranslations = {} 

554 pending = defaultdict(list) 

555 cache_keys: dict[str, str | None] = {} 

556 result: list[TranslationResultDict] | None 

557 for text, unit in sources: 

558 original_source = text 

559 text, replacements = self.cleanup_text(text, unit) 

560 

561 if not text or self.is_rate_limited(): 

562 output[original_source] = [] 

563 continue 

564 

565 # Try cached results 

566 cache_keys[text], result = self.get_cached( 

567 unit, source_language, target_language, text, threshold, replacements 

568 ) 

569 if result is not None: 

570 output[original_source] = result 

571 continue 

572 

573 pending[text].append((unit, original_source, replacements)) 

574 

575 # Fetch pending strings to translate 

576 if pending: 

577 # Unit is only used in WeblateMemory and it is used only to get a project 

578 # so it doesn't matter we potentially flatten this. 

579 try: 

580 translations = self.download_multiple_translations( 

581 source_language, 

582 target_language, 

583 [ 

584 (text, occurrences[0][0]) 

585 for text, occurrences in pending.items() 

586 ], 

587 user, 

588 threshold, 

589 ) 

590 except Exception as exc: 

591 if self.is_rate_limit_error(exc): 

592 self.set_rate_limit() 

593 

594 self.report_error("Could not fetch translations") 

595 if isinstance(exc, MachineTranslationError): 

596 raise 

597 raise MachineTranslationError(self.get_error_message(exc)) from exc 

598 

599 # Postprocess translations 

600 for text, result in translations.items(): 

601 for _unit, original_source, replacements in pending[text]: 

602 # Always operate on copy of the dictionaries 

603 partial = [x.copy() for x in result] 

604 

605 for item in partial: 

606 item["original_source"] = original_source 

607 if cache_key := cache_keys[text]: 

608 cache.set(cache_key, partial, self.cache_expiry) 

609 if replacements or self.force_uncleanup: 

610 self.uncleanup_results(replacements, partial) 

611 output[original_source] = partial 

612 return output 

613 

614 def get_error_message(self, exc: Exception) -> str: 

615 if isinstance(exc, RequestException) and exc.response and exc.response.text: 

616 return f"{exc.__class__.__name__}: {exc}: {exc.response.text}" 

617 return f"{exc.__class__.__name__}: {exc}" 

618 

619 def signed_salt(self, appid, secret, text): 

620 """Generate salt and sign as used by Chinese services.""" 

621 salt = str(random.randint(0, 10000000000)) # noqa: S311 

622 

623 payload = appid + text + salt + secret 

624 digest = md5(payload.encode(), usedforsecurity=False).hexdigest() 

625 

626 return salt, digest 

627 

628 def batch_translate( 

629 self, 

630 units: list[Unit] | UnitQuerySet, 

631 user: User | None = None, 

632 threshold: int = 75, 

633 *, 

634 source_language: Language | None = None, 

635 ) -> None: 

636 try: 

637 translation = units[0].translation 

638 except IndexError: 

639 return 

640 

641 if source_language is None: 

642 # Fall back to component source language 

643 source_language = self.get_source_language(translation) 

644 

645 translating_from_source: bool = ( 

646 translation.component.source_language == source_language 

647 ) 

648 

649 try: 

650 source, language = self.get_languages(source_language, translation.language) 

651 except UnsupportedLanguageError: 

652 return 

653 

654 self.account_usage(translation.component.project, delta=len(units)) 

655 

656 source_plural = source_language.plural 

657 target_plural = translation.plural 

658 plural_mapper = PluralMapper(source_plural, target_plural) 

659 alternate_units: dict[int, Unit] | None = None 

660 if not translating_from_source: 

661 alternate_units = plural_mapper.get_other_units(units, source_language) 

662 plural_mapper.map_units(units, alternate_units) 

663 

664 # TODO: fetch source from other units 

665 sources = [(text, unit) for unit in units for text in unit.plural_map] 

666 translations = self._translate(source, language, sources, user, threshold) 

667 

668 for unit in units: 

669 result: UnitMemoryResultDict = unit.machinery 

670 if min(result.get("quality", ()), default=0) >= self.max_score: 

671 continue 

672 translation_lists = [translations[text] for text in unit.plural_map] 

673 plural_count = len(translation_lists) 

674 translation = result.setdefault("translation", [""] * plural_count) 

675 quality = result.setdefault("quality", [0] * plural_count) 

676 origin = result.setdefault("origin", [None] * plural_count) 

677 for plural, possible_translations in enumerate(translation_lists): 

678 for item in possible_translations: 

679 if quality[plural] > item["quality"]: 

680 continue 

681 quality[plural] = item["quality"] 

682 translation[plural] = item["text"] 

683 origin[plural] = self 

684 

685 @cached_property 

686 def user(self): 

687 """Weblate user used to track changes by this engine.""" 

688 from weblate.auth.models import User 

689 

690 return User.objects.get_or_create_bot( 

691 scope="mt", 

692 name=self.get_identifier(), 

693 verbose=self.name, 

694 ) 

695 

696 

697class MachineTranslation(BatchMachineTranslation): 

698 def download_translations( 

699 self, 

700 source_language, 

701 target_language, 

702 text: str, 

703 unit: Unit | None, 

704 user: User | None, 

705 threshold: int = 75, 

706 ) -> DownloadTranslations: 

707 """ 

708 Download list of possible translations from a service. 

709 

710 Should return dict with translation text, translation quality, source of 

711 translation, source string. 

712 

713 You can use self.name as source of translation, if you can not give 

714 better hint and text parameter as source string if you do no fuzzy 

715 matching. 

716 """ 

717 raise NotImplementedError 

718 

719 def download_multiple_translations( 

720 self, 

721 source_language, 

722 target_language, 

723 sources: list[tuple[str, Unit | None]], 

724 user: User | None = None, 

725 threshold: int = 75, 

726 ) -> DownloadMultipleTranslations: 

727 return { 

728 text: list( 

729 self.download_translations( 

730 source_language, 

731 target_language, 

732 text, 

733 unit, 

734 user, 

735 threshold=threshold, 

736 ) 

737 ) 

738 for text, unit in sources 

739 } 

740 

741 

742class InternalMachineTranslation(MachineTranslation): 

743 do_cleanup = False 

744 accounting_key = "internal" 

745 cache_translations = False 

746 settings_form: type[BaseMachineryForm] | None = None 

747 

748 def is_supported( 

749 self, source_language: Language, target_language: Language 

750 ) -> bool: 

751 """Any language is supported.""" 

752 return True 

753 

754 def is_rate_limited(self) -> bool: 

755 """Disable rate limiting.""" 

756 return False 

757 

758 def get_language_possibilities(self, language: Language) -> Iterator[Language]: # type: ignore[override] 

759 yield get_machinery_language(language) 

760 

761 

762class GlossaryMachineTranslationMixin(MachineTranslation): 

763 glossary_name_format = ( 

764 "weblate:{project}:{source_language}:{target_language}:{checksum}" 

765 ) 

766 glossary_name_format_pattern = ( 

767 r"weblate:(\d+):([A-z0-9@_-]+):([A-z0-9@_-]+):([a-f0-9]+)" 

768 ) 

769 glossary_support = True 

770 

771 glossary_count_limit = 0 

772 

773 def delete_cache(self) -> None: 

774 """Delete general caches and glossary cache.""" 

775 super().delete_cache() 

776 cache.delete(self.get_cache_key("glossaries")) 

777 

778 def is_glossary_supported(self, source_language: str, target_language: str) -> bool: 

779 return True 

780 

781 def list_glossaries(self) -> dict[str, str]: 

782 """ 

783 List glossaries from the service. 

784 

785 Returns dictionary with names and id. 

786 """ 

787 raise NotImplementedError 

788 

789 def delete_glossary(self, glossary_id: str) -> None: 

790 raise NotImplementedError 

791 

792 def delete_oldest_glossary(self) -> None: 

793 raise NotImplementedError 

794 

795 def create_glossary( 

796 self, source_language: str, target_language: str, name: str, tsv: str 

797 ) -> None: 

798 """ 

799 Create glossary in the service. 

800 

801 - Creates the glossary in the service 

802 - May raise GlossaryAlreadyExists if creation fails 

803 - Performs any other necessary operation, e.g uploading TSV file to bucket 

804 """ 

805 raise NotImplementedError 

806 

807 def get_glossaries(self, use_cache: bool = True) -> dict[str, str]: 

808 cache_key = self.get_cache_key("glossaries") 

809 if use_cache: 

810 cached = cache.get(cache_key) 

811 if cached is not None: 

812 return cached 

813 

814 result = self.list_glossaries() 

815 

816 cache.set(cache_key, result, 24 * 3600) 

817 return result 

818 

819 def tsv_checksum(self, tsv: str) -> str: 

820 """Calculate checksum of given TSV glossary.""" 

821 return hash_to_checksum(calculate_hash(tsv)) if tsv else "" 

822 

823 def get_cached( 

824 self, 

825 unit, 

826 source_language, 

827 target_language, 

828 text, 

829 threshold, 

830 replacements, 

831 *extra_parts, 

832 ): 

833 """Retrieve cached translation with glossary checksum.""" 

834 from weblate.glossary.models import get_glossary_tsv 

835 

836 return super().get_cached( 

837 unit, 

838 source_language, 

839 target_language, 

840 text, 

841 threshold, 

842 replacements, 

843 self.tsv_checksum(get_glossary_tsv(unit.translation)), 

844 *extra_parts, 

845 ) 

846 

847 def get_glossary_count_limit(self) -> int: 

848 return self.glossary_count_limit 

849 

850 def get_glossary_id( 

851 self, source_language: str, target_language: str, unit: Unit | None 

852 ) -> str | None: 

853 from weblate.glossary.models import get_glossary_tsv 

854 

855 if unit is None: 

856 return None 

857 

858 translation = unit.translation 

859 

860 # Check glossary support for a language pair 

861 if not self.is_glossary_supported(source_language, target_language): 

862 return None 

863 

864 # Check if there is a glossary 

865 glossary_tsv = get_glossary_tsv(translation) 

866 if not glossary_tsv: 

867 return None 

868 

869 # Calculate hash to check for changes 

870 glossary_checksum = self.tsv_checksum(glossary_tsv) 

871 glossary_name = self.glossary_name_format.format( 

872 project=translation.component.project.id, 

873 source_language=source_language, 

874 target_language=target_language, 

875 checksum=glossary_checksum, 

876 ) 

877 

878 # Fetch list of glossaries 

879 glossaries = self.get_glossaries() 

880 if glossary_name in glossaries: 

881 return glossaries[glossary_name] 

882 

883 # Remove stale glossaries for this language pair 

884 hashless_name = self.glossary_name_format.format( 

885 project=translation.component.project.id, 

886 source_language=source_language, 

887 target_language=target_language, 

888 checksum="", 

889 ) 

890 for name, glossary_id in glossaries.items(): 

891 if name.startswith(hashless_name): 

892 translation.log_debug( 

893 "%s: removing stale glossary %s (%s)", self.mtid, name, glossary_id 

894 ) 

895 with contextlib.suppress(GlossaryDoesNotExistError): 

896 self.delete_glossary(glossary_id) 

897 

898 # Ensure we are in service limits 

899 glossary_count_limit = self.get_glossary_count_limit() 

900 if glossary_count_limit and len(glossaries) + 1 >= glossary_count_limit: 

901 translation.log_debug( 

902 "%s: approached limit of %d glossaries, removing oldest glossary", 

903 self.mtid, 

904 self.glossary_count_limit, 

905 ) 

906 with contextlib.suppress(GlossaryDoesNotExistError): 

907 self.delete_oldest_glossary() 

908 

909 # Create new glossary 

910 translation.log_debug("%s: creating glossary %s", self.mtid, glossary_name) 

911 with contextlib.suppress(GlossaryAlreadyExistsError): 

912 self.create_glossary( 

913 source_language, target_language, glossary_name, glossary_tsv 

914 ) 

915 

916 # Fetch glossaries again, without using cache 

917 glossaries = self.get_glossaries(use_cache=False) 

918 return glossaries[glossary_name] 

919 

920 def match_name_format(self, string: str) -> re.Match | None: 

921 """ 

922 Match glossary name against format. 

923 

924 Only way so far to identify glossaries from memories 

925 """ 

926 return re.match(self.glossary_name_format_pattern, string) 

927 

928 

929class XMLMachineTranslationMixin(BatchMachineTranslation): 

930 highlight_syntax = True 

931 force_uncleanup = True 

932 

933 def unescape_text(self, text: str) -> str: 

934 """Unescaping of the text with replacements.""" 

935 return unescape(text) 

936 

937 def escape_text(self, text: str) -> str: 

938 """Escaping of the text with replacements.""" 

939 return escape(text) 

940 

941 def format_replacement( 

942 self, h_start: int, h_end: int, h_text: str, h_kind: Unit | None 

943 ) -> str: 

944 """Generate a single replacement.""" 

945 raise NotImplementedError 

946 

947 def make_re_placeholder(self, text: str) -> str: 

948 return re.escape(text) 

949 

950 

951class ResponseStatusMachineTranslation(MachineTranslation): 

952 def check_failure(self, response) -> None: 

953 payload = response.json() 

954 

955 # Check response status 

956 response_status = payload.get("responseStatus", payload.get("code", None)) 

957 if response_status and response_status != 200: 

958 error_text = payload.get( 

959 "responseDetails", 

960 payload.get( 

961 "message", 

962 payload.get("status", f"Response status {response_status}"), 

963 ), 

964 ) 

965 if response_status == 429: 

966 raise MachineryRateLimitError(error_text) 

967 raise MachineTranslationError(error_text) 

968 

969 super().check_failure(response)