Coverage for app/venv/lib/python3.14/site-packages/weblate/machinery/base.py: 24%
480 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5"""Base code for machine translation services."""
7from __future__ import annotations
9import contextlib
10import random
11import re
12import time
13from collections import defaultdict
14from hashlib import md5
15from html import escape, unescape
16from itertools import chain
17from typing import TYPE_CHECKING, ClassVar
18from urllib.parse import quote
20from django.core.cache import cache
21from django.core.exceptions import ValidationError
22from django.utils.functional import cached_property
23from django.utils.translation import gettext
24from requests.exceptions import HTTPError, JSONDecodeError, RequestException
26from weblate.checks.utils import highlight_string
27from weblate.lang.models import Language, PluralMapper
28from weblate.machinery.forms import BaseMachineryForm
29from weblate.utils.errors import report_error
30from weblate.utils.hash import calculate_dict_hash, calculate_hash, hash_to_checksum
31from weblate.utils.requests import request
32from weblate.utils.similarity import Comparer
33from weblate.utils.site import get_site_url
35from .types import (
36 SourceLanguageChoices,
37)
39if TYPE_CHECKING: 39 ↛ 40line 39 didn't jump to line 40 because the condition on line 39 was never true
40 from collections.abc import Iterable, Iterator
42 from requests.auth import AuthBase
44 from weblate.auth.models import User
45 from weblate.trans.models import Translation, Unit
46 from weblate.trans.models.unit import UnitQuerySet
48 from .types import (
49 DownloadMultipleTranslations,
50 DownloadTranslations,
51 SettingsDict,
52 TranslationResultDict,
53 UnitMemoryResultDict,
54 )
57def get_machinery_language(language: Language) -> Language:
58 if language.code.endswith("_devel"):
59 return Language.objects.get(code=language.code[:-6])
60 return language
63class MachineTranslationError(Exception):
64 """Generic Machine translation error."""
67class MachineryRateLimitError(MachineTranslationError):
68 """Raised when rate limiting is detected."""
71class UnsupportedLanguageError(MachineTranslationError):
72 """Raised when language is not supported."""
75class GlossaryAlreadyExistsError(MachineTranslationError):
76 """Raised when glossary creation fails because it already exists."""
79class GlossaryDoesNotExistError(MachineTranslationError):
80 """Raised when glossary deletion fails because it does not exist."""
83class BatchMachineTranslation:
84 """Generic object for machine translation services."""
86 name = "MT"
87 max_score = 100
88 rank_boost = 0
89 cache_translations = True
90 language_map: ClassVar[dict[str, str]] = {}
91 same_languages = False
92 do_cleanup = True
93 # Batch size is currently used in autotranslate
94 batch_size = 20
95 accounting_key = "external"
96 force_uncleanup = False
97 highlight_syntax = False
98 glossary_support = False
99 settings_form: type[BaseMachineryForm] | None = BaseMachineryForm
100 request_timeout = 5
101 is_available = True
102 replacement_start = "[X"
103 replacement_end = "X]"
104 # Cache results for 30 days
105 cache_expiry = 30 * 24 * 3600
107 @classmethod
108 def get_rank(cls):
109 return cls.max_score + cls.rank_boost
111 def __init__(self, settings: SettingsDict) -> None:
112 """Create new machine translation object."""
113 self.mtid = self.get_identifier()
114 self.rate_limit_cache = f"{self.mtid}-rate-limit"
115 self.languages_cache = f"{self.mtid}-languages"
116 self.comparer = Comparer()
117 self.supported_languages_error: Exception | None = None
118 self.supported_languages_error_age: float = 0
119 self.settings = settings
121 def delete_cache(self) -> None:
122 cache.delete_many([self.rate_limit_cache, self.languages_cache])
124 def validate_settings(self) -> None:
125 try:
126 self.download_languages()
127 except Exception as error:
128 raise ValidationError(
129 gettext("Could not fetch supported languages: %s") % error
130 ) from error
131 try:
132 self.download_multiple_translations("en", "de", [("test", None)], None, 75)
133 except Exception as error:
134 raise ValidationError(
135 gettext("Could not fetch translation: %s") % error
136 ) from error
138 @property
139 def api_base_url(self):
140 base = self.settings["url"]
141 if base.endswith("/"):
142 base = base.rstrip("/")
143 return base
145 def get_api_url(self, *parts):
146 """Generate service URL gracefully handle trailing slashes."""
147 return "/".join(
148 chain([self.api_base_url], (quote(part, b"") for part in parts))
149 )
151 @classmethod
152 def get_identifier(cls):
153 return cls.name.lower().replace(" ", "-")
155 @classmethod
156 def get_doc_anchor(cls) -> str:
157 return f"mt-{cls.get_identifier()}"
159 def account_usage(self, project, delta: int = 1) -> None:
160 key = f"machinery-accounting:{self.accounting_key}:{project.id}"
161 try:
162 cache.incr(key, delta=delta)
163 except ValueError:
164 cache.set(key, delta, 24 * 3600)
166 def get_headers(self) -> dict[str, str]:
167 """Add authentication headers to request."""
168 return {}
170 def get_auth(self) -> tuple[str, str] | AuthBase | None:
171 return None
173 def check_failure(self, response) -> None:
174 # Directly raise error as last resort, subclass can prepend this
175 # with something more clever
176 try:
177 response.raise_for_status()
178 except HTTPError as error:
179 detail = response.text
180 try:
181 payload = response.json()
182 except JSONDecodeError:
183 pass
184 else:
185 if isinstance(payload, dict) and payload:
186 if detail_error := payload.get("error"):
187 if isinstance(detail_error, str):
188 detail = detail_error
189 elif isinstance(detail_error, dict):
190 if "message" in detail_error:
191 detail = detail_error["message"]
192 else:
193 detail = str(detail_error)
194 else:
195 detail = str(payload)
197 if detail:
198 message = f"{error.args[0]}: {detail[:200]}"
199 raise HTTPError(message, response=response) from error
200 raise
202 def request(self, method, url, skip_auth=False, **kwargs):
203 """Perform JSON request."""
204 # Create custom headers
205 headers = {
206 "Referer": get_site_url(),
207 "Accept": "application/json; charset=utf-8",
208 }
209 if "headers" in kwargs:
210 headers.update(kwargs.pop("headers"))
211 # Optional authentication
212 if not skip_auth:
213 headers.update(self.get_headers())
215 # Fire request
216 response = request(
217 method,
218 url,
219 headers=headers,
220 timeout=self.request_timeout,
221 auth=self.get_auth(),
222 raise_for_status=False,
223 **kwargs,
224 )
226 self.check_failure(response)
228 return response
230 def download_languages(self):
231 """Download list of supported languages from a service."""
232 return []
234 def map_language_code(self, code: str) -> str:
235 """Map language code to service specific."""
236 code = code.removesuffix("_devel")
237 if code in self.language_map:
238 return self.language_map[code]
239 return code
241 def report_error(
242 self, cause: str, extra_log: str | None = None, message: bool = False
243 ) -> None:
244 """Report error situations."""
245 report_error(
246 f"machinery[{self.name}]: {cause}", extra_log=extra_log, message=message
247 )
249 @cached_property
250 def supported_languages(self):
251 """Return list of supported languages."""
252 # Try using list from cache
253 languages_cache = cache.get(self.languages_cache)
254 if languages_cache is not None:
255 # hiredis-py 3 makes list from set
256 return set(languages_cache)
258 if self.is_rate_limited():
259 return set()
261 # Download
262 try:
263 languages = set(self.download_languages())
264 except Exception as exc:
265 self.supported_languages_error = exc
266 self.supported_languages_error_age = time.time()
267 self.report_error("Could not fetch languages, using defaults")
268 return set()
270 # Update cache
271 cache.set(self.languages_cache, languages, 3600 * 48)
272 return languages
274 def is_supported(self, source, language):
275 """Check whether given language combination is supported."""
276 return (
277 language in self.supported_languages
278 and source in self.supported_languages
279 and source != language
280 )
282 def is_rate_limited(self):
283 return cache.get(self.rate_limit_cache, False)
285 def set_rate_limit(self):
286 return cache.set(self.rate_limit_cache, True, 1800)
288 def is_rate_limit_error(self, exc) -> bool:
289 if isinstance(exc, MachineryRateLimitError):
290 return True
291 if not isinstance(exc, HTTPError):
292 return False
293 # Apply rate limiting for following status codes:
294 # HTTP 456 Client Error: Quota Exceeded (DeepL)
295 # HTTP 429 Too Many Requests
296 # HTTP 401 Unauthorized
297 # HTTP 403 Forbidden
298 # HTTP 503 Service Unavailable
299 return exc.response.status_code in {456, 429, 401, 403, 503}
301 def get_cache_key(
302 self, scope: str, *, parts: Iterable[str | int] = (), text: str | None = None
303 ) -> str:
304 """
305 Cache key for caching translations.
307 Used to avoid fetching same translations again.
309 This includes project ID for project scoped entries via
310 Project.get_machinery_settings.
311 """
312 key = [
313 "mt",
314 self.mtid,
315 scope,
316 calculate_dict_hash(self.settings),
317 *parts,
318 ]
319 if text is not None:
320 key.append(calculate_hash(text))
322 return ":".join(str(part) for part in key)
324 def unescape_text(self, text: str):
325 """Unescaping of the text with replacements."""
326 return text
328 def escape_text(self, text: str):
329 """Escaping of the text with replacements."""
330 return text
332 def make_re_placeholder(self, text: str):
333 """Convert placeholder into a regular expression."""
334 # Allow additional space before ]
335 return re.escape(text[:-1]) + " *" + re.escape(text[-1:])
337 def format_replacement(
338 self, h_start: int, h_end: int, h_text: str, h_kind: Unit | None
339 ) -> str:
340 """Generate a single replacement."""
341 return f"{self.replacement_start}{h_start}{self.replacement_end}"
343 def get_highlights(
344 self, text: str, unit
345 ) -> Iterable[tuple[int, int, str, Unit | None]]:
346 for h_start, h_end, h_text in highlight_string(
347 text, unit, highlight_syntax=self.highlight_syntax
348 ):
349 yield h_start, h_end, h_text, None
351 def cleanup_text(self, text: str, unit: Unit) -> tuple[str, dict[str, str]]:
352 """Remove placeholder to avoid confusing the machine translation."""
353 replacements: dict[str, str] = {}
354 if not self.do_cleanup:
355 return text, replacements
357 parts = []
358 start = 0
359 for h_start, h_end, h_text, h_kind in self.get_highlights(text, unit):
360 parts.append(self.escape_text(text[start:h_start]))
361 h_text = self.escape_text(h_text)
362 placeholder = self.format_replacement(h_start, h_end, h_text, h_kind)
363 replacements[placeholder] = h_text
364 parts.append(placeholder)
365 start = h_end
367 parts.append(self.escape_text(text[start:]))
369 return "".join(parts), replacements
371 def uncleanup_text(self, replacements: dict[str, str], text: str) -> str:
372 for source, target in replacements.items():
373 text = re.sub(self.make_re_placeholder(source), target, text)
374 return self.unescape_text(text)
376 def uncleanup_results(
377 self, replacements: dict[str, str], results: list[TranslationResultDict]
378 ) -> None:
379 """Reverts replacements done by cleanup_text."""
380 for result in results:
381 result["text"] = self.uncleanup_text(replacements, result["text"])
382 result["source"] = self.uncleanup_text(replacements, result["source"])
384 def get_language_possibilities(self, language: Language) -> Iterator[str]:
385 code = language.code
386 mapped_code = self.map_language_code(code)
387 if not mapped_code:
388 return
389 yield mapped_code
390 code = code.replace("-", "_")
391 while "_" in code:
392 code = code.rsplit("_", 1)[0]
393 yield self.map_language_code(code)
395 def get_languages(
396 self, source_language: Language, target_language: Language
397 ) -> tuple[str, str]:
398 if source_language == target_language and not self.same_languages:
399 msg = "Same languages"
400 raise UnsupportedLanguageError(msg)
402 for source in self.get_language_possibilities(source_language):
403 for target in self.get_language_possibilities(target_language):
404 if self.is_supported(source, target):
405 return source, target
407 if self.supported_languages_error:
408 if self.supported_languages_error_age + 3600 > time.time():
409 raise MachineTranslationError(repr(self.supported_languages_error))
410 self.supported_languages_error = None
411 self.supported_languages_error_age = 0
413 msg = "Not supported"
414 raise UnsupportedLanguageError(msg)
416 def get_cached(
417 self,
418 unit,
419 source_language,
420 target_language,
421 text,
422 threshold,
423 replacements,
424 *extra_parts,
425 ) -> tuple[str | None, list[TranslationResultDict] | None]:
426 if not self.cache_translations:
427 return None, None
428 cache_key = self.get_cache_key(
429 "translation",
430 parts=(source_language, target_language, threshold, *extra_parts),
431 text=text,
432 )
433 result = cache.get(cache_key)
434 if result and (replacements or self.force_uncleanup):
435 self.uncleanup_results(replacements, result)
436 return cache_key, result
438 def search(self, unit, text, user: User | None):
439 """Search for known translations of `text`."""
440 translation = unit.translation
441 try:
442 source_language, target_language = self.get_languages(
443 translation.component.source_language, translation.language
444 )
445 except UnsupportedLanguageError:
446 unit.translation.log_debug(
447 "machinery failed: not supported language pair: %s - %s",
448 translation.component.source_language.code,
449 translation.language.code,
450 )
451 return []
453 self.account_usage(translation.component.project)
454 return self._translate(
455 source_language, target_language, [(text, unit)], user, threshold=10
456 )[text]
458 def get_default_source_language(self, translation: Translation) -> Language:
459 """Return default source language for the translation."""
460 return translation.component.source_language
462 def get_source_language(self, translation: Translation) -> Language:
463 selection = self.settings.get("source_language", SourceLanguageChoices.AUTO)
465 if selection == SourceLanguageChoices.SOURCE:
466 return translation.component.source_language
468 if selection == SourceLanguageChoices.SECONDARY:
469 # Use secondary if configured
470 if translation.component.secondary_language:
471 return translation.component.secondary_language
472 if translation.component.project.secondary_language:
473 return translation.component.project.secondary_language
475 return self.get_default_source_language(translation)
477 def translate(
478 self,
479 unit: Unit,
480 user: User | None = None,
481 threshold: int = 75,
482 *,
483 source_language: Language | None = None,
484 ):
485 """Return list of machine translations."""
486 translation = unit.translation
487 if source_language is None:
488 # Fall back to component source language
489 source_language = self.get_source_language(translation)
490 translating_from_source: bool = (
491 translation.component.source_language == source_language
492 )
494 try:
495 mapped_source_language, target_language = self.get_languages(
496 source_language, translation.language
497 )
498 except UnsupportedLanguageError:
499 unit.translation.log_debug(
500 "machinery failed: not supported language pair: %s - %s",
501 source_language.code,
502 translation.language.code,
503 )
504 return []
506 self.account_usage(translation.component.project)
508 source_plural = source_language.plural
509 target_plural = translation.plural
510 plural_mapper = PluralMapper(source_plural, target_plural)
511 alternate_units: dict[int, Unit] | None = None
512 if not translating_from_source:
513 alternate_units = plural_mapper.get_other_units([unit], source_language)
515 plural_mapper.map_units([unit], alternate_units)
516 translations = self._translate(
517 mapped_source_language,
518 target_language,
519 [(text, unit) for text in unit.plural_map],
520 user,
521 threshold=threshold,
522 )
523 return [translations[text] for text in unit.plural_map]
525 def download_multiple_translations(
526 self,
527 source_language,
528 target_language,
529 sources: list[tuple[str, Unit | None]],
530 user: User | None = None,
531 threshold: int = 75,
532 ) -> DownloadMultipleTranslations:
533 """
534 Download dictionary of a lists of possible translations from a service.
536 Should return dict with translation text, translation quality, source of
537 translation, source string.
539 You can use self.name as source of translation, if you can not give
540 better hint and text parameter as source string if you do no fuzzy
541 matching.
542 """
543 raise NotImplementedError
545 def _translate(
546 self,
547 source_language,
548 target_language,
549 sources: list[tuple[str, Unit]],
550 user=None,
551 threshold: int = 75,
552 ) -> DownloadMultipleTranslations:
553 output: DownloadMultipleTranslations = {}
554 pending = defaultdict(list)
555 cache_keys: dict[str, str | None] = {}
556 result: list[TranslationResultDict] | None
557 for text, unit in sources:
558 original_source = text
559 text, replacements = self.cleanup_text(text, unit)
561 if not text or self.is_rate_limited():
562 output[original_source] = []
563 continue
565 # Try cached results
566 cache_keys[text], result = self.get_cached(
567 unit, source_language, target_language, text, threshold, replacements
568 )
569 if result is not None:
570 output[original_source] = result
571 continue
573 pending[text].append((unit, original_source, replacements))
575 # Fetch pending strings to translate
576 if pending:
577 # Unit is only used in WeblateMemory and it is used only to get a project
578 # so it doesn't matter we potentially flatten this.
579 try:
580 translations = self.download_multiple_translations(
581 source_language,
582 target_language,
583 [
584 (text, occurrences[0][0])
585 for text, occurrences in pending.items()
586 ],
587 user,
588 threshold,
589 )
590 except Exception as exc:
591 if self.is_rate_limit_error(exc):
592 self.set_rate_limit()
594 self.report_error("Could not fetch translations")
595 if isinstance(exc, MachineTranslationError):
596 raise
597 raise MachineTranslationError(self.get_error_message(exc)) from exc
599 # Postprocess translations
600 for text, result in translations.items():
601 for _unit, original_source, replacements in pending[text]:
602 # Always operate on copy of the dictionaries
603 partial = [x.copy() for x in result]
605 for item in partial:
606 item["original_source"] = original_source
607 if cache_key := cache_keys[text]:
608 cache.set(cache_key, partial, self.cache_expiry)
609 if replacements or self.force_uncleanup:
610 self.uncleanup_results(replacements, partial)
611 output[original_source] = partial
612 return output
614 def get_error_message(self, exc: Exception) -> str:
615 if isinstance(exc, RequestException) and exc.response and exc.response.text:
616 return f"{exc.__class__.__name__}: {exc}: {exc.response.text}"
617 return f"{exc.__class__.__name__}: {exc}"
619 def signed_salt(self, appid, secret, text):
620 """Generate salt and sign as used by Chinese services."""
621 salt = str(random.randint(0, 10000000000)) # noqa: S311
623 payload = appid + text + salt + secret
624 digest = md5(payload.encode(), usedforsecurity=False).hexdigest()
626 return salt, digest
628 def batch_translate(
629 self,
630 units: list[Unit] | UnitQuerySet,
631 user: User | None = None,
632 threshold: int = 75,
633 *,
634 source_language: Language | None = None,
635 ) -> None:
636 try:
637 translation = units[0].translation
638 except IndexError:
639 return
641 if source_language is None:
642 # Fall back to component source language
643 source_language = self.get_source_language(translation)
645 translating_from_source: bool = (
646 translation.component.source_language == source_language
647 )
649 try:
650 source, language = self.get_languages(source_language, translation.language)
651 except UnsupportedLanguageError:
652 return
654 self.account_usage(translation.component.project, delta=len(units))
656 source_plural = source_language.plural
657 target_plural = translation.plural
658 plural_mapper = PluralMapper(source_plural, target_plural)
659 alternate_units: dict[int, Unit] | None = None
660 if not translating_from_source:
661 alternate_units = plural_mapper.get_other_units(units, source_language)
662 plural_mapper.map_units(units, alternate_units)
664 # TODO: fetch source from other units
665 sources = [(text, unit) for unit in units for text in unit.plural_map]
666 translations = self._translate(source, language, sources, user, threshold)
668 for unit in units:
669 result: UnitMemoryResultDict = unit.machinery
670 if min(result.get("quality", ()), default=0) >= self.max_score:
671 continue
672 translation_lists = [translations[text] for text in unit.plural_map]
673 plural_count = len(translation_lists)
674 translation = result.setdefault("translation", [""] * plural_count)
675 quality = result.setdefault("quality", [0] * plural_count)
676 origin = result.setdefault("origin", [None] * plural_count)
677 for plural, possible_translations in enumerate(translation_lists):
678 for item in possible_translations:
679 if quality[plural] > item["quality"]:
680 continue
681 quality[plural] = item["quality"]
682 translation[plural] = item["text"]
683 origin[plural] = self
685 @cached_property
686 def user(self):
687 """Weblate user used to track changes by this engine."""
688 from weblate.auth.models import User
690 return User.objects.get_or_create_bot(
691 scope="mt",
692 name=self.get_identifier(),
693 verbose=self.name,
694 )
697class MachineTranslation(BatchMachineTranslation):
698 def download_translations(
699 self,
700 source_language,
701 target_language,
702 text: str,
703 unit: Unit | None,
704 user: User | None,
705 threshold: int = 75,
706 ) -> DownloadTranslations:
707 """
708 Download list of possible translations from a service.
710 Should return dict with translation text, translation quality, source of
711 translation, source string.
713 You can use self.name as source of translation, if you can not give
714 better hint and text parameter as source string if you do no fuzzy
715 matching.
716 """
717 raise NotImplementedError
719 def download_multiple_translations(
720 self,
721 source_language,
722 target_language,
723 sources: list[tuple[str, Unit | None]],
724 user: User | None = None,
725 threshold: int = 75,
726 ) -> DownloadMultipleTranslations:
727 return {
728 text: list(
729 self.download_translations(
730 source_language,
731 target_language,
732 text,
733 unit,
734 user,
735 threshold=threshold,
736 )
737 )
738 for text, unit in sources
739 }
742class InternalMachineTranslation(MachineTranslation):
743 do_cleanup = False
744 accounting_key = "internal"
745 cache_translations = False
746 settings_form: type[BaseMachineryForm] | None = None
748 def is_supported(
749 self, source_language: Language, target_language: Language
750 ) -> bool:
751 """Any language is supported."""
752 return True
754 def is_rate_limited(self) -> bool:
755 """Disable rate limiting."""
756 return False
758 def get_language_possibilities(self, language: Language) -> Iterator[Language]: # type: ignore[override]
759 yield get_machinery_language(language)
762class GlossaryMachineTranslationMixin(MachineTranslation):
763 glossary_name_format = (
764 "weblate:{project}:{source_language}:{target_language}:{checksum}"
765 )
766 glossary_name_format_pattern = (
767 r"weblate:(\d+):([A-z0-9@_-]+):([A-z0-9@_-]+):([a-f0-9]+)"
768 )
769 glossary_support = True
771 glossary_count_limit = 0
773 def delete_cache(self) -> None:
774 """Delete general caches and glossary cache."""
775 super().delete_cache()
776 cache.delete(self.get_cache_key("glossaries"))
778 def is_glossary_supported(self, source_language: str, target_language: str) -> bool:
779 return True
781 def list_glossaries(self) -> dict[str, str]:
782 """
783 List glossaries from the service.
785 Returns dictionary with names and id.
786 """
787 raise NotImplementedError
789 def delete_glossary(self, glossary_id: str) -> None:
790 raise NotImplementedError
792 def delete_oldest_glossary(self) -> None:
793 raise NotImplementedError
795 def create_glossary(
796 self, source_language: str, target_language: str, name: str, tsv: str
797 ) -> None:
798 """
799 Create glossary in the service.
801 - Creates the glossary in the service
802 - May raise GlossaryAlreadyExists if creation fails
803 - Performs any other necessary operation, e.g uploading TSV file to bucket
804 """
805 raise NotImplementedError
807 def get_glossaries(self, use_cache: bool = True) -> dict[str, str]:
808 cache_key = self.get_cache_key("glossaries")
809 if use_cache:
810 cached = cache.get(cache_key)
811 if cached is not None:
812 return cached
814 result = self.list_glossaries()
816 cache.set(cache_key, result, 24 * 3600)
817 return result
819 def tsv_checksum(self, tsv: str) -> str:
820 """Calculate checksum of given TSV glossary."""
821 return hash_to_checksum(calculate_hash(tsv)) if tsv else ""
823 def get_cached(
824 self,
825 unit,
826 source_language,
827 target_language,
828 text,
829 threshold,
830 replacements,
831 *extra_parts,
832 ):
833 """Retrieve cached translation with glossary checksum."""
834 from weblate.glossary.models import get_glossary_tsv
836 return super().get_cached(
837 unit,
838 source_language,
839 target_language,
840 text,
841 threshold,
842 replacements,
843 self.tsv_checksum(get_glossary_tsv(unit.translation)),
844 *extra_parts,
845 )
847 def get_glossary_count_limit(self) -> int:
848 return self.glossary_count_limit
850 def get_glossary_id(
851 self, source_language: str, target_language: str, unit: Unit | None
852 ) -> str | None:
853 from weblate.glossary.models import get_glossary_tsv
855 if unit is None:
856 return None
858 translation = unit.translation
860 # Check glossary support for a language pair
861 if not self.is_glossary_supported(source_language, target_language):
862 return None
864 # Check if there is a glossary
865 glossary_tsv = get_glossary_tsv(translation)
866 if not glossary_tsv:
867 return None
869 # Calculate hash to check for changes
870 glossary_checksum = self.tsv_checksum(glossary_tsv)
871 glossary_name = self.glossary_name_format.format(
872 project=translation.component.project.id,
873 source_language=source_language,
874 target_language=target_language,
875 checksum=glossary_checksum,
876 )
878 # Fetch list of glossaries
879 glossaries = self.get_glossaries()
880 if glossary_name in glossaries:
881 return glossaries[glossary_name]
883 # Remove stale glossaries for this language pair
884 hashless_name = self.glossary_name_format.format(
885 project=translation.component.project.id,
886 source_language=source_language,
887 target_language=target_language,
888 checksum="",
889 )
890 for name, glossary_id in glossaries.items():
891 if name.startswith(hashless_name):
892 translation.log_debug(
893 "%s: removing stale glossary %s (%s)", self.mtid, name, glossary_id
894 )
895 with contextlib.suppress(GlossaryDoesNotExistError):
896 self.delete_glossary(glossary_id)
898 # Ensure we are in service limits
899 glossary_count_limit = self.get_glossary_count_limit()
900 if glossary_count_limit and len(glossaries) + 1 >= glossary_count_limit:
901 translation.log_debug(
902 "%s: approached limit of %d glossaries, removing oldest glossary",
903 self.mtid,
904 self.glossary_count_limit,
905 )
906 with contextlib.suppress(GlossaryDoesNotExistError):
907 self.delete_oldest_glossary()
909 # Create new glossary
910 translation.log_debug("%s: creating glossary %s", self.mtid, glossary_name)
911 with contextlib.suppress(GlossaryAlreadyExistsError):
912 self.create_glossary(
913 source_language, target_language, glossary_name, glossary_tsv
914 )
916 # Fetch glossaries again, without using cache
917 glossaries = self.get_glossaries(use_cache=False)
918 return glossaries[glossary_name]
920 def match_name_format(self, string: str) -> re.Match | None:
921 """
922 Match glossary name against format.
924 Only way so far to identify glossaries from memories
925 """
926 return re.match(self.glossary_name_format_pattern, string)
929class XMLMachineTranslationMixin(BatchMachineTranslation):
930 highlight_syntax = True
931 force_uncleanup = True
933 def unescape_text(self, text: str) -> str:
934 """Unescaping of the text with replacements."""
935 return unescape(text)
937 def escape_text(self, text: str) -> str:
938 """Escaping of the text with replacements."""
939 return escape(text)
941 def format_replacement(
942 self, h_start: int, h_end: int, h_text: str, h_kind: Unit | None
943 ) -> str:
944 """Generate a single replacement."""
945 raise NotImplementedError
947 def make_re_placeholder(self, text: str) -> str:
948 return re.escape(text)
951class ResponseStatusMachineTranslation(MachineTranslation):
952 def check_failure(self, response) -> None:
953 payload = response.json()
955 # Check response status
956 response_status = payload.get("responseStatus", payload.get("code", None))
957 if response_status and response_status != 200:
958 error_text = payload.get(
959 "responseDetails",
960 payload.get(
961 "message",
962 payload.get("status", f"Response status {response_status}"),
963 ),
964 )
965 if response_status == 429:
966 raise MachineryRateLimitError(error_text)
967 raise MachineTranslationError(error_text)
969 super().check_failure(response)