Coverage for app/venv/lib/python3.14/site-packages/weblate/glossary/models.py: 13%
145 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5from __future__ import annotations
7import re
8from collections import defaultdict
9from copy import copy
10from itertools import chain
11from typing import TYPE_CHECKING, cast
13import ahocorasick_rs
14import sentry_sdk
15from django.core.cache import cache
16from django.db.models import Prefetch, Q, Value
17from django.db.models.functions import MD5, Lower
19from weblate.trans.models.unit import Unit
20from weblate.utils.csv import PROHIBITED_INITIAL_CHARS
21from weblate.utils.state import STATE_TRANSLATED
22from weblate.utils.unicodechars import CONTROLCHARS
24if TYPE_CHECKING: 24 ↛ 25line 24 didn't jump to line 25 because the condition on line 24 was never true
25 from weblate.trans.models.translation import Translation
27SPLIT_RE = re.compile(r"[\s,.:!?]+")
28NON_WORD_RE = re.compile(r"\W")
29CONTROLCHARS_TRANS = str.maketrans(dict.fromkeys(CONTROLCHARS))
32def get_glossary_sources(component):
33 # Fetch list of terms defined in a translation
34 return list(
35 component.source_translation.unit_set.filter(state__gte=STATE_TRANSLATED)
36 .values_list(Lower("source"), flat=True)
37 .distinct()
38 )
41def get_glossary_automaton(project):
42 from weblate.trans.models.component import prefetch_glossary_terms
44 with sentry_sdk.start_span(op="glossary.automaton", name=project.slug):
45 # Chain terms
46 prefetch_glossary_terms(project.glossaries)
47 terms = set(
48 chain.from_iterable(
49 glossary.glossary_sources for glossary in project.glossaries
50 )
51 )
52 # Remove blank string as that is not really reasonable to match
53 terms.discard("")
54 # Build automaton for efficient Aho-Corasick search
55 return ahocorasick_rs.AhoCorasick(
56 terms,
57 implementation=ahocorasick_rs.Implementation.ContiguousNFA,
58 store_patterns=False,
59 )
62def get_glossary_units(project, source_language, target_language):
63 return Unit.objects.filter(
64 translation__component__in=project.glossaries,
65 translation__component__source_language=source_language,
66 translation__language=target_language,
67 )
70def get_glossary_terms(
71 unit: Unit, *, full: bool = False, include_variants: bool = True
72) -> list[Unit]:
73 """Return list of term pairs for an unit."""
74 if unit.glossary_terms is None:
75 fetch_glossary_terms([unit], full=full, include_variants=include_variants)
76 return cast("list[Unit]", unit.glossary_terms)
79def fetch_glossary_terms( # noqa: C901
80 units: list[Unit], *, full: bool = False, include_variants: bool = True
81) -> None:
82 """Fetch glossary terms for list of units."""
83 from weblate.trans.models import Component, Project
85 if len(units) == 0:
86 return
88 translations: dict[int, Translation] = {}
89 translation_units: dict[int, list[Unit]] = defaultdict(list)
91 for unit in units:
92 translations[unit.translation.id] = unit.translation
93 translation_units[unit.translation.id].append(unit)
94 # Initialize glossary terms
95 unit.glossary_terms = []
97 for translation_id, translation in translations.items():
98 language = translation.language
99 component = translation.component
100 # Do not get glossary matches when display is disabled
101 if component.hide_glossary_matches:
102 continue
103 project = component.project
104 source_language = component.source_language
106 # Short circuit source language
107 if language == source_language:
108 continue
110 # Extract all source strings
111 sources = [unit.source.lower() for unit in translation_units[translation_id]]
113 # Match word boundaries if needed
114 uses_whitespace = source_language.uses_whitespace()
115 boundaries: list[set[int]] = [set() for i in range(len(sources))]
116 if uses_whitespace:
117 # Get list of word boundaries
118 for i, source in enumerate(sources):
119 boundaries[i] = {
120 match.span()[0] for match in NON_WORD_RE.finditer(source)
121 }
122 boundaries[i].add(-1)
123 boundaries[i].add(len(source))
125 automaton = project.glossary_automaton
126 positions: list[dict[str, list[tuple[int, int]]]] = [
127 defaultdict(list) for i in range(len(sources))
128 ]
129 terms: set[str] = set()
130 # Extract terms present in the source
131 with sentry_sdk.start_span(op="glossary.match", name=project.slug):
132 for i, source in enumerate(sources):
133 for _termno, start, end in automaton.find_matches_as_indexes(
134 source, overlapping=True
135 ):
136 if not uses_whitespace or (
137 (start - 1 in boundaries[i]) and (end in boundaries[i])
138 ):
139 term = source[start:end].lower()
140 terms.add(term)
141 positions[i][term].append((start, end))
143 # Skip processing when there are no matches
144 if not terms:
145 continue
147 base_units = get_glossary_units(project, source_language, language)
148 # Variant is used for variant grouping below, source unit for flags
149 base_units = base_units.select_related("source_unit", "variant")
151 # Exclude currently edited unit items to prevent self-referencing glossary items
152 current_unit_ids = [u.pk for u in translation_units[translation_id] if u.pk]
153 if current_unit_ids:
154 base_units = base_units.exclude(pk__in=current_unit_ids)
156 if full:
157 # Include full details needed for rendering
158 base_units = base_units.prefetch()
159 else:
160 # Component priority is needed for ordering, file format and flags for flags
161 base_units = base_units.prefetch_related(
162 Prefetch(
163 "translation__component",
164 queryset=Component.objects.only(
165 "priority",
166 "file_format",
167 "check_flags",
168 "project",
169 ),
170 ),
171 Prefetch(
172 "translation__component__project",
173 queryset=Project.objects.only(
174 "check_flags",
175 ),
176 ),
177 )
179 glossary_units = list(
180 base_units.filter(
181 Q(source__lower__md5__in=[MD5(Value(term)) for term in terms]),
182 )
183 )
185 # Add variants manually. This could be done by adding filtering on
186 # variant__unit__source in the above query, but this slows down the query
187 # considerably and variants are rarely used.
188 glossary_variants: dict[int, dict[int, Unit]] = defaultdict(dict)
189 if include_variants:
190 processed_variants = set()
192 for match in glossary_units:
193 if not match.variant_id or match.variant_id in processed_variants:
194 continue
195 processed_variants.add(match.variant_id)
196 for child in base_units.filter(variant_id=match.variant_id).exclude(
197 pk=match.pk
198 ):
199 glossary_variants[match.pk][child.pk] = child
201 # Prepare term lookup
202 glossary_lookup: dict[str, list[Unit]] = defaultdict(list)
203 for match in glossary_units:
204 glossary_lookup[match.source.lower()].append(match)
206 # Inject matches back to the units
207 for i, unit in enumerate(translation_units[translation_id]):
208 result: dict[int, Unit] = {}
209 for term, glossary_positions in positions[i].items():
210 try:
211 matches = glossary_lookup[term]
212 except KeyError:
213 continue
215 for match in matches:
216 item = copy(match)
217 item.glossary_positions = tuple(glossary_positions)
218 result[item.pk] = item
219 for variant in glossary_variants[match.pk].values():
220 item = copy(variant)
221 item.glossary_positions = tuple(glossary_positions)
222 result[item.pk] = item
224 # Store sorted results in a unit cache
225 unit.glossary_terms = sorted(
226 result.values(), key=lambda x: x.glossary_sort_key
227 )
230def render_glossary_units_tsv(units) -> str:
231 r"""
232 Build a tab separated glossary.
234 Based on the DeepL specification:
236 - duplicate source entries are not allowed
237 - neither source nor target entry may be empty
238 - source and target entries must not contain any C0 or C1 control characters (including, e.g., "\t" or "\n") or any Unicode newline
239 - source and target entries must not contain any leading or trailing Unicode whitespace character
240 - source/target entry pairs are separated by a newline
241 - source entries and target entries are separated by a tab
242 """
243 from weblate.trans.models.component import Component
245 def cleanup(text):
246 """
247 Clean up the provided text by removing unwanted characters.
249 - Translates and removes control characters using CONTROLCHARS_TRANS.
250 - Strips leading and trailing whitespace.
251 - Removes leading characters from PROHIBITED_INITIAL_CHARS if present.
252 """
253 text = text.translate(CONTROLCHARS_TRANS)
254 prohibited_initial_chars_pattern = (
255 "^(" + "|".join(re.escape(char) for char in PROHIBITED_INITIAL_CHARS) + ")*"
256 )
258 return re.sub(prohibited_initial_chars_pattern, "", text).strip()
260 # We can get list or iterator as well
261 if hasattr(units, "prefetch_related"):
262 units = units.prefetch_related(
263 "source_unit",
264 "translation",
265 Prefetch("translation__component", queryset=Component.objects.defer_huge()),
266 )
268 included = set()
269 output = []
270 for unit in units:
271 # Skip forbidden term
272 if "forbidden" in unit.all_flags:
273 continue
275 if not unit.translated and "read-only" not in unit.all_flags:
276 continue
278 # Cleanup strings
279 source = cleanup(unit.source)
280 target = source if "read-only" in unit.all_flags else cleanup(unit.target)
282 # Skip blanks and duplicates
283 if not source or not target or source in included:
284 continue
286 # Memoize included
287 included.add(source)
289 # Render TSV
290 output.append(f"{source}\t{target}")
292 return "\n".join(output)
295def get_glossary_tsv(translation) -> str:
296 project = translation.component.project
297 source_language = translation.component.source_language
298 language = translation.language
300 cache_key = project.get_glossary_tsv_cache_key(source_language, language)
302 cached = cache.get(cache_key)
303 if cached is not None:
304 return cached
306 # Get glossary units
307 units = get_glossary_units(project, source_language, language)
309 # Render as tsv
310 result = render_glossary_units_tsv(units.filter(state__gte=STATE_TRANSLATED))
312 cache.set(cache_key, result, 24 * 3600)
314 return result