Coverage for app/venv/lib/python3.14/site-packages/weblate/glossary/models.py: 13%

145 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7import re 

8from collections import defaultdict 

9from copy import copy 

10from itertools import chain 

11from typing import TYPE_CHECKING, cast 

12 

13import ahocorasick_rs 

14import sentry_sdk 

15from django.core.cache import cache 

16from django.db.models import Prefetch, Q, Value 

17from django.db.models.functions import MD5, Lower 

18 

19from weblate.trans.models.unit import Unit 

20from weblate.utils.csv import PROHIBITED_INITIAL_CHARS 

21from weblate.utils.state import STATE_TRANSLATED 

22from weblate.utils.unicodechars import CONTROLCHARS 

23 

24if TYPE_CHECKING: 24 ↛ 25line 24 didn't jump to line 25 because the condition on line 24 was never true

25 from weblate.trans.models.translation import Translation 

26 

27SPLIT_RE = re.compile(r"[\s,.:!?]+") 

28NON_WORD_RE = re.compile(r"\W") 

29CONTROLCHARS_TRANS = str.maketrans(dict.fromkeys(CONTROLCHARS)) 

30 

31 

32def get_glossary_sources(component): 

33 # Fetch list of terms defined in a translation 

34 return list( 

35 component.source_translation.unit_set.filter(state__gte=STATE_TRANSLATED) 

36 .values_list(Lower("source"), flat=True) 

37 .distinct() 

38 ) 

39 

40 

41def get_glossary_automaton(project): 

42 from weblate.trans.models.component import prefetch_glossary_terms 

43 

44 with sentry_sdk.start_span(op="glossary.automaton", name=project.slug): 

45 # Chain terms 

46 prefetch_glossary_terms(project.glossaries) 

47 terms = set( 

48 chain.from_iterable( 

49 glossary.glossary_sources for glossary in project.glossaries 

50 ) 

51 ) 

52 # Remove blank string as that is not really reasonable to match 

53 terms.discard("") 

54 # Build automaton for efficient Aho-Corasick search 

55 return ahocorasick_rs.AhoCorasick( 

56 terms, 

57 implementation=ahocorasick_rs.Implementation.ContiguousNFA, 

58 store_patterns=False, 

59 ) 

60 

61 

62def get_glossary_units(project, source_language, target_language): 

63 return Unit.objects.filter( 

64 translation__component__in=project.glossaries, 

65 translation__component__source_language=source_language, 

66 translation__language=target_language, 

67 ) 

68 

69 

70def get_glossary_terms( 

71 unit: Unit, *, full: bool = False, include_variants: bool = True 

72) -> list[Unit]: 

73 """Return list of term pairs for an unit.""" 

74 if unit.glossary_terms is None: 

75 fetch_glossary_terms([unit], full=full, include_variants=include_variants) 

76 return cast("list[Unit]", unit.glossary_terms) 

77 

78 

79def fetch_glossary_terms( # noqa: C901 

80 units: list[Unit], *, full: bool = False, include_variants: bool = True 

81) -> None: 

82 """Fetch glossary terms for list of units.""" 

83 from weblate.trans.models import Component, Project 

84 

85 if len(units) == 0: 

86 return 

87 

88 translations: dict[int, Translation] = {} 

89 translation_units: dict[int, list[Unit]] = defaultdict(list) 

90 

91 for unit in units: 

92 translations[unit.translation.id] = unit.translation 

93 translation_units[unit.translation.id].append(unit) 

94 # Initialize glossary terms 

95 unit.glossary_terms = [] 

96 

97 for translation_id, translation in translations.items(): 

98 language = translation.language 

99 component = translation.component 

100 # Do not get glossary matches when display is disabled 

101 if component.hide_glossary_matches: 

102 continue 

103 project = component.project 

104 source_language = component.source_language 

105 

106 # Short circuit source language 

107 if language == source_language: 

108 continue 

109 

110 # Extract all source strings 

111 sources = [unit.source.lower() for unit in translation_units[translation_id]] 

112 

113 # Match word boundaries if needed 

114 uses_whitespace = source_language.uses_whitespace() 

115 boundaries: list[set[int]] = [set() for i in range(len(sources))] 

116 if uses_whitespace: 

117 # Get list of word boundaries 

118 for i, source in enumerate(sources): 

119 boundaries[i] = { 

120 match.span()[0] for match in NON_WORD_RE.finditer(source) 

121 } 

122 boundaries[i].add(-1) 

123 boundaries[i].add(len(source)) 

124 

125 automaton = project.glossary_automaton 

126 positions: list[dict[str, list[tuple[int, int]]]] = [ 

127 defaultdict(list) for i in range(len(sources)) 

128 ] 

129 terms: set[str] = set() 

130 # Extract terms present in the source 

131 with sentry_sdk.start_span(op="glossary.match", name=project.slug): 

132 for i, source in enumerate(sources): 

133 for _termno, start, end in automaton.find_matches_as_indexes( 

134 source, overlapping=True 

135 ): 

136 if not uses_whitespace or ( 

137 (start - 1 in boundaries[i]) and (end in boundaries[i]) 

138 ): 

139 term = source[start:end].lower() 

140 terms.add(term) 

141 positions[i][term].append((start, end)) 

142 

143 # Skip processing when there are no matches 

144 if not terms: 

145 continue 

146 

147 base_units = get_glossary_units(project, source_language, language) 

148 # Variant is used for variant grouping below, source unit for flags 

149 base_units = base_units.select_related("source_unit", "variant") 

150 

151 # Exclude currently edited unit items to prevent self-referencing glossary items 

152 current_unit_ids = [u.pk for u in translation_units[translation_id] if u.pk] 

153 if current_unit_ids: 

154 base_units = base_units.exclude(pk__in=current_unit_ids) 

155 

156 if full: 

157 # Include full details needed for rendering 

158 base_units = base_units.prefetch() 

159 else: 

160 # Component priority is needed for ordering, file format and flags for flags 

161 base_units = base_units.prefetch_related( 

162 Prefetch( 

163 "translation__component", 

164 queryset=Component.objects.only( 

165 "priority", 

166 "file_format", 

167 "check_flags", 

168 "project", 

169 ), 

170 ), 

171 Prefetch( 

172 "translation__component__project", 

173 queryset=Project.objects.only( 

174 "check_flags", 

175 ), 

176 ), 

177 ) 

178 

179 glossary_units = list( 

180 base_units.filter( 

181 Q(source__lower__md5__in=[MD5(Value(term)) for term in terms]), 

182 ) 

183 ) 

184 

185 # Add variants manually. This could be done by adding filtering on 

186 # variant__unit__source in the above query, but this slows down the query 

187 # considerably and variants are rarely used. 

188 glossary_variants: dict[int, dict[int, Unit]] = defaultdict(dict) 

189 if include_variants: 

190 processed_variants = set() 

191 

192 for match in glossary_units: 

193 if not match.variant_id or match.variant_id in processed_variants: 

194 continue 

195 processed_variants.add(match.variant_id) 

196 for child in base_units.filter(variant_id=match.variant_id).exclude( 

197 pk=match.pk 

198 ): 

199 glossary_variants[match.pk][child.pk] = child 

200 

201 # Prepare term lookup 

202 glossary_lookup: dict[str, list[Unit]] = defaultdict(list) 

203 for match in glossary_units: 

204 glossary_lookup[match.source.lower()].append(match) 

205 

206 # Inject matches back to the units 

207 for i, unit in enumerate(translation_units[translation_id]): 

208 result: dict[int, Unit] = {} 

209 for term, glossary_positions in positions[i].items(): 

210 try: 

211 matches = glossary_lookup[term] 

212 except KeyError: 

213 continue 

214 

215 for match in matches: 

216 item = copy(match) 

217 item.glossary_positions = tuple(glossary_positions) 

218 result[item.pk] = item 

219 for variant in glossary_variants[match.pk].values(): 

220 item = copy(variant) 

221 item.glossary_positions = tuple(glossary_positions) 

222 result[item.pk] = item 

223 

224 # Store sorted results in a unit cache 

225 unit.glossary_terms = sorted( 

226 result.values(), key=lambda x: x.glossary_sort_key 

227 ) 

228 

229 

230def render_glossary_units_tsv(units) -> str: 

231 r""" 

232 Build a tab separated glossary. 

233 

234 Based on the DeepL specification: 

235 

236 - duplicate source entries are not allowed 

237 - neither source nor target entry may be empty 

238 - source and target entries must not contain any C0 or C1 control characters (including, e.g., "\t" or "\n") or any Unicode newline 

239 - source and target entries must not contain any leading or trailing Unicode whitespace character 

240 - source/target entry pairs are separated by a newline 

241 - source entries and target entries are separated by a tab 

242 """ 

243 from weblate.trans.models.component import Component 

244 

245 def cleanup(text): 

246 """ 

247 Clean up the provided text by removing unwanted characters. 

248 

249 - Translates and removes control characters using CONTROLCHARS_TRANS. 

250 - Strips leading and trailing whitespace. 

251 - Removes leading characters from PROHIBITED_INITIAL_CHARS if present. 

252 """ 

253 text = text.translate(CONTROLCHARS_TRANS) 

254 prohibited_initial_chars_pattern = ( 

255 "^(" + "|".join(re.escape(char) for char in PROHIBITED_INITIAL_CHARS) + ")*" 

256 ) 

257 

258 return re.sub(prohibited_initial_chars_pattern, "", text).strip() 

259 

260 # We can get list or iterator as well 

261 if hasattr(units, "prefetch_related"): 

262 units = units.prefetch_related( 

263 "source_unit", 

264 "translation", 

265 Prefetch("translation__component", queryset=Component.objects.defer_huge()), 

266 ) 

267 

268 included = set() 

269 output = [] 

270 for unit in units: 

271 # Skip forbidden term 

272 if "forbidden" in unit.all_flags: 

273 continue 

274 

275 if not unit.translated and "read-only" not in unit.all_flags: 

276 continue 

277 

278 # Cleanup strings 

279 source = cleanup(unit.source) 

280 target = source if "read-only" in unit.all_flags else cleanup(unit.target) 

281 

282 # Skip blanks and duplicates 

283 if not source or not target or source in included: 

284 continue 

285 

286 # Memoize included 

287 included.add(source) 

288 

289 # Render TSV 

290 output.append(f"{source}\t{target}") 

291 

292 return "\n".join(output) 

293 

294 

295def get_glossary_tsv(translation) -> str: 

296 project = translation.component.project 

297 source_language = translation.component.source_language 

298 language = translation.language 

299 

300 cache_key = project.get_glossary_tsv_cache_key(source_language, language) 

301 

302 cached = cache.get(cache_key) 

303 if cached is not None: 

304 return cached 

305 

306 # Get glossary units 

307 units = get_glossary_units(project, source_language, language) 

308 

309 # Render as tsv 

310 result = render_glossary_units_tsv(units.filter(state__gte=STATE_TRANSLATED)) 

311 

312 cache.set(cache_key, result, 24 * 3600) 

313 

314 return result