Coverage for app/venv/lib/python3.14/site-packages/weblate/trans/util.py: 47%

206 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7import locale 

8import os 

9import platform 

10import re 

11import sys 

12from operator import itemgetter 

13from types import GeneratorType 

14from typing import TYPE_CHECKING, Any, TypeVar, cast 

15from urllib.parse import urlparse 

16 

17import django.shortcuts 

18from django.core.cache import cache 

19from django.http import HttpResponseRedirect 

20from django.shortcuts import redirect, resolve_url 

21from django.utils.http import url_has_allowed_host_and_scheme 

22from django.utils.translation import gettext, gettext_lazy 

23from lxml import etree 

24from packaging.version import Version 

25from translate.misc.multistring import multistring 

26from translate.storage.placeables.lisa import parse_xliff, strelem_to_xml 

27 

28from weblate.auth.results import Denied 

29from weblate.utils.data import data_dir 

30 

31if TYPE_CHECKING: 31 ↛ 32line 31 didn't jump to line 32 because the condition on line 31 was never true

32 from collections.abc import Callable, Generator, Iterable 

33 

34 from django.db.models import Model 

35 from django.shortcuts import SupportsGetAbsoluteUrl 

36 

37 from weblate.auth.models import User 

38 from weblate.auth.results import PermissionResult 

39 from weblate.lang.models import Language 

40 from weblate.trans.models import Project, Translation, Unit 

41 

42 

43def detect_strxfrm() -> bool: 

44 # macOS problematic behavior 

45 if platform.system() == "Darwin": 45 ↛ 46line 45 didn't jump to line 46 because the condition on line 45 was never true

46 version = Version(platform.mac_ver()[0]) 

47 if version > Version("15.0") and version < Version("15.6"): 

48 # Avoid triggering strxfrm on macOS 15 until 15.6 where it either 

49 # crashes with OSError or causes Python segmentation fault. 

50 return False 

51 

52 if locale.strxfrm("a") == "a": 52 ↛ 70line 52 didn't jump to line 70 because the condition on line 52 was always true

53 # Initialize to sane Unicode locales for strxfrm 

54 try: 

55 locale.setlocale(locale.LC_ALL, ("en_US", "UTF-8")) 

56 except locale.Error: 

57 return False 

58 

59 # Try whether strxfrm works 

60 try: 

61 locale.strxfrm("zkouška") 

62 except OSError: 

63 # Crashes on macOS 15 and some FreeBSD derivatives, see 

64 # https://github.com/python/cpython/issues/130567 

65 return False 

66 

67 return True 

68 

69 # Assume it is not working 

70 return False 

71 

72 

73PLURAL_SEPARATOR = "\x1e\x1e" 

74USE_STRXFRM = detect_strxfrm() 

75 

76PRIORITY_CHOICES = ( 

77 (60, gettext_lazy("Very high")), 

78 (80, gettext_lazy("High")), 

79 (100, gettext_lazy("Medium")), 

80 (120, gettext_lazy("Low")), 

81 (140, gettext_lazy("Very low")), 

82) 

83 

84# Generated by scripts/generate-cjk-regexp.py 

85CJK_PATTERN = re.compile( 

86 r"([\u1100-\u11ff\u2e80-\u2fdf\u2ff0-\u9fff\ua960-\ua97f\uac00-\ud7ff\uf900-\ufaff\ufe30-\ufe4f\uff00-\uffef\U0001aff0-\U0001b16f\U0001f200-\U0001f2ff\U00020000-\U0003FFFF]+)" 

87) 

88 

89 

90def is_plural(text: str) -> bool: 

91 """Check whether string is plural form.""" 

92 return text.find(PLURAL_SEPARATOR) != -1 

93 

94 

95def split_plural(text: str) -> list[str]: 

96 return text.split(PLURAL_SEPARATOR) 

97 

98 

99def join_plural(plurals: Iterable[str]) -> str: 

100 return PLURAL_SEPARATOR.join(plurals) 

101 

102 

103def get_string( 

104 text: str | multistring | list | Generator[str, None, None] | None, 

105) -> str: 

106 """Return correctly formatted string from ttkit unit data.""" 

107 # Check for null target (happens with XLIFF) 

108 if text is None: 108 ↛ 109line 108 didn't jump to line 109 because the condition on line 108 was never true

109 return "" 

110 if isinstance(text, multistring): 110 ↛ 111line 110 didn't jump to line 111 because the condition on line 110 was never true

111 return join_plural(get_string(str(item)) for item in text.strings) 

112 if isinstance(text, (list, GeneratorType)): 112 ↛ 113line 112 didn't jump to line 113 because the condition on line 112 was never true

113 return join_plural(get_string(str(item)) for item in text) 

114 if isinstance(text, str): 114 ↛ 121line 114 didn't jump to line 121 because the condition on line 114 was always true

115 # Remove possible surrogates in the string. There doesn't seem to be 

116 # a cheap way to detect this, so do the conversion in both cases. In 

117 # case of failure, this at least fails when parsing the file instead 

118 # being that later when inserting the data to the database. 

119 return text.encode("utf-16", "surrogatepass").decode("utf-16") 

120 # We might get integer or float in some formats 

121 return str(text) 

122 

123 

124def is_repo_link(val: str) -> bool: 

125 """Check whether repository is just a link for other one.""" 

126 return val.startswith("weblate://") 

127 

128 

129def get_distinct_translations(units: Iterable[Unit]) -> list[Unit]: 

130 """ 

131 Return list of distinct translations. 

132 

133 It should be possible to use distinct('target') since Django 1.4, but it is not 

134 supported with MySQL, so let's emulate that based on presumption we won't get too 

135 many results. 

136 """ 

137 targets = {} 

138 result = [] 

139 for unit in units: 

140 if unit.target in targets: 

141 continue 

142 targets[unit.target] = 1 

143 result.append(unit) 

144 return result 

145 

146 

147def translation_percent( 

148 translated: int, total: int, zero_complete: bool = True 

149) -> float: 

150 """Return translation percentage.""" 

151 if total == 0: 

152 return 100.0 if zero_complete else 0.0 

153 if total is None: 153 ↛ 154line 153 didn't jump to line 154 because the condition on line 153 was never true

154 return 0.0 

155 perc = (1000 * translated // total) / 10.0 

156 # Avoid displaying misleading rounded 0.0% or 100.0% 

157 if perc == 0.0 and translated != 0: 157 ↛ 158line 157 didn't jump to line 158 because the condition on line 157 was never true

158 return 0.1 

159 if perc == 100.0 and translated < total: 159 ↛ 160line 159 didn't jump to line 160 because the condition on line 159 was never true

160 return 99.9 

161 return perc 

162 

163 

164def get_clean_env( 

165 extra: dict[str, str] | None = None, extra_path: str | None = None 

166) -> dict[str, str]: 

167 """Return cleaned up environment for subprocess execution.""" 

168 environ = { 

169 "LANG": "C.UTF-8", 

170 "LC_ALL": "C.UTF-8", 

171 "HOME": data_dir("home"), 

172 "PATH": "/bin:/usr/bin:/usr/local/bin", 

173 } 

174 if extra is not None: 

175 environ.update(extra) 

176 variables = ( 

177 # Keep PATH setup 

178 "PATH", 

179 # Keep Python search path 

180 "PYTHONPATH", 

181 # Keep linker configuration 

182 "LD_LIBRARY_PATH", 

183 "LD_PRELOAD", 

184 # Fontconfig configuration by weblate.fonts 

185 "FONTCONFIG_FILE", 

186 # Needed by Git on Windows 

187 "SystemRoot", 

188 # Pass proxy configuration 

189 "http_proxy", 

190 "https_proxy", 

191 "HTTPS_PROXY", 

192 "NO_PROXY", 

193 # below two are needed for openshift3 deployment, 

194 # where nss_wrapper is used 

195 # more on the topic on below link: 

196 # https://docs.openshift.com/enterprise/3.2/creating_images/guidelines.html 

197 "NSS_WRAPPER_GROUP", 

198 "NSS_WRAPPER_PASSWD", 

199 ) 

200 for var in variables: 

201 if var in os.environ: 

202 environ[var] = os.environ[var] 

203 # Extend path to include virtualenv, avoid insert already existing ones to 

204 # not break existing ordering (for example PATH injection used in tests) 

205 venv_path = os.path.join(sys.exec_prefix, "bin") 

206 if venv_path not in environ["PATH"]: 206 ↛ 207line 206 didn't jump to line 207 because the condition on line 206 was never true

207 environ["PATH"] = "{}:{}".format(venv_path, environ["PATH"]) 

208 if extra_path and extra_path not in environ["PATH"]: 

209 environ["PATH"] = "{}:{}".format(extra_path, environ["PATH"]) 

210 return environ 

211 

212 

213def cleanup_repo_url(url: str, text: str | None = None) -> str: 

214 """Remove credentials from repository URL.""" 

215 if text is None: 

216 text = url 

217 try: 

218 parsed = urlparse(url) 

219 except ValueError: 

220 # The URL can not be parsed, so avoid stripping 

221 return text 

222 if parsed.username and parsed.password: 222 ↛ 223line 222 didn't jump to line 223 because the condition on line 222 was never true

223 return text.replace(f"{parsed.username}:{parsed.password}@", "") 

224 if parsed.username: 224 ↛ 225line 224 didn't jump to line 225 because the condition on line 224 was never true

225 return text.replace(f"{parsed.username}@", "") 

226 return text 

227 

228 

229def redirect_param(location, params, *args, **kwargs): 

230 """Redirect to a URL with parameters.""" 

231 return HttpResponseRedirect(resolve_url(location, *args, **kwargs) + params) 

232 

233 

234def cleanup_path(path: str) -> str: 

235 """Remove leading ./ or / from path.""" 

236 if not path: 

237 return path 

238 

239 # interpret absolute pathname as relative, remove drive letter or 

240 # UNC path, redundant separators, "." and ".." components. 

241 path = os.path.splitdrive(path)[1] 

242 invalid_path_parts = ("", os.path.curdir, os.path.pardir) 

243 path = os.path.sep.join( 

244 x for x in path.split(os.path.sep) if x not in invalid_path_parts 

245 ) 

246 

247 return os.path.normpath(path) 

248 

249 

250def get_project_description(project: Project) -> str: 

251 """Return verbose description for project translation.""" 

252 # Cache the count as it might be expensive to calculate (it pull 

253 # all project stats) and there is no need to always have up to date 

254 # count here 

255 cache_key = f"project-lang-count-{project.id}" 

256 count = cache.get(cache_key) 

257 if count is None: 

258 count = project.stats.languages 

259 cache.set(cache_key, count, 6 * 3600) 

260 return gettext( 

261 "{0} is being translated into {1} languages using Weblate. " 

262 "Join the translation or start translating your own project." 

263 ).format(project, count) 

264 

265 

266def render( 

267 request, 

268 template_name: str, 

269 context: dict[str, Any] | None = None, 

270 content_type: str | None = None, 

271 status: int | None = None, 

272 using=None, 

273): 

274 """Render template with Weblate extended context.""" 

275 if context is None: 275 ↛ 276line 275 didn't jump to line 276 because the condition on line 275 was never true

276 context = {} 

277 if "project" in context and context["project"] is not None: 277 ↛ 278line 277 didn't jump to line 278 because the condition on line 277 was never true

278 context["description"] = get_project_description(context["project"]) 

279 

280 return django.shortcuts.render( 

281 request, 

282 template_name=template_name, 

283 context=context, 

284 content_type=content_type, 

285 status=status, 

286 using=using, 

287 ) 

288 

289 

290def path_separator(path: str) -> str: 

291 """ 

292 Consolidate path separator. 

293 

294 Always use / as path separator for consistency. 

295 """ 

296 if os.path.sep != "/": 296 ↛ 297line 296 didn't jump to line 297 because the condition on line 296 was never true

297 return path.replace(os.path.sep, "/") 

298 return path 

299 

300 

301T = TypeVar("T") 

302 

303 

304def sort_unicode(choices: list[T], key: Callable[[T], str]) -> list[T]: 

305 """Unicode aware sorting if available.""" 

306 

307 def sort_strxfrm(item: T) -> str: 

308 return locale.strxfrm(key(item)) 

309 

310 return sorted(choices, key=sort_strxfrm if USE_STRXFRM else key) 

311 

312 

313def sort_choices(choices: list[tuple[str, str]]) -> list[tuple[str, str]]: 

314 """Sort choices alphabetically.""" 

315 return sort_unicode(choices, itemgetter(1)) 

316 

317 

318def sort_objects(objects: list[Model]) -> list[Model]: 

319 """Sort objects alphabetically.""" 

320 return sort_unicode(objects, str) 

321 

322 

323def redirect_next( 

324 next_url: str | None, fallback: str | SupportsGetAbsoluteUrl 

325) -> HttpResponseRedirect: 

326 """Redirect to next URL from request after validating it.""" 

327 if ( 

328 next_url is None 

329 or not url_has_allowed_host_and_scheme(next_url, allowed_hosts=None) 

330 or not next_url.startswith("/") 

331 ): 

332 return redirect(fallback) 

333 return HttpResponseRedirect(next_url) 

334 

335 

336def xliff_string_to_rich(string: str): 

337 """ 

338 Convert XLIFF string to StringElement. 

339 

340 Transform a string containing XLIFF placeholders as XML into a rich content 

341 (StringElement) 

342 """ 

343 if isinstance(string, list): 

344 return [parse_xliff(s) for s in string] 

345 return [parse_xliff(string)] 

346 

347 

348def rich_to_xliff_string(string_elements): 

349 """ 

350 Convert StringElement to XLIFF string. 

351 

352 Transform rich content (StringElement) into a string with placeholder kept as XML 

353 """ 

354 # Create dummy root element 

355 xml = etree.Element("e") 

356 for string_element in string_elements: 

357 # Inject placeable from translate-toolkit 

358 strelem_to_xml(xml, string_element) 

359 

360 # Remove any possible namespace 

361 for child in xml: 

362 tag = cast("str", child.tag) 

363 if tag.startswith("{"): 

364 child.tag = tag[tag.index("}") + 1 :] 

365 etree.cleanup_namespaces(xml) 

366 

367 # Convert to string 

368 string_xml = etree.tostring(xml, encoding="unicode") 

369 

370 # Strip dummy root element 

371 return get_string(string_xml[3:][:-4]) 

372 

373 

374def check_upload_method_permissions( 

375 user: User, translation: Translation, method: str 

376) -> PermissionResult | bool: 

377 """Check whether user has permission to perform upload method.""" 

378 from weblate.formats.base import BilingualUpdateMixin 

379 

380 if method == "source": 

381 if not translation.is_source: 

382 return Denied( 

383 gettext("Source upload is only supported on the source language.") 

384 ) 

385 if translation.is_template: 

386 return Denied( 

387 gettext( 

388 "Source upload is only supported for bilingual translations, you might want to use replace upload instead." 

389 ) 

390 ) 

391 if not issubclass(translation.component.file_format_cls, BilingualUpdateMixin): 

392 return Denied( 

393 gettext( 

394 "Update source strings upload is not supported with this format." 

395 ) 

396 ) 

397 return user.has_perm("upload.perform", translation) 

398 if method == "add": 

399 return user.has_perm("unit.add", translation) 

400 if method in {"translate", "fuzzy"}: 

401 return user.has_perm("unit.edit", translation) 

402 if method == "suggest": 

403 return user.has_perm("suggestion.add", translation) 

404 if method == "approve": 

405 return user.has_perm("unit.review", translation) 

406 if method == "replace": 

407 return bool(translation.filename) and ( 

408 user.has_perm("component.edit", translation) 

409 or user.has_perms(["unit.add", "unit.delete", "unit.edit"], translation) 

410 ) 

411 msg = f"Invalid method: {method}" 

412 raise ValueError(msg) 

413 

414 

415def is_unused_string(string: str) -> bool: 

416 """Check whether string should not be used.""" 

417 return string.startswith("<unused singular") 

418 

419 

420def count_words(string: str, language: Language | None = None) -> int: 

421 """Count number of words in a string.""" 

422 if language is not None and language.is_cjk(): 422 ↛ 423line 422 didn't jump to line 423 because the condition on line 422 was never true

423 count = 0 

424 for s in split_plural(string): 

425 if is_unused_string(s): 

426 continue 

427 even = True 

428 for sec in CJK_PATTERN.split(string): 

429 if even: 

430 count += len(sec.split()) 

431 else: 

432 count += len(sec) 

433 even = not even 

434 return count 

435 return sum(len(s.split()) for s in split_plural(string) if not is_unused_string(s))