Coverage for app/venv/lib/python3.14/site-packages/weblate/trans/util.py: 47%
206 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5from __future__ import annotations
7import locale
8import os
9import platform
10import re
11import sys
12from operator import itemgetter
13from types import GeneratorType
14from typing import TYPE_CHECKING, Any, TypeVar, cast
15from urllib.parse import urlparse
17import django.shortcuts
18from django.core.cache import cache
19from django.http import HttpResponseRedirect
20from django.shortcuts import redirect, resolve_url
21from django.utils.http import url_has_allowed_host_and_scheme
22from django.utils.translation import gettext, gettext_lazy
23from lxml import etree
24from packaging.version import Version
25from translate.misc.multistring import multistring
26from translate.storage.placeables.lisa import parse_xliff, strelem_to_xml
28from weblate.auth.results import Denied
29from weblate.utils.data import data_dir
31if TYPE_CHECKING: 31 ↛ 32line 31 didn't jump to line 32 because the condition on line 31 was never true
32 from collections.abc import Callable, Generator, Iterable
34 from django.db.models import Model
35 from django.shortcuts import SupportsGetAbsoluteUrl
37 from weblate.auth.models import User
38 from weblate.auth.results import PermissionResult
39 from weblate.lang.models import Language
40 from weblate.trans.models import Project, Translation, Unit
43def detect_strxfrm() -> bool:
44 # macOS problematic behavior
45 if platform.system() == "Darwin": 45 ↛ 46line 45 didn't jump to line 46 because the condition on line 45 was never true
46 version = Version(platform.mac_ver()[0])
47 if version > Version("15.0") and version < Version("15.6"):
48 # Avoid triggering strxfrm on macOS 15 until 15.6 where it either
49 # crashes with OSError or causes Python segmentation fault.
50 return False
52 if locale.strxfrm("a") == "a": 52 ↛ 70line 52 didn't jump to line 70 because the condition on line 52 was always true
53 # Initialize to sane Unicode locales for strxfrm
54 try:
55 locale.setlocale(locale.LC_ALL, ("en_US", "UTF-8"))
56 except locale.Error:
57 return False
59 # Try whether strxfrm works
60 try:
61 locale.strxfrm("zkouška")
62 except OSError:
63 # Crashes on macOS 15 and some FreeBSD derivatives, see
64 # https://github.com/python/cpython/issues/130567
65 return False
67 return True
69 # Assume it is not working
70 return False
73PLURAL_SEPARATOR = "\x1e\x1e"
74USE_STRXFRM = detect_strxfrm()
76PRIORITY_CHOICES = (
77 (60, gettext_lazy("Very high")),
78 (80, gettext_lazy("High")),
79 (100, gettext_lazy("Medium")),
80 (120, gettext_lazy("Low")),
81 (140, gettext_lazy("Very low")),
82)
84# Generated by scripts/generate-cjk-regexp.py
85CJK_PATTERN = re.compile(
86 r"([\u1100-\u11ff\u2e80-\u2fdf\u2ff0-\u9fff\ua960-\ua97f\uac00-\ud7ff\uf900-\ufaff\ufe30-\ufe4f\uff00-\uffef\U0001aff0-\U0001b16f\U0001f200-\U0001f2ff\U00020000-\U0003FFFF]+)"
87)
90def is_plural(text: str) -> bool:
91 """Check whether string is plural form."""
92 return text.find(PLURAL_SEPARATOR) != -1
95def split_plural(text: str) -> list[str]:
96 return text.split(PLURAL_SEPARATOR)
99def join_plural(plurals: Iterable[str]) -> str:
100 return PLURAL_SEPARATOR.join(plurals)
103def get_string(
104 text: str | multistring | list | Generator[str, None, None] | None,
105) -> str:
106 """Return correctly formatted string from ttkit unit data."""
107 # Check for null target (happens with XLIFF)
108 if text is None: 108 ↛ 109line 108 didn't jump to line 109 because the condition on line 108 was never true
109 return ""
110 if isinstance(text, multistring): 110 ↛ 111line 110 didn't jump to line 111 because the condition on line 110 was never true
111 return join_plural(get_string(str(item)) for item in text.strings)
112 if isinstance(text, (list, GeneratorType)): 112 ↛ 113line 112 didn't jump to line 113 because the condition on line 112 was never true
113 return join_plural(get_string(str(item)) for item in text)
114 if isinstance(text, str): 114 ↛ 121line 114 didn't jump to line 121 because the condition on line 114 was always true
115 # Remove possible surrogates in the string. There doesn't seem to be
116 # a cheap way to detect this, so do the conversion in both cases. In
117 # case of failure, this at least fails when parsing the file instead
118 # being that later when inserting the data to the database.
119 return text.encode("utf-16", "surrogatepass").decode("utf-16")
120 # We might get integer or float in some formats
121 return str(text)
124def is_repo_link(val: str) -> bool:
125 """Check whether repository is just a link for other one."""
126 return val.startswith("weblate://")
129def get_distinct_translations(units: Iterable[Unit]) -> list[Unit]:
130 """
131 Return list of distinct translations.
133 It should be possible to use distinct('target') since Django 1.4, but it is not
134 supported with MySQL, so let's emulate that based on presumption we won't get too
135 many results.
136 """
137 targets = {}
138 result = []
139 for unit in units:
140 if unit.target in targets:
141 continue
142 targets[unit.target] = 1
143 result.append(unit)
144 return result
147def translation_percent(
148 translated: int, total: int, zero_complete: bool = True
149) -> float:
150 """Return translation percentage."""
151 if total == 0:
152 return 100.0 if zero_complete else 0.0
153 if total is None: 153 ↛ 154line 153 didn't jump to line 154 because the condition on line 153 was never true
154 return 0.0
155 perc = (1000 * translated // total) / 10.0
156 # Avoid displaying misleading rounded 0.0% or 100.0%
157 if perc == 0.0 and translated != 0: 157 ↛ 158line 157 didn't jump to line 158 because the condition on line 157 was never true
158 return 0.1
159 if perc == 100.0 and translated < total: 159 ↛ 160line 159 didn't jump to line 160 because the condition on line 159 was never true
160 return 99.9
161 return perc
164def get_clean_env(
165 extra: dict[str, str] | None = None, extra_path: str | None = None
166) -> dict[str, str]:
167 """Return cleaned up environment for subprocess execution."""
168 environ = {
169 "LANG": "C.UTF-8",
170 "LC_ALL": "C.UTF-8",
171 "HOME": data_dir("home"),
172 "PATH": "/bin:/usr/bin:/usr/local/bin",
173 }
174 if extra is not None:
175 environ.update(extra)
176 variables = (
177 # Keep PATH setup
178 "PATH",
179 # Keep Python search path
180 "PYTHONPATH",
181 # Keep linker configuration
182 "LD_LIBRARY_PATH",
183 "LD_PRELOAD",
184 # Fontconfig configuration by weblate.fonts
185 "FONTCONFIG_FILE",
186 # Needed by Git on Windows
187 "SystemRoot",
188 # Pass proxy configuration
189 "http_proxy",
190 "https_proxy",
191 "HTTPS_PROXY",
192 "NO_PROXY",
193 # below two are needed for openshift3 deployment,
194 # where nss_wrapper is used
195 # more on the topic on below link:
196 # https://docs.openshift.com/enterprise/3.2/creating_images/guidelines.html
197 "NSS_WRAPPER_GROUP",
198 "NSS_WRAPPER_PASSWD",
199 )
200 for var in variables:
201 if var in os.environ:
202 environ[var] = os.environ[var]
203 # Extend path to include virtualenv, avoid insert already existing ones to
204 # not break existing ordering (for example PATH injection used in tests)
205 venv_path = os.path.join(sys.exec_prefix, "bin")
206 if venv_path not in environ["PATH"]: 206 ↛ 207line 206 didn't jump to line 207 because the condition on line 206 was never true
207 environ["PATH"] = "{}:{}".format(venv_path, environ["PATH"])
208 if extra_path and extra_path not in environ["PATH"]:
209 environ["PATH"] = "{}:{}".format(extra_path, environ["PATH"])
210 return environ
213def cleanup_repo_url(url: str, text: str | None = None) -> str:
214 """Remove credentials from repository URL."""
215 if text is None:
216 text = url
217 try:
218 parsed = urlparse(url)
219 except ValueError:
220 # The URL can not be parsed, so avoid stripping
221 return text
222 if parsed.username and parsed.password: 222 ↛ 223line 222 didn't jump to line 223 because the condition on line 222 was never true
223 return text.replace(f"{parsed.username}:{parsed.password}@", "")
224 if parsed.username: 224 ↛ 225line 224 didn't jump to line 225 because the condition on line 224 was never true
225 return text.replace(f"{parsed.username}@", "")
226 return text
229def redirect_param(location, params, *args, **kwargs):
230 """Redirect to a URL with parameters."""
231 return HttpResponseRedirect(resolve_url(location, *args, **kwargs) + params)
234def cleanup_path(path: str) -> str:
235 """Remove leading ./ or / from path."""
236 if not path:
237 return path
239 # interpret absolute pathname as relative, remove drive letter or
240 # UNC path, redundant separators, "." and ".." components.
241 path = os.path.splitdrive(path)[1]
242 invalid_path_parts = ("", os.path.curdir, os.path.pardir)
243 path = os.path.sep.join(
244 x for x in path.split(os.path.sep) if x not in invalid_path_parts
245 )
247 return os.path.normpath(path)
250def get_project_description(project: Project) -> str:
251 """Return verbose description for project translation."""
252 # Cache the count as it might be expensive to calculate (it pull
253 # all project stats) and there is no need to always have up to date
254 # count here
255 cache_key = f"project-lang-count-{project.id}"
256 count = cache.get(cache_key)
257 if count is None:
258 count = project.stats.languages
259 cache.set(cache_key, count, 6 * 3600)
260 return gettext(
261 "{0} is being translated into {1} languages using Weblate. "
262 "Join the translation or start translating your own project."
263 ).format(project, count)
266def render(
267 request,
268 template_name: str,
269 context: dict[str, Any] | None = None,
270 content_type: str | None = None,
271 status: int | None = None,
272 using=None,
273):
274 """Render template with Weblate extended context."""
275 if context is None: 275 ↛ 276line 275 didn't jump to line 276 because the condition on line 275 was never true
276 context = {}
277 if "project" in context and context["project"] is not None: 277 ↛ 278line 277 didn't jump to line 278 because the condition on line 277 was never true
278 context["description"] = get_project_description(context["project"])
280 return django.shortcuts.render(
281 request,
282 template_name=template_name,
283 context=context,
284 content_type=content_type,
285 status=status,
286 using=using,
287 )
290def path_separator(path: str) -> str:
291 """
292 Consolidate path separator.
294 Always use / as path separator for consistency.
295 """
296 if os.path.sep != "/": 296 ↛ 297line 296 didn't jump to line 297 because the condition on line 296 was never true
297 return path.replace(os.path.sep, "/")
298 return path
301T = TypeVar("T")
304def sort_unicode(choices: list[T], key: Callable[[T], str]) -> list[T]:
305 """Unicode aware sorting if available."""
307 def sort_strxfrm(item: T) -> str:
308 return locale.strxfrm(key(item))
310 return sorted(choices, key=sort_strxfrm if USE_STRXFRM else key)
313def sort_choices(choices: list[tuple[str, str]]) -> list[tuple[str, str]]:
314 """Sort choices alphabetically."""
315 return sort_unicode(choices, itemgetter(1))
318def sort_objects(objects: list[Model]) -> list[Model]:
319 """Sort objects alphabetically."""
320 return sort_unicode(objects, str)
323def redirect_next(
324 next_url: str | None, fallback: str | SupportsGetAbsoluteUrl
325) -> HttpResponseRedirect:
326 """Redirect to next URL from request after validating it."""
327 if (
328 next_url is None
329 or not url_has_allowed_host_and_scheme(next_url, allowed_hosts=None)
330 or not next_url.startswith("/")
331 ):
332 return redirect(fallback)
333 return HttpResponseRedirect(next_url)
336def xliff_string_to_rich(string: str):
337 """
338 Convert XLIFF string to StringElement.
340 Transform a string containing XLIFF placeholders as XML into a rich content
341 (StringElement)
342 """
343 if isinstance(string, list):
344 return [parse_xliff(s) for s in string]
345 return [parse_xliff(string)]
348def rich_to_xliff_string(string_elements):
349 """
350 Convert StringElement to XLIFF string.
352 Transform rich content (StringElement) into a string with placeholder kept as XML
353 """
354 # Create dummy root element
355 xml = etree.Element("e")
356 for string_element in string_elements:
357 # Inject placeable from translate-toolkit
358 strelem_to_xml(xml, string_element)
360 # Remove any possible namespace
361 for child in xml:
362 tag = cast("str", child.tag)
363 if tag.startswith("{"):
364 child.tag = tag[tag.index("}") + 1 :]
365 etree.cleanup_namespaces(xml)
367 # Convert to string
368 string_xml = etree.tostring(xml, encoding="unicode")
370 # Strip dummy root element
371 return get_string(string_xml[3:][:-4])
374def check_upload_method_permissions(
375 user: User, translation: Translation, method: str
376) -> PermissionResult | bool:
377 """Check whether user has permission to perform upload method."""
378 from weblate.formats.base import BilingualUpdateMixin
380 if method == "source":
381 if not translation.is_source:
382 return Denied(
383 gettext("Source upload is only supported on the source language.")
384 )
385 if translation.is_template:
386 return Denied(
387 gettext(
388 "Source upload is only supported for bilingual translations, you might want to use replace upload instead."
389 )
390 )
391 if not issubclass(translation.component.file_format_cls, BilingualUpdateMixin):
392 return Denied(
393 gettext(
394 "Update source strings upload is not supported with this format."
395 )
396 )
397 return user.has_perm("upload.perform", translation)
398 if method == "add":
399 return user.has_perm("unit.add", translation)
400 if method in {"translate", "fuzzy"}:
401 return user.has_perm("unit.edit", translation)
402 if method == "suggest":
403 return user.has_perm("suggestion.add", translation)
404 if method == "approve":
405 return user.has_perm("unit.review", translation)
406 if method == "replace":
407 return bool(translation.filename) and (
408 user.has_perm("component.edit", translation)
409 or user.has_perms(["unit.add", "unit.delete", "unit.edit"], translation)
410 )
411 msg = f"Invalid method: {method}"
412 raise ValueError(msg)
415def is_unused_string(string: str) -> bool:
416 """Check whether string should not be used."""
417 return string.startswith("<unused singular")
420def count_words(string: str, language: Language | None = None) -> int:
421 """Count number of words in a string."""
422 if language is not None and language.is_cjk(): 422 ↛ 423line 422 didn't jump to line 423 because the condition on line 422 was never true
423 count = 0
424 for s in split_plural(string):
425 if is_unused_string(s):
426 continue
427 even = True
428 for sec in CJK_PATTERN.split(string):
429 if even:
430 count += len(sec.split())
431 else:
432 count += len(sec)
433 even = not even
434 return count
435 return sum(len(s.split()) for s in split_plural(string) if not is_unused_string(s))