Coverage for app/venv/lib/python3.14/site-packages/weblate/formats/exporters.py: 55%
221 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 07:15 +0000
1# Copyright © Michal Čihař <michal@weblate.org>
2#
3# SPDX-License-Identifier: GPL-3.0-or-later
5"""Exporter using translate-toolkit."""
7from __future__ import annotations
9import re
10from itertools import chain
11from typing import TYPE_CHECKING, ClassVar
13from django.utils.translation import gettext_lazy
14from lxml.etree import XMLSyntaxError
15from translate.misc.multistring import multistring
16from translate.storage.aresource import AndroidResourceFile
17from translate.storage.csvl10n import csvfile
18from translate.storage.jsonl10n import JsonFile, JsonNestedFile
19from translate.storage.mo import mofile
20from translate.storage.poxliff import PoXliffFile
21from translate.storage.properties import stringsutf8file
22from translate.storage.pypo import pofile
23from translate.storage.tbx import tbxfile
24from translate.storage.tmx import tmxfile
25from translate.storage.xliff import xlifffile
27import weblate.utils.version
28from weblate.formats.external import XlsxFormat
29from weblate.trans.util import split_plural, xliff_string_to_rich
30from weblate.utils.csv import PROHIBITED_INITIAL_CHARS
32from .base import BaseExporter
34if TYPE_CHECKING: 34 ↛ 35line 34 didn't jump to line 35 because the condition on line 34 was never true
35 from translate.storage.base import TranslationStore
36 from translate.storage.lisa import LISAfile
38 from weblate.trans.models import Translation
40# Map to remove control characters except newlines and tabs
41# Based on lxml - src/lxml/apihelpers.pxi _is_valid_xml_utf8
42XML_REPLACE_CHARMAP = dict.fromkeys(
43 chain(
44 (x for x in range(32) if x not in {9, 10, 13}),
45 [0xFFFE, 0xFFFF],
46 range(0xD800, 0xDFFF + 1),
47 )
48)
50DASHES = re.compile(r"--+")
53class PoExporter(BaseExporter):
54 name = "po"
55 content_type = "text/x-po"
56 extension = "po"
57 verbose = gettext_lazy("gettext PO")
58 storage_class: ClassVar[type[TranslationStore]] = pofile
60 def store_flags(self, output, flags) -> None:
61 for flag in flags.items():
62 output.settypecomment(flags.format_flag(flag))
64 def get_storage(self):
65 store = super().get_storage()
66 plural = self.plural
68 # Set po file header
69 store.updateheader(
70 add=True,
71 language=self.language.code,
72 x_generator=f"Weblate {weblate.utils.version.VERSION}",
73 project_id_version=f"{self.language.name} ({self.project.name})",
74 plural_forms=plural.plural_form,
75 language_team=f"{self.language.name} <{self.url}>",
76 )
77 return store
80class XMLFilterMixin(BaseExporter):
81 def string_filter(self, text):
82 return super().string_filter(text).translate(XML_REPLACE_CHARMAP)
85class XMLExporter(XMLFilterMixin, BaseExporter):
86 """Wrapper for XML based exporters to strip control characters."""
88 storage_class: ClassVar[type[LISAfile]]
90 def get_storage(self):
91 return self.storage_class(
92 sourcelanguage=self.source_language.code,
93 targetlanguage=self.language.code,
94 )
96 def add(self, unit, word) -> None:
97 unit.settarget(word, self.language.code)
100class PoXliffExporter(XMLExporter):
101 name = "xliff"
102 content_type = "application/x-xliff+xml"
103 extension = "xlf"
104 set_id = True
105 verbose = gettext_lazy("XLIFF 1.1 with gettext extensions")
106 storage_class: ClassVar[type[LISAfile]] = PoXliffFile
108 def store_flags(self, output, flags) -> None:
109 if flags.has_value("max-length"):
110 output.xmlelement.set("maxwidth", str(flags.get_value("max-length")))
112 output.xmlelement.set("weblate-flags", flags.format())
114 def handle_plurals(self, plurals):
115 if len(plurals) == 1:
116 return self.string_filter(plurals[0])
117 return multistring([self.string_filter(plural) for plural in plurals])
119 def build_unit(self, unit):
120 output = super().build_unit(unit)
121 try:
122 converted_source = xliff_string_to_rich(unit.get_source_plurals())
123 converted_target = xliff_string_to_rich(unit.get_target_plurals())
124 except (XMLSyntaxError, TypeError, KeyError):
125 return output
126 output.set_rich_source(converted_source, self.source_language.code)
127 output.set_rich_target(converted_target, self.language.code)
128 return output
131class XliffExporter(PoXliffExporter):
132 name = "xliff11"
133 content_type = "application/x-xliff+xml"
134 extension = "xlf"
135 set_id = True
136 verbose = gettext_lazy("XLIFF 1.1")
137 storage_class = xlifffile
140class TBXExporter(XMLExporter):
141 name = "tbx"
142 content_type = "application/x-tbx"
143 extension = "tbx"
144 verbose = gettext_lazy("TBX")
145 storage_class = tbxfile
148class TMXExporter(XMLExporter):
149 name = "tmx"
150 content_type = "application/x-tmx"
151 extension = "tmx"
152 verbose = gettext_lazy("TMX")
153 storage_class = tmxfile
156class MoExporter(PoExporter):
157 name = "mo"
158 content_type = "application/x-gettext-catalog"
159 extension = "mo"
160 verbose = gettext_lazy("gettext MO")
161 storage_class = mofile
163 def __init__(
164 self,
165 project=None,
166 source_language=None,
167 language=None,
168 url=None,
169 translation=None,
170 fieldnames=None,
171 ) -> None:
172 super().__init__(
173 project=project,
174 source_language=source_language,
175 language=language,
176 url=url,
177 translation=translation,
178 fieldnames=fieldnames,
179 )
180 # Detect storage properties
181 self.monolingual = False
182 self.use_context = False
183 if translation:
184 self.monolingual = translation.component.has_template()
185 if self.monolingual:
186 try:
187 unit = translation.store.content_units[0]
188 self.use_context = not unit.template.source
189 except IndexError:
190 pass
192 def store_flags(self, output, flags) -> None:
193 return
195 def add_unit(self, unit) -> None:
196 # Parse properties from unit
197 if self.monolingual:
198 if self.use_context:
199 source = ""
200 context = unit.context
201 else:
202 source = unit.context
203 context = ""
204 else:
205 source = self.handle_plurals(unit.get_source_plurals())
206 context = unit.context
207 # Actually create the unit and set attributes
208 output = self.create_unit(source)
209 output.target = self.handle_plurals(unit.get_target_plurals())
210 if context:
211 output.setcontext(context)
212 # Add unit to the storage
213 self.storage.addunit(output)
215 @staticmethod
216 def supports(translation: Translation) -> bool:
217 return translation.component.file_format in {"po", "po-mono"}
220class CVSBaseExporter(BaseExporter):
221 storage_class = csvfile
223 def get_storage(self):
224 storage = self.storage_class(fieldnames=self.fieldnames)
225 # Use Excel dialect instead of translate-toolkit "default" to avoid
226 # unnecessary escaping with backslash which later confuses our importer
227 # at it is typically used occasionally.
228 storage.dialect = "excel"
229 return storage
232class CSVExporter(CVSBaseExporter):
233 name = "csv"
234 content_type = "text/csv"
235 extension = "csv"
236 verbose = gettext_lazy("CSV")
238 def string_filter(self, text):
239 """
240 Avoid Excel interpreting text as formula.
242 This is really bad idea, implemented in Excel, as this change leads to
243 displaying additional ' in all other tools, but this seems to be what most
244 people have gotten used to. Hopefully these characters are not widely used at
245 first position of translatable strings, so that harm is reduced.
247 Reverse for this is in weblate.formats.ttkit.CSVUnit.unescape_csv
248 """
249 if text and text[0] in PROHIBITED_INITIAL_CHARS: 249 ↛ 250line 249 didn't jump to line 250 because the condition on line 249 was never true
250 return "'{}'".format(text.replace("|", "\\|"))
251 return text
254class MultiCSVExporter(CVSBaseExporter):
255 name = "csv-multi"
256 content_type = "text/csv"
257 extension = "csv"
258 verbose = gettext_lazy("Multivalue CSV")
260 @staticmethod
261 def supports(translation: Translation) -> bool:
262 return translation.component.file_format_cls.has_multiple_strings
264 def string_filter(self, text):
265 """
266 Avoid Excel interpreting text as formula.
268 This is really bad idea, implemented in Excel, as this change leads to
269 displaying additional ' in all other tools, but this seems to be what most
270 people have gotten used to. Hopefully these characters are not widely used at
271 first position of translatable strings, so that harm is reduced.
273 Reverse for this is in weblate.formats.ttkit.CSVUnit.unescape_csv
274 """
275 if text and text[0] in PROHIBITED_INITIAL_CHARS:
276 return "'{}'".format(text.replace("|", "\\|"))
277 return text
279 def add_units(self, units):
280 # Override add_units to handle multivalue units by adding each translation
281 # as a separate row in the CSV export.
283 for unit in units:
284 if not unit.target:
285 continue
287 # Split the target into plural forms
288 targets = split_plural(unit.target)
290 if len(targets) > 1:
291 # Multiple translations for this unit - add each one separately
292 for target in targets:
293 if target.strip(): # Only add non-empty targets
294 # Create a temporary unit with this specific target
295 temp_unit = type(unit)(
296 translation=unit.translation,
297 source=unit.source,
298 target=target,
299 context=unit.context,
300 location=unit.location,
301 note=unit.note,
302 flags=unit.flags,
303 explanation=unit.explanation,
304 state=unit.state,
305 position=unit.position,
306 id_hash=unit.id_hash,
307 )
308 # Set source_unit to the original unit's source_unit to avoid None issues
309 temp_unit.source_unit = unit.source_unit
310 # Set unresolved_comments to empty list to avoid query issues
311 temp_unit.__dict__["unresolved_comments"] = []
312 # Set suggestions to empty list to avoid database query issues
313 temp_unit.__dict__["suggestions"] = []
315 self.add_unit(temp_unit)
316 else:
317 # Single translation - add normally
318 self.add_unit(unit)
321class XlsxExporter(XMLFilterMixin, CVSBaseExporter):
322 name = "xlsx"
323 content_type = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
324 extension = "xlsx"
325 verbose = gettext_lazy("XLSX")
327 def serialize(self):
328 """Return storage content."""
329 return XlsxFormat.serialize(self.storage)
332class MonolingualExporter(BaseExporter):
333 """Base class for monolingual exports."""
335 @staticmethod
336 def supports(translation: Translation) -> bool:
337 return translation.component.has_template()
339 def build_unit(self, unit):
340 output = self.create_unit(unit.context)
341 output.setid(unit.context)
342 self.add(output, self.handle_plurals(unit.get_target_plurals()))
343 return output
346class JSONExporter(MonolingualExporter):
347 storage_class = JsonFile
348 name = "json"
349 content_type = "application/json"
350 extension = "json"
351 verbose = gettext_lazy("JSON")
354class JSONNestedExporter(JSONExporter):
355 name = "json-nested"
356 verbose = gettext_lazy("JSON nested structure file")
357 storage_class = JsonNestedFile
360class AndroidResourceExporter(XMLFilterMixin, MonolingualExporter):
361 storage_class = AndroidResourceFile
362 name = "aresource"
363 content_type = "application/xml"
364 extension = "xml"
365 verbose = gettext_lazy("Android String Resource")
367 def add(self, unit, word) -> None:
368 # Need to have storage to handle plurals
369 unit._store = self.storage # noqa: SLF001
370 super().add(unit, word)
372 def add_note(self, output, note: str, origin: str) -> None:
373 # Remove -- from the comment or - at the end as that is not
374 # allowed inside XML comment
375 note = DASHES.sub("-", note)
376 if note.endswith("-"):
377 note += " "
378 super().add_note(output, note, origin)
381class StringsExporter(MonolingualExporter):
382 storage_class = stringsutf8file
383 name = "strings"
384 content_type = "text/plain"
385 extension = "strings"
386 verbose = gettext_lazy("iOS strings")
388 def create_unit(self, source):
389 return self.storage.UnitClass(source, self.storage.personality.name)