Coverage for app/venv/lib/python3.14/site-packages/weblate/formats/exporters.py: 55%

221 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5"""Exporter using translate-toolkit.""" 

6 

7from __future__ import annotations 

8 

9import re 

10from itertools import chain 

11from typing import TYPE_CHECKING, ClassVar 

12 

13from django.utils.translation import gettext_lazy 

14from lxml.etree import XMLSyntaxError 

15from translate.misc.multistring import multistring 

16from translate.storage.aresource import AndroidResourceFile 

17from translate.storage.csvl10n import csvfile 

18from translate.storage.jsonl10n import JsonFile, JsonNestedFile 

19from translate.storage.mo import mofile 

20from translate.storage.poxliff import PoXliffFile 

21from translate.storage.properties import stringsutf8file 

22from translate.storage.pypo import pofile 

23from translate.storage.tbx import tbxfile 

24from translate.storage.tmx import tmxfile 

25from translate.storage.xliff import xlifffile 

26 

27import weblate.utils.version 

28from weblate.formats.external import XlsxFormat 

29from weblate.trans.util import split_plural, xliff_string_to_rich 

30from weblate.utils.csv import PROHIBITED_INITIAL_CHARS 

31 

32from .base import BaseExporter 

33 

34if TYPE_CHECKING: 34 ↛ 35line 34 didn't jump to line 35 because the condition on line 34 was never true

35 from translate.storage.base import TranslationStore 

36 from translate.storage.lisa import LISAfile 

37 

38 from weblate.trans.models import Translation 

39 

40# Map to remove control characters except newlines and tabs 

41# Based on lxml - src/lxml/apihelpers.pxi _is_valid_xml_utf8 

42XML_REPLACE_CHARMAP = dict.fromkeys( 

43 chain( 

44 (x for x in range(32) if x not in {9, 10, 13}), 

45 [0xFFFE, 0xFFFF], 

46 range(0xD800, 0xDFFF + 1), 

47 ) 

48) 

49 

50DASHES = re.compile(r"--+") 

51 

52 

53class PoExporter(BaseExporter): 

54 name = "po" 

55 content_type = "text/x-po" 

56 extension = "po" 

57 verbose = gettext_lazy("gettext PO") 

58 storage_class: ClassVar[type[TranslationStore]] = pofile 

59 

60 def store_flags(self, output, flags) -> None: 

61 for flag in flags.items(): 

62 output.settypecomment(flags.format_flag(flag)) 

63 

64 def get_storage(self): 

65 store = super().get_storage() 

66 plural = self.plural 

67 

68 # Set po file header 

69 store.updateheader( 

70 add=True, 

71 language=self.language.code, 

72 x_generator=f"Weblate {weblate.utils.version.VERSION}", 

73 project_id_version=f"{self.language.name} ({self.project.name})", 

74 plural_forms=plural.plural_form, 

75 language_team=f"{self.language.name} <{self.url}>", 

76 ) 

77 return store 

78 

79 

80class XMLFilterMixin(BaseExporter): 

81 def string_filter(self, text): 

82 return super().string_filter(text).translate(XML_REPLACE_CHARMAP) 

83 

84 

85class XMLExporter(XMLFilterMixin, BaseExporter): 

86 """Wrapper for XML based exporters to strip control characters.""" 

87 

88 storage_class: ClassVar[type[LISAfile]] 

89 

90 def get_storage(self): 

91 return self.storage_class( 

92 sourcelanguage=self.source_language.code, 

93 targetlanguage=self.language.code, 

94 ) 

95 

96 def add(self, unit, word) -> None: 

97 unit.settarget(word, self.language.code) 

98 

99 

100class PoXliffExporter(XMLExporter): 

101 name = "xliff" 

102 content_type = "application/x-xliff+xml" 

103 extension = "xlf" 

104 set_id = True 

105 verbose = gettext_lazy("XLIFF 1.1 with gettext extensions") 

106 storage_class: ClassVar[type[LISAfile]] = PoXliffFile 

107 

108 def store_flags(self, output, flags) -> None: 

109 if flags.has_value("max-length"): 

110 output.xmlelement.set("maxwidth", str(flags.get_value("max-length"))) 

111 

112 output.xmlelement.set("weblate-flags", flags.format()) 

113 

114 def handle_plurals(self, plurals): 

115 if len(plurals) == 1: 

116 return self.string_filter(plurals[0]) 

117 return multistring([self.string_filter(plural) for plural in plurals]) 

118 

119 def build_unit(self, unit): 

120 output = super().build_unit(unit) 

121 try: 

122 converted_source = xliff_string_to_rich(unit.get_source_plurals()) 

123 converted_target = xliff_string_to_rich(unit.get_target_plurals()) 

124 except (XMLSyntaxError, TypeError, KeyError): 

125 return output 

126 output.set_rich_source(converted_source, self.source_language.code) 

127 output.set_rich_target(converted_target, self.language.code) 

128 return output 

129 

130 

131class XliffExporter(PoXliffExporter): 

132 name = "xliff11" 

133 content_type = "application/x-xliff+xml" 

134 extension = "xlf" 

135 set_id = True 

136 verbose = gettext_lazy("XLIFF 1.1") 

137 storage_class = xlifffile 

138 

139 

140class TBXExporter(XMLExporter): 

141 name = "tbx" 

142 content_type = "application/x-tbx" 

143 extension = "tbx" 

144 verbose = gettext_lazy("TBX") 

145 storage_class = tbxfile 

146 

147 

148class TMXExporter(XMLExporter): 

149 name = "tmx" 

150 content_type = "application/x-tmx" 

151 extension = "tmx" 

152 verbose = gettext_lazy("TMX") 

153 storage_class = tmxfile 

154 

155 

156class MoExporter(PoExporter): 

157 name = "mo" 

158 content_type = "application/x-gettext-catalog" 

159 extension = "mo" 

160 verbose = gettext_lazy("gettext MO") 

161 storage_class = mofile 

162 

163 def __init__( 

164 self, 

165 project=None, 

166 source_language=None, 

167 language=None, 

168 url=None, 

169 translation=None, 

170 fieldnames=None, 

171 ) -> None: 

172 super().__init__( 

173 project=project, 

174 source_language=source_language, 

175 language=language, 

176 url=url, 

177 translation=translation, 

178 fieldnames=fieldnames, 

179 ) 

180 # Detect storage properties 

181 self.monolingual = False 

182 self.use_context = False 

183 if translation: 

184 self.monolingual = translation.component.has_template() 

185 if self.monolingual: 

186 try: 

187 unit = translation.store.content_units[0] 

188 self.use_context = not unit.template.source 

189 except IndexError: 

190 pass 

191 

192 def store_flags(self, output, flags) -> None: 

193 return 

194 

195 def add_unit(self, unit) -> None: 

196 # Parse properties from unit 

197 if self.monolingual: 

198 if self.use_context: 

199 source = "" 

200 context = unit.context 

201 else: 

202 source = unit.context 

203 context = "" 

204 else: 

205 source = self.handle_plurals(unit.get_source_plurals()) 

206 context = unit.context 

207 # Actually create the unit and set attributes 

208 output = self.create_unit(source) 

209 output.target = self.handle_plurals(unit.get_target_plurals()) 

210 if context: 

211 output.setcontext(context) 

212 # Add unit to the storage 

213 self.storage.addunit(output) 

214 

215 @staticmethod 

216 def supports(translation: Translation) -> bool: 

217 return translation.component.file_format in {"po", "po-mono"} 

218 

219 

220class CVSBaseExporter(BaseExporter): 

221 storage_class = csvfile 

222 

223 def get_storage(self): 

224 storage = self.storage_class(fieldnames=self.fieldnames) 

225 # Use Excel dialect instead of translate-toolkit "default" to avoid 

226 # unnecessary escaping with backslash which later confuses our importer 

227 # at it is typically used occasionally. 

228 storage.dialect = "excel" 

229 return storage 

230 

231 

232class CSVExporter(CVSBaseExporter): 

233 name = "csv" 

234 content_type = "text/csv" 

235 extension = "csv" 

236 verbose = gettext_lazy("CSV") 

237 

238 def string_filter(self, text): 

239 """ 

240 Avoid Excel interpreting text as formula. 

241 

242 This is really bad idea, implemented in Excel, as this change leads to 

243 displaying additional ' in all other tools, but this seems to be what most 

244 people have gotten used to. Hopefully these characters are not widely used at 

245 first position of translatable strings, so that harm is reduced. 

246 

247 Reverse for this is in weblate.formats.ttkit.CSVUnit.unescape_csv 

248 """ 

249 if text and text[0] in PROHIBITED_INITIAL_CHARS: 249 ↛ 250line 249 didn't jump to line 250 because the condition on line 249 was never true

250 return "'{}'".format(text.replace("|", "\\|")) 

251 return text 

252 

253 

254class MultiCSVExporter(CVSBaseExporter): 

255 name = "csv-multi" 

256 content_type = "text/csv" 

257 extension = "csv" 

258 verbose = gettext_lazy("Multivalue CSV") 

259 

260 @staticmethod 

261 def supports(translation: Translation) -> bool: 

262 return translation.component.file_format_cls.has_multiple_strings 

263 

264 def string_filter(self, text): 

265 """ 

266 Avoid Excel interpreting text as formula. 

267 

268 This is really bad idea, implemented in Excel, as this change leads to 

269 displaying additional ' in all other tools, but this seems to be what most 

270 people have gotten used to. Hopefully these characters are not widely used at 

271 first position of translatable strings, so that harm is reduced. 

272 

273 Reverse for this is in weblate.formats.ttkit.CSVUnit.unescape_csv 

274 """ 

275 if text and text[0] in PROHIBITED_INITIAL_CHARS: 

276 return "'{}'".format(text.replace("|", "\\|")) 

277 return text 

278 

279 def add_units(self, units): 

280 # Override add_units to handle multivalue units by adding each translation 

281 # as a separate row in the CSV export. 

282 

283 for unit in units: 

284 if not unit.target: 

285 continue 

286 

287 # Split the target into plural forms 

288 targets = split_plural(unit.target) 

289 

290 if len(targets) > 1: 

291 # Multiple translations for this unit - add each one separately 

292 for target in targets: 

293 if target.strip(): # Only add non-empty targets 

294 # Create a temporary unit with this specific target 

295 temp_unit = type(unit)( 

296 translation=unit.translation, 

297 source=unit.source, 

298 target=target, 

299 context=unit.context, 

300 location=unit.location, 

301 note=unit.note, 

302 flags=unit.flags, 

303 explanation=unit.explanation, 

304 state=unit.state, 

305 position=unit.position, 

306 id_hash=unit.id_hash, 

307 ) 

308 # Set source_unit to the original unit's source_unit to avoid None issues 

309 temp_unit.source_unit = unit.source_unit 

310 # Set unresolved_comments to empty list to avoid query issues 

311 temp_unit.__dict__["unresolved_comments"] = [] 

312 # Set suggestions to empty list to avoid database query issues 

313 temp_unit.__dict__["suggestions"] = [] 

314 

315 self.add_unit(temp_unit) 

316 else: 

317 # Single translation - add normally 

318 self.add_unit(unit) 

319 

320 

321class XlsxExporter(XMLFilterMixin, CVSBaseExporter): 

322 name = "xlsx" 

323 content_type = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" 

324 extension = "xlsx" 

325 verbose = gettext_lazy("XLSX") 

326 

327 def serialize(self): 

328 """Return storage content.""" 

329 return XlsxFormat.serialize(self.storage) 

330 

331 

332class MonolingualExporter(BaseExporter): 

333 """Base class for monolingual exports.""" 

334 

335 @staticmethod 

336 def supports(translation: Translation) -> bool: 

337 return translation.component.has_template() 

338 

339 def build_unit(self, unit): 

340 output = self.create_unit(unit.context) 

341 output.setid(unit.context) 

342 self.add(output, self.handle_plurals(unit.get_target_plurals())) 

343 return output 

344 

345 

346class JSONExporter(MonolingualExporter): 

347 storage_class = JsonFile 

348 name = "json" 

349 content_type = "application/json" 

350 extension = "json" 

351 verbose = gettext_lazy("JSON") 

352 

353 

354class JSONNestedExporter(JSONExporter): 

355 name = "json-nested" 

356 verbose = gettext_lazy("JSON nested structure file") 

357 storage_class = JsonNestedFile 

358 

359 

360class AndroidResourceExporter(XMLFilterMixin, MonolingualExporter): 

361 storage_class = AndroidResourceFile 

362 name = "aresource" 

363 content_type = "application/xml" 

364 extension = "xml" 

365 verbose = gettext_lazy("Android String Resource") 

366 

367 def add(self, unit, word) -> None: 

368 # Need to have storage to handle plurals 

369 unit._store = self.storage # noqa: SLF001 

370 super().add(unit, word) 

371 

372 def add_note(self, output, note: str, origin: str) -> None: 

373 # Remove -- from the comment or - at the end as that is not 

374 # allowed inside XML comment 

375 note = DASHES.sub("-", note) 

376 if note.endswith("-"): 

377 note += " " 

378 super().add_note(output, note, origin) 

379 

380 

381class StringsExporter(MonolingualExporter): 

382 storage_class = stringsutf8file 

383 name = "strings" 

384 content_type = "text/plain" 

385 extension = "strings" 

386 verbose = gettext_lazy("iOS strings") 

387 

388 def create_unit(self, source): 

389 return self.storage.UnitClass(source, self.storage.personality.name)