Coverage for app/venv/lib/python3.14/site-packages/weblate/utils/validators.py: 55%

176 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 07:15 +0000

1# Copyright © Michal Čihař <michal@weblate.org> 

2# 

3# SPDX-License-Identifier: GPL-3.0-or-later 

4 

5from __future__ import annotations 

6 

7import base64 

8import binascii 

9import os 

10import re 

11import sys 

12from email.errors import HeaderDefect 

13from email.headerregistry import Address 

14from gettext import c2py # type: ignore[attr-defined] 

15from io import BytesIO 

16from pathlib import Path 

17from typing import cast 

18from urllib.parse import urlparse 

19 

20from disposable_email_domains import blocklist 

21from django.conf import settings 

22from django.core.exceptions import ValidationError 

23from django.core.validators import EmailValidator as EmailValidatorDjango 

24from django.core.validators import URLValidator, validate_ipv46_address 

25from django.utils.translation import gettext, gettext_lazy 

26 

27from weblate.trans.util import cleanup_path 

28from weblate.utils.data import data_dir 

29 

30USERNAME_MATCHER = re.compile(r"^[\w@+-][\w.@+-]*$") 

31 

32# Reject some suspicious e-mail addresses, based on checks enforced by Exim MTA 

33EMAIL_BLACKLIST = re.compile(r"^([./|]|.*([@%!`#&?]|/\.\./))") 

34 

35# Matches Git condition on "name consists only of disallowed characters" 

36CRUD_RE = re.compile(r"^[.,;:<>\"'\\]+$") 

37 

38ALLOWED_IMAGES = {"image/jpeg", "image/png", "image/apng", "image/gif", "image/webp"} 

39 

40# File formats we do not accept on translation/glossary upload 

41FORBIDDEN_EXTENSIONS = { 

42 ".png", 

43 ".jpg", 

44 ".gif", 

45 ".svg", 

46 ".doc", 

47 ".rtf", 

48 ".xls", 

49 ".docx", 

50 ".py", 

51 ".js", 

52 ".exe", 

53 ".dll", 

54 ".zip", 

55} 

56 

57 

58def validate_re(value, groups=None, allow_empty=True) -> None: 

59 try: 

60 compiled = re.compile(value) 

61 except re.error as error: 

62 raise ValidationError( 

63 gettext("Compilation failed: {0}").format(error) 

64 ) from error 

65 if not allow_empty and compiled.match(""): 65 ↛ 66line 65 didn't jump to line 66 because the condition on line 65 was never true

66 raise ValidationError( 

67 gettext("The regular expression can not match an empty string.") 

68 ) 

69 if not groups: 69 ↛ 71line 69 didn't jump to line 71 because the condition on line 69 was always true

70 return 

71 for group in groups: 

72 if group not in compiled.groupindex: 

73 raise ValidationError( 

74 gettext( 

75 'Regular expression is missing named group "{0}", ' 

76 "the simplest way to define it is {1}." 

77 ).format(group, f"(?P<{group}>.*)") 

78 ) 

79 

80 

81def validate_re_nonempty(value): 

82 return validate_re(value, allow_empty=False) 

83 

84 

85def validate_bitmap(value) -> None: 

86 """Validate bitmap, based on django.forms.fields.ImageField.""" 

87 from PIL import Image 

88 

89 if value is None: 

90 return 

91 

92 # Ensure we have image object and content type 

93 # Pretty much copy from django.forms.fields.ImageField: 

94 

95 # We need to get a file object for Pillow. We might have a path or we 

96 # might have to read the data into memory. 

97 if hasattr(value, "temporary_file_path"): 

98 content = value.temporary_file_path() 

99 elif hasattr(value, "read"): 

100 content = BytesIO(value.read()) 

101 else: 

102 content = BytesIO(value["content"]) 

103 

104 try: 

105 # load() could spot a truncated JPEG, but it loads the entire 

106 # image in memory, which is a DoS vector. See #3848 and #18520. 

107 image = Image.open(content) 

108 # verify() must be called immediately after the constructor. 

109 image.verify() 

110 

111 # Pillow doesn't detect the MIME type of all formats. In those 

112 # cases, content_type will be None. 

113 value.file.content_type = Image.MIME.get(cast("str", image.format)) 

114 except Exception as exc: 

115 # Pillow doesn't recognize it as an image. 

116 raise ValidationError( 

117 gettext("The uploaded image was invalid."), code="invalid_image" 

118 ).with_traceback(sys.exc_info()[2]) from exc 

119 if hasattr(value.file, "seek") and callable(value.file.seek): 

120 value.file.seek(0) 

121 

122 # Check image type 

123 if value.file.content_type not in ALLOWED_IMAGES: 

124 image.close() 

125 raise ValidationError( 

126 gettext("Unsupported image type: %s") % value.file.content_type 

127 ) 

128 

129 # Check dimensions 

130 width, height = image.size 

131 if width > 2000 or height > 2000: 

132 image.close() 

133 raise ValidationError( 

134 gettext("The image is too big, please crop or scale it down.") 

135 ) 

136 

137 image.close() 

138 

139 

140def clean_fullname(val): 

141 """Remove special characters from user full name.""" 

142 if not val: 142 ↛ 143line 142 didn't jump to line 143 because the condition on line 142 was never true

143 return val 

144 val = val.strip() 

145 for i in range(0x20): 

146 val = val.replace(chr(i), "") 

147 return val 

148 

149 

150def validate_fullname(val): 

151 if val != clean_fullname(val): 

152 raise ValidationError( 

153 gettext("Please avoid using special characters in the full name.") 

154 ) 

155 # Validates full name that would be rejected by Git 

156 if CRUD_RE.match(val): 

157 raise ValidationError(gettext("Name consists only of disallowed characters.")) 

158 

159 return val 

160 

161 

162def validate_file_extension(value): 

163 """Validate file upload based on extension.""" 

164 ext = os.path.splitext(value.name)[1] 

165 if ext.lower() in FORBIDDEN_EXTENSIONS: 

166 raise ValidationError(gettext("Unsupported file format.")) 

167 return value 

168 

169 

170def validate_username(value) -> None: 

171 if value.startswith("."): 171 ↛ 172line 171 didn't jump to line 172 because the condition on line 171 was never true

172 raise ValidationError(gettext("The username can not start with a full stop.")) 

173 if not USERNAME_MATCHER.match(value): 

174 raise ValidationError( 

175 gettext( 

176 "Username may only contain letters, " 

177 "numbers or the following characters: @ . + - _" 

178 ) 

179 ) 

180 

181 

182class EmailValidator(EmailValidatorDjango): 

183 message = gettext_lazy("Enter a valid e-mail address.") 

184 

185 def __call__(self, value: str | None): 

186 super().__call__(value) 

187 if value is None: 187 ↛ 188line 187 didn't jump to line 188 because the condition on line 187 was never true

188 return 

189 user_part = value.rsplit("@", 1)[0] 

190 if EMAIL_BLACKLIST.match(user_part): 

191 raise ValidationError(gettext("Enter a valid e-mail address.")) 

192 if not re.match(settings.REGISTRATION_EMAIL_MATCH, value): 192 ↛ 193line 192 didn't jump to line 193 because the condition on line 192 was never true

193 raise ValidationError(gettext("This e-mail address is disallowed.")) 

194 try: 

195 address = Address(addr_spec=value) 

196 except HeaderDefect as error: 

197 raise ValidationError( 

198 gettext("Invalid e-mail address: {}").format(error) 

199 ) from error 

200 

201 if address.domain in blocklist: 201 ↛ 202line 201 didn't jump to line 202 because the condition on line 201 was never true

202 raise ValidationError(gettext("Disposable e-mail domains are disallowed.")) 

203 

204 

205validate_email = EmailValidator() 

206 

207 

208def validate_plural_formula(value) -> None: 

209 try: 

210 c2py(value or "0") 

211 except ValueError as error: 

212 raise ValidationError( 

213 gettext("Could not evaluate plural formula: {}").format(error) 

214 ) from error 

215 

216 

217def validate_filename(value) -> None: 

218 if "../" in value or "..\\" in value: 218 ↛ 219line 218 didn't jump to line 219 because the condition on line 218 was never true

219 raise ValidationError( 

220 gettext("The filename can not contain reference to a parent directory.") 

221 ) 

222 if os.path.isabs(value): 222 ↛ 223line 222 didn't jump to line 223 because the condition on line 222 was never true

223 raise ValidationError(gettext("The filename can not be an absolute path.")) 

224 

225 cleaned = cleanup_path(value) 

226 if value != cleaned: 226 ↛ 227line 226 didn't jump to line 227 because the condition on line 226 was never true

227 raise ValidationError( 

228 gettext( 

229 "The filename should be as simple as possible. " 

230 "Maybe you want to use: {}" 

231 ).format(cleaned) 

232 ) 

233 

234 

235def validate_backup_path(value: str) -> None: 

236 # Lazily import borg as it pulls quite a lot of memory usage 

237 from borg.helpers import Location 

238 

239 try: 

240 loc = Location(value) 

241 except ValueError as err: 

242 raise ValidationError(str(err)) from err 

243 

244 if loc.archive: 

245 msg = "No archive can be specified in backup location." 

246 raise ValidationError(msg) 

247 

248 if loc.proto == "file": 

249 # The path is already normalized here 

250 path = Path(loc.path) 

251 

252 # Restrict relative paths as the cwd might change 

253 if not path.is_absolute(): 

254 msg = "Backup location has to be an absolute path." 

255 raise ValidationError(msg) 

256 

257 # Restrict placing under Weblate backups as that will produce mess 

258 data_backups = Path(data_dir("backups")) 

259 if data_backups == path or data_backups in path.parents: 

260 msg = "Backup location should be outside Weblate backups in DATA_DIR." 

261 raise ValidationError(msg) 

262 

263 

264def validate_slug(value) -> None: 

265 """Prohibits some special values.""" 

266 # This one is used as wildcard in the URL for widgets and translate pages 

267 if value == "-": 

268 raise ValidationError(gettext("This name is prohibited")) 

269 

270 

271def validate_language_aliases(value) -> None: 

272 """Validate language aliases - comma separated semi colon values.""" 

273 if not value: 273 ↛ 274line 273 didn't jump to line 274 because the condition on line 273 was never true

274 return 

275 for part in value.split(","): 275 ↛ exitline 275 didn't return from function 'validate_language_aliases' because the loop on line 275 didn't complete

276 if part.count(":") != 1: 276 ↛ 275line 276 didn't jump to line 275 because the condition on line 276 was always true

277 raise ValidationError(gettext("Syntax error in language aliases.")) 

278 

279 

280def validate_project_name(value) -> None: 

281 """Prohibits some special values.""" 

282 if settings.PROJECT_NAME_RESTRICT_RE is not None and re.match( 282 ↛ 285line 282 didn't jump to line 285 because the condition on line 282 was never true

283 settings.PROJECT_NAME_RESTRICT_RE, value 

284 ): 

285 raise ValidationError(gettext("This name is prohibited")) 

286 

287 

288def validate_project_web(value) -> None: 

289 # Regular expression filtering 

290 if settings.PROJECT_WEB_RESTRICT_RE is not None and re.match( 290 ↛ 293line 290 didn't jump to line 293 because the condition on line 290 was never true

291 settings.PROJECT_WEB_RESTRICT_RE, value 

292 ): 

293 raise ValidationError(gettext("This URL is prohibited")) 

294 parsed = urlparse(value) 

295 hostname = parsed.hostname or "" 

296 hostname = hostname.lower() 

297 

298 # Hostname filtering 

299 if any( 

300 hostname.endswith(blocked) for blocked in settings.PROJECT_WEB_RESTRICT_HOST 

301 ): 

302 raise ValidationError(gettext("This URL is prohibited")) 

303 

304 # Numeric address filtering 

305 if settings.PROJECT_WEB_RESTRICT_NUMERIC: 305 ↛ exitline 305 didn't return from function 'validate_project_web' because the condition on line 305 was always true

306 try: 

307 validate_ipv46_address(hostname) 

308 except ValidationError: 

309 pass 

310 else: 

311 raise ValidationError(gettext("This URL is prohibited")) 

312 

313 

314def validate_base64_encoded_string(value: str) -> None: 

315 """Validate that the given string is a valid base64 encoded string.""" 

316 try: 

317 base64.b64decode(value) 

318 except binascii.Error as error: 

319 raise ValidationError(gettext("Invalid base64 encoded string")) from error 

320 

321 

322class WeblateURLValidator(URLValidator): 

323 """Validator for http and https URLs only.""" 

324 

325 schemes: list[str] = [ # noqa: RUF012 

326 "http", 

327 "https", 

328 ] 

329 

330 

331class WeblateEditorURLValidator(URLValidator): 

332 schemes: list[str] = [ # noqa: RUF012 

333 "editor", 

334 "netbeans", 

335 "txmt", 

336 "pycharm", 

337 "phpstorm", 

338 "idea", 

339 "jetbrains", 

340 ] 

341 

342 regex = re.compile( 

343 r"^(?:[a-z0-9.+-]*)://" # scheme is validated separately 

344 r"(?:" + WeblateURLValidator.hostname_re + ")" 

345 r"(?:[/?#][^\s]*)?" # resource path 

346 r"\Z", 

347 re.IGNORECASE, 

348 ) 

349 

350 

351class WeblateServiceURLValidator(WeblateURLValidator): 

352 """ 

353 Validator allowing local URLs like http://domain:5000. 

354 

355 This is useful for using dockerized services. 

356 """ 

357 

358 host_re = ( 

359 "(" 

360 + WeblateURLValidator.hostname_re 

361 + WeblateURLValidator.domain_re 

362 + WeblateURLValidator.tld_re 

363 + "|" 

364 + WeblateURLValidator.hostname_re 

365 + ")" 

366 ) 

367 regex = re.compile( 

368 r"^(?:[a-z0-9.+-]*)://" # scheme is validated separately 

369 r"(?:[^\s:@/]+(?::[^\s:@/]*)?@)?" # user:pass authentication 

370 r"(?:" 

371 + WeblateURLValidator.ipv4_re 

372 + "|" 

373 + WeblateURLValidator.ipv6_re 

374 + "|" 

375 + host_re 

376 + ")" 

377 r"(?::[0-9]{1,5})?" # port 

378 r"(?:[/?#][^\s]*)?" # resource path 

379 r"\Z", 

380 re.IGNORECASE, 

381 )