Coverage for api/models/media.py: 66%

200 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 06:14 +0000

1import mimetypes 

2from textwrap import dedent 

3 

4from django.conf import settings 

5from django.core.exceptions import ValidationError 

6from django.db import models 

7from django.urls import reverse 

8from django.utils.html import format_html 

9 

10import structlog 

11from elasticsearch import Elasticsearch, NotFoundError, helpers 

12from openverse_attribution.license import License 

13 

14from api.constants.moderation import DecisionAction 

15from api.models.base import OpenLedgerModel 

16from api.models.mixins import ForeignIdentifierMixin, IdentifierMixin, MediaMixin 

17 

18 

19MATURE = "mature" 

20DMCA = "dmca" 

21OTHER = "other" 

22 

23logger = structlog.get_logger(__name__) 

24 

25 

26class AbstractMedia( 

27 IdentifierMixin, ForeignIdentifierMixin, MediaMixin, OpenLedgerModel 

28): 

29 """ 

30 Generic model from which to inherit all media classes. 

31 

32 This class stores information common to all media types indexed by Openverse. 

33 

34 Properties 

35 ========== 

36 id: >- 

37 This is Django's automatic primary key, used for models that do not 

38 define one explicitly. 

39 """ 

40 

41 watermarked = models.BooleanField( 

42 blank=True, 

43 null=True, 

44 help_text="Whether the media contains a watermark. Not currently leveraged.", 

45 ) 

46 

47 license = models.CharField( 

48 max_length=50, 

49 help_text="The name of license for the media.", 

50 ) 

51 license_version = models.CharField( 

52 max_length=25, 

53 blank=True, 

54 null=True, 

55 help_text="The version of the media license.", 

56 ) 

57 

58 source = models.CharField( 

59 max_length=80, 

60 blank=True, 

61 null=True, 

62 db_index=True, 

63 help_text="The source of the data, meaning a particular dataset. " 

64 "Source and provider can be different. Eg: the Google Open " 

65 "Images dataset is source=openimages, but provider=flickr.", 

66 ) 

67 last_synced_with_source = models.DateTimeField( 

68 blank=True, 

69 null=True, 

70 db_index=True, 

71 help_text="The date the media was last updated from the upstream source.", 

72 ) 

73 removed_from_source = models.BooleanField( 

74 default=False, 

75 help_text="Whether the media has been removed from the upstream source.", 

76 ) 

77 

78 view_count = models.IntegerField( 

79 blank=True, null=True, default=0, help_text="Vestigial field, purpose unknown." 

80 ) 

81 

82 tags = models.JSONField( 

83 blank=True, 

84 null=True, 

85 help_text=dedent(""" 

86 JSON array of objects containing tags for the media. Each tag object 

87 is expected to have: 

88 

89 - `name`: The tag itself (e.g. "dog") 

90 - `provider`: The source of the tag 

91 - `accuracy`: If the tag was added using a machine-labeler, the confidence 

92 for that label expressed as a value between 0 and 1. 

93 

94 Note that only `name` and `accuracy` are presently surfaced in API results. 

95 """), 

96 ) 

97 

98 category = models.CharField( 

99 max_length=80, 

100 blank=True, 

101 null=True, 

102 db_index=True, 

103 help_text="The top-level classification of this media file.", 

104 ) 

105 

106 meta_data = models.JSONField( 

107 blank=True, 

108 null=True, 

109 help_text=dedent(""" 

110 JSON object containing extra data about the media item. No fields are expected, 

111 but if the `license_url` field is available, it will be used for determining 

112 the license URL for the media item. The `description` field, if available, is 

113 also indexed into Elasticsearch and as a search field on queries. 

114 """), 

115 ) 

116 

117 @property 

118 def license_url(self) -> str | None: 

119 """A direct link to the license deed or legal terms.""" 

120 

121 if self.meta_data and (url := self.meta_data.get("license_url")): 121 ↛ 124line 121 didn't jump to line 124 because the condition on line 121 was always true

122 return url 

123 

124 logger.warning( 

125 "Media item missing `license_url` in `meta_data`", 

126 media_class=self.__class__.__name__, 

127 identifier=self.identifier, 

128 ) 

129 try: 

130 return License(self.license.lower(), self.license_version).url 

131 except ValueError: 

132 return None 

133 

134 @property 

135 def attribution(self) -> str | None: 

136 """Legally valid attribution for the media item in plain-text English.""" 

137 

138 try: 

139 return License( 

140 self.license.lower(), 

141 self.license_version, 

142 ).get_attribution_text( 

143 self.title, 

144 self.creator, 

145 self.license_url, 

146 ) 

147 except ValueError: 

148 return None 

149 

150 class Meta: 

151 """ 

152 Meta class for all media types indexed by Openverse. 

153 

154 All concrete media classes should inherit their Meta class from this. 

155 """ 

156 

157 ordering = ["-created_on"] 

158 abstract = True 

159 constraints = [ 

160 models.UniqueConstraint( 

161 fields=["foreign_identifier", "provider"], 

162 name="unique_provider_%(class)s", # populated by concrete model 

163 ), 

164 ] 

165 

166 def __str__(self): 

167 """ 

168 Return the string representation of the model, used in the Django admin site. 

169 

170 :return: the string representation of the model 

171 """ 

172 

173 return f"{self.__class__.__name__}: {self.identifier}" 

174 

175 

176class AbstractMediaReport(models.Model): 

177 """ 

178 Generic model from which to inherit all reported media classes. 

179 

180 'Reported' here refers to content reports such as sensitive, copyright-violating or 

181 deleted content. Subclasses must populate the field ``media_class``. 

182 """ 

183 

184 media_class: type[models.Model] = None 

185 """the model class associated with this media type e.g. ``Image`` or ``Audio``""" 

186 

187 REPORT_CHOICES = [(MATURE, MATURE), (DMCA, DMCA), (OTHER, OTHER)] 

188 

189 created_at = models.DateTimeField(auto_now_add=True) 

190 

191 media_obj = models.ForeignKey( 

192 to="AbstractMedia", 

193 to_field="identifier", 

194 on_delete=models.DO_NOTHING, 

195 db_constraint=False, 

196 db_column="identifier", 

197 related_name="abstract_media_report", 

198 help_text="The reference to the media being reported.", 

199 ) 

200 """ 

201 There can be many reports associated with a single media item, hence foreign key. 

202 Sub-classes must override this field to point to a concrete sub-class of 

203 ``AbstractMedia``. 

204 """ 

205 

206 reason = models.CharField( 

207 max_length=20, 

208 choices=REPORT_CHOICES, 

209 help_text="The reason to report media to Openverse.", 

210 ) 

211 description = models.TextField( 

212 max_length=500, 

213 blank=True, 

214 null=True, 

215 help_text="The explanation on why media is being reported.", 

216 ) 

217 decision = models.ForeignKey( 

218 to="AbstractMediaDecision", 

219 on_delete=models.SET_NULL, 

220 blank=True, 

221 null=True, 

222 help_text="The moderation decision for this report.", 

223 ) 

224 

225 class Meta: 

226 abstract = True 

227 

228 def clean(self): 

229 """Clean fields and raise errors that can be handled by Django Admin.""" 

230 

231 if not self.media_class.objects.filter(identifier=self.media_obj_id).exists(): 231 ↛ 232line 231 didn't jump to line 232 because the condition on line 231 was never true

232 raise ValidationError( 

233 f"No '{self.media_class.__name__}' instance " 

234 f"with identifier '{self.media_obj_id}'." 

235 ) 

236 

237 def media_url(self, request=None) -> str: 

238 """ 

239 Build the URL of the media item. This uses ``reverse`` and 

240 ``request.build_absolute_uri`` to build the URL without having to worry 

241 about canonical URL or trailing slashes. 

242 

243 :param request: the current request object, to get absolute URLs 

244 :return: the URL of the media item 

245 """ 

246 

247 url = reverse( 

248 f"{self.media_class.__name__.lower()}-detail", 

249 args=[self.media_obj_id], 

250 ) 

251 if request is not None: 

252 url = request.build_absolute_uri(url) 

253 return url 

254 

255 def url(self, request=None) -> str: 

256 url = self.media_url(request) 

257 return format_html(f"<a href={url}>{url}</a>") 

258 

259 @property 

260 def is_pending(self) -> bool: 

261 """ 

262 Determine if the report has not been moderated and does not have an 

263 associated decision. Use the inverse of this function to determine 

264 if a report has been reviewed and moderated. 

265 

266 :return: whether the report is in the "pending" state 

267 """ 

268 

269 return self.decision_id is None 

270 

271 def save(self, *args, **kwargs): 

272 """Perform a clean, and then save changes to the DB.""" 

273 

274 self.clean() 

275 super().save(*args, **kwargs) 

276 

277 

278class AbstractMediaDecision(OpenLedgerModel): 

279 """Generic model from which to inherit all moderation decision classes.""" 

280 

281 media_class: type[models.Model] = None 

282 """the model class associated with this media type e.g. ``Image`` or ``Audio``""" 

283 

284 moderator = models.ForeignKey( 

285 to="auth.User", 

286 on_delete=models.DO_NOTHING, 

287 help_text="The moderator who undertook this decision.", 

288 ) 

289 """ 

290 The ``User`` referenced by this field must be a part of the moderators' 

291 group. 

292 """ 

293 

294 media_objs = models.ManyToManyField( 

295 to="AbstractMedia", 

296 through="AbstractMediaDecisionThrough", 

297 help_text="The media items being moderated.", 

298 ) 

299 """ 

300 This is a many-to-many relation, using a bridge table, to enable bulk 

301 moderation which applies a single action to more than one media items. 

302 """ 

303 

304 notes = models.TextField( 

305 max_length=500, 

306 blank=True, 

307 null=True, 

308 help_text="The moderator's explanation for the decision or additional notes.", 

309 ) 

310 

311 action = models.CharField( 

312 max_length=32, 

313 choices=DecisionAction.choices, 

314 help_text="Action taken by the moderator.", 

315 ) 

316 

317 class Meta: 

318 abstract = True 

319 

320 

321class AbstractMediaDecisionThrough(models.Model): 

322 """ 

323 Generic model for the many-to-many reference table between media and decisions. 

324 

325 This is made explicit (rather than using Django's default) so that the media can 

326 be referenced by `identifier` rather than an arbitrary `id`. 

327 """ 

328 

329 media_class: type[models.Model] = None 

330 """the model class associated with this media type e.g. ``Image`` or ``Audio``""" 

331 sensitive_media_class: type[models.Model] = None 

332 """the model class associated with this media type e.g. ``SensitiveImage`` or ``SensitiveAudio``""" 

333 deleted_media_class: type[models.Model] = None 

334 """the model class associated with this media type e.g. ``DeletedImage`` or ``DeletedAudio``""" 

335 

336 media_obj = models.ForeignKey( 

337 AbstractMedia, 

338 to_field="identifier", 

339 on_delete=models.DO_NOTHING, 

340 db_column="identifier", 

341 db_constraint=False, 

342 ) 

343 decision = models.ForeignKey(AbstractMediaDecision, on_delete=models.CASCADE) 

344 

345 class Meta: 

346 abstract = True 

347 

348 def perform_action(self, action=None): 

349 """Perform the action specified in the decision.""" 

350 

351 action = self.decision.action if action is None else action 

352 

353 if action in { 

354 DecisionAction.DEINDEXED_SENSITIVE, 

355 DecisionAction.DEINDEXED_COPYRIGHT, 

356 }: 

357 self.deleted_media_class.objects.create(media_obj_id=self.media_obj_id) 

358 

359 if action == DecisionAction.MARKED_SENSITIVE: 

360 self.sensitive_media_class.objects.create(media_obj_id=self.media_obj_id) 

361 

362 def save(self, *args, **kwargs): 

363 super().save(*args, **kwargs) 

364 self.perform_action() 

365 

366 

367class PerformIndexUpdateMixin: 

368 @classmethod 

369 def indexes(cls): 

370 return [cls.es_index, f"{cls.es_index}-filtered"] 

371 

372 def _perform_index_update(self, method: str, raise_errors: bool, **es_method_args): 

373 """ 

374 Call ``method`` on the Elasticsearch client. 

375 

376 Automatically handles ``DoesNotExist`` warnings, forces a refresh, 

377 and calls the method for origin and filtered indexes. 

378 """ 

379 es: Elasticsearch = settings.ES 

380 

381 try: 

382 document_id = self.media_obj.id 

383 except self.media_class.DoesNotExist: 

384 if raise_errors: 

385 raise ValidationError( 

386 f"No '{self.media_class.__name__}' instance " 

387 f"with identifier {self.media_obj.identifier}." 

388 ) 

389 

390 for index in self.indexes(): 

391 try: 

392 getattr(es, method)( 

393 index=index, 

394 id=document_id, 

395 refresh=True, 

396 **es_method_args, 

397 ) 

398 except NotFoundError: 

399 # This is expected for the filtered index, but we should still 

400 # log, just in case. 

401 logger.warning( 

402 f"Document with _id {document_id} not found " 

403 f"in {index} index. No update performed." 

404 ) 

405 continue 

406 

407 @classmethod 

408 def _bulk_perform_index_update( 

409 cls, 

410 method: str, 

411 document_ids: list[str], 

412 **es_method_args, 

413 ): 

414 """ 

415 Call ``method`` on the Elasticsearch client in a bulk operation. 

416 

417 Automatically handles 404 errors for documents, forces a refresh, 

418 and calls the method for origin and filtered indexes. 

419 

420 Unlike the single-document behaviour, this function does not 

421 provide validation to check if the media objects exist. 

422 """ 

423 

424 es: Elasticsearch = settings.ES 

425 

426 actions = [ 

427 { 

428 "_op_type": method, 

429 "_index": index, 

430 "_id": document_id, 

431 **es_method_args, 

432 } 

433 for index in cls.indexes() 

434 for document_id in document_ids 

435 ] 

436 

437 # Perform all actions in bulk, while allowing for missing 

438 # documents, similar to the single-document behaviour. In all 

439 # other cases, this raises ``BulkIndexError``. 

440 helpers.bulk(es, actions, ignore_status=(404,)) 

441 es.indices.refresh(index=cls.indexes()) 

442 

443 

444class AbstractDeletedMedia(PerformIndexUpdateMixin, OpenLedgerModel): 

445 """ 

446 Generic model from which to inherit all deleted media classes. 

447 

448 'Deleted' here refers to media which has been deleted at the source or intentionally 

449 de-indexed by us. Unlike sensitive reports, this action is irreversible. Subclasses 

450 must populate ``media_class`` and ``es_index`` fields. 

451 """ 

452 

453 media_class: type[models.Model] = None 

454 """the model class associated with this media type e.g. ``Image`` or ``Audio``""" 

455 es_index: str = None 

456 """the name of the ES index from ``settings.MEDIA_INDEX_MAPPING``""" 

457 

458 media_obj = models.OneToOneField( 

459 to="AbstractMedia", 

460 to_field="identifier", 

461 on_delete=models.DO_NOTHING, 

462 primary_key=True, 

463 db_constraint=False, 

464 db_column="identifier", 

465 related_name="deleted_abstract_media", 

466 help_text="The reference to the deleted media.", 

467 ) 

468 """ 

469 Sub-classes must override this field to point to a concrete sub-class of 

470 ``AbstractMedia``. 

471 

472 Note that unlike ``AbstractSensitiveMedia``, this does not provide 

473 a ``delete()`` method to undo the effects of ``save()``. Deindexed 

474 media can only be restored through a data refresh. 

475 """ 

476 

477 class Meta: 

478 abstract = True 

479 

480 def save(self, *args, **kwargs): 

481 super().save(*args, **kwargs) 

482 self.perform_action() 

483 

484 def _update_es(self, raise_errors: bool): 

485 self._perform_index_update( 

486 "delete", 

487 raise_errors, 

488 ) 

489 

490 def perform_action(self): 

491 self._update_es(True) 

492 self.media_obj.delete() # remove the actual model instance 

493 

494 @classmethod 

495 def _bulk_update_es(cls, media_item_ids: list[str]): 

496 cls._bulk_perform_index_update( 

497 "delete", 

498 media_item_ids, 

499 ) 

500 

501 @classmethod 

502 def bulk_perform_action(cls, media_items: list[type[AbstractMedia]]): 

503 cls._bulk_update_es(media_items.values_list("id", flat=True)) 

504 media_items.delete() # remove the actual model instances 

505 

506 

507class AbstractSensitiveMedia(PerformIndexUpdateMixin, models.Model): 

508 """ 

509 Generic model from which to inherit all sensitive media classes. 

510 

511 Subclasses must populate ``media_class`` and ``es_index`` fields. 

512 """ 

513 

514 media_class: type[models.Model] = None 

515 """the model class associated with this media type e.g. ``Image`` or ``Audio``""" 

516 es_index: str = None 

517 """the name of the ES index from ``settings.MEDIA_INDEX_MAPPING``""" 

518 

519 created_on = models.DateTimeField(auto_now_add=True) 

520 

521 media_obj = models.OneToOneField( 

522 to="AbstractMedia", 

523 to_field="identifier", 

524 on_delete=models.DO_NOTHING, 

525 primary_key=True, 

526 db_constraint=False, 

527 db_column="identifier", 

528 related_name="sensitive_abstract_media", 

529 help_text="The reference to the sensitive media.", 

530 ) 

531 """ 

532 Sub-classes must override this field to point to a concrete sub-class of 

533 ``AbstractMedia``. 

534 """ 

535 

536 class Meta: 

537 abstract = True 

538 

539 def save(self, *args, **kwargs): 

540 self._update_es(True, True) 

541 super().save(*args, **kwargs) 

542 

543 def delete(self, *args, **kwargs): 

544 self._update_es(False, False) 

545 super().delete(*args, **kwargs) 

546 

547 def _update_es(self, is_mature: bool, raise_errors: bool): 

548 """ 

549 Update the Elasticsearch document associated with the given model. 

550 

551 :param is_mature: whether to mark the media item as mature 

552 :param raise_errors: whether to raise an error if the no media item is found 

553 """ 

554 self._perform_index_update( 

555 "update", 

556 raise_errors, 

557 doc={"mature": is_mature}, 

558 ) 

559 

560 @classmethod 

561 def _bulk_update_es(cls, is_mature: bool, media_item_ids: list[str]): 

562 cls._bulk_perform_index_update( 

563 "update", 

564 media_item_ids, 

565 doc={"mature": is_mature}, 

566 ) 

567 

568 @classmethod 

569 def bulk_perform_action( 

570 cls, 

571 is_mature: bool, 

572 media_items: list[type[AbstractMedia]], 

573 ): 

574 cls._bulk_update_es(is_mature, media_items.values_list("id", flat=True)) 

575 

576 

577class AbstractMediaList(OpenLedgerModel): 

578 """ 

579 Generic model from which to inherit media lists. 

580 

581 Each subclass should define its own `ManyToManyField` to point to a subclass of 

582 `AbstractMedia`. 

583 """ 

584 

585 title = models.CharField(max_length=2000, help_text="Display name") 

586 slug = models.CharField( 

587 max_length=200, 

588 help_text="A unique identifier used to make a friendly URL for " 

589 "downstream API consumers.", 

590 unique=True, 

591 db_index=True, 

592 ) 

593 auth = models.CharField( 

594 max_length=64, 

595 help_text="A randomly generated string assigned upon list creation. " 

596 "Used to authenticate updates and deletions.", 

597 ) 

598 

599 class Meta: 

600 abstract = True 

601 

602 

603class AbstractAltFile: 

604 """ 

605 This is not a Django model. 

606 

607 This Python class serves as the schema for an alternative file. An alt file 

608 provides alternative qualities, formats and resolutions that are available 

609 from the provider that are not canonical. 

610 

611 The schema of the class must correspond to that of the 

612 :py:class:`api.models.mixins.FileMixin` class. 

613 """ 

614 

615 def __init__(self, attrs): 

616 self.url = attrs.get("url") 

617 self.filesize = attrs.get("filesize") 

618 self.filetype = attrs.get("filetype") 

619 

620 @property 

621 def size_in_mib(self): # ~ MiB or mibibytes 

622 return self.filesize / 2**20 

623 

624 @property 

625 def size_in_mb(self): # ~ MB or megabytes 

626 return self.filesize / 1e6 

627 

628 @property 

629 def mime_type(self): 

630 """ 

631 Get the MIME type of the file inferred from the extension of the file. 

632 

633 :return: the inferred MIME type of the file 

634 """ 

635 

636 return mimetypes.types_map[f".{self.filetype}"]