Coverage for api/models/media.py: 66%
200 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 06:14 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 06:14 +0000
1import mimetypes
2from textwrap import dedent
4from django.conf import settings
5from django.core.exceptions import ValidationError
6from django.db import models
7from django.urls import reverse
8from django.utils.html import format_html
10import structlog
11from elasticsearch import Elasticsearch, NotFoundError, helpers
12from openverse_attribution.license import License
14from api.constants.moderation import DecisionAction
15from api.models.base import OpenLedgerModel
16from api.models.mixins import ForeignIdentifierMixin, IdentifierMixin, MediaMixin
19MATURE = "mature"
20DMCA = "dmca"
21OTHER = "other"
23logger = structlog.get_logger(__name__)
26class AbstractMedia(
27 IdentifierMixin, ForeignIdentifierMixin, MediaMixin, OpenLedgerModel
28):
29 """
30 Generic model from which to inherit all media classes.
32 This class stores information common to all media types indexed by Openverse.
34 Properties
35 ==========
36 id: >-
37 This is Django's automatic primary key, used for models that do not
38 define one explicitly.
39 """
41 watermarked = models.BooleanField(
42 blank=True,
43 null=True,
44 help_text="Whether the media contains a watermark. Not currently leveraged.",
45 )
47 license = models.CharField(
48 max_length=50,
49 help_text="The name of license for the media.",
50 )
51 license_version = models.CharField(
52 max_length=25,
53 blank=True,
54 null=True,
55 help_text="The version of the media license.",
56 )
58 source = models.CharField(
59 max_length=80,
60 blank=True,
61 null=True,
62 db_index=True,
63 help_text="The source of the data, meaning a particular dataset. "
64 "Source and provider can be different. Eg: the Google Open "
65 "Images dataset is source=openimages, but provider=flickr.",
66 )
67 last_synced_with_source = models.DateTimeField(
68 blank=True,
69 null=True,
70 db_index=True,
71 help_text="The date the media was last updated from the upstream source.",
72 )
73 removed_from_source = models.BooleanField(
74 default=False,
75 help_text="Whether the media has been removed from the upstream source.",
76 )
78 view_count = models.IntegerField(
79 blank=True, null=True, default=0, help_text="Vestigial field, purpose unknown."
80 )
82 tags = models.JSONField(
83 blank=True,
84 null=True,
85 help_text=dedent("""
86 JSON array of objects containing tags for the media. Each tag object
87 is expected to have:
89 - `name`: The tag itself (e.g. "dog")
90 - `provider`: The source of the tag
91 - `accuracy`: If the tag was added using a machine-labeler, the confidence
92 for that label expressed as a value between 0 and 1.
94 Note that only `name` and `accuracy` are presently surfaced in API results.
95 """),
96 )
98 category = models.CharField(
99 max_length=80,
100 blank=True,
101 null=True,
102 db_index=True,
103 help_text="The top-level classification of this media file.",
104 )
106 meta_data = models.JSONField(
107 blank=True,
108 null=True,
109 help_text=dedent("""
110 JSON object containing extra data about the media item. No fields are expected,
111 but if the `license_url` field is available, it will be used for determining
112 the license URL for the media item. The `description` field, if available, is
113 also indexed into Elasticsearch and as a search field on queries.
114 """),
115 )
117 @property
118 def license_url(self) -> str | None:
119 """A direct link to the license deed or legal terms."""
121 if self.meta_data and (url := self.meta_data.get("license_url")): 121 ↛ 124line 121 didn't jump to line 124 because the condition on line 121 was always true
122 return url
124 logger.warning(
125 "Media item missing `license_url` in `meta_data`",
126 media_class=self.__class__.__name__,
127 identifier=self.identifier,
128 )
129 try:
130 return License(self.license.lower(), self.license_version).url
131 except ValueError:
132 return None
134 @property
135 def attribution(self) -> str | None:
136 """Legally valid attribution for the media item in plain-text English."""
138 try:
139 return License(
140 self.license.lower(),
141 self.license_version,
142 ).get_attribution_text(
143 self.title,
144 self.creator,
145 self.license_url,
146 )
147 except ValueError:
148 return None
150 class Meta:
151 """
152 Meta class for all media types indexed by Openverse.
154 All concrete media classes should inherit their Meta class from this.
155 """
157 ordering = ["-created_on"]
158 abstract = True
159 constraints = [
160 models.UniqueConstraint(
161 fields=["foreign_identifier", "provider"],
162 name="unique_provider_%(class)s", # populated by concrete model
163 ),
164 ]
166 def __str__(self):
167 """
168 Return the string representation of the model, used in the Django admin site.
170 :return: the string representation of the model
171 """
173 return f"{self.__class__.__name__}: {self.identifier}"
176class AbstractMediaReport(models.Model):
177 """
178 Generic model from which to inherit all reported media classes.
180 'Reported' here refers to content reports such as sensitive, copyright-violating or
181 deleted content. Subclasses must populate the field ``media_class``.
182 """
184 media_class: type[models.Model] = None
185 """the model class associated with this media type e.g. ``Image`` or ``Audio``"""
187 REPORT_CHOICES = [(MATURE, MATURE), (DMCA, DMCA), (OTHER, OTHER)]
189 created_at = models.DateTimeField(auto_now_add=True)
191 media_obj = models.ForeignKey(
192 to="AbstractMedia",
193 to_field="identifier",
194 on_delete=models.DO_NOTHING,
195 db_constraint=False,
196 db_column="identifier",
197 related_name="abstract_media_report",
198 help_text="The reference to the media being reported.",
199 )
200 """
201 There can be many reports associated with a single media item, hence foreign key.
202 Sub-classes must override this field to point to a concrete sub-class of
203 ``AbstractMedia``.
204 """
206 reason = models.CharField(
207 max_length=20,
208 choices=REPORT_CHOICES,
209 help_text="The reason to report media to Openverse.",
210 )
211 description = models.TextField(
212 max_length=500,
213 blank=True,
214 null=True,
215 help_text="The explanation on why media is being reported.",
216 )
217 decision = models.ForeignKey(
218 to="AbstractMediaDecision",
219 on_delete=models.SET_NULL,
220 blank=True,
221 null=True,
222 help_text="The moderation decision for this report.",
223 )
225 class Meta:
226 abstract = True
228 def clean(self):
229 """Clean fields and raise errors that can be handled by Django Admin."""
231 if not self.media_class.objects.filter(identifier=self.media_obj_id).exists(): 231 ↛ 232line 231 didn't jump to line 232 because the condition on line 231 was never true
232 raise ValidationError(
233 f"No '{self.media_class.__name__}' instance "
234 f"with identifier '{self.media_obj_id}'."
235 )
237 def media_url(self, request=None) -> str:
238 """
239 Build the URL of the media item. This uses ``reverse`` and
240 ``request.build_absolute_uri`` to build the URL without having to worry
241 about canonical URL or trailing slashes.
243 :param request: the current request object, to get absolute URLs
244 :return: the URL of the media item
245 """
247 url = reverse(
248 f"{self.media_class.__name__.lower()}-detail",
249 args=[self.media_obj_id],
250 )
251 if request is not None:
252 url = request.build_absolute_uri(url)
253 return url
255 def url(self, request=None) -> str:
256 url = self.media_url(request)
257 return format_html(f"<a href={url}>{url}</a>")
259 @property
260 def is_pending(self) -> bool:
261 """
262 Determine if the report has not been moderated and does not have an
263 associated decision. Use the inverse of this function to determine
264 if a report has been reviewed and moderated.
266 :return: whether the report is in the "pending" state
267 """
269 return self.decision_id is None
271 def save(self, *args, **kwargs):
272 """Perform a clean, and then save changes to the DB."""
274 self.clean()
275 super().save(*args, **kwargs)
278class AbstractMediaDecision(OpenLedgerModel):
279 """Generic model from which to inherit all moderation decision classes."""
281 media_class: type[models.Model] = None
282 """the model class associated with this media type e.g. ``Image`` or ``Audio``"""
284 moderator = models.ForeignKey(
285 to="auth.User",
286 on_delete=models.DO_NOTHING,
287 help_text="The moderator who undertook this decision.",
288 )
289 """
290 The ``User`` referenced by this field must be a part of the moderators'
291 group.
292 """
294 media_objs = models.ManyToManyField(
295 to="AbstractMedia",
296 through="AbstractMediaDecisionThrough",
297 help_text="The media items being moderated.",
298 )
299 """
300 This is a many-to-many relation, using a bridge table, to enable bulk
301 moderation which applies a single action to more than one media items.
302 """
304 notes = models.TextField(
305 max_length=500,
306 blank=True,
307 null=True,
308 help_text="The moderator's explanation for the decision or additional notes.",
309 )
311 action = models.CharField(
312 max_length=32,
313 choices=DecisionAction.choices,
314 help_text="Action taken by the moderator.",
315 )
317 class Meta:
318 abstract = True
321class AbstractMediaDecisionThrough(models.Model):
322 """
323 Generic model for the many-to-many reference table between media and decisions.
325 This is made explicit (rather than using Django's default) so that the media can
326 be referenced by `identifier` rather than an arbitrary `id`.
327 """
329 media_class: type[models.Model] = None
330 """the model class associated with this media type e.g. ``Image`` or ``Audio``"""
331 sensitive_media_class: type[models.Model] = None
332 """the model class associated with this media type e.g. ``SensitiveImage`` or ``SensitiveAudio``"""
333 deleted_media_class: type[models.Model] = None
334 """the model class associated with this media type e.g. ``DeletedImage`` or ``DeletedAudio``"""
336 media_obj = models.ForeignKey(
337 AbstractMedia,
338 to_field="identifier",
339 on_delete=models.DO_NOTHING,
340 db_column="identifier",
341 db_constraint=False,
342 )
343 decision = models.ForeignKey(AbstractMediaDecision, on_delete=models.CASCADE)
345 class Meta:
346 abstract = True
348 def perform_action(self, action=None):
349 """Perform the action specified in the decision."""
351 action = self.decision.action if action is None else action
353 if action in {
354 DecisionAction.DEINDEXED_SENSITIVE,
355 DecisionAction.DEINDEXED_COPYRIGHT,
356 }:
357 self.deleted_media_class.objects.create(media_obj_id=self.media_obj_id)
359 if action == DecisionAction.MARKED_SENSITIVE:
360 self.sensitive_media_class.objects.create(media_obj_id=self.media_obj_id)
362 def save(self, *args, **kwargs):
363 super().save(*args, **kwargs)
364 self.perform_action()
367class PerformIndexUpdateMixin:
368 @classmethod
369 def indexes(cls):
370 return [cls.es_index, f"{cls.es_index}-filtered"]
372 def _perform_index_update(self, method: str, raise_errors: bool, **es_method_args):
373 """
374 Call ``method`` on the Elasticsearch client.
376 Automatically handles ``DoesNotExist`` warnings, forces a refresh,
377 and calls the method for origin and filtered indexes.
378 """
379 es: Elasticsearch = settings.ES
381 try:
382 document_id = self.media_obj.id
383 except self.media_class.DoesNotExist:
384 if raise_errors:
385 raise ValidationError(
386 f"No '{self.media_class.__name__}' instance "
387 f"with identifier {self.media_obj.identifier}."
388 )
390 for index in self.indexes():
391 try:
392 getattr(es, method)(
393 index=index,
394 id=document_id,
395 refresh=True,
396 **es_method_args,
397 )
398 except NotFoundError:
399 # This is expected for the filtered index, but we should still
400 # log, just in case.
401 logger.warning(
402 f"Document with _id {document_id} not found "
403 f"in {index} index. No update performed."
404 )
405 continue
407 @classmethod
408 def _bulk_perform_index_update(
409 cls,
410 method: str,
411 document_ids: list[str],
412 **es_method_args,
413 ):
414 """
415 Call ``method`` on the Elasticsearch client in a bulk operation.
417 Automatically handles 404 errors for documents, forces a refresh,
418 and calls the method for origin and filtered indexes.
420 Unlike the single-document behaviour, this function does not
421 provide validation to check if the media objects exist.
422 """
424 es: Elasticsearch = settings.ES
426 actions = [
427 {
428 "_op_type": method,
429 "_index": index,
430 "_id": document_id,
431 **es_method_args,
432 }
433 for index in cls.indexes()
434 for document_id in document_ids
435 ]
437 # Perform all actions in bulk, while allowing for missing
438 # documents, similar to the single-document behaviour. In all
439 # other cases, this raises ``BulkIndexError``.
440 helpers.bulk(es, actions, ignore_status=(404,))
441 es.indices.refresh(index=cls.indexes())
444class AbstractDeletedMedia(PerformIndexUpdateMixin, OpenLedgerModel):
445 """
446 Generic model from which to inherit all deleted media classes.
448 'Deleted' here refers to media which has been deleted at the source or intentionally
449 de-indexed by us. Unlike sensitive reports, this action is irreversible. Subclasses
450 must populate ``media_class`` and ``es_index`` fields.
451 """
453 media_class: type[models.Model] = None
454 """the model class associated with this media type e.g. ``Image`` or ``Audio``"""
455 es_index: str = None
456 """the name of the ES index from ``settings.MEDIA_INDEX_MAPPING``"""
458 media_obj = models.OneToOneField(
459 to="AbstractMedia",
460 to_field="identifier",
461 on_delete=models.DO_NOTHING,
462 primary_key=True,
463 db_constraint=False,
464 db_column="identifier",
465 related_name="deleted_abstract_media",
466 help_text="The reference to the deleted media.",
467 )
468 """
469 Sub-classes must override this field to point to a concrete sub-class of
470 ``AbstractMedia``.
472 Note that unlike ``AbstractSensitiveMedia``, this does not provide
473 a ``delete()`` method to undo the effects of ``save()``. Deindexed
474 media can only be restored through a data refresh.
475 """
477 class Meta:
478 abstract = True
480 def save(self, *args, **kwargs):
481 super().save(*args, **kwargs)
482 self.perform_action()
484 def _update_es(self, raise_errors: bool):
485 self._perform_index_update(
486 "delete",
487 raise_errors,
488 )
490 def perform_action(self):
491 self._update_es(True)
492 self.media_obj.delete() # remove the actual model instance
494 @classmethod
495 def _bulk_update_es(cls, media_item_ids: list[str]):
496 cls._bulk_perform_index_update(
497 "delete",
498 media_item_ids,
499 )
501 @classmethod
502 def bulk_perform_action(cls, media_items: list[type[AbstractMedia]]):
503 cls._bulk_update_es(media_items.values_list("id", flat=True))
504 media_items.delete() # remove the actual model instances
507class AbstractSensitiveMedia(PerformIndexUpdateMixin, models.Model):
508 """
509 Generic model from which to inherit all sensitive media classes.
511 Subclasses must populate ``media_class`` and ``es_index`` fields.
512 """
514 media_class: type[models.Model] = None
515 """the model class associated with this media type e.g. ``Image`` or ``Audio``"""
516 es_index: str = None
517 """the name of the ES index from ``settings.MEDIA_INDEX_MAPPING``"""
519 created_on = models.DateTimeField(auto_now_add=True)
521 media_obj = models.OneToOneField(
522 to="AbstractMedia",
523 to_field="identifier",
524 on_delete=models.DO_NOTHING,
525 primary_key=True,
526 db_constraint=False,
527 db_column="identifier",
528 related_name="sensitive_abstract_media",
529 help_text="The reference to the sensitive media.",
530 )
531 """
532 Sub-classes must override this field to point to a concrete sub-class of
533 ``AbstractMedia``.
534 """
536 class Meta:
537 abstract = True
539 def save(self, *args, **kwargs):
540 self._update_es(True, True)
541 super().save(*args, **kwargs)
543 def delete(self, *args, **kwargs):
544 self._update_es(False, False)
545 super().delete(*args, **kwargs)
547 def _update_es(self, is_mature: bool, raise_errors: bool):
548 """
549 Update the Elasticsearch document associated with the given model.
551 :param is_mature: whether to mark the media item as mature
552 :param raise_errors: whether to raise an error if the no media item is found
553 """
554 self._perform_index_update(
555 "update",
556 raise_errors,
557 doc={"mature": is_mature},
558 )
560 @classmethod
561 def _bulk_update_es(cls, is_mature: bool, media_item_ids: list[str]):
562 cls._bulk_perform_index_update(
563 "update",
564 media_item_ids,
565 doc={"mature": is_mature},
566 )
568 @classmethod
569 def bulk_perform_action(
570 cls,
571 is_mature: bool,
572 media_items: list[type[AbstractMedia]],
573 ):
574 cls._bulk_update_es(is_mature, media_items.values_list("id", flat=True))
577class AbstractMediaList(OpenLedgerModel):
578 """
579 Generic model from which to inherit media lists.
581 Each subclass should define its own `ManyToManyField` to point to a subclass of
582 `AbstractMedia`.
583 """
585 title = models.CharField(max_length=2000, help_text="Display name")
586 slug = models.CharField(
587 max_length=200,
588 help_text="A unique identifier used to make a friendly URL for "
589 "downstream API consumers.",
590 unique=True,
591 db_index=True,
592 )
593 auth = models.CharField(
594 max_length=64,
595 help_text="A randomly generated string assigned upon list creation. "
596 "Used to authenticate updates and deletions.",
597 )
599 class Meta:
600 abstract = True
603class AbstractAltFile:
604 """
605 This is not a Django model.
607 This Python class serves as the schema for an alternative file. An alt file
608 provides alternative qualities, formats and resolutions that are available
609 from the provider that are not canonical.
611 The schema of the class must correspond to that of the
612 :py:class:`api.models.mixins.FileMixin` class.
613 """
615 def __init__(self, attrs):
616 self.url = attrs.get("url")
617 self.filesize = attrs.get("filesize")
618 self.filetype = attrs.get("filetype")
620 @property
621 def size_in_mib(self): # ~ MiB or mibibytes
622 return self.filesize / 2**20
624 @property
625 def size_in_mb(self): # ~ MB or megabytes
626 return self.filesize / 1e6
628 @property
629 def mime_type(self):
630 """
631 Get the MIME type of the file inferred from the extension of the file.
633 :return: the inferred MIME type of the file
634 """
636 return mimetypes.types_map[f".{self.filetype}"]