Coverage for api/serializers/media_serializers.py: 88%

274 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 06:14 +0000

1from collections import namedtuple 

2from math import floor 

3from typing import TypedDict 

4 

5from django.conf import settings 

6from django.core.exceptions import ValidationError as DjangoValidationError 

7from django.core.validators import MaxValueValidator 

8from django.urls import reverse 

9from rest_framework import serializers 

10from rest_framework.exceptions import NotAuthenticated, ValidationError 

11from rest_framework.request import Request 

12 

13from adrf.serializers import Serializer 

14from drf_spectacular.utils import extend_schema_serializer 

15from elasticsearch_dsl.response import Hit 

16from openverse_attribution.license import License 

17 

18from api.constants import restricted_features, sensitivity 

19from api.constants.licenses import LICENSE_GROUPS 

20from api.constants.media_types import MediaType 

21from api.constants.parameters import COLLECTION, TAG 

22from api.constants.search import COLLECTIONS 

23from api.constants.sorting import DESCENDING, RELEVANCE, SORT_DIRECTIONS, SORT_FIELDS 

24from api.controllers import search_controller 

25from api.models.media import AbstractMedia 

26from api.serializers.base import BaseModelSerializer 

27from api.serializers.docs import ( 

28 COLLECTION_HELP_TEXT, 

29 CREATOR_HELP_TEXT, 

30 EXCLUDED_SOURCE_HELP_TEXT, 

31 SOURCE_HELP_TEXT, 

32 TAG_HELP_TEXT, 

33 UNSTABLE_WARNING, 

34) 

35from api.serializers.fields import SchemableHyperlinkedIdentityField 

36from api.utils.help_text import make_comma_separated_help_text 

37from api.utils.url import add_protocol 

38 

39 

40####################### 

41# Request serializers # 

42####################### 

43 

44 

45class PaginatedRequestSerializer(Serializer): 

46 """This serializer passes pagination parameters from the query string.""" 

47 

48 _SUBJECT_TO_PAGINATION_LIMITS = ( 

49 "This parameter is subject to limitations based on authentication " 

50 "and access level. For details, refer to [the authentication " 

51 "documentation](#tag/auth)." 

52 ) 

53 

54 field_names = [ 

55 "page_size", 

56 "page", 

57 ] 

58 page_size = serializers.IntegerField( 

59 label="page_size", 

60 help_text=f"Number of results to return per page. {_SUBJECT_TO_PAGINATION_LIMITS}", 

61 required=False, 

62 default=restricted_features.MAX_PAGE_SIZE.anonymous, 

63 min_value=1, 

64 ) 

65 page = serializers.IntegerField( 

66 label="page", 

67 help_text=f"The page of results to retrieve. {_SUBJECT_TO_PAGINATION_LIMITS}", 

68 required=False, 

69 default=1, 

70 min_value=1, 

71 ) 

72 

73 def validate_page_size(self, value): 

74 level, max_value = restricted_features.MAX_PAGE_SIZE.request_level( 

75 self.context.get("request") 

76 ) 

77 

78 validator = MaxValueValidator( 

79 max_value, 

80 message=serializers.IntegerField.default_error_messages["max_value"].format( 

81 max_value=max_value 

82 ), 

83 ) 

84 

85 try: 

86 validator(value) 

87 except (ValidationError, DjangoValidationError) as e: 

88 if level == restricted_features.PRIVILEGED: 88 ↛ 89line 88 didn't jump to line 89 because the condition on line 88 was never true

89 raise 

90 

91 raise NotAuthenticated( 

92 detail=f"page_size may not exceed {max_value} for {level} requests", 

93 code=e.code, 

94 ) 

95 

96 return value 

97 

98 def clamp_result_count(self, real_result_count): 

99 _, max_depth = restricted_features.MAX_RESULT_COUNT.request_level( 

100 self.context.get("request") 

101 ) 

102 

103 if real_result_count > max_depth: 

104 return max_depth 

105 

106 return real_result_count 

107 

108 def clamp_page_count(self, real_page_count): 

109 _, max_depth = restricted_features.MAX_RESULT_COUNT.request_level( 

110 self.context.get("request") 

111 ) 

112 

113 page_size = self.data["page_size"] 

114 max_possible_page_count = max_depth / page_size 

115 

116 if real_page_count > max_possible_page_count: 

117 return floor(max_possible_page_count) 

118 

119 return real_page_count 

120 

121 def validate(self, data): 

122 data = super().validate(data) 

123 

124 # pagination depth is validated as a combination of page and page size, 

125 # and so cannot be validated in the individual field validation methods 

126 level, max_depth = restricted_features.MAX_RESULT_COUNT.request_level( 

127 self.context.get("request") 

128 ) 

129 

130 requested_result_depth = data["page"] * data["page_size"] 

131 

132 result_depth_validator = MaxValueValidator( 

133 max_depth, 

134 message=serializers.IntegerField.default_error_messages["max_value"].format( 

135 max_value=max_depth 

136 ), 

137 ) 

138 

139 try: 

140 result_depth_validator(requested_result_depth) 

141 except (ValidationError, DjangoValidationError) as e: 

142 if level == restricted_features.PRIVILEGED: 142 ↛ 143line 142 didn't jump to line 143 because the condition on line 142 was never true

143 raise 

144 

145 raise NotAuthenticated( 

146 detail=f"pagination depth may not exceed {max_depth} for {level} requests", 

147 code=e.code, 

148 ) 

149 

150 return data 

151 

152 

153@extend_schema_serializer( 

154 # Hide internal fields from documentation. 

155 # Also see `field_names` below. 

156 exclude_fields=[ 

157 "internal__index", 

158 ], 

159) 

160class MediaSearchRequestSerializer(PaginatedRequestSerializer): 

161 """This serializer parses and validates search query string parameters.""" 

162 

163 DeprecatedParam = namedtuple("DeprecatedParam", ["original", "successor"]) 

164 deprecated_params = [ 

165 DeprecatedParam("li", "license"), 

166 DeprecatedParam("lt", "license_type"), 

167 DeprecatedParam("pagesize", "page_size"), 

168 DeprecatedParam("provider", "source"), 

169 ] 

170 field_names = [ 

171 "q", 

172 "source", 

173 "excluded_source", 

174 "license", 

175 "license_type", 

176 "creator", 

177 "tags", 

178 COLLECTION, 

179 TAG, 

180 "title", 

181 "filter_dead", 

182 "extension", 

183 "mature", 

184 "unstable__sort_by", 

185 "unstable__sort_dir", 

186 "unstable__authority", 

187 "unstable__authority_boost", 

188 "unstable__include_sensitive_results", 

189 ] 

190 field_names.extend(PaginatedRequestSerializer.field_names) 

191 """ 

192 Keep the fields names in sync with the actual fields below as this list is 

193 used to generate Swagger documentation. 

194 """ 

195 

196 q = serializers.CharField( 

197 label="query", 

198 help_text="A query string that should not exceed 200 characters in length", 

199 required=False, 

200 ) 

201 source = serializers.CharField( 

202 label="provider", 

203 required=False, 

204 ) 

205 excluded_source = serializers.CharField( 

206 label="excluded_provider", 

207 required=False, 

208 ) 

209 tags = serializers.CharField( 

210 label="tags", 

211 help_text="Search by tag only. Cannot be used with `q`. The search " 

212 "is fuzzy, so `tags=cat` will match any value that includes the word " 

213 "`cat`. If the value contains space, items that contain any of the " 

214 "words in the value will match. To search for several values, join " 

215 "them with a comma.", 

216 required=False, 

217 max_length=200, 

218 ) 

219 title = serializers.CharField( 

220 label="title", 

221 help_text="Search by title only. Cannot be used with `q`. The search is fuzzy," 

222 " so `title=photo` will match any value that includes the word `photo`. " 

223 "If the value contains space, items that contain any of the words in the " 

224 "value will match. To search for several values, join them with a comma.", 

225 required=False, 

226 max_length=200, 

227 ) 

228 creator = serializers.CharField( 

229 label="creator", 

230 help_text=CREATOR_HELP_TEXT, 

231 required=False, 

232 max_length=200, 

233 ) 

234 unstable__collection = serializers.ChoiceField( 

235 source="collection", 

236 label="collection", 

237 choices=COLLECTIONS, 

238 help_text=COLLECTION_HELP_TEXT, 

239 required=False, 

240 ) 

241 unstable__tag = serializers.CharField( 

242 label="tag", 

243 source="tag", 

244 help_text=TAG_HELP_TEXT, 

245 required=False, 

246 max_length=200, 

247 ) 

248 license = serializers.CharField( 

249 label="licenses", 

250 help_text=make_comma_separated_help_text(LICENSE_GROUPS["all"], "licenses"), 

251 required=False, 

252 ) 

253 license_type = serializers.CharField( 

254 label="license type", 

255 help_text=make_comma_separated_help_text( 

256 LICENSE_GROUPS.keys(), "license types" 

257 ), 

258 required=False, 

259 ) 

260 filter_dead = serializers.BooleanField( 

261 label="filter_dead", 

262 help_text="Control whether 404 links are filtered out.", 

263 required=False, 

264 default=settings.FILTER_DEAD_LINKS_BY_DEFAULT, 

265 ) 

266 extension = serializers.CharField( 

267 label="extension", 

268 help_text="A comma separated list of desired file extensions.", 

269 required=False, 

270 ) 

271 mature = serializers.BooleanField( 

272 label="mature", 

273 default=False, 

274 required=False, 

275 help_text="Whether to include sensitive content.", 

276 ) 

277 

278 # The ``unstable__`` prefix is used in the query params. 

279 # The validated data does not contain the ``unstable__`` prefix. 

280 # If you rename these fields, update the following references: 

281 # - ``field_names`` in ``MediaSearchRequestSerializer`` 

282 # - validators for these fields in ``MediaSearchRequestSerializer`` 

283 unstable__sort_by = serializers.ChoiceField( 

284 source="sort_by", 

285 help_text=f"{UNSTABLE_WARNING}The field which should be the basis for sorting results.", 

286 choices=SORT_FIELDS, 

287 required=False, 

288 default=RELEVANCE, 

289 ) 

290 unstable__sort_dir = serializers.ChoiceField( 

291 source="sort_dir", 

292 help_text=f"{UNSTABLE_WARNING}The direction of sorting. Cannot be applied when sorting by " 

293 "`relevance`.", 

294 choices=SORT_DIRECTIONS, 

295 required=False, 

296 default=DESCENDING, 

297 ) 

298 unstable__authority = serializers.BooleanField( 

299 label="authority", 

300 help_text=f"{UNSTABLE_WARNING}If enabled, the search will add a boost to results that are " 

301 "from authoritative sources.", 

302 required=False, 

303 default=False, 

304 ) 

305 unstable__authority_boost = serializers.FloatField( 

306 label="authority_boost", 

307 help_text=f"{UNSTABLE_WARNING}The boost coefficient to apply to authoritative sources, " 

308 "multiplied with the popularity boost.", 

309 required=False, 

310 default=1.0, 

311 min_value=0.0, 

312 max_value=10.0, 

313 ) 

314 unstable__include_sensitive_results = serializers.BooleanField( 

315 source="include_sensitive_results", 

316 label="include_sensitive_results", 

317 help_text=f"{UNSTABLE_WARNING}Whether to include results considered sensitive.", 

318 required=False, 

319 default=False, 

320 ) 

321 

322 # The ``internal__`` prefix is used in the query params. 

323 # If you rename these fields, update the following references: 

324 # - ``field_names`` in ``MediaSearchRequestSerializer`` 

325 # - validators for these fields in ``MediaSearchRequestSerializer`` 

326 internal__index = serializers.CharField( 

327 source="index", 

328 help_text="The index against which to perform the search.", 

329 required=False, 

330 ) 

331 

332 class Context(TypedDict, total=True): 

333 warnings: list[dict] 

334 media_type: MediaType 

335 request: Request 

336 

337 context: Context 

338 

339 def __init__(self, *args, **kwargs): 

340 super().__init__(*args, **kwargs) 

341 self.context["warnings"] = [] 

342 self.media_type = self.context.get("media_type") 

343 if not self.media_type: 343 ↛ 344line 343 didn't jump to line 344 because the condition on line 343 was never true

344 raise ValueError( 

345 "The media request serializer's `media_type` context variable must be set." 

346 ) 

347 media_path = {"image": "images", "audio": "audio"}[self.media_type] 

348 variables = { 

349 "origin": settings.CANONICAL_ORIGIN, 

350 "media_path": media_path, 

351 "collection_param": COLLECTION, 

352 } 

353 

354 self.fields["source"].help_text = SOURCE_HELP_TEXT.format(**variables) 

355 self.fields["excluded_source"].help_text = EXCLUDED_SOURCE_HELP_TEXT.format( 

356 **variables 

357 ) 

358 

359 def is_request_anonymous(self): 

360 request = self.context.get("request") 

361 return getattr(request, "auth", None) is None 

362 

363 @staticmethod 

364 def _truncate(value): 

365 max_length = 200 

366 return value if len(value) <= max_length else value[:max_length] 

367 

368 def validate_q(self, value): 

369 return self._truncate(value) 

370 

371 @staticmethod 

372 def validate_license(value): 

373 """Check whether license is a valid license code.""" 

374 

375 licenses = value.lower().split(",") 

376 for _license in licenses: 

377 if _license not in LICENSE_GROUPS["all"]: 

378 raise serializers.ValidationError( 

379 f"License '{_license}' does not exist." 

380 ) 

381 # lowers the case of the value before returning 

382 return value.lower() 

383 

384 @staticmethod 

385 def validate_license_type(value): 

386 """Check whether license type is a known collection of licenses.""" 

387 

388 license_types = value.lower().split(",") 

389 license_groups = [] 

390 for _type in license_types: 390 ↛ 396line 390 didn't jump to line 396 because the loop on line 390 didn't complete

391 if _type not in LICENSE_GROUPS: 391 ↛ 395line 391 didn't jump to line 395 because the condition on line 391 was always true

392 raise serializers.ValidationError( 

393 f"License type '{_type}' does not exist." 

394 ) 

395 license_groups.append(LICENSE_GROUPS[_type]) 

396 intersected = set.intersection(*license_groups) 

397 return ",".join(intersected) 

398 

399 def validate_unstable__collection(self, value): 

400 if self.initial_data.get("q", None) is not None: 

401 raise serializers.ValidationError( 

402 "The `collection` parameter cannot be used with the `q` parameter." 

403 ) 

404 if value == "tag" and not self.initial_data.get(TAG): 

405 raise serializers.ValidationError( 

406 f"The `{TAG}` parameter is required when `{COLLECTION}` is set to `tag`." 

407 ) 

408 if value == "source" and not self.initial_data.get("source"): 

409 raise serializers.ValidationError( 

410 f"The `source` parameter is required when `{COLLECTION}` is set to `source`." 

411 ) 

412 if value == "creator" and not ( 

413 self.initial_data.get("creator") and self.initial_data.get("source") 

414 ): 

415 raise serializers.ValidationError( 

416 f"The `creator` and `source` parameters are required when `{COLLECTION}` is set to `creator`." 

417 ) 

418 return value 

419 

420 def validate_source(self, value): 

421 """ 

422 For the regular searches, split the value.lower() by comma and only return the 

423 source names that are in the search controller's sources list for the media type. 

424 For the collection=tag, return the value as is. It is ignored in the query builder. 

425 For source and creator collections, accept the value as is, without lower-casing 

426 or splitting, and check if it's in the valid source name list. 

427 This function validates the source and excluded_source fields, but `excluded_source` 

428 is ignored for collection requests. 

429 """ 

430 allowed_sources = list(search_controller.get_sources(self.media_type).keys()) 

431 sources_list = ", ".join([f"'{s}'" for s in allowed_sources]) 

432 collection = self.initial_data.get(COLLECTION) 

433 

434 # For collection=tag, return the value as is. It is ignored in the query builder. 

435 if collection == "tag": 

436 return value 

437 

438 if collection: 

439 if value not in allowed_sources: 

440 raise serializers.ValidationError( 

441 f"Invalid source parameter '{value}'. Use one of the valid sources: {sources_list}" 

442 ) 

443 return value 

444 else: 

445 sources = set(value.lower().split(",")) 

446 valid_sources = {source for source in sources if source in allowed_sources} 

447 if not valid_sources: 

448 # Raise only if there are _no_ valid sources selected 

449 # If the requester passed only `mispelled_museum_name1,mispelled_musesum_name2` 

450 # the request cannot move forward, as all the top responses will likely be from Flickr 

451 # which provides radically different responses than most other providers. 

452 # If even one source is valid, it won't be a problem, in which case we'll issue a warning 

453 raise serializers.ValidationError( 

454 f"Invalid source parameter '{value}'. No valid sources selected. " 

455 f"Refer to the source list for valid options: {sources_list}." 

456 ) 

457 elif invalid_sources := (sources - valid_sources): 457 ↛ 458line 457 didn't jump to line 458 because the condition on line 457 was never true

458 available_sources_uri = self.context.get("request").build_absolute_uri( 

459 reverse(f"{self.media_type}-stats") 

460 ) 

461 self.context["warnings"].append( 

462 { 

463 "code": "partially invalid source parameter", 

464 "message": ( 

465 "The source parameter included non-existent sources. " 

466 f"For a list of available sources, see {available_sources_uri}" 

467 ), 

468 "invalid_sources": invalid_sources, 

469 "valid_sources": valid_sources, 

470 } 

471 ) 

472 

473 return ",".join(valid_sources) 

474 

475 def validate_excluded_source(self, input_sources): 

476 if "source" in self.initial_data: 

477 raise serializers.ValidationError( 

478 "Cannot set both 'source' and 'excluded_source'. " 

479 "Use exactly one of these." 

480 ) 

481 return self.validate_source(input_sources) 

482 

483 def validate_creator(self, value): 

484 return self._truncate(value) 

485 

486 def validate_tags(self, value): 

487 return self._truncate(value) 

488 

489 def validate_title(self, value): 

490 return self._truncate(value) 

491 

492 def validate_unstable__sort_by(self, value): 

493 return RELEVANCE if self.is_request_anonymous() else value 

494 

495 def validate_unstable__sort_dir(self, value): 

496 return DESCENDING if self.is_request_anonymous() else value 

497 

498 def validate_unstable__authority(self, value): 

499 return False if self.is_request_anonymous() else value 

500 

501 def validate_unstable__include_sensitive_results( 

502 self, 

503 value, 

504 ): 

505 exclusive_fields = ("mature", "unstable__include_sensitive_results") 

506 if all(f in self.initial_data for f in exclusive_fields): 

507 raise serializers.ValidationError( 

508 "`mature` and `unstable__include_sensitive_results` " 

509 "must not both be defined." 

510 ) 

511 

512 return self.initial_data.get("mature") or value 

513 

514 def validate_internal__index(self, value): 

515 """ 

516 Check whether the given index name is a valid index or alias. However, 

517 for unauthenticated requests, no check is performed and ``None`` is 

518 returned immediately. 

519 

520 :param value: the provided index name to check 

521 :return: ``None`` if request is anonymous, the provided name if it is valid 

522 :raise: ``serializers.ValidationError`` if not anonymous and invalid index name 

523 """ 

524 

525 if self.is_request_anonymous(): 

526 return None 

527 if not settings.ES.indices.exists(value): # ``exists`` includes aliases. 

528 raise serializers.ValidationError(f"Invalid index name `{value}`.") 

529 

530 if not value.startswith(self.media_type): 

531 raise serializers.ValidationError(f"Invalid index name `{value}`.") 

532 return value 

533 

534 @staticmethod 

535 def validate_extension(value): 

536 return value.lower() 

537 

538 def validate(self, data): 

539 data = super().validate(data) 

540 errors = {} 

541 for param, successor in self.deprecated_params: 

542 if param in self.initial_data: 542 ↛ 543line 542 didn't jump to line 543 because the condition on line 542 was never true

543 errors[param] = ( 

544 f"Parameter '{param}' is deprecated in this release of the API. " 

545 f"Use '{successor}' instead." 

546 ) 

547 if errors: 547 ↛ 548line 547 didn't jump to line 548 because the condition on line 547 was never true

548 raise serializers.ValidationError(errors) 

549 

550 return data 

551 

552 

553class MediaThumbnailRequestSerializer(Serializer): 

554 """This serializer parses and validates thumbnail query string parameters.""" 

555 

556 full_size = serializers.BooleanField( 

557 source="is_full_size", 

558 allow_null=True, 

559 required=False, 

560 default=False, 

561 help_text="whether to render the actual image and not a thumbnail version", 

562 ) 

563 compressed = serializers.BooleanField( 

564 source="is_compressed", 

565 allow_null=True, 

566 default=None, 

567 required=False, 

568 help_text="whether to compress the output image to reduce file size," 

569 "defaults to opposite of `full_size`", 

570 ) 

571 

572 def validate(self, data): 

573 if data.get("is_compressed") is None: 

574 data["is_compressed"] = not data["is_full_size"] 

575 return data 

576 

577 

578class MediaReportRequestSerializer(serializers.ModelSerializer): 

579 class Meta: 

580 model = None 

581 fields = ["identifier", "reason", "description"] 

582 read_only_fields = ["identifier"] 

583 

584 def to_internal_value(self, data): 

585 """ 

586 Map data before validation. 

587 

588 See ``MediaReportRequestSerializer::_map_reason`` docstring for 

589 further explanation. 

590 """ 

591 

592 data["reason"] = self._map_reason(data.get("reason")) 

593 return super().to_internal_value(data) 

594 

595 def validate(self, attrs): 

596 if ( 

597 attrs["reason"] == "other" 

598 and ("description" not in attrs or len(attrs["description"])) < 20 

599 ): 

600 raise serializers.ValidationError( 

601 "Description must be at least be 20 characters long" 

602 ) 

603 

604 return attrs 

605 

606 def _map_reason(self, value): 

607 """ 

608 Map `sensitive` to `mature` for forwards compatibility. 

609 

610 This is an interim implementation until the API is updated 

611 to use the new "sensitive" terminology. 

612 

613 Once the API is updated to use "sensitive" as the designator 

614 rather than the current "mature" term, this function should 

615 be updated to reverse the mapping, that is, map `mature` to 

616 `sensitive`, for backwards compatibility. 

617 

618 Note: This cannot be implemented as a simpler `validate_reason` method 

619 on the serializer because field validation runs _before_ validators 

620 declared on the serializer. This means the choice field's validation 

621 will complain about `reason` set to the incorrect value before we have 

622 a chance to map it to the correct value. 

623 

624 This could be mitigated by adding all values, current, future, and 

625 deprecated, to the model field. However, that requires a migration 

626 each time we make that change, and would send an incorrect message 

627 about our data expectations. It's cleaner and more consistent to map 

628 the data up-front, at serialization time, to prevent any confusion at 

629 the data model level. 

630 """ 

631 

632 return "mature" if value == "sensitive" else value 

633 

634 

635######################## 

636# Response serializers # 

637######################## 

638 

639 

640class TagSerializer(Serializer): 

641 """This output serializer serializes a singular tag.""" 

642 

643 name = serializers.CharField( 

644 help_text="The name of a detailed tag.", 

645 ) 

646 accuracy = serializers.FloatField( 

647 default=None, 

648 allow_null=True, 

649 help_text="The accuracy of a machine-generated tag. Human-generated " 

650 "tags have a null accuracy field.", 

651 ) 

652 unstable__provider = serializers.CharField( 

653 label="provider", 

654 source="provider", 

655 # Provider is present in the database but not in Elasticsearch, so it may 

656 # not always be present during serialization 

657 allow_null=True, 

658 help_text="The source of the tag. When this field matches the provider for the " 

659 "record, the tag originated from the upstream provider. Otherwise, the tag " 

660 "was added with an external machine-generated labeling processes.", 

661 ) 

662 

663 

664@extend_schema_serializer( 

665 exclude_fields=[ 

666 "unstable__sensitivity", 

667 ], 

668) 

669class MediaSerializer(BaseModelSerializer): 

670 """ 

671 This serializer serializes a single media file. 

672 

673 The class should be inherited by all individual media serializers. 

674 """ 

675 

676 class Meta: 

677 model = AbstractMedia 

678 fields = [ 

679 "id", 

680 "indexed_on", 

681 "title", 

682 "foreign_landing_url", 

683 "url", 

684 "creator", 

685 "creator_url", 

686 "license", 

687 "license_version", 

688 "license_url", # property 

689 "provider", 

690 "source", 

691 "category", 

692 "filesize", 

693 "filetype", 

694 "tags", 

695 "attribution", # property 

696 "fields_matched", 

697 "mature", 

698 "unstable__sensitivity", 

699 ] 

700 """ 

701 Keep the fields names in sync with the actual fields below as this list is 

702 used to generate Swagger documentation. 

703 """ 

704 

705 id = serializers.CharField( 

706 help_text="Our unique identifier for an open-licensed work.", 

707 source="identifier", 

708 ) 

709 

710 indexed_on = serializers.DateTimeField( 

711 source="created_on", 

712 help_text="The timestamp of when the media was indexed by Openverse.", 

713 ) 

714 

715 tags = TagSerializer( 

716 allow_null=True, # replaced with ``[]`` in ``to_representation`` below 

717 many=True, 

718 help_text="Tags with detailed metadata, such as accuracy.", 

719 ) 

720 

721 fields_matched = serializers.ListField( 

722 allow_null=True, # replaced with ``[]`` in ``to_representation`` below 

723 help_text="List the fields that matched the query for this result.", 

724 ) 

725 

726 mature = serializers.BooleanField( 

727 help_text="Whether the media item is marked as mature", 

728 source="sensitive", 

729 ) 

730 

731 # This should be promoted to a stable field alongside 

732 # `include_sensitive_results` 

733 unstable__sensitivity = serializers.SerializerMethodField( 

734 help_text=( 

735 "An array of sensitivity annotations. " 

736 "May contain the following values: 'sensitive_text', " 

737 "'user_reported_sensitive', or 'provider_supplied_sensitive'" 

738 ) 

739 ) 

740 

741 def get_unstable__sensitivity(self, obj: Hit | AbstractMedia) -> list[str]: 

742 result = [] 

743 

744 # obj.identifier needs to be cast to a string because 

745 # Django UUID fields return UUID objects by default 

746 # and UUID comparison fails against _any_ string object, 

747 # even if the string matches the UUID. 

748 if str(obj.identifier) in self.context.get( 

749 "sensitive_text_result_identifiers", set() 

750 ): 

751 result.append(sensitivity.TEXT) 

752 

753 # ``obj.sensitive`` will either be `mature` from the ES document (see below) 

754 # or the ``sensitive`` property on the Image or Audio model. 

755 if obj.sensitive: 755 ↛ 777line 755 didn't jump to line 777 because the condition on line 755 was never true

756 # We do not currently have any documents marked `mature=true` 

757 # that were not marked so as a result of a confirmed user report. 

758 # This is despite the fact that the ingestion server _does_ copy 

759 # the mature field from record `meta_data`. If you query for 

760 # documents in the production image and audio indexes that have 

761 # `mature=true` but do not have confirmed reports, you will get 

762 # 0 results. Whether this is because we truly do not have results 

763 # that providers have themselves marked as mature, unsafe, etc, 

764 # it isn't clear (aside from Flickr, where we use "safe search"). 

765 # What is clear is that we do not need to handle provider reported 

766 # sensitivity here because it simply does not occur in our database. 

767 # That is _very_ convenient because provider supplied maturity is 

768 # much more complex to derive. Its condition is that the result 

769 # has no report _and_ has `mature=true` on the document in ES. 

770 # The only way to derive that is to query both the database and 

771 # Elasticsearch (or have access to both of those individually). 

772 # Due to the flexibility of this serializer in being able to 

773 # handle both Elasticsearch ``Hit``s _and_ media model instances, 

774 # trying to handle provider supplied maturity significantly 

775 # increases complexity here and, in order to prevent redundant 

776 # queries to either Postgres or ES, in other parts of the codebase. 

777 result.append(sensitivity.USER_REPORTED) 

778 

779 return result 

780 

781 def to_representation(self, *args, **kwargs): 

782 # This serializer adapts both ES Hits *and* Media instances. Currently, 

783 # ES has a `mature` field on it which represents if maturity was present on 

784 # the record in the database. The attributes in the code have been renamed 

785 # to `sensitive`, but for the time being this flag still exists on the ES index. 

786 # In order to prevent failures in serialization (since the serializer is looking 

787 # for the `sensitive` attribute), we rename it here. 

788 obj = args[0] 

789 if isinstance(obj, Hit): 789 ↛ 790line 789 didn't jump to line 790 because the condition on line 789 was never true

790 obj.sensitive = obj.mature 

791 

792 output = super().to_representation(*args, **kwargs) 

793 

794 # Ensure lists are ``[]`` instead of ``None`` 

795 # TODO: These fields are still marked 'Nullable' in the API docs 

796 list_fields = ["tags", "fields_matched"] 

797 for list_field in list_fields: 

798 if output[list_field] is None: 

799 output[list_field] = [] 

800 

801 # Ensure license is lowercase 

802 output["license"] = output["license"].lower() 

803 

804 if output.get("license_url") is None: 804 ↛ 805line 804 didn't jump to line 805 because the condition on line 804 was never true

805 try: 

806 lic = License(output["license"], output["license_version"]) 

807 output["license_url"] = lic.url 

808 except ValueError: 

809 pass 

810 

811 # Ensure URLs have scheme 

812 url_fields = ["url", "creator_url", "foreign_landing_url"] 

813 for url_field in url_fields: 

814 output[url_field] = add_protocol(output[url_field]) 

815 

816 return output 

817 

818 

819####################### 

820# Dynamic serializers # 

821####################### 

822 

823 

824def get_hyperlinks_serializer(media_type): 

825 class MediaHyperlinksSerializer(Serializer): 

826 """ 

827 This serializer creates URLs pointing to other endpoints for this media item. 

828 

829 These URLs include the thumbnail, details page and list of related media. 

830 """ 

831 

832 field_names = [ 

833 "thumbnail", # Not suffixed with `_url` because it points to an image 

834 "detail_url", 

835 "related_url", 

836 ] 

837 """ 

838 Keep the fields names in sync with the actual fields below as this list is 

839 used to generate Swagger documentation. 

840 """ 

841 

842 thumbnail = SchemableHyperlinkedIdentityField( 

843 read_only=True, 

844 view_name=f"{media_type}-thumb", 

845 lookup_field="identifier", 

846 help_text="A direct link to the miniature artwork.", 

847 # Some audio results do not have thumbnails 

848 allow_null=media_type == "audio", 

849 ) 

850 detail_url = SchemableHyperlinkedIdentityField( 

851 read_only=True, 

852 view_name=f"{media_type}-detail", 

853 lookup_field="identifier", 

854 help_text="A direct link to the detail view of this audio file.", 

855 ) 

856 related_url = SchemableHyperlinkedIdentityField( 

857 read_only=True, 

858 view_name=f"{media_type}-related", 

859 lookup_field="identifier", 

860 help_text="A link to an endpoint that provides similar audio files.", 

861 ) 

862 

863 return MediaHyperlinksSerializer