Coverage for api/models/audio.py: 82%

124 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 06:14 +0000

1from textwrap import dedent as d 

2 

3from django.conf import settings 

4from django.contrib.postgres.fields import ArrayField 

5from django.db import models 

6 

7from uuslug import uuslug 

8 

9from api.constants.media_types import AUDIO_TYPE 

10from api.models import OpenLedgerModel 

11from api.models.media import ( 

12 AbstractAltFile, 

13 AbstractDeletedMedia, 

14 AbstractMedia, 

15 AbstractMediaDecision, 

16 AbstractMediaDecisionThrough, 

17 AbstractMediaList, 

18 AbstractMediaReport, 

19 AbstractSensitiveMedia, 

20) 

21from api.models.mixins import FileMixin, ForeignIdentifierMixin, MediaMixin 

22from api.utils.waveform import generate_peaks 

23 

24 

25class AltAudioFile(AbstractAltFile): 

26 def __init__(self, attrs): 

27 self.bit_rate = attrs.get("bit_rate") 

28 self.sample_rate = attrs.get("sample_rate") 

29 super().__init__(attrs) 

30 

31 @property 

32 def sample_rate_in_khz(self): 

33 return self.sample_rate / 1e3 

34 

35 @property 

36 def bit_rate_in_kbps(self): 

37 return self.bit_rate / 1e3 

38 

39 def __str__(self): 

40 br = self.bit_rate_in_kbps 

41 sr = self.sample_rate_in_khz 

42 return f"<AltAudioFile {br}kbps / {sr}kHz>" 

43 

44 def __repr__(self): 

45 return str(self) 

46 

47 

48class AudioSet(ForeignIdentifierMixin, MediaMixin, FileMixin, OpenLedgerModel): 

49 """ 

50 This is an ordered collection of audio files, such as a podcast series or an album. 

51 

52 Not to be confused with ``AudioList`` which is a many-to-many collection of audio 

53 files, like a playlist or favourites library. 

54 

55 The FileMixin inherited by this model refers not to audio but album art. 

56 """ 

57 

58 class Meta: 

59 db_table = "audioset" # drop the `api_` prefix 

60 constraints = [ 

61 models.UniqueConstraint( 

62 fields=["foreign_identifier", "provider"], 

63 name="unique_foreign_identifier_provider", 

64 ), 

65 ] 

66 

67 @property 

68 def identifier(self): 

69 return f"{self.provider}--{self.foreign_identifier}" 

70 

71 @property 

72 def tracks(self): 

73 return Audio.objects.filter( 

74 provider=self.provider, 

75 audio_set_foreign_identifier=self.foreign_identifier, 

76 ) 

77 

78 

79class AudioFileMixin(FileMixin): 

80 """ 

81 This mixin adds fields related to audio quality to the standard file mixin. 

82 

83 Do not use this as the sole base class. 

84 """ 

85 

86 bit_rate = models.IntegerField( 

87 blank=True, 

88 null=True, 

89 help_text="Number in bits per second, eg. 128000.", 

90 ) 

91 sample_rate = models.IntegerField( 

92 blank=True, 

93 null=True, 

94 help_text="Number in hertz, eg. 44100.", 

95 ) 

96 

97 @property 

98 def sample_rate_in_khz(self): 

99 return self.sample_rate / 1e3 

100 

101 @property 

102 def bit_rate_in_kbps(self): 

103 return self.bit_rate / 1e3 

104 

105 class Meta: 

106 abstract = True 

107 

108 

109class AudioAddOn(OpenLedgerModel): 

110 audio_identifier = models.UUIDField( 

111 primary_key=True, 

112 help_text=("The identifier of the audio object."), 

113 ) 

114 """ 

115 This cannot be a "ForeignKey" or "OneToOneRel" because the refresh process 

116 wipes out the Audio table completely and recreates it. If we made these a FK 

117 or OneToOneRel there'd be foreign key constraint added that would be violated 

118 when the Audio table is recreated. 

119 

120 The index is necessary as this column is used by the Audio object to query 

121 for the relevant add on. 

122 

123 The refresh process will also eventually include cleaning up any potentially 

124 dangling audio_add_on rows. 

125 """ 

126 

127 waveform_peaks = ArrayField( 

128 base_field=models.FloatField(), 

129 # The approximate resolution of waveform generation 

130 # results in _about_ 1000 peaks. We use 1500 to give 

131 # sufficient wiggle room should we have any outlier 

132 # files pop up. 

133 # https://github.com/WordPress/openverse-api/blob/a7955c86d43bff504e8d41454f68717d79dd3a44/api/catalog/api/utils/waveform.py#L71 

134 size=1500, 

135 help_text=( 

136 "The waveform peaks. A list of floats in the range of 0 -> 1 inclusively." 

137 ), 

138 null=True, 

139 ) 

140 

141 

142class Audio(AudioFileMixin, AbstractMedia): 

143 """ 

144 One audio media instance. 

145 

146 Inherited fields 

147 ================ 

148 category: eg. music, sound_effect, podcast, news & audiobook 

149 

150 Properties 

151 ========== 

152 audioset: >- 

153 This is a virtual foreign-key to `AudioSet` built on top of the fields 

154 `audio_set_foreign_identifier` and `provider`. 

155 """ 

156 

157 audioset = models.ForeignObject( 

158 to="AudioSet", 

159 on_delete=models.DO_NOTHING, 

160 from_fields=["audio_set_foreign_identifier", "provider"], 

161 to_fields=["foreign_identifier", "provider"], 

162 null=True, 

163 ) 

164 

165 # Replaces the foreign key to AudioSet 

166 audio_set_foreign_identifier = models.TextField( 

167 blank=True, 

168 null=True, 

169 help_text="Reference to set of which this track is a part.", 

170 ) 

171 audio_set_position = models.IntegerField( 

172 blank=True, null=True, help_text="Ordering of the audio in the set." 

173 ) 

174 

175 genres = ArrayField( 

176 base_field=models.CharField( 

177 max_length=80, 

178 blank=True, 

179 ), 

180 null=True, 

181 db_index=True, 

182 help_text="An array of audio genres such as " 

183 "`rock`, `electronic` for `music` category, or " 

184 "`politics`, `sport`, `education` for `podcast` category", 

185 ) 

186 

187 duration = models.IntegerField( 

188 blank=True, 

189 null=True, 

190 help_text="The time length of the audio file in milliseconds.", 

191 ) 

192 

193 alt_files = models.JSONField( 

194 blank=True, 

195 null=True, 

196 help_text=d(""" 

197 JSON object containing information on alternative audio files. Each object 

198 is expected to contain: 

199 

200 - `url`: URL reference to the file 

201 - `filesize`: File size in bytes 

202 - `filetype`: Extension of the file 

203 - `bit_rate`: Bitrate of the file in bits/second 

204 - `sample_rate`: Sample rate of the file in bits/second 

205 """), 

206 ) 

207 

208 @property 

209 def sensitive(self) -> bool: 

210 return hasattr(self, "sensitive_audio") 

211 

212 @property 

213 def alternative_files(self): 

214 if hasattr(self.alt_files, "__iter__"): 

215 return [AltAudioFile(alt_file) for alt_file in self.alt_files] 

216 return None 

217 

218 @property 

219 def duration_in_s(self): 

220 return self.duration / 1e3 

221 

222 @property 

223 def audio_set(self): 

224 return getattr(self, "audioset") 

225 

226 def get_or_create_waveform(self): 

227 add_on, _ = AudioAddOn.objects.get_or_create(audio_identifier=self.identifier) 

228 

229 if add_on.waveform_peaks is not None: 

230 return add_on.waveform_peaks 

231 

232 add_on.waveform_peaks = generate_peaks(self) 

233 add_on.save() 

234 

235 return add_on.waveform_peaks 

236 

237 class Meta(AbstractMedia.Meta): 

238 db_table = "audio" 

239 verbose_name = "audio track" 

240 verbose_name_plural = "audio tracks" 

241 

242 def get_absolute_url(self): 

243 """Enable the "View on site" link in the Django Admin.""" 

244 

245 from django.urls import reverse 

246 

247 return reverse("audio-detail", args=[str(self.identifier)]) 

248 

249 

250class DeletedAudio(AbstractDeletedMedia): 

251 """ 

252 Audio tracks deleted from the upstream source. 

253 

254 Do not create instances of this model manually. Create an ``AudioReport`` instance 

255 instead. 

256 """ 

257 

258 media_class = Audio 

259 es_index = settings.MEDIA_INDEX_MAPPING[AUDIO_TYPE] 

260 

261 media_obj = models.OneToOneField( 

262 to="Audio", 

263 to_field="identifier", 

264 on_delete=models.DO_NOTHING, 

265 primary_key=True, 

266 db_constraint=False, 

267 db_column="identifier", 

268 related_name="deleted_audio", 

269 help_text="The reference to the deleted audio.", 

270 ) 

271 

272 class Meta: 

273 verbose_name = "deleted audio track" 

274 verbose_name_plural = "deleted audio tracks" 

275 

276 

277class SensitiveAudio(AbstractSensitiveMedia): 

278 """ 

279 Audio tracks with verified sensitivity reports. 

280 

281 Do not create instances of this model manually. Create an ``AudioReport`` instance 

282 instead. 

283 """ 

284 

285 media_class = Audio 

286 es_index = settings.MEDIA_INDEX_MAPPING[AUDIO_TYPE] 

287 

288 media_obj = models.OneToOneField( 

289 to="Audio", 

290 to_field="identifier", 

291 on_delete=models.DO_NOTHING, 

292 primary_key=True, 

293 db_constraint=False, 

294 db_column="identifier", 

295 related_name="sensitive_audio", 

296 help_text="The reference to the sensitive audio.", 

297 ) 

298 

299 class Meta: 

300 db_table = "api_matureaudio" 

301 verbose_name = "sensitive audio track" 

302 verbose_name_plural = "sensitive audio tracks" 

303 

304 

305class AudioReport(AbstractMediaReport): 

306 """ 

307 User-submitted reports of audio tracks. 

308 

309 ``AudioDecision`` is populated only if moderators have made a decision 

310 for this report. 

311 """ 

312 

313 media_class = Audio 

314 

315 media_obj = models.ForeignKey( 

316 to="Audio", 

317 to_field="identifier", 

318 on_delete=models.DO_NOTHING, 

319 db_constraint=False, 

320 db_column="identifier", 

321 related_name="audio_report", 

322 help_text="The reference to the audio being reported.", 

323 ) 

324 decision = models.ForeignKey( 

325 to="AudioDecision", 

326 on_delete=models.SET_NULL, 

327 blank=True, 

328 null=True, 

329 help_text="The moderation decision for this report.", 

330 ) 

331 

332 class Meta: 

333 db_table = "nsfw_reports_audio" 

334 

335 

336class AudioDecision(AbstractMediaDecision): 

337 """Moderation decisions taken for audio tracks.""" 

338 

339 media_class = Audio 

340 

341 media_objs = models.ManyToManyField( 

342 to="Audio", 

343 through="AudioDecisionThrough", 

344 help_text="The audio items being moderated.", 

345 ) 

346 

347 

348class AudioDecisionThrough(AbstractMediaDecisionThrough): 

349 """ 

350 Many-to-many reference table for audio decisions. 

351 

352 This is made explicit (rather than using Django's default) so that the audio can 

353 be referenced by `identifier` rather than an arbitrary `id`. 

354 """ 

355 

356 media_class = Audio 

357 sensitive_media_class = SensitiveAudio 

358 deleted_media_class = DeletedAudio 

359 

360 media_obj = models.ForeignKey( 

361 Audio, 

362 to_field="identifier", 

363 on_delete=models.DO_NOTHING, 

364 db_column="identifier", 

365 db_constraint=False, 

366 ) 

367 decision = models.ForeignKey(AudioDecision, on_delete=models.CASCADE) 

368 

369 

370class AudioList(AbstractMediaList): 

371 """A list of audio files. Currently unused.""" 

372 

373 audios = models.ManyToManyField( 

374 Audio, 

375 related_name="lists", 

376 help_text="A list of identifier keys corresponding to audios.", 

377 ) 

378 

379 class Meta: 

380 db_table = "audiolist" 

381 

382 def save(self, *args, **kwargs): 

383 self.slug = uuslug(self.title, instance=self) 

384 super().save(*args, **kwargs)