From 94eba4c156af080e87caf10cf8ffbea03bd17407 Mon Sep 17 00:00:00 2001 From: Lillie <91358136+LillieH1000@users.noreply.github.com> Date: Wed, 26 Aug 2026 19:44:14 -0400 Subject: [PATCH] [ie/globalplayer] Fix extractors (#17442) Closes #17215, Closes #17429 Authored by: LillieH1000 --- yt_dlp/extractor/globalplayer.py | 221 +++++++++++++++---------------- 1 file changed, 105 insertions(+), 116 deletions(-) diff --git a/yt_dlp/extractor/globalplayer.py b/yt_dlp/extractor/globalplayer.py index 3d4a9304ca..0a51e02079 100644 --- a/yt_dlp/extractor/globalplayer.py +++ b/yt_dlp/extractor/globalplayer.py @@ -1,14 +1,6 @@ from .common import InfoExtractor -from ..utils import ( - clean_html, - join_nonempty, - parse_duration, - str_or_none, - traverse_obj, - unified_strdate, - unified_timestamp, - urlhandle_detect_ext, -) +from ..utils import url_or_none +from ..utils.traversal import require, traverse_obj class GlobalPlayerBaseIE(InfoExtractor): @@ -16,29 +8,11 @@ class GlobalPlayerBaseIE(InfoExtractor): webpage = self._download_webpage(url, video_id) return self._search_nextjs_data(webpage, video_id)['props']['pageProps'] - def _request_ext(self, url, video_id): - return urlhandle_detect_ext(self._request_webpage( # Server rejects HEAD requests - url, video_id, note='Determining source extension')) - - def _extract_audio(self, episode, series): - return { - 'vcodec': 'none', - **traverse_obj(series, { - 'series': 'title', - 'series_id': 'id', - 'thumbnail': 'imageUrl', - 'uploader': 'itunesAuthor', # podcasts only - }), - **traverse_obj(episode, { - 'id': 'id', - 'description': ('description', {clean_html}), - 'duration': ('duration', {parse_duration}), - 'thumbnail': 'imageUrl', - 'url': 'streamUrl', - 'timestamp': (('pubDate', 'startDate'), {unified_timestamp}), - 'title': 'title', - }, get_all=False), - } + @staticmethod + def _get_playback_url(data): + return traverse_obj(data, ( + 'playback', lambda _, v: v['canUse'] == 'true', + 'url', {url_or_none}, any, {require('playback URL')})) class GlobalPlayerLiveIE(GlobalPlayerBaseIE): @@ -48,11 +22,10 @@ class GlobalPlayerLiveIE(GlobalPlayerBaseIE): 'info_dict': { 'id': '2mx1E', 'ext': 'aac', - 'display_id': 'smoothchill-uk', - 'title': 're:^Smooth Chill.+$', - 'thumbnail': 'https://herald.musicradio.com/media/f296ade8-50c9-4f60-911f-924e96873620.png', - 'description': 'Music To Chill To', 'live_status': 'is_live', + 'thumbnail': 'md5:d5040f26c7c4061014a44866129b900e', + 'description': 'md5:6e183929da9001778895f32ae85124bc', + 'title': 're:^Smooth Chill.+$', }, }, { # national station @@ -60,11 +33,10 @@ class GlobalPlayerLiveIE(GlobalPlayerBaseIE): 'info_dict': { 'id': '2mwx4', 'ext': 'aac', - 'description': 'turn up the feel good!', - 'thumbnail': 'https://herald.musicradio.com/media/49b9e8cb-15bf-4bf2-8c28-a4850cc6b0f3.png', 'live_status': 'is_live', + 'description': 'md5:492d07dfea8addadd15650ef40c10d02', + 'thumbnail': 'md5:6f13378a53ce55bcf57365a654e1b490', 'title': 're:^Heart UK.+$', - 'display_id': 'heart-uk', }, }, { # regional variation @@ -72,110 +44,131 @@ class GlobalPlayerLiveIE(GlobalPlayerBaseIE): 'info_dict': { 'id': 'AMqg', 'ext': 'aac', - 'thumbnail': 'https://herald.musicradio.com/media/49b9e8cb-15bf-4bf2-8c28-a4850cc6b0f3.png', - 'title': 're:^Heart London.+$', 'live_status': 'is_live', - 'display_id': 'heart-london', - 'description': 'turn up the feel good!', + 'description': 'md5:492d07dfea8addadd15650ef40c10d02', + 'thumbnail': 'md5:6f13378a53ce55bcf57365a654e1b490', + 'title': 're:^Heart London.+$', }, }] def _real_extract(self, url): video_id = self._match_id(url) - station = self._get_page_props(url, video_id)['station'] - stream_url = station['streamUrl'] + meta = self._get_page_props(url, video_id)['station'] + station_id = meta['id'] + + data = self._download_json(f'https://bff-web-guacamole.musicradio.com/playables/{station_id}', video_id) return { - 'id': station['id'], - 'display_id': join_nonempty('brandSlug', 'slug', from_dict=station) or station.get('legacyStationPrefix'), - 'url': stream_url, - 'ext': self._request_ext(stream_url, video_id), + 'id': station_id, + 'url': self._get_playback_url(data), + 'ext': 'aac', 'vcodec': 'none', 'is_live': True, - **traverse_obj(station, { - 'title': (('name', 'brandName'), {str_or_none}), - 'description': 'tagline', - 'thumbnail': 'brandLogo', - }, get_all=False), + **traverse_obj(meta, { + 'thumbnail': ('brandLogo', {url_or_none}), + 'description': ('tagline', {str}), + 'title': ('name', {str}), + }), } class GlobalPlayerLivePlaylistIE(GlobalPlayerBaseIE): _VALID_URL = r'https?://www\.globalplayer\.com/playlists/(?P\w+)' _TESTS = [{ - # "live playlist" + # live playlist 'url': 'https://www.globalplayer.com/playlists/8bLk/', 'info_dict': { 'id': '8bLk', 'ext': 'aac', 'live_status': 'is_live', - 'description': 'md5:e10f5e10b01a7f2c14ba815509fbb38d', - 'thumbnail': 'https://images.globalplayer.com/images/551379?width=450&signature=oMLPZIoi5_dBSHnTMREW0Xg76mA=', + 'thumbnail': 'md5:391a13cc087b42f626e9e65bbeaf0a11', + 'description': 'md5:f015f2f6c6f6a807669ebcc9a0ca147c', 'title': 're:^Classic FM Hall of Fame.+$', }, }] def _real_extract(self, url): video_id = self._match_id(url) - station = self._get_page_props(url, video_id)['playlistData'] - stream_url = station['streamUrl'] + meta = self._get_page_props(url, video_id)['playlistData'] return { - 'id': video_id, - 'url': stream_url, - 'ext': self._request_ext(stream_url, video_id), + 'url': meta['streamUrl'], + 'ext': 'aac', 'vcodec': 'none', + 'id': video_id, 'is_live': True, - **traverse_obj(station, { - 'title': 'title', - 'description': 'description', - 'thumbnail': 'image', + **traverse_obj(meta, { + 'thumbnail': ('image', {url_or_none}), + 'description': ('description', {str}), + 'title': ('title', {str}), }), } class GlobalPlayerAudioIE(GlobalPlayerBaseIE): - _VALID_URL = r'https?://www\.globalplayer\.com/(?:(?Ppodcasts)/|catchup/\w+/\w+/)(?P\w+)/?(?:$|[?#])' + _VALID_URL = r'https?://www\.globalplayer\.com/(?P(?Ppodcasts)/|catchup/\w+/\w+/)(?P\w+)/?(?:$|[?#])' _TESTS = [{ # podcast 'url': 'https://www.globalplayer.com/podcasts/42KuaM/', - 'playlist_mincount': 5, + 'playlist_mincount': 2, 'info_dict': { 'id': '42KuaM', - 'title': 'Filthy Ritual', 'thumbnail': 'md5:60286e7d12d795bd1bbc9efc6cee643e', - 'categories': ['Society & Culture', 'True Crime'], - 'uploader': 'Global', - 'description': 'md5:da5b918eac9ae319454a10a563afacf9', + 'description': 'md5:17b7b9e3c76b2f4d9e31ccc4f0b66e32', + 'title': 'Filthy Ritual', }, }, { # radio catchup 'url': 'https://www.globalplayer.com/catchup/lbc/uk/46vyD7z/', - 'playlist_mincount': 3, + 'playlist_mincount': 2, 'info_dict': { 'id': '46vyD7z', - 'description': 'Nick Ferrari At Breakfast is Leading Britain\'s Conversation.', + 'thumbnail': 'md5:664ad62a8fb920a2b8e264ed780eee3d', + 'description': 'md5:53b6fa5ef71a3cff6628551bcc416384', 'title': 'Nick Ferrari', - 'thumbnail': 'md5:4df24d8a226f5b2508efbcc6ae874ebf', }, }] def _real_extract(self, url): - video_id, podcast = self._match_valid_url(url).group('id', 'podcast') + video_id, path, podcast = self._match_valid_url(url).group('id', 'path', 'podcast') props = self._get_page_props(url, video_id) - series = props['podcastInfo'] if podcast else props['catchupInfo'] + if podcast: + meta = props['podcastInfo']['metadata'] + blocks = props['podcastInfo']['blocks'][1]['items'] + else: + catchup = props['catchupShow'] if 'catchupShow' in props else props['catchupInfo'] + meta = catchup['metadata'] + blocks = catchup['blocks'][1]['items'] + + def _entries(): + for block in blocks: + entry_id = block['id'] + data = self._download_json( + f'https://bff-web-guacamole.musicradio.com/playables/{entry_id}', + video_id, f'Downloading metadata JSON for {entry_id}') + + yield { + 'id': entry_id, + 'url': self._get_playback_url(data), + 'vcodec': 'none', + 'extractor': GlobalPlayerAudioEpisodeIE.IE_NAME, + 'extractor_key': GlobalPlayerAudioEpisodeIE.ie_key(), + 'webpage_url': f'https://www.globalplayer.com/{path}episodes/{entry_id}', + **traverse_obj(block, { + 'thumbnail': ('image', 'url', {url_or_none}), + 'description': ('description', {str}), + 'title': ('title', {str}), + }), + } return { '_type': 'playlist', 'id': video_id, - 'entries': [self._extract_audio(ep, series) for ep in traverse_obj( - series, ('episodes', lambda _, v: v['id'] and v['streamUrl']))], - 'categories': traverse_obj(series, ('categories', ..., 'name')) or None, - **traverse_obj(series, { - 'description': 'description', - 'thumbnail': 'imageUrl', - 'title': 'title', - 'uploader': 'itunesAuthor', # podcasts only + 'entries': _entries(), + **traverse_obj(meta, { + 'thumbnail': ('image', 'url', {url_or_none}), + 'description': ('description', {str}), + 'title': ('title', {str}), }), } @@ -184,44 +177,42 @@ class GlobalPlayerAudioEpisodeIE(GlobalPlayerBaseIE): _VALID_URL = r'https?://www\.globalplayer\.com/(?:(?Ppodcasts)|catchup/\w+/\w+)/episodes/(?P\w+)/?(?:$|[?#])' _TESTS = [{ # podcast - 'url': 'https://www.globalplayer.com/podcasts/episodes/7DrfNnE/', + 'url': 'https://www.globalplayer.com/podcasts/episodes/7DrorSc/', 'info_dict': { - 'id': '7DrfNnE', + 'id': '7DrorSc', 'ext': 'mp3', - 'title': 'Filthy Ritual - Trailer', - 'description': 'md5:1f1562fd0f01b4773b590984f94223e0', 'thumbnail': 'md5:60286e7d12d795bd1bbc9efc6cee643e', - 'duration': 225.0, - 'timestamp': 1681254900, - 'series': 'Filthy Ritual', - 'series_id': '42KuaM', - 'upload_date': '20230411', - 'uploader': 'Global', + 'description': 'md5:372e5aa2b531f9eba863dfc67d007c1c', + 'title': 'Filthy Ritual - Trailer', }, }, { - # radio catchup - 'url': 'https://www.globalplayer.com/catchup/lbc/uk/episodes/2zGq26Vcv1fCWhddC4JAwETXWe/', + # radio catchup - test urls are removed after 7 days + 'url': 'https://www.globalplayer.com/catchup/lbc/uk/episodes/2zGmrV6DnvogKkNCXkwkQ8HQTA/', 'info_dict': { - 'id': '2zGq26Vcv1fCWhddC4JAwETXWe', + 'id': '2zGmrV6DnvogKkNCXkwkQ8HQTA', 'ext': 'm4a', - 'timestamp': 1682056800, - 'series': 'Nick Ferrari', - 'thumbnail': 'md5:4df24d8a226f5b2508efbcc6ae874ebf', - 'upload_date': '20230421', - 'series_id': '46vyD7z', - 'description': 'Nick Ferrari At Breakfast is Leading Britain\'s Conversation.', + 'thumbnail': 'md5:664ad62a8fb920a2b8e264ed780eee3d', + 'description': 'md5:53b6fa5ef71a3cff6628551bcc416384', 'title': 'Nick Ferrari', - 'duration': 10800.0, }, }] def _real_extract(self, url): video_id, podcast = self._match_valid_url(url).group('id', 'podcast') props = self._get_page_props(url, video_id) - episode = props['podcastEpisode'] if podcast else props['catchupEpisode'] + meta = props['podcastEpisode']['metadata'] if podcast else props['catchupEpisode']['metadata'] + data = self._download_json(f'https://bff-web-guacamole.musicradio.com/playables/{video_id}', video_id) - return self._extract_audio( - episode, traverse_obj(episode, 'podcast', 'show', expected_type=dict) or {}) + return { + 'id': video_id, + 'url': self._get_playback_url(data), + 'vcodec': 'none', + **traverse_obj(meta, { + 'thumbnail': ('image', 'url', {url_or_none}), + 'description': ('description', {str}), + 'title': ('title', {str}), + }), + } class GlobalPlayerVideoIE(GlobalPlayerBaseIE): @@ -231,9 +222,8 @@ class GlobalPlayerVideoIE(GlobalPlayerBaseIE): 'info_dict': { 'id': '2JsSZ7Gm2uP', 'ext': 'mp4', - 'description': 'md5:6a9f063c67c42f218e42eee7d0298bfd', 'thumbnail': 'md5:d4498af48e15aae4839ce77b97d39550', - 'upload_date': '20230420', + 'description': 'md5:6a9f063c67c42f218e42eee7d0298bfd', 'title': 'Treble Malakai Bayoh sings a sublime Handel aria at Classic FM Live', }, }] @@ -245,10 +235,9 @@ class GlobalPlayerVideoIE(GlobalPlayerBaseIE): return { 'id': video_id, **traverse_obj(meta, { - 'url': 'url', - 'thumbnail': ('image', 'url'), - 'title': 'title', - 'upload_date': ('publish_date', {unified_strdate}), - 'description': 'description', + 'url': ('url', {url_or_none}), + 'thumbnail': ('image', 'url', {url_or_none}), + 'description': ('description', {str}), + 'title': ('title', {str}), }), }