yt-dlp/yt_dlp/extractor/radiofrance.py

import re

from .common import InfoExtractor
from ..utils import parse_duration, unified_strdate


class RadioFranceIE(InfoExtractor):
    _VALID_URL = r'^https?://maison\.radiofrance\.fr/radiovisions/(?P<id>[^?#]+)'
    IE_NAME = 'radiofrance'

    _TEST = {
        'url': 'http://maison.radiofrance.fr/radiovisions/one-one',
        'md5': 'bdbb28ace95ed0e04faab32ba3160daf',
        'info_dict': {
            'id': 'one-one',
            'ext': 'ogg',
            'title': 'One to one',
            'description': "Plutôt que d'imaginer la radio de demain comme technologie ou comme création de contenu, je veux montrer que quelles que soient ses évolutions, j'ai l'intime conviction que la radio continuera d'être un grand média de proximité pour les auditeurs.",
            'uploader': 'Thomas Hercouët',
        },
    }

    def _real_extract(self, url):
        m = self._match_valid_url(url)
        video_id = m.group('id')

        webpage = self._download_webpage(url, video_id)
        title = self._html_search_regex(r'<h1>(.*?)</h1>', webpage, 'title')
        description = self._html_search_regex(
            r'<div class="bloc_page_wrapper"><div class="text">(.*?)</div>',
            webpage, 'description', fatal=False)
        uploader = self._html_search_regex(
            r'<div class="credit">&nbsp;&nbsp;&copy;&nbsp;(.*?)</div>',
            webpage, 'uploader', fatal=False)

        formats_str = self._html_search_regex(
            r'class="jp-jplayer[^"]*" data-source="([^"]+)">',
            webpage, 'audio URLs')
        formats = [
            {
                'format_id': fm[0],
                'url': fm[1],
                'vcodec': 'none',
                'quality': i,
            }
            for i, fm in
            enumerate(re.findall(r"([a-z0-9]+)\s*:\s*'([^']+)'", formats_str))
        ]

        return {
            'id': video_id,
            'title': title,
            'formats': formats,
            'description': description,
            'uploader': uploader,
        }


class FranceCultureIE(InfoExtractor):
    _VALID_URL = r'https?://(?:www\.)?radiofrance\.fr/(?:franceculture|fip|francemusique|mouv|franceinter)/podcasts/(?:[^?#]+/)?(?P<display_id>[^?#]+)-(?P<id>\d+)($|[?#])'
    _TESTS = [
        {
            'url': 'https://www.radiofrance.fr/franceculture/podcasts/science-en-questions/la-physique-d-einstein-aiderait-elle-a-comprendre-le-cerveau-8440487',
            'info_dict': {
                'id': '8440487',
                'display_id': 'la-physique-d-einstein-aiderait-elle-a-comprendre-le-cerveau',
                'ext': 'mp3',
                'title': 'La physique d’Einstein aiderait-elle à comprendre le cerveau ?',
                'description': 'Existerait-il un pont conceptuel entre la physique de l’espace-temps et les neurosciences ?',
                'thumbnail': 'https://cdn.radiofrance.fr/s3/cruiser-production/2022/05/d184e7a3-4827-4494-bf94-04ed7b120db4/1200x630_gettyimages-200171095-001.jpg',
                'upload_date': '20220514',
                'duration': 2750,
            },
        },
        {
            'url': 'https://www.radiofrance.fr/franceinter/podcasts/la-rafle-du-vel-d-hiv-une-affaire-d-etat/les-racines-du-crime-episode-1-3715507',
            'only_matching': True,
        }
    ]

    def _real_extract(self, url):
        video_id, display_id = self._match_valid_url(url).group('id', 'display_id')
        webpage = self._download_webpage(url, display_id)

        # _search_json_ld doesn't correctly handle this. See https://github.com/yt-dlp/yt-dlp/pull/3874#discussion_r891903846
        video_data = self._search_json('', webpage, 'audio data', display_id, contains_pattern=r'{\s*"@type"\s*:\s*"AudioObject".+}')

        return {
            'id': video_id,
            'display_id': display_id,
            'url': video_data['contentUrl'],
            'ext': video_data.get('encodingFormat'),
            'vcodec': 'none' if video_data.get('encodingFormat') == 'mp3' else None,
            'duration': parse_duration(video_data.get('duration')),
            'title': self._html_search_regex(r'(?s)<h1[^>]*itemprop="[^"]*name[^"]*"[^>]*>(.+?)</h1>',
                                             webpage, 'title', default=self._og_search_title(webpage)),
            'description': self._html_search_regex(
                r'(?s)<meta name="description"\s*content="([^"]+)', webpage, 'description', default=None),
            'thumbnail': self._og_search_thumbnail(webpage),
            'uploader': self._html_search_regex(
                r'(?s)<span class="author">(.*?)</span>', webpage, 'uploader', default=None),
            'upload_date': unified_strdate(self._search_regex(
                r'"datePublished"\s*:\s*"([^"]+)', webpage, 'timestamp', fatal=False))
        }
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
+								import re
 								from .common import InfoExtractor
-												[cleanup] Misc fixes

Closes #4027

											
										
										
											2022-06-10 19:03:54 +00:00
+								from ..utils import parse_duration, unified_strdate
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
 								class RadioFranceIE(InfoExtractor):
 								    _VALID_URL = r'^https?://maison\.radiofrance\.fr/radiovisions/(?P<id>[^?#]+)'
-												[radiofrance] Modernize

											
										
										
											2014-03-23 16:43:33 +00:00
+								    IE_NAME = 'radiofrance'
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
 								    _TEST = {
-												[radiofrance] Modernize

											
										
										
											2014-03-23 16:43:33 +00:00
+								        'url': 'http://maison.radiofrance.fr/radiovisions/one-one',
 								        'md5': 'bdbb28ace95ed0e04faab32ba3160daf',
 								        'info_dict': {
 								            'id': 'one-one',
 								            'ext': 'ogg',
-												[refactor] Single quotes consistency

											
										
										
											2016-02-14 09:37:17 +00:00
+								            'title': 'One to one',
 								            'description': "Plutôt que d'imaginer la radio de demain comme technologie ou comme création de contenu, je veux montrer que quelles que soient ses évolutions, j'ai l'intime conviction que la radio continuera d'être un grand média de proximité pour les auditeurs.",
 								            'uploader': 'Thomas Hercouët',
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
+								        },
 								    }
 								    def _real_extract(self, url):
-												[extractor] Common function `_match_valid_url`

											
										
										
											2021-08-19 01:41:24 +00:00
+								        m = self._match_valid_url(url)
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
+								        video_id = m.group('id')
 								        webpage = self._download_webpage(url, video_id)
-												[radiofrance] Modernize

											
										
										
											2014-03-23 16:43:33 +00:00
+								        title = self._html_search_regex(r'<h1>(.*?)</h1>', webpage, 'title')
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
+								        description = self._html_search_regex(
 								            r'<div class="bloc_page_wrapper"><div class="text">(.*?)</div>',
-												[radiofrance] Modernize

											
										
										
											2014-03-23 16:43:33 +00:00
+								            webpage, 'description', fatal=False)
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
+								        uploader = self._html_search_regex(
 								            r'<div class="credit">&nbsp;&nbsp;&copy;&nbsp;(.*?)</div>',
-												[radiofrance] Modernize

											
										
										
											2014-03-23 16:43:33 +00:00
+								            webpage, 'uploader', fatal=False)
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
 								        formats_str = self._html_search_regex(
 								            r'class="jp-jplayer[^"]*" data-source="([^"]+)">',
-												[radiofrance] Modernize

											
										
										
											2014-03-23 16:43:33 +00:00
+								            webpage, 'audio URLs')
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
+								        formats = [
 								            {
-												[radiofrance] remove unused imports

											
										
										
											2013-12-17 11:35:16 +00:00
+								                'format_id': fm[0],
 								                'url': fm[1],
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
+								                'vcodec': 'none',
-												[formatsort] Remove misuse of 'preference'

'preference' is to be used only when the format is better that ALL qualities of a lower preference irrespective of ANY sorting order the user requests. See deezer.py for correct use of this

In the older sorting method, `preference`, `quality` and `language_preference` were functionally almost equivalent. So these disparities doesn't really matter there

Also, despite what the documentation says, the default for `preference` was actually 0 and not -1. I have tried to correct this and also account for it when converting `preference` to `quality`

											
										
										
											2021-02-18 22:03:16 +00:00
+								                'quality': i,
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
+								            }
-												[radiofrance] Modernize

											
										
										
											2014-03-23 16:43:33 +00:00
+								            for i, fm in
 								            enumerate(re.findall(r"([a-z0-9]+)\s*:\s*'([^']+)'", formats_str))
-												[radiofrance] Add support (Fixes #1942)

											
										
										
											2013-12-16 20:34:41 +00:00
+								        ]
 								        return {
 								            'id': video_id,
 								            'title': title,
 								            'formats': formats,
 								            'description': description,
 								            'uploader': uploader,
 								        }
-												[cleanup] Misc fixes

Closes #4027

											
										
										
											2022-06-10 19:03:54 +00:00
 								class FranceCultureIE(InfoExtractor):
-												[extractor/radiofrance] Add more radios (#4065)

Closes #4087 
Authored by: bubbleguuum
											
										
										
											2022-06-19 01:36:14 +00:00
+								    _VALID_URL = r'https?://(?:www\.)?radiofrance\.fr/(?:franceculture|fip|francemusique|mouv|franceinter)/podcasts/(?:[^?#]+/)?(?P<display_id>[^?#]+)-(?P<id>\d+)($|[?#])'
-												[cleanup] Misc fixes

Closes #4027

											
										
										
											2022-06-10 19:03:54 +00:00
+								    _TESTS = [
 								        {
 								            'url': 'https://www.radiofrance.fr/franceculture/podcasts/science-en-questions/la-physique-d-einstein-aiderait-elle-a-comprendre-le-cerveau-8440487',
 								            'info_dict': {
 								                'id': '8440487',
 								                'display_id': 'la-physique-d-einstein-aiderait-elle-a-comprendre-le-cerveau',
 								                'ext': 'mp3',
 								                'title': 'La physique d’Einstein aiderait-elle à comprendre le cerveau ?',
 								                'description': 'Existerait-il un pont conceptuel entre la physique de l’espace-temps et les neurosciences ?',
 								                'thumbnail': 'https://cdn.radiofrance.fr/s3/cruiser-production/2022/05/d184e7a3-4827-4494-bf94-04ed7b120db4/1200x630_gettyimages-200171095-001.jpg',
 								                'upload_date': '20220514',
 								                'duration': 2750,
 								            },
 								        },
-												[extractor/radiofrance] Add more radios (#4065)

Closes #4087 
Authored by: bubbleguuum
											
										
										
											2022-06-19 01:36:14 +00:00
+								        {
 								            'url': 'https://www.radiofrance.fr/franceinter/podcasts/la-rafle-du-vel-d-hiv-une-affaire-d-etat/les-racines-du-crime-episode-1-3715507',
 								            'only_matching': True,
 								        }
-												[cleanup] Misc fixes

Closes #4027

											
										
										
											2022-06-10 19:03:54 +00:00
+								    ]
 								    def _real_extract(self, url):
 								        video_id, display_id = self._match_valid_url(url).group('id', 'display_id')
 								        webpage = self._download_webpage(url, display_id)
 								        # _search_json_ld doesn't correctly handle this. See https://github.com/yt-dlp/yt-dlp/pull/3874#discussion_r891903846
-												[extractor] Make search_json able to parse lists

Now `contains_pattern` can be set to `\[.+\]`

											
										
										
											2022-10-03 11:20:27 +00:00
+								        video_data = self._search_json('', webpage, 'audio data', display_id, contains_pattern=r'{\s*"@type"\s*:\s*"AudioObject".+}')
-												[cleanup] Misc fixes

Closes #4027

											
										
										
											2022-06-10 19:03:54 +00:00
 								        return {
 								            'id': video_id,
 								            'display_id': display_id,
 								            'url': video_data['contentUrl'],
 								            'ext': video_data.get('encodingFormat'),
 								            'vcodec': 'none' if video_data.get('encodingFormat') == 'mp3' else None,
 								            'duration': parse_duration(video_data.get('duration')),
 								            'title': self._html_search_regex(r'(?s)<h1[^>]*itemprop="[^"]*name[^"]*"[^>]*>(.+?)</h1>',
 								                                             webpage, 'title', default=self._og_search_title(webpage)),
 								            'description': self._html_search_regex(
 								                r'(?s)<meta name="description"\s*content="([^"]+)', webpage, 'description', default=None),
 								            'thumbnail': self._og_search_thumbnail(webpage),
 								            'uploader': self._html_search_regex(
 								                r'(?s)<span class="author">(.*?)</span>', webpage, 'uploader', default=None),
 								            'upload_date': unified_strdate(self._search_regex(
 								                r'"datePublished"\s*:\s*"([^"]+)', webpage, 'timestamp', fatal=False))
 								        }