[Alsace20TV] Add new extractors Alsace20TVIE, Alsace20TVEmbedIE

[CPAC] Add extractor for Canadian Parliament
CPACIE: single episode CPACPlaylistIE: playlists and searches
2025-07-13 15:04:14 +09:00 · 2022-02-24 18:43:47 +00:00 · 2022-02-24 18:27:57 +00:00 · 2022-02-24 18:26:58 +00:00 · 2022-02-24 13:44:52 +00:00 · 2022-02-24 13:34:32 +00:00
6 changed files with 309 additions and 2 deletions
--- a/youtube_dl/extractor/aliexpress.py
+++ b/youtube_dl/extractor/aliexpress.py
@ -18,7 +18,7 @@ class AliExpressLiveIE(InfoExtractor):
            'id': '2800002704436634',
            'ext': 'mp4',
            'title': 'CASIMA7.22',
-            'thumbnail': r're:http://.*\.jpg',
+            'thumbnail': r're:https?://.*\.jpg',
            'uploader': 'CASIMA Official Store',
            'timestamp': 1500717600,
            'upload_date': '20170722',
--- a/youtube_dl/extractor/alsace20tv.py
+++ b/youtube_dl/extractor/alsace20tv.py
@ -0,0 +1,89 @@
 # coding: utf-8
 from __future__ import unicode_literals
 from .common import InfoExtractor
 from ..utils import (
    clean_html,
    dict_get,
    get_element_by_class,
    int_or_none,
    unified_strdate,
    url_or_none,
 )
 class Alsace20TVIE(InfoExtractor):
    _VALID_URL = r'https?://(?:www\.)?alsace20\.tv/(?:[\w-]+/)+[\w-]+-(?P<id>[\w]+)'
    _TESTS = [{
        'url': 'https://www.alsace20.tv/VOD/Actu/JT/Votre-JT-jeudi-3-fevrier-lyNHCXpYJh.html',
        # 'md5': 'd91851bf9af73c0ad9b2cdf76c127fbb',
        'info_dict': {
            'id': 'lyNHCXpYJh',
            'ext': 'mp4',
            'description': 'md5:fc0bc4a0692d3d2dba4524053de4c7b7',
            'title': 'Votre JT du jeudi 3 février',
            'upload_date': '20220203',
            'thumbnail': r're:https?://.+\.jpg',
            'duration': 1073,
            'view_count': int,
        },
        'params': {
            'format': 'bestvideo',
        },
    }]
    def _extract_video(self, video_id, url=None):
        info = self._download_json(
            'https://www.alsace20.tv/visionneuse/visio_v9_js.php?key=%s&habillage=0&mode=html' % (video_id, ),
            video_id) or {}
        title = info['titre']
        formats = []
        for res, fmt_url in (info.get('files') or {}).items():
            formats.extend(
                self._extract_smil_formats(fmt_url, video_id, fatal=False)
                if '/smil:_' in fmt_url
                else self._extract_mpd_formats(fmt_url, video_id, mpd_id=res, fatal=False))
        self._sort_formats(formats)
        webpage = (url and self._download_webpage(url, video_id, fatal=False)) or ''
        thumbnail = url_or_none(dict_get(info, ('image', 'preview', )) or self._og_search_thumbnail(webpage))
        upload_date = self._search_regex(r'/(\d{6})_', thumbnail, 'upload_date', default=None)
        upload_date = unified_strdate('20%s-%s-%s' % (upload_date[:2], upload_date[2:4], upload_date[4:])) if upload_date else None
        return {
            'id': video_id,
            'title': title,
            'formats': formats,
            'description': clean_html(get_element_by_class('wysiwyg', webpage)),
            'upload_date': upload_date,
            'thumbnail': thumbnail,
            'duration': int_or_none(self._og_search_property('video:duration', webpage) if webpage else None),
            'view_count': int_or_none(info.get('nb_vues')),
        }
    def _real_extract(self, url):
        video_id = self._match_id(url)
        return self._extract_video(video_id, url)
 class Alsace20TVEmbedIE(Alsace20TVIE):
    _VALID_URL = r'https?://(?:www\.)?alsace20\.tv/emb/(?P<id>[\w]+)'
    _TESTS = [{
        'url': 'https://www.alsace20.tv/emb/lyNHCXpYJh',
        # 'md5': 'd91851bf9af73c0ad9b2cdf76c127fbb',
        'info_dict': {
            'id': 'lyNHCXpYJh',
            'ext': 'mp4',
            'title': 'Votre JT du jeudi 3 février',
            'upload_date': '20220203',
            'thumbnail': r're:https?://.+\.jpg',
            'view_count': int,
        },
        'params': {
            'format': 'bestvideo',
        },
    }]
    def _real_extract(self, url):
        video_id = self._match_id(url)
        return self._extract_video(video_id)
--- a/youtube_dl/extractor/bigo.py
+++ b/youtube_dl/extractor/bigo.py
@ -0,0 +1,59 @@
 # coding: utf-8
 from __future__ import unicode_literals
 from .common import InfoExtractor
 from ..utils import ExtractorError, urlencode_postdata
 class BigoIE(InfoExtractor):
    _VALID_URL = r'https?://(?:www\.)?bigo\.tv/(?:[a-z]{2,}/)?(?P<id>[^/]+)'
    _TESTS = [{
        'url': 'https://www.bigo.tv/ja/221338632',
        'info_dict': {
            'id': '6576287577575737440',
            'title': '土よ〜💁‍♂️ 休憩室/REST room',
            'thumbnail': r're:https?://.+',
            'uploader': '✨Shin💫',
            'uploader_id': '221338632',
            'is_live': True,
        },
        'skip': 'livestream',
    }, {
        'url': 'https://www.bigo.tv/th/Tarlerm1304',
        'only_matching': True,
    }, {
        'url': 'https://bigo.tv/115976881',
        'only_matching': True,
    }]
    def _real_extract(self, url):
        user_id = self._match_id(url)
        info_raw = self._download_json(
            'https://bigo.tv/studio/getInternalStudioInfo',
            user_id, data=urlencode_postdata({'siteId': user_id}))
        if not isinstance(info_raw, dict):
            raise ExtractorError('Received invalid JSON data')
        if info_raw.get('code'):
            raise ExtractorError(
                'Bigo says: %s (code %s)' % (info_raw.get('msg'), info_raw.get('code')), expected=True)
        info = info_raw.get('data') or {}
        if not info.get('alive'):
            raise ExtractorError('This user is offline.', expected=True)
        return {
            'id': info.get('roomId') or user_id,
            'title': info.get('roomTopic') or info.get('nick_name') or user_id,
            'formats': [{
                'url': info.get('hls_src'),
                'ext': 'mp4',
                'protocol': 'm3u8',
            }],
            'thumbnail': info.get('snapshot'),
            'uploader': info.get('nick_name'),
            'uploader_id': user_id,
            'is_live': True,
        }
--- a/youtube_dl/extractor/cpac.py
+++ b/youtube_dl/extractor/cpac.py
@ -0,0 +1,148 @@
 # coding: utf-8
 from __future__ import unicode_literals
 from .common import InfoExtractor
 from ..compat import compat_str
 from ..utils import (
    int_or_none,
    str_or_none,
    try_get,
    unified_timestamp,
    update_url_query,
    urljoin,
 )
 # compat_range
 try:
    if callable(xrange):
        range = xrange
 except (NameError, TypeError):
    pass
 class CPACIE(InfoExtractor):
    IE_NAME = 'cpac'
    _VALID_URL = r'https?://(?:www\.)?cpac\.ca/(?P<fr>l-)?episode\?id=(?P<id>[\da-f]{8}(?:-[\da-f]{4}){3}-[\da-f]{12})'
    _TEST = {
        # 'url': 'http://www.cpac.ca/en/programs/primetime-politics/episodes/65490909',
        'url': 'https://www.cpac.ca/episode?id=fc7edcae-4660-47e1-ba61-5b7f29a9db0f',
        'md5': 'e46ad699caafd7aa6024279f2614e8fa',
        'info_dict': {
            'id': 'fc7edcae-4660-47e1-ba61-5b7f29a9db0f',
            'ext': 'mp4',
            'upload_date': '20220215',
            'title': 'News Conference to Celebrate National Kindness Week – February 15, 2022',
            'description': 'md5:466a206abd21f3a6f776cdef290c23fb',
            'timestamp': 1644901200,
        },
        'params': {
            'format': 'bestvideo',
            'hls_prefer_native': True,
        },
    }
    def _real_extract(self, url):
        video_id = self._match_id(url)
        url_lang = 'fr' if '/l-episode?' in url else 'en'
        content = self._download_json(
            'https://www.cpac.ca/api/1/services/contentModel.json?url=/site/website/episode/index.xml&crafterSite=cpacca&id=' + video_id,
            video_id)
        video_url = try_get(content, lambda x: x['page']['details']['videoUrl'], compat_str)
        formats = []
        if video_url:
            content = content['page']
            title = str_or_none(content['details']['title_%s_t' % (url_lang, )])
            formats = self._extract_m3u8_formats(video_url, video_id, m3u8_id='hls', ext='mp4')
            for fmt in formats:
                # prefer language to match URL
                fmt_lang = fmt.get('language')
                if fmt_lang == url_lang:
                    fmt['language_preference'] = 10
                elif not fmt_lang:
                    fmt['language_preference'] = -1
                else:
                    fmt['language_preference'] = -10
        self._sort_formats(formats)
        category = str_or_none(content['details']['category_%s_t' % (url_lang, )])
        def is_live(v_type):
            return (v_type == 'live') if v_type is not None else None
        return {
            'id': video_id,
            'formats': formats,
            'title': title,
            'description': str_or_none(content['details'].get('description_%s_t' % (url_lang, ))),
            'timestamp': unified_timestamp(content['details'].get('liveDateTime')),
            'category': [category] if category else None,
            'thumbnail': urljoin(url, str_or_none(content['details'].get('image_%s_s' % (url_lang, )))),
            'is_live': is_live(content['details'].get('type')),
        }
 class CPACPlaylistIE(InfoExtractor):
    IE_NAME = 'cpac:playlist'
    _VALID_URL = r'(?i)https?://(?:www\.)?cpac\.ca/(?:program|search|(?P<fr>emission|rechercher))\?(?:[^&]+&)*?(?P<id>(?:id=\d+|programId=\d+|key=[^&]+))'
    _TESTS = [{
        'url': 'https://www.cpac.ca/program?id=6',
        'info_dict': {
            'id': 'id=6',
            'title': 'Headline Politics',
            'description': 'Watch CPAC’s signature long-form coverage of the day’s pressing political events as they unfold.',
        },
        'playlist_count': 10,
    }, {
        'url': 'https://www.cpac.ca/search?key=hudson&type=all&order=desc',
        'info_dict': {
            'id': 'key=hudson',
            'title': 'hudson',
        },
        'playlist_count': 22,
    }, {
        'url': 'https://www.cpac.ca/search?programId=50',
        'info_dict': {
            'id': 'programId=50',
            'title': '50',
        },
        'playlist_count': 9,
    }, {
        'url': 'https://www.cpac.ca/emission?id=6',
        'only_matching': True,
    }, {
        'url': 'https://www.cpac.ca/rechercher?key=hudson&type=all&order=desc',
        'only_matching': True,
    }]
    def _real_extract(self, url):
        video_id = self._match_id(url)
        url_lang = 'fr' if any(x in url for x in ('/emission?', '/rechercher?')) else 'en'
        pl_type, list_type = ('program', 'itemList') if any(x in url for x in ('/program?', '/emission?')) else ('search', 'searchResult')
        api_url = (
            'https://www.cpac.ca/api/1/services/contentModel.json?url=/site/website/%s/index.xml&crafterSite=cpacca&%s'
            % (pl_type, video_id, ))
        content = self._download_json(api_url, video_id)
        entries = []
        total_pages = int_or_none(try_get(content, lambda x: x['page'][list_type]['totalPages']), default=1)
        for page in range(1, total_pages + 1):
            if page > 1:
                api_url = update_url_query(api_url, {'page': '%d' % (page, ), })
                content = self._download_json(
                    api_url, video_id,
                    note='Downloading continuation - %d' % (page, ),
                    fatal=False)
            for item in try_get(content, lambda x: x['page'][list_type]['item'], list) or []:
                episode_url = urljoin(url, try_get(item, lambda x: x['url_%s_s' % (url_lang, )]))
                if episode_url:
                    entries.append(episode_url)
        return self.playlist_result(
            (self.url_result(entry) for entry in entries),
            playlist_id=video_id,
            playlist_title=try_get(content, lambda x: x['page']['program']['title_%s_t' % (url_lang, )]) or video_id.split('=')[-1],
            playlist_description=try_get(content, lambda x: x['page']['program']['description_%s_t' % (url_lang, )]),
        )
--- a/youtube_dl/extractor/extractors.py
+++ b/youtube_dl/extractor/extractors.py
@ -51,6 +51,10 @@ from .anvato import AnvatoIE
 from .aol import AolIE
 from .allocine import AllocineIE
 from .aliexpress import AliExpressLiveIE
 from .alsace20tv import (
    Alsace20TVIE,
    Alsace20TVEmbedIE,
 )
 from .apa import APAIE
 from .aparat import AparatIE
 from .appleconnect import AppleConnectIE
@ -115,6 +119,7 @@ from .bfmtv import (
 )
 from .bibeltv import BibelTVIE
 from .bigflix import BigflixIE
 from .bigo import BigoIE
 from .bild import BildIE
 from .bilibili import (
    BiliBiliIE,
@ -254,6 +259,10 @@ from .commonprotocols import (
 from .condenast import CondeNastIE
 from .contv import CONtvIE
 from .corus import CorusIE
 from .cpac import (
    CPACIE,
    CPACPlaylistIE,
 )
 from .cracked import CrackedIE
 from .crackle import CrackleIE
 from .crooksandliars import CrooksAndLiarsIE
--- a/youtube_dl/extractor/myspass.py
+++ b/youtube_dl/extractor/myspass.py
@ -35,7 +35,9 @@ class MySpassIE(InfoExtractor):
        title = xpath_text(metadata, 'title', fatal=True)
        video_url = xpath_text(metadata, 'url_flv', 'download url', True)
        video_id_int = int(video_id)
-        for group in re.search(r'/myspass2009/\d+/(\d+)/(\d+)/(\d+)/', video_url).groups():
+
        grps = re.search(r'/myspass2009/\d+/(\d+)/(\d+)/(\d+)/', video_url)
        for group in grps.groups() if grps else []:
            group_int = int(group)
            if group_int > video_id_int:
                video_url = video_url.replace(
Author	SHA1	Message	Date
dirkf	f8e543c906	[Alsace20TV] Add new extractors Alsace20TVIE, Alsace20TVEmbedIE	2022-02-24 18:43:47 +00:00
dirkf	c4d1738316	[CPAC] Add extractor for Canadian Parliament CPACIE: single episode CPACPlaylistIE: playlists and searches	2022-02-24 18:27:57 +00:00
dirkf	1f13ccfd7f	Fixed groups() call on potentially empty regex search object (#30676 ) * Fixed groups() call on potentially empty regex search object. - https://github.com/ytdl-org/youtube-dl/issues/30521 * minimising lines changed Co-authored-by: yayorbitgum <50963144+yayorbitgum@users.noreply.github.com>	2022-02-24 18:26:58 +00:00
marieell	923292ba64	[aliexpress] Fix test case	2022-02-24 13:44:52 +00:00
Lesmiscore (Naoya Ozaki)	782bfd26db	[bigo] add support for bigo.tv (#30635 ) * [bigo] add support for bigo.tv * [bigo] prepend "Bigo says" * title fallback * add error for invalid json data	2022-02-24 13:34:32 +00:00