youtube-dl/youtube_dl/extractor/wat.py

# coding: utf-8
from __future__ import unicode_literals

from .common import InfoExtractor
from ..compat import compat_str
from ..utils import (
    unified_strdate,
    HEADRequest,
    int_or_none,
)


class WatIE(InfoExtractor):
    _VALID_URL = r'(?:wat:|https?://(?:www\.)?wat\.tv/video/.*-)(?P<id>[0-9a-z]+)'
    IE_NAME = 'wat.tv'
    _TESTS = [
        {
            'url': 'http://www.wat.tv/video/soupe-figues-l-orange-aux-epices-6z1uz_2hvf7_.html',
            'info_dict': {
                'id': '11713067',
                'ext': 'mp4',
                'title': 'Soupe de figues à l\'orange et aux épices',
                'description': 'Retrouvez l\'émission "Petits plats en équilibre", diffusée le 18 août 2014.',
                'upload_date': '20140819',
                'duration': 120,
            },
            'params': {
                # m3u8 download
                'skip_download': True,
            },
            'expected_warnings': ['HTTP Error 404'],
        },
        {
            'url': 'http://www.wat.tv/video/gregory-lemarchal-voix-ange-6z1v7_6ygkj_.html',
            'md5': 'b16574df2c3cd1a36ca0098f2a791925',
            'info_dict': {
                'id': '11713075',
                'ext': 'mp4',
                'title': 'Grégory Lemarchal, une voix d\'ange depuis 10 ans (1/3)',
                'upload_date': '20140816',
            },
            'expected_warnings': ["Ce contenu n'est pas disponible pour l'instant."],
        },
    ]

    def _real_extract(self, url):
        video_id = self._match_id(url)
        video_id = video_id if video_id.isdigit() and len(video_id) > 6 else compat_str(int(video_id, 36))

        # 'contentv4' is used in the website, but it also returns the related
        # videos, we don't need them
        video_data = self._download_json(
            'http://www.wat.tv/interface/contentv4s/' + video_id, video_id)
        video_info = video_data['media']

        error_desc = video_info.get('error_desc')
        if error_desc:
            self.report_warning(
                '%s returned error: %s' % (self.IE_NAME, error_desc))

        chapters = video_info['chapters']
        if chapters:
            first_chapter = chapters[0]

            def video_id_for_chapter(chapter):
                return chapter['tc_start'].split('-')[0]

            if video_id_for_chapter(first_chapter) != video_id:
                self.to_screen('Multipart video detected')
                entries = [self.url_result('wat:%s' % video_id_for_chapter(chapter)) for chapter in chapters]
                return self.playlist_result(entries, video_id, video_info['title'])
            # Otherwise we can continue and extract just one part, we have to use
            # the video id for getting the video url
        else:
            first_chapter = video_info

        title = first_chapter['title']

        def extract_url(path_template, url_type):
            req_url = 'http://www.wat.tv/get/%s' % (path_template % video_id)
            head = self._request_webpage(HEADRequest(req_url), video_id, 'Extracting %s url' % url_type, fatal=False)
            if head:
                red_url = head.geturl()
                if req_url != red_url:
                    return red_url
            return None

        formats = []
        manifest_urls = self._download_json(
            'http://www.wat.tv/get/webhtml/' + video_id, video_id)
        m3u8_url = manifest_urls.get('hls')
        if m3u8_url:
            formats.extend(self._extract_m3u8_formats(
                m3u8_url, video_id, 'mp4',
                'm3u8_native', m3u8_id='hls', fatal=False))
        mpd_url = manifest_urls.get('mpd')
        if mpd_url:
            formats.extend(self._extract_mpd_formats(
                mpd_url.replace('://das-q1.tf1.fr/', '://das-q1-ssl.tf1.fr/'),
                video_id, mpd_id='dash', fatal=False))
        self._sort_formats(formats)

        date_diffusion = first_chapter.get('date_diffusion') or video_data.get('configv4', {}).get('estatS4')
        upload_date = unified_strdate(date_diffusion) if date_diffusion else None
        duration = None
        files = video_info['files']
        if files:
            duration = int_or_none(files[0].get('duration'))

        return {
            'id': video_id,
            'title': title,
            'thumbnail': first_chapter.get('preview'),
            'description': first_chapter.get('description'),
            'view_count': int_or_none(video_info.get('views')),
            'upload_date': upload_date,
            'duration': duration,
            'formats': formats,
        }
WatIE: support videos divided in multiple parts (closes #222 and #659) The id for the videos is now the full id, no the one in the webpage url. Also extract more information: description, view_count and upload_date 2013-06-29 18:22:03 +02:00			`# coding: utf-8`
[wat] Modernize 2014-03-29 15:15:16 +01:00			`from __future__ import unicode_literals`
WatIE: support videos divided in multiple parts (closes #222 and #659) The id for the videos is now the full id, no the one in the webpage url. Also extract more information: description, view_count and upload_date 2013-06-29 18:22:03 +02:00
Add WatIE 2013-06-28 22:01:47 +02:00			`from .common import InfoExtractor`
[wat] extract all formats 2016-04-22 10:36:14 +02:00			`from ..compat import compat_str`
[wat] Capture and output error message 2014-09-23 14:58:35 +02:00			`from ..utils import (`
			`unified_strdate,`
[wat] extract all formats 2016-04-22 10:36:14 +02:00			`HEADRequest,`
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00			`int_or_none,`
[wat] Capture and output error message 2014-09-23 14:58:35 +02:00			`)`
Add WatIE 2013-06-28 22:01:47 +02:00

			`class WatIE(InfoExtractor):`
[wat] extract all formats 2016-04-22 10:36:14 +02:00			`_VALID_URL = r'(?:wat:\|https?://(?:www\.)?wat\.tv/video/.*-)(?P<id>[0-9a-z]+)'`
Add WatIE 2013-06-28 22:01:47 +02:00			`IE_NAME = 'wat.tv'`
[wat] Use server time and pass country argument (Closes #3579) 2014-08-25 15:21:33 +02:00			`_TESTS = [`
			`{`
			`'url': 'http://www.wat.tv/video/soupe-figues-l-orange-aux-epices-6z1uz_2hvf7_.html',`
			`'info_dict': {`
			`'id': '11713067',`
			`'ext': 'mp4',`
			`'title': 'Soupe de figues à l\'orange et aux épices',`
			`'description': 'Retrouvez l\'émission "Petits plats en équilibre", diffusée le 18 août 2014.',`
			`'upload_date': '20140819',`
			`'duration': 120,`
			`},`
[wat] try all supported adaptive urls 2018-06-17 16:56:52 +02:00			`'params': {`
			`# m3u8 download`
			`'skip_download': True,`
			`},`
			`'expected_warnings': ['HTTP Error 404'],`
[wat] Use server time and pass country argument (Closes #3579) 2014-08-25 15:21:33 +02:00			`},`
			`{`
			`'url': 'http://www.wat.tv/video/gregory-lemarchal-voix-ange-6z1v7_6ygkj_.html',`
[wat] try all supported adaptive urls 2018-06-17 16:56:52 +02:00			`'md5': 'b16574df2c3cd1a36ca0098f2a791925',`
[wat] Use server time and pass country argument (Closes #3579) 2014-08-25 15:21:33 +02:00			`'info_dict': {`
			`'id': '11713075',`
			`'ext': 'mp4',`
			`'title': 'Grégory Lemarchal, une voix d\'ange depuis 10 ans (1/3)',`
			`'upload_date': '20140816',`
			`},`
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00			`'expected_warnings': ["Ce contenu n'est pas disponible pour l'instant."],`
Disable way and tf1 tests, the whole videos are served sometimes, so the md5 sum doesn't match. 2013-07-30 11:19:07 +02:00			`},`
[wat] Use server time and pass country argument (Closes #3579) 2014-08-25 15:21:33 +02:00			`]`
[wat] Modernize 2014-03-29 15:15:16 +01:00
Add WatIE 2013-06-28 22:01:47 +02:00			`def _real_extract(self, url):`
[wat] extract all formats 2016-04-22 10:36:14 +02:00			`video_id = self._match_id(url)`
			`video_id = video_id if video_id.isdigit() and len(video_id) > 6 else compat_str(int(video_id, 36))`
WatIE: support videos divided in multiple parts (closes #222 and #659) The id for the videos is now the full id, no the one in the webpage url. Also extract more information: description, view_count and upload_date 2013-06-29 18:22:03 +02:00
[wat] extract all formats 2016-04-22 10:36:14 +02:00			`# 'contentv4' is used in the website, but it also returns the related`
			`# videos, we don't need them`
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00			`video_data = self._download_json(`
			`'http://www.wat.tv/interface/contentv4s/' + video_id, video_id)`
			`video_info = video_data['media']`
[wat] Add support for SD and HD videos (Closes #3558) 2014-08-23 21:22:10 +02:00
[wat] Capture and output error message 2014-09-23 14:58:35 +02:00			`error_desc = video_info.get('error_desc')`
			`if error_desc:`
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00			`self.report_warning(`
			`'%s returned error: %s' % (self.IE_NAME, error_desc))`
[wat] Capture and output error message 2014-09-23 14:58:35 +02:00
WatIE: support videos divided in multiple parts (closes #222 and #659) The id for the videos is now the full id, no the one in the webpage url. Also extract more information: description, view_count and upload_date 2013-06-29 18:22:03 +02:00			`chapters = video_info['chapters']`
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00			`if chapters:`
			`first_chapter = chapters[0]`
Add WatIE 2013-06-28 22:01:47 +02:00
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00			`def video_id_for_chapter(chapter):`
			`return chapter['tc_start'].split('-')[0]`
WatIE: support videos divided in multiple parts (closes #222 and #659) The id for the videos is now the full id, no the one in the webpage url. Also extract more information: description, view_count and upload_date 2013-06-29 18:22:03 +02:00
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00			`if video_id_for_chapter(first_chapter) != video_id:`
			`self.to_screen('Multipart video detected')`
			`entries = [self.url_result('wat:%s' % video_id_for_chapter(chapter)) for chapter in chapters]`
			`return self.playlist_result(entries, video_id, video_info['title'])`
			`# Otherwise we can continue and extract just one part, we have to use`
			`# the video id for getting the video url`
			`else:`
			`first_chapter = video_info`
[wat] extract all formats 2016-04-22 10:36:14 +02:00
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00			`title = first_chapter['title']`
[wat] extract all formats 2016-04-22 10:36:14 +02:00
			`def extract_url(path_template, url_type):`
			`req_url = 'http://www.wat.tv/get/%s' % (path_template % video_id)`
[wat] extract dash formats 2016-09-06 21:44:45 +02:00			`head = self._request_webpage(HEADRequest(req_url), video_id, 'Extracting %s url' % url_type, fatal=False)`
			`if head:`
			`red_url = head.geturl()`
			`if req_url != red_url:`
			`return red_url`
			`return None`

[wat] extract all formats 2016-04-22 10:36:14 +02:00			`formats = []`
[wat] fix format extraction(closes #27901) 2021-01-21 17:20:32 +01:00			`manifest_urls = self._download_json(`
			`'http://www.wat.tv/get/webhtml/' + video_id, video_id)`
			`m3u8_url = manifest_urls.get('hls')`
			`if m3u8_url:`
			`formats.extend(self._extract_m3u8_formats(`
			`m3u8_url, video_id, 'mp4',`
			`'m3u8_native', m3u8_id='hls', fatal=False))`
			`mpd_url = manifest_urls.get('mpd')`
			`if mpd_url:`
			`formats.extend(self._extract_mpd_formats(`
			`mpd_url.replace('://das-q1.tf1.fr/', '://das-q1-ssl.tf1.fr/'),`
			`video_id, mpd_id='dash', fatal=False))`
			`self._sort_formats(formats)`
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00
			`date_diffusion = first_chapter.get('date_diffusion') or video_data.get('configv4', {}).get('estatS4')`
			`upload_date = unified_strdate(date_diffusion) if date_diffusion else None`
			`duration = None`
			`files = video_info['files']`
			`if files:`
			`duration = int_or_none(files[0].get('duration'))`
[wat] Add support for SD and HD videos (Closes #3558) 2014-08-23 21:22:10 +02:00
[wat] Modernize 2014-03-29 15:15:16 +01:00			`return {`
[wat] extract all formats 2016-04-22 10:36:14 +02:00			`'id': video_id,`
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00			`'title': title,`
			`'thumbnail': first_chapter.get('preview'),`
			`'description': first_chapter.get('description'),`
			`'view_count': int_or_none(video_info.get('views')),`
[wat] Modernize 2014-03-29 15:15:16 +01:00			`'upload_date': upload_date,`
[wat] improve extraction(#10281) add alternative method to extract http formats works even if the video is geo-restricted or removed from public access(most of the cases) 2016-08-10 15:17:22 +02:00			`'duration': duration,`
[wat] Add support for SD and HD videos (Closes #3558) 2014-08-23 21:22:10 +02:00			`'formats': formats,`
[wat] Modernize 2014-03-29 15:15:16 +01:00			`}`