yt-dlp/youtube_dl/extractor/bloomberg.py

from __future__ import unicode_literals

import re

from .common import InfoExtractor


class BloombergIE(InfoExtractor):
    _VALID_URL = r'https?://(?:www\.)?bloomberg\.com/(?:[^/]+/)*(?P<id>[^/?#]+)'

    _TESTS = [{
        'url': 'http://www.bloomberg.com/news/videos/b/aaeae121-5949-481e-a1ce-4562db6f5df2',
        # The md5 checksum changes
        'info_dict': {
            'id': 'qurhIVlJSB6hzkVi229d8g',
            'ext': 'flv',
            'title': 'Shah\'s Presentation on Foreign-Exchange Strategies',
            'description': 'md5:a8ba0302912d03d246979735c17d2761',
        },
    }, {
        'url': 'http://www.bloomberg.com/news/articles/2015-11-12/five-strange-things-that-have-been-happening-in-financial-markets',
        'only_matching': True,
    }, {
        'url': 'http://www.bloomberg.com/politics/videos/2015-11-25/karl-rove-on-jeb-bush-s-struggles-stopping-trump',
        'only_matching': True,
    }]

    def _real_extract(self, url):
        name = self._match_id(url)
        webpage = self._download_webpage(url, name)
        video_id = self._search_regex(
            r'["\']bmmrId["\']\s*:\s*(["\'])(?P<url>.+?)\1',
            webpage, 'id', group='url')
        title = re.sub(': Video$', '', self._og_search_title(webpage))

        embed_info = self._download_json(
            'http://www.bloomberg.com/api/embed?id=%s' % video_id, video_id)
        formats = []
        for stream in embed_info['streams']:
            if stream['muxing_format'] == 'TS':
                formats.extend(self._extract_m3u8_formats(stream['url'], video_id))
            else:
                formats.extend(self._extract_f4m_formats(stream['url'], video_id))
        self._sort_formats(formats)

        return {
            'id': video_id,
            'title': title,
            'formats': formats,
            'description': self._og_search_description(webpage),
            'thumbnail': self._og_search_thumbnail(webpage),
        }
[bloomberg] Fix extraction (fixes #2154) Stop using the OoyalaIE, extract the f4m url instead. 2014-03-29 11:55:12 +01:00			`from __future__ import unicode_literals`

Add an extractor for Bloomberg (closes #1436) 2013-09-16 19:39:39 +02:00			`import re`

			`from .common import InfoExtractor`


			`class BloombergIE(InfoExtractor):`
[bloomberg] Relax _VALID_URL even more (Closes #7685) 2015-11-28 17:39:36 +01:00			`_VALID_URL = r'https?://(?:www\.)?bloomberg\.com/(?:[^/]+/)*(?P<id>[^/?#]+)'`
Add an extractor for Bloomberg (closes #1436) 2013-09-16 19:39:39 +02:00
[bloomberg] Reax _VALID_URL (Closes #7546) 2015-11-19 17:55:06 +01:00			`_TESTS = [{`
[bloomberg] Adapt to website changes (fixes #5347) 2015-04-03 15:01:17 +02:00			`'url': 'http://www.bloomberg.com/news/videos/b/aaeae121-5949-481e-a1ce-4562db6f5df2',`
[bloomberg] Extract the available formats (closes #2776) It uses a helper method in the InfoExtractor class. The downloader will pick the requested formats using the bitrate in the info dict. 2014-07-28 15:25:56 +02:00			`# The md5 checksum changes`
[bloomberg] Fix extraction (fixes #2154) Stop using the OoyalaIE, extract the f4m url instead. 2014-03-29 11:55:12 +01:00			`'info_dict': {`
			`'id': 'qurhIVlJSB6hzkVi229d8g',`
			`'ext': 'flv',`
			`'title': 'Shah\'s Presentation on Foreign-Exchange Strategies',`
[bloomberg] Adapt to website changes (fixes #5347) 2015-04-03 15:01:17 +02:00			`'description': 'md5:a8ba0302912d03d246979735c17d2761',`
Add an extractor for Bloomberg (closes #1436) 2013-09-16 19:39:39 +02:00			`},`
[bloomberg] Reax _VALID_URL (Closes #7546) 2015-11-19 17:55:06 +01:00			`}, {`
			`'url': 'http://www.bloomberg.com/news/articles/2015-11-12/five-strange-things-that-have-been-happening-in-financial-markets',`
			`'only_matching': True,`
[bloomberg] Relax _VALID_URL even more (Closes #7685) 2015-11-28 17:39:36 +01:00			`}, {`
			`'url': 'http://www.bloomberg.com/politics/videos/2015-11-25/karl-rove-on-jeb-bush-s-struggles-stopping-trump',`
			`'only_matching': True,`
[bloomberg] Reax _VALID_URL (Closes #7546) 2015-11-19 17:55:06 +01:00			`}]`
Add an extractor for Bloomberg (closes #1436) 2013-09-16 19:39:39 +02:00
			`def _real_extract(self, url):`
[bloomberg] Modernize 2015-02-24 11:08:00 +01:00			`name = self._match_id(url)`
Add an extractor for Bloomberg (closes #1436) 2013-09-16 19:39:39 +02:00			`webpage = self._download_webpage(url, name)`
[bloomberg] Improve video id regex 2015-11-28 17:41:39 +01:00			`video_id = self._search_regex(`
			`r'["\']bmmrId["\']\s:\s(["\'])(?P<url>.+?)\1',`
			`webpage, 'id', group='url')`
[bloomberg] Fix extraction (fixes #2154) Stop using the OoyalaIE, extract the f4m url instead. 2014-03-29 11:55:12 +01:00			`title = re.sub(': Video$', '', self._og_search_title(webpage))`

[bloomberg] Adapt to website changes (fixes #5347) 2015-04-03 15:01:17 +02:00			`embed_info = self._download_json(`
			`'http://www.bloomberg.com/api/embed?id=%s' % video_id, video_id)`
			`formats = []`
			`for stream in embed_info['streams']:`
[bloomberg] Modernize 2015-11-28 17:40:29 +01:00			`if stream['muxing_format'] == 'TS':`
[bloomberg] Adapt to website changes (fixes #5347) 2015-04-03 15:01:17 +02:00			`formats.extend(self._extract_m3u8_formats(stream['url'], video_id))`
			`else:`
			`formats.extend(self._extract_f4m_formats(stream['url'], video_id))`
			`self._sort_formats(formats)`

[bloomberg] Fix extraction (fixes #2154) Stop using the OoyalaIE, extract the f4m url instead. 2014-03-29 11:55:12 +01:00			`return {`
[bloomberg] Adapt to website changes (fixes #5347) 2015-04-03 15:01:17 +02:00			`'id': video_id,`
[bloomberg] Fix extraction (fixes #2154) Stop using the OoyalaIE, extract the f4m url instead. 2014-03-29 11:55:12 +01:00			`'title': title,`
[bloomberg] Adapt to website changes (fixes #5347) 2015-04-03 15:01:17 +02:00			`'formats': formats,`
[bloomberg] Fix extraction (fixes #2154) Stop using the OoyalaIE, extract the f4m url instead. 2014-03-29 11:55:12 +01:00			`'description': self._og_search_description(webpage),`
			`'thumbnail': self._og_search_thumbnail(webpage),`
			`}`