youtube-dl/youtube_dl/extractor/kakao.py

# coding: utf-8

from __future__ import unicode_literals

from .common import InfoExtractor
from ..compat import compat_HTTPError
from ..utils import (
    ExtractorError,
    int_or_none,
    str_or_none,
    strip_or_none,
    try_get,
    unified_timestamp,
    update_url_query,
)


class KakaoIE(InfoExtractor):
    _VALID_URL = r'https?://(?:play-)?tv\.kakao\.com/(?:channel/\d+|embed/player)/cliplink/(?P<id>\d+|[^?#&]+@my)'
    _API_BASE_TMPL = 'http://tv.kakao.com/api/v1/ft/cliplinks/%s/'

    _TESTS = [{
        'url': 'http://tv.kakao.com/channel/2671005/cliplink/301965083',
        'md5': '702b2fbdeb51ad82f5c904e8c0766340',
        'info_dict': {
            'id': '301965083',
            'ext': 'mp4',
            'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動！顔高低差GPも！」 『乃木坂工事中』',
            'uploader_id': '2671005',
            'uploader': '그랑그랑이',
            'timestamp': 1488160199,
            'upload_date': '20170227',
        }
    }, {
        'url': 'http://tv.kakao.com/channel/2653210/cliplink/300103180',
        'md5': 'a8917742069a4dd442516b86e7d66529',
        'info_dict': {
            'id': '300103180',
            'ext': 'mp4',
            'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
            'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
            'uploader_id': '2653210',
            'uploader': '쇼! 음악중심',
            'timestamp': 1485684628,
            'upload_date': '20170129',
        }
    }, {
        # geo restricted
        'url': 'https://tv.kakao.com/channel/3643855/cliplink/412069491',
        'only_matching': True,
    }]

    def _real_extract(self, url):
        video_id = self._match_id(url)
        display_id = video_id.rstrip('@my')
        api_base = self._API_BASE_TMPL % video_id

        player_header = {
            'Referer': update_url_query(
                'http://tv.kakao.com/embed/player/cliplink/%s' % video_id, {
                    'service': 'kakao_tv',
                    'autoplay': '1',
                    'profile': 'HIGH',
                    'wmode': 'transparent',
                })
        }

        query = {
            'player': 'monet_html5',
            'referer': url,
            'uuid': '',
            'service': 'kakao_tv',
            'section': '',
            'dteType': 'PC',
            'fields': ','.join([
                '-*', 'tid', 'clipLink', 'displayTitle', 'clip', 'title',
                'description', 'channelId', 'createTime', 'duration', 'playCount',
                'likeCount', 'commentCount', 'tagList', 'channel', 'name', 'thumbnailUrl',
                'videoOutputList', 'width', 'height', 'kbps', 'profile', 'label'])
        }

        impress = self._download_json(
            api_base + 'impress', display_id, 'Downloading video info',
            query=query, headers=player_header)

        clip_link = impress['clipLink']
        clip = clip_link['clip']

        title = clip.get('title') or clip_link.get('displayTitle')

        query.update({
            'fields': '-*,code,message,url',
            'tid': impress.get('tid') or '',
        })

        formats = []
        for fmt in (clip.get('videoOutputList') or []):
            try:
                profile_name = fmt['profile']
                if profile_name == 'AUDIO':
                    continue
                query['profile'] = profile_name
                try:
                    fmt_url_json = self._download_json(
                        api_base + 'raw/videolocation', display_id,
                        'Downloading video URL for profile %s' % profile_name,
                        query=query, headers=player_header)
                except ExtractorError as e:
                    if isinstance(e.cause, compat_HTTPError) and e.cause.code == 403:
                        resp = self._parse_json(e.cause.read().decode(), video_id)
                        if resp.get('code') == 'GeoBlocked':
                            self.raise_geo_restricted()
                    continue

                fmt_url = fmt_url_json['url']
                formats.append({
                    'url': fmt_url,
                    'format_id': profile_name,
                    'width': int_or_none(fmt.get('width')),
                    'height': int_or_none(fmt.get('height')),
                    'format_note': fmt.get('label'),
                    'filesize': int_or_none(fmt.get('filesize')),
                    'tbr': int_or_none(fmt.get('kbps')),
                })
            except KeyError:
                pass
        self._sort_formats(formats)

        return {
            'id': display_id,
            'title': title,
            'description': strip_or_none(clip.get('description')),
            'uploader': try_get(clip_link, lambda x: x['channel']['name']),
            'uploader_id': str_or_none(clip_link.get('channelId')),
            'thumbnail': clip.get('thumbnailUrl'),
            'timestamp': unified_timestamp(clip_link.get('createTime')),
            'duration': int_or_none(clip.get('duration')),
            'view_count': int_or_none(clip.get('playCount')),
            'like_count': int_or_none(clip.get('likeCount')),
            'comment_count': int_or_none(clip.get('commentCount')),
            'formats': formats,
            'tags': clip.get('tagList'),
        }
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								# coding: utf-8
 								from __future__ import unicode_literals
 								from .common import InfoExtractor
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								from ..compat import compat_HTTPError
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								from ..utils import (
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								    ExtractorError,
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								    int_or_none,
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								    str_or_none,
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								    strip_or_none,
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								    try_get,
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								    unified_timestamp,
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								    update_url_query,
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								)
 								class KakaoIE(InfoExtractor):
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								    _VALID_URL = r'https?://(?:play-)?tv\.kakao\.com/(?:channel/\d+|embed/player)/cliplink/(?P<id>\d+|[^?#&]+@my)'
 								    _API_BASE_TMPL = 'http://tv.kakao.com/api/v1/ft/cliplinks/%s/'
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
 								    _TESTS = [{
 								        'url': 'http://tv.kakao.com/channel/2671005/cliplink/301965083',
 								        'md5': '702b2fbdeb51ad82f5c904e8c0766340',
 								        'info_dict': {
 								            'id': '301965083',
 								            'ext': 'mp4',
 								            'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動！顔高低差GPも！」 『乃木坂工事中』',
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								            'uploader_id': '2671005',
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								            'uploader': '그랑그랑이',
 								            'timestamp': 1488160199,
 								            'upload_date': '20170227',
 								        }
 								    }, {
 								        'url': 'http://tv.kakao.com/channel/2653210/cliplink/300103180',
 								        'md5': 'a8917742069a4dd442516b86e7d66529',
 								        'info_dict': {
 								            'id': '300103180',
 								            'ext': 'mp4',
 								            'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
 								            'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								            'uploader_id': '2653210',
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								            'uploader': '쇼! 음악중심',
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								            'timestamp': 1485684628,
 								            'upload_date': '20170129',
 								        }
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								    }, {
 								        # geo restricted
 								        'url': 'https://tv.kakao.com/channel/3643855/cliplink/412069491',
 								        'only_matching': True,
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								    }]
 								    def _real_extract(self, url):
 								        video_id = self._match_id(url)
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								        display_id = video_id.rstrip('@my')
 								        api_base = self._API_BASE_TMPL % video_id
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								        player_header = {
 								            'Referer': update_url_query(
 								                'http://tv.kakao.com/embed/player/cliplink/%s' % video_id, {
 								                    'service': 'kakao_tv',
 								                    'autoplay': '1',
 								                    'profile': 'HIGH',
 								                    'wmode': 'transparent',
 								                })
 								        }
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								        query = {
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								            'player': 'monet_html5',
 								            'referer': url,
 								            'uuid': '',
 								            'service': 'kakao_tv',
 								            'section': '',
 								            'dteType': 'PC',
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								            'fields': ','.join([
 								                '-*', 'tid', 'clipLink', 'displayTitle', 'clip', 'title',
 								                'description', 'channelId', 'createTime', 'duration', 'playCount',
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								                'likeCount', 'commentCount', 'tagList', 'channel', 'name', 'thumbnailUrl',
-												[kakao] remove raw request and extract format total bitrate

											
										
										
											2019-11-01 12:40:41 +01:00
+								                'videoOutputList', 'width', 'height', 'kbps', 'profile', 'label'])
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								        }
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
 								        impress = self._download_json(
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								            api_base + 'impress', display_id, 'Downloading video info',
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								            query=query, headers=player_header)
 								        clip_link = impress['clipLink']
 								        clip = clip_link['clip']
 								        title = clip.get('title') or clip_link.get('displayTitle')
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								        query.update({
 								            'fields': '-*,code,message,url',
 								            'tid': impress.get('tid') or '',
 								        })
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
 								        formats = []
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								        for fmt in (clip.get('videoOutputList') or []):
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								            try:
 								                profile_name = fmt['profile']
-												[kakao] remove raw request and extract format total bitrate

											
										
										
											2019-11-01 12:40:41 +01:00
+								                if profile_name == 'AUDIO':
 								                    continue
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								                query['profile'] = profile_name
 								                try:
 								                    fmt_url_json = self._download_json(
 								                        api_base + 'raw/videolocation', display_id,
 								                        'Downloading video URL for profile %s' % profile_name,
 								                        query=query, headers=player_header)
 								                except ExtractorError as e:
 								                    if isinstance(e.cause, compat_HTTPError) and e.cause.code == 403:
 								                        resp = self._parse_json(e.cause.read().decode(), video_id)
 								                        if resp.get('code') == 'GeoBlocked':
 								                            self.raise_geo_restricted()
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								                    continue
 								                fmt_url = fmt_url_json['url']
 								                formats.append({
 								                    'url': fmt_url,
 								                    'format_id': profile_name,
 								                    'width': int_or_none(fmt.get('width')),
 								                    'height': int_or_none(fmt.get('height')),
 								                    'format_note': fmt.get('label'),
-												[kakao] remove raw request and extract format total bitrate

											
										
										
											2019-11-01 12:40:41 +01:00
+								                    'filesize': int_or_none(fmt.get('filesize')),
 								                    'tbr': int_or_none(fmt.get('kbps')),
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								                })
 								            except KeyError:
 								                pass
 								        self._sort_formats(formats)
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								        return {
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								            'id': display_id,
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								            'title': title,
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								            'description': strip_or_none(clip.get('description')),
-												[kakao] improve info extraction and detect geo restriction(closes #26577)

											
										
										
											2021-02-14 19:48:26 +01:00
+								            'uploader': try_get(clip_link, lambda x: x['channel']['name']),
 								            'uploader_id': str_or_none(clip_link.get('channelId')),
 								            'thumbnail': clip.get('thumbnailUrl'),
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								            'timestamp': unified_timestamp(clip_link.get('createTime')),
 								            'duration': int_or_none(clip.get('duration')),
 								            'view_count': int_or_none(clip.get('playCount')),
 								            'like_count': int_or_none(clip.get('likeCount')),
 								            'comment_count': int_or_none(clip.get('commentCount')),
 								            'formats': formats,
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								            'tags': clip.get('tagList'),
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								        }