yt-dlp/yt_dlp/extractor/kakao.py

# coding: utf-8

from __future__ import unicode_literals

from .common import InfoExtractor
from ..compat import compat_str
from ..utils import (
    int_or_none,
    strip_or_none,
    traverse_obj,
    unified_timestamp,
)


class KakaoIE(InfoExtractor):
    _VALID_URL = r'https?://(?:play-)?tv\.kakao\.com/(?:channel/\d+|embed/player)/cliplink/(?P<id>\d+|[^?#&]+@my)'
    _API_BASE_TMPL = 'http://tv.kakao.com/api/v1/ft/playmeta/cliplink/%s/'
    _CDN_API = 'https://tv.kakao.com/katz/v1/ft/cliplink/%s/readyNplay?'

    _TESTS = [{
        'url': 'http://tv.kakao.com/channel/2671005/cliplink/301965083',
        'md5': '702b2fbdeb51ad82f5c904e8c0766340',
        'info_dict': {
            'id': '301965083',
            'ext': 'mp4',
            'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動！顔高低差GPも！」 『乃木坂工事中』',
            'uploader_id': 2671005,
            'uploader': '그랑그랑이',
            'timestamp': 1488160199,
            'upload_date': '20170227',
        }
    }, {
        'url': 'http://tv.kakao.com/channel/2653210/cliplink/300103180',
        'md5': 'a8917742069a4dd442516b86e7d66529',
        'info_dict': {
            'id': '300103180',
            'ext': 'mp4',
            'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
            'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
            'uploader_id': 2653210,
            'uploader': '쇼! 음악중심',
            'timestamp': 1485684628,
            'upload_date': '20170129',
        }
    }]

    def _real_extract(self, url):
        video_id = self._match_id(url)
        api_base = self._API_BASE_TMPL % video_id
        cdn_api_base = self._CDN_API % video_id

        query = {
            'player': 'monet_html5',
            'referer': url,
            'uuid': '',
            'service': 'kakao_tv',
            'section': '',
            'dteType': 'PC',
            'fields': ','.join([
                '-*', 'tid', 'clipLink', 'displayTitle', 'clip', 'title',
                'description', 'channelId', 'createTime', 'duration', 'playCount',
                'likeCount', 'commentCount', 'tagList', 'channel', 'name',
                'clipChapterThumbnailList', 'thumbnailUrl', 'timeInSec', 'isDefault',
                'videoOutputList', 'width', 'height', 'kbps', 'profile', 'label'])
        }

        api_json = self._download_json(
            api_base, video_id, 'Downloading video info')

        clip_link = api_json['clipLink']
        clip = clip_link['clip']

        title = clip.get('title') or clip_link.get('displayTitle')

        formats = []
        for fmt in clip.get('videoOutputList', []):
            profile_name = fmt.get('profile')
            if not profile_name or profile_name == 'AUDIO':
                continue
            query.update({
                'profile': profile_name,
                'fields': '-*,url',
            })

            fmt_url_json = self._download_json(
                cdn_api_base, video_id,
                'Downloading video URL for profile %s' % profile_name,
                query=query, fatal=False)
            fmt_url = traverse_obj(fmt_url_json, ('videoLocation', 'url'))
            if not fmt_url:
                continue

            formats.append({
                'url': fmt_url,
                'format_id': profile_name,
                'width': int_or_none(fmt.get('width')),
                'height': int_or_none(fmt.get('height')),
                'format_note': fmt.get('label'),
                'filesize': int_or_none(fmt.get('filesize')),
                'tbr': int_or_none(fmt.get('kbps')),
            })
        self._sort_formats(formats)

        thumbs = []
        for thumb in clip.get('clipChapterThumbnailList') or []:
            thumbs.append({
                'url': thumb.get('thumbnailUrl'),
                'id': compat_str(thumb.get('timeInSec')),
                'preference': -1 if thumb.get('isDefault') else 0
            })
        top_thumbnail = clip.get('thumbnailUrl')
        if top_thumbnail:
            thumbs.append({
                'url': top_thumbnail,
                'preference': 10,
            })

        return {
            'id': video_id,
            'title': title,
            'description': strip_or_none(clip.get('description')),
            'uploader': traverse_obj(clip_link, ('channel', 'name')),
            'uploader_id': clip_link.get('channelId'),
            'thumbnails': thumbs,
            'timestamp': unified_timestamp(clip_link.get('createTime')),
            'duration': int_or_none(clip.get('duration')),
            'view_count': int_or_none(clip.get('playCount')),
            'like_count': int_or_none(clip.get('likeCount')),
            'comment_count': int_or_none(clip.get('commentCount')),
            'formats': formats,
            'tags': clip.get('tagList'),
        }
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
+								# coding: utf-8
 								from __future__ import unicode_literals
 								from .common import InfoExtractor
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
+								from ..compat import compat_str
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
+								from ..utils import (
 								    int_or_none,
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											5 years ago
+								    strip_or_none,
-												[kakao] Fix extractor
Closes #699

											
										
										
											3 years ago
+								    traverse_obj,
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
+								    unified_timestamp,
 								)
 								class KakaoIE(InfoExtractor):
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											5 years ago
+								    _VALID_URL = r'https?://(?:play-)?tv\.kakao\.com/(?:channel/\d+|embed/player)/cliplink/(?P<id>\d+|[^?#&]+@my)'
-												[kakao] new apis

there are also ageLimit and GeoBlock attributes provided by api_json if needed
											
										
										
											4 years ago
+								    _API_BASE_TMPL = 'http://tv.kakao.com/api/v1/ft/playmeta/cliplink/%s/'
 								    _CDN_API = 'https://tv.kakao.com/katz/v1/ft/cliplink/%s/readyNplay?'
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
 								    _TESTS = [{
 								        'url': 'http://tv.kakao.com/channel/2671005/cliplink/301965083',
 								        'md5': '702b2fbdeb51ad82f5c904e8c0766340',
 								        'info_dict': {
 								            'id': '301965083',
 								            'ext': 'mp4',
 								            'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動！顔高低差GPも！」 『乃木坂工事中』',
 								            'uploader_id': 2671005,
 								            'uploader': '그랑그랑이',
 								            'timestamp': 1488160199,
 								            'upload_date': '20170227',
 								        }
 								    }, {
 								        'url': 'http://tv.kakao.com/channel/2653210/cliplink/300103180',
 								        'md5': 'a8917742069a4dd442516b86e7d66529',
 								        'info_dict': {
 								            'id': '300103180',
 								            'ext': 'mp4',
 								            'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
 								            'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
 								            'uploader_id': 2653210,
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											5 years ago
+								            'uploader': '쇼! 음악중심',
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
+								            'timestamp': 1485684628,
 								            'upload_date': '20170129',
 								        }
 								    }]
 								    def _real_extract(self, url):
 								        video_id = self._match_id(url)
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											5 years ago
+								        api_base = self._API_BASE_TMPL % video_id
-												[kakao] new apis

there are also ageLimit and GeoBlock attributes provided by api_json if needed
											
										
										
											4 years ago
+								        cdn_api_base = self._CDN_API % video_id
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											5 years ago
+								        query = {
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
+								            'player': 'monet_html5',
 								            'referer': url,
 								            'uuid': '',
 								            'service': 'kakao_tv',
 								            'section': '',
 								            'dteType': 'PC',
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											5 years ago
+								            'fields': ','.join([
 								                '-*', 'tid', 'clipLink', 'displayTitle', 'clip', 'title',
 								                'description', 'channelId', 'createTime', 'duration', 'playCount',
 								                'likeCount', 'commentCount', 'tagList', 'channel', 'name',
-												[kakao] remove raw request and extract format total bitrate

											
										
										
											5 years ago
+								                'clipChapterThumbnailList', 'thumbnailUrl', 'timeInSec', 'isDefault',
 								                'videoOutputList', 'width', 'height', 'kbps', 'profile', 'label'])
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
+								        }
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
-												[kakao] new apis

there are also ageLimit and GeoBlock attributes provided by api_json if needed
											
										
										
											4 years ago
+								        api_json = self._download_json(
 								            api_base, video_id, 'Downloading video info')
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
-												[kakao] new apis

there are also ageLimit and GeoBlock attributes provided by api_json if needed
											
										
										
											4 years ago
+								        clip_link = api_json['clipLink']
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
+								        clip = clip_link['clip']
 								        title = clip.get('title') or clip_link.get('displayTitle')
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
 								        formats = []
-												[kakao] remove raw request and extract format total bitrate

											
										
										
											5 years ago
+								        for fmt in clip.get('videoOutputList', []):
-												[kakao] Fix extractor
Closes #699

											
										
										
											3 years ago
+								            profile_name = fmt.get('profile')
 								            if not profile_name or profile_name == 'AUDIO':
 								                continue
 								            query.update({
 								                'profile': profile_name,
 								                'fields': '-*,url',
 								            })
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
-												[kakao] Fix extractor
Closes #699

											
										
										
											3 years ago
+								            fmt_url_json = self._download_json(
 								                cdn_api_base, video_id,
 								                'Downloading video URL for profile %s' % profile_name,
 								                query=query, fatal=False)
 								            fmt_url = traverse_obj(fmt_url_json, ('videoLocation', 'url'))
 								            if not fmt_url:
 								                continue
 								            formats.append({
 								                'url': fmt_url,
 								                'format_id': profile_name,
 								                'width': int_or_none(fmt.get('width')),
 								                'height': int_or_none(fmt.get('height')),
 								                'format_note': fmt.get('label'),
 								                'filesize': int_or_none(fmt.get('filesize')),
 								                'tbr': int_or_none(fmt.get('kbps')),
 								            })
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
+								        self._sort_formats(formats)
 								        thumbs = []
-												[kakao] Fix extractor
Closes #699

											
										
										
											3 years ago
+								        for thumb in clip.get('clipChapterThumbnailList') or []:
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
+								            thumbs.append({
 								                'url': thumb.get('thumbnailUrl'),
 								                'id': compat_str(thumb.get('timeInSec')),
 								                'preference': -1 if thumb.get('isDefault') else 0
 								            })
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
+								        top_thumbnail = clip.get('thumbnailUrl')
 								        if top_thumbnail:
 								            thumbs.append({
 								                'url': top_thumbnail,
 								                'preference': 10,
 								            })
-												[kakao] Add extractor (closes #12298)

											
										
										
											7 years ago
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
+								        return {
-												[kakao] new apis

there are also ageLimit and GeoBlock attributes provided by api_json if needed
											
										
										
											4 years ago
+								            'id': video_id,
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
+								            'title': title,
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											5 years ago
+								            'description': strip_or_none(clip.get('description')),
-												[kakao] Fix extractor
Closes #699

											
										
										
											3 years ago
+								            'uploader': traverse_obj(clip_link, ('channel', 'name')),
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
+								            'uploader_id': clip_link.get('channelId'),
 								            'thumbnails': thumbs,
 								            'timestamp': unified_timestamp(clip_link.get('createTime')),
 								            'duration': int_or_none(clip.get('duration')),
 								            'view_count': int_or_none(clip.get('playCount')),
 								            'like_count': int_or_none(clip.get('likeCount')),
 								            'comment_count': int_or_none(clip.get('commentCount')),
 								            'formats': formats,
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											5 years ago
+								            'tags': clip.get('tagList'),
-												[kakao] Improve (closes #14007)

											
										
										
											7 years ago
+								        }