yt-dlp/yt_dlp/extractor/kakao.py

from .common import InfoExtractor
from ..compat import compat_HTTPError
from ..utils import (
    ExtractorError,
    int_or_none,
    strip_or_none,
    str_or_none,
    traverse_obj,
    unified_timestamp,
)


class KakaoIE(InfoExtractor):
    _VALID_URL = r'https?://(?:play-)?tv\.kakao\.com/(?:channel/\d+|embed/player)/cliplink/(?P<id>\d+|[^?#&]+@my)'
    _API_BASE_TMPL = 'http://tv.kakao.com/api/v1/ft/playmeta/cliplink/%s/'
    _CDN_API = 'https://tv.kakao.com/katz/v1/ft/cliplink/%s/readyNplay?'

    _TESTS = [{
        'url': 'http://tv.kakao.com/channel/2671005/cliplink/301965083',
        'md5': '702b2fbdeb51ad82f5c904e8c0766340',
        'info_dict': {
            'id': '301965083',
            'ext': 'mp4',
            'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動！顔高低差GPも！」 『乃木坂工事中』',
            'description': '',
            'uploader_id': '2671005',
            'uploader': '그랑그랑이',
            'timestamp': 1488160199,
            'upload_date': '20170227',
            'like_count': int,
            'thumbnail': r're:http://.+/thumb\.png',
            'tags': ['乃木坂'],
            'view_count': int,
            'duration': 1503,
            'comment_count': int,
        }
    }, {
        'url': 'http://tv.kakao.com/channel/2653210/cliplink/300103180',
        'md5': 'a8917742069a4dd442516b86e7d66529',
        'info_dict': {
            'id': '300103180',
            'ext': 'mp4',
            'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
            'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
            'uploader_id': '2653210',
            'uploader': '쇼! 음악중심',
            'timestamp': 1485684628,
            'upload_date': '20170129',
            'like_count': int,
            'thumbnail': r're:http://.+/thumb\.png',
            'tags': 'count:28',
            'view_count': int,
            'duration': 184,
            'comment_count': int,
        }
    }, {
        # geo restricted
        'url': 'https://tv.kakao.com/channel/3643855/cliplink/412069491',
        'only_matching': True,
    }]

    def _real_extract(self, url):
        video_id = self._match_id(url)
        api_base = self._API_BASE_TMPL % video_id
        cdn_api_base = self._CDN_API % video_id

        query = {
            'player': 'monet_html5',
            'referer': url,
            'uuid': '',
            'service': 'kakao_tv',
            'section': '',
            'dteType': 'PC',
            'fields': ','.join([
                '-*', 'tid', 'clipLink', 'displayTitle', 'clip', 'title',
                'description', 'channelId', 'createTime', 'duration', 'playCount',
                'likeCount', 'commentCount', 'tagList', 'channel', 'name',
                'clipChapterThumbnailList', 'thumbnailUrl', 'timeInSec', 'isDefault',
                'videoOutputList', 'width', 'height', 'kbps', 'profile', 'label'])
        }

        api_json = self._download_json(
            api_base, video_id, 'Downloading video info')

        clip_link = api_json['clipLink']
        clip = clip_link['clip']

        title = clip.get('title') or clip_link.get('displayTitle')

        formats = []
        for fmt in clip.get('videoOutputList') or []:
            profile_name = fmt.get('profile')
            if not profile_name or profile_name == 'AUDIO':
                continue
            query.update({
                'profile': profile_name,
                'fields': '-*,code,message,url',
            })
            try:
                fmt_url_json = self._download_json(
                    cdn_api_base, video_id, query=query,
                    note='Downloading video URL for profile %s' % profile_name)
            except ExtractorError as e:
                if isinstance(e.cause, compat_HTTPError) and e.cause.code == 403:
                    resp = self._parse_json(e.cause.read().decode(), video_id)
                    if resp.get('code') == 'GeoBlocked':
                        self.raise_geo_restricted()

            fmt_url = traverse_obj(fmt_url_json, ('videoLocation', 'url'))
            if not fmt_url:
                continue

            formats.append({
                'url': fmt_url,
                'format_id': profile_name,
                'width': int_or_none(fmt.get('width')),
                'height': int_or_none(fmt.get('height')),
                'format_note': fmt.get('label'),
                'filesize': int_or_none(fmt.get('filesize')),
                'tbr': int_or_none(fmt.get('kbps')),
            })
        self._sort_formats(formats)

        thumbs = []
        for thumb in clip.get('clipChapterThumbnailList') or []:
            thumbs.append({
                'url': thumb.get('thumbnailUrl'),
                'id': str(thumb.get('timeInSec')),
                'preference': -1 if thumb.get('isDefault') else 0
            })
        top_thumbnail = clip.get('thumbnailUrl')
        if top_thumbnail:
            thumbs.append({
                'url': top_thumbnail,
                'preference': 10,
            })

        return {
            'id': video_id,
            'title': title,
            'description': strip_or_none(clip.get('description')),
            'uploader': traverse_obj(clip_link, ('channel', 'name')),
            'uploader_id': str_or_none(clip_link.get('channelId')),
            'thumbnails': thumbs,
            'timestamp': unified_timestamp(clip_link.get('createTime')),
            'duration': int_or_none(clip.get('duration')),
            'view_count': int_or_none(clip.get('playCount')),
            'like_count': int_or_none(clip.get('likeCount')),
            'comment_count': int_or_none(clip.get('commentCount')),
            'formats': formats,
            'tags': clip.get('tagList'),
        }
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								from .common import InfoExtractor
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								from ..compat import compat_HTTPError
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								from ..utils import (
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								    ExtractorError,
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								    int_or_none,
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								    strip_or_none,
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								    str_or_none,
-												[kakao] Fix extractor
Closes #699

											
										
										
											2021-08-15 10:57:44 +02:00
+								    traverse_obj,
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								    unified_timestamp,
 								)
 								class KakaoIE(InfoExtractor):
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								    _VALID_URL = r'https?://(?:play-)?tv\.kakao\.com/(?:channel/\d+|embed/player)/cliplink/(?P<id>\d+|[^?#&]+@my)'
-												[kakao] new apis

there are also ageLimit and GeoBlock attributes provided by api_json if needed
											
										
										
											2020-09-13 12:31:36 +02:00
+								    _API_BASE_TMPL = 'http://tv.kakao.com/api/v1/ft/playmeta/cliplink/%s/'
 								    _CDN_API = 'https://tv.kakao.com/katz/v1/ft/cliplink/%s/readyNplay?'
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
 								    _TESTS = [{
 								        'url': 'http://tv.kakao.com/channel/2671005/cliplink/301965083',
 								        'md5': '702b2fbdeb51ad82f5c904e8c0766340',
 								        'info_dict': {
 								            'id': '301965083',
 								            'ext': 'mp4',
 								            'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動！顔高低差GPも！」 『乃木坂工事中』',
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								            'description': '',
 								            'uploader_id': '2671005',
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								            'uploader': '그랑그랑이',
 								            'timestamp': 1488160199,
 								            'upload_date': '20170227',
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								            'like_count': int,
 								            'thumbnail': r're:http://.+/thumb\.png',
 								            'tags': ['乃木坂'],
 								            'view_count': int,
 								            'duration': 1503,
 								            'comment_count': int,
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								        }
 								    }, {
 								        'url': 'http://tv.kakao.com/channel/2653210/cliplink/300103180',
 								        'md5': 'a8917742069a4dd442516b86e7d66529',
 								        'info_dict': {
 								            'id': '300103180',
 								            'ext': 'mp4',
 								            'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
 								            'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								            'uploader_id': '2653210',
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								            'uploader': '쇼! 음악중심',
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								            'timestamp': 1485684628,
 								            'upload_date': '20170129',
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								            'like_count': int,
 								            'thumbnail': r're:http://.+/thumb\.png',
 								            'tags': 'count:28',
 								            'view_count': int,
 								            'duration': 184,
 								            'comment_count': int,
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								        }
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								    }, {
 								        # geo restricted
 								        'url': 'https://tv.kakao.com/channel/3643855/cliplink/412069491',
 								        'only_matching': True,
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								    }]
 								    def _real_extract(self, url):
 								        video_id = self._match_id(url)
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								        api_base = self._API_BASE_TMPL % video_id
-												[kakao] new apis

there are also ageLimit and GeoBlock attributes provided by api_json if needed
											
										
										
											2020-09-13 12:31:36 +02:00
+								        cdn_api_base = self._CDN_API % video_id
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								        query = {
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								            'player': 'monet_html5',
 								            'referer': url,
 								            'uuid': '',
 								            'service': 'kakao_tv',
 								            'section': '',
 								            'dteType': 'PC',
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								            'fields': ','.join([
 								                '-*', 'tid', 'clipLink', 'displayTitle', 'clip', 'title',
 								                'description', 'channelId', 'createTime', 'duration', 'playCount',
 								                'likeCount', 'commentCount', 'tagList', 'channel', 'name',
-												[kakao] remove raw request and extract format total bitrate

											
										
										
											2019-11-01 12:40:41 +01:00
+								                'clipChapterThumbnailList', 'thumbnailUrl', 'timeInSec', 'isDefault',
 								                'videoOutputList', 'width', 'height', 'kbps', 'profile', 'label'])
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								        }
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
-												[kakao] new apis

there are also ageLimit and GeoBlock attributes provided by api_json if needed
											
										
										
											2020-09-13 12:31:36 +02:00
+								        api_json = self._download_json(
 								            api_base, video_id, 'Downloading video info')
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
-												[kakao] new apis

there are also ageLimit and GeoBlock attributes provided by api_json if needed
											
										
										
											2020-09-13 12:31:36 +02:00
+								        clip_link = api_json['clipLink']
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								        clip = clip_link['clip']
 								        title = clip.get('title') or clip_link.get('displayTitle')
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
 								        formats = []
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								        for fmt in clip.get('videoOutputList') or []:
-												[kakao] Fix extractor
Closes #699

											
										
										
											2021-08-15 10:57:44 +02:00
+								            profile_name = fmt.get('profile')
 								            if not profile_name or profile_name == 'AUDIO':
 								                continue
 								            query.update({
 								                'profile': profile_name,
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								                'fields': '-*,code,message,url',
-												[kakao] Fix extractor
Closes #699

											
										
										
											2021-08-15 10:57:44 +02:00
+								            })
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								            try:
 								                fmt_url_json = self._download_json(
 								                    cdn_api_base, video_id, query=query,
 								                    note='Downloading video URL for profile %s' % profile_name)
 								            except ExtractorError as e:
 								                if isinstance(e.cause, compat_HTTPError) and e.cause.code == 403:
 								                    resp = self._parse_json(e.cause.read().decode(), video_id)
 								                    if resp.get('code') == 'GeoBlocked':
 								                        self.raise_geo_restricted()
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
-												[kakao] Fix extractor
Closes #699

											
										
										
											2021-08-15 10:57:44 +02:00
+								            fmt_url = traverse_obj(fmt_url_json, ('videoLocation', 'url'))
 								            if not fmt_url:
 								                continue
 								            formats.append({
 								                'url': fmt_url,
 								                'format_id': profile_name,
 								                'width': int_or_none(fmt.get('width')),
 								                'height': int_or_none(fmt.get('height')),
 								                'format_note': fmt.get('label'),
 								                'filesize': int_or_none(fmt.get('filesize')),
 								                'tbr': int_or_none(fmt.get('kbps')),
 								            })
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								        self._sort_formats(formats)
 								        thumbs = []
-												[kakao] Fix extractor
Closes #699

											
										
										
											2021-08-15 10:57:44 +02:00
+								        for thumb in clip.get('clipChapterThumbnailList') or []:
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								            thumbs.append({
 								                'url': thumb.get('thumbnailUrl'),
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								                'id': str(thumb.get('timeInSec')),
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
+								                'preference': -1 if thumb.get('isDefault') else 0
 								            })
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								        top_thumbnail = clip.get('thumbnailUrl')
 								        if top_thumbnail:
 								            thumbs.append({
 								                'url': top_thumbnail,
 								                'preference': 10,
 								            })
-												[kakao] Add extractor (closes #12298)

											
										
										
											2017-08-24 04:32:24 +02:00
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								        return {
-												[kakao] new apis

there are also ageLimit and GeoBlock attributes provided by api_json if needed
											
										
										
											2020-09-13 12:31:36 +02:00
+								            'id': video_id,
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								            'title': title,
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								            'description': strip_or_none(clip.get('description')),
-												[kakao] Fix extractor
Closes #699

											
										
										
											2021-08-15 10:57:44 +02:00
+								            'uploader': traverse_obj(clip_link, ('channel', 'name')),
-												[kakao] Detect geo-restriction

Code from: https://github.com/ytdl-org/youtube-dl/commit/d8085580f63ad3b146a31712ff76cf41d5a4558a

											
										
										
											2022-01-11 18:11:12 +01:00
+								            'uploader_id': str_or_none(clip_link.get('channelId')),
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								            'thumbnails': thumbs,
 								            'timestamp': unified_timestamp(clip_link.get('createTime')),
 								            'duration': int_or_none(clip.get('duration')),
 								            'view_count': int_or_none(clip.get('playCount')),
 								            'like_count': int_or_none(clip.get('likeCount')),
 								            'comment_count': int_or_none(clip.get('commentCount')),
 								            'formats': formats,
-												[kakao] improve extraction

- support embed URLs
- support Kakao Legacy vid based embed URLs
- only extract fields used for extraction
- strip description and extract tags

											
										
										
											2019-11-01 11:37:41 +01:00
+								            'tags': clip.get('tagList'),
-												[kakao] Improve (closes #14007)

											
										
										
											2017-09-23 02:25:15 +02:00
+								        }