Przeglądaj źródła

[kakao] improve info extraction and detect geo restriction(closes #26577)

Remita Amine 4 lat temu
rodzic
commit
d8085580f6
1 zmienionych plików z 30 dodań i 34 usunięć
  1. 30 34
      youtube_dl/extractor/kakao.py

+ 30 - 34
youtube_dl/extractor/kakao.py

@@ -3,10 +3,13 @@
 from __future__ import unicode_literals
 from __future__ import unicode_literals
 
 
 from .common import InfoExtractor
 from .common import InfoExtractor
-from ..compat import compat_str
+from ..compat import compat_HTTPError
 from ..utils import (
 from ..utils import (
+    ExtractorError,
     int_or_none,
     int_or_none,
+    str_or_none,
     strip_or_none,
     strip_or_none,
+    try_get,
     unified_timestamp,
     unified_timestamp,
     update_url_query,
     update_url_query,
 )
 )
@@ -23,7 +26,7 @@ class KakaoIE(InfoExtractor):
             'id': '301965083',
             'id': '301965083',
             'ext': 'mp4',
             'ext': 'mp4',
             'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動!顔高低差GPも!」 『乃木坂工事中』',
             'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動!顔高低差GPも!」 『乃木坂工事中』',
-            'uploader_id': 2671005,
+            'uploader_id': '2671005',
             'uploader': '그랑그랑이',
             'uploader': '그랑그랑이',
             'timestamp': 1488160199,
             'timestamp': 1488160199,
             'upload_date': '20170227',
             'upload_date': '20170227',
@@ -36,11 +39,15 @@ class KakaoIE(InfoExtractor):
             'ext': 'mp4',
             'ext': 'mp4',
             'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
             'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
             'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
             'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
-            'uploader_id': 2653210,
+            'uploader_id': '2653210',
             'uploader': '쇼! 음악중심',
             'uploader': '쇼! 음악중심',
             'timestamp': 1485684628,
             'timestamp': 1485684628,
             'upload_date': '20170129',
             'upload_date': '20170129',
         }
         }
+    }, {
+        # geo restricted
+        'url': 'https://tv.kakao.com/channel/3643855/cliplink/412069491',
+        'only_matching': True,
     }]
     }]
 
 
     def _real_extract(self, url):
     def _real_extract(self, url):
@@ -68,8 +75,7 @@ class KakaoIE(InfoExtractor):
             'fields': ','.join([
             'fields': ','.join([
                 '-*', 'tid', 'clipLink', 'displayTitle', 'clip', 'title',
                 '-*', 'tid', 'clipLink', 'displayTitle', 'clip', 'title',
                 'description', 'channelId', 'createTime', 'duration', 'playCount',
                 'description', 'channelId', 'createTime', 'duration', 'playCount',
-                'likeCount', 'commentCount', 'tagList', 'channel', 'name',
-                'clipChapterThumbnailList', 'thumbnailUrl', 'timeInSec', 'isDefault',
+                'likeCount', 'commentCount', 'tagList', 'channel', 'name', 'thumbnailUrl',
                 'videoOutputList', 'width', 'height', 'kbps', 'profile', 'label'])
                 'videoOutputList', 'width', 'height', 'kbps', 'profile', 'label'])
         }
         }
 
 
@@ -82,24 +88,28 @@ class KakaoIE(InfoExtractor):
 
 
         title = clip.get('title') or clip_link.get('displayTitle')
         title = clip.get('title') or clip_link.get('displayTitle')
 
 
-        query['tid'] = impress.get('tid', '')
+        query.update({
+            'fields': '-*,code,message,url',
+            'tid': impress.get('tid') or '',
+        })
 
 
         formats = []
         formats = []
-        for fmt in clip.get('videoOutputList', []):
+        for fmt in (clip.get('videoOutputList') or []):
             try:
             try:
                 profile_name = fmt['profile']
                 profile_name = fmt['profile']
                 if profile_name == 'AUDIO':
                 if profile_name == 'AUDIO':
                     continue
                     continue
-                query.update({
-                    'profile': profile_name,
-                    'fields': '-*,url',
-                })
-                fmt_url_json = self._download_json(
-                    api_base + 'raw/videolocation', display_id,
-                    'Downloading video URL for profile %s' % profile_name,
-                    query=query, headers=player_header, fatal=False)
-
-                if fmt_url_json is None:
+                query['profile'] = profile_name
+                try:
+                    fmt_url_json = self._download_json(
+                        api_base + 'raw/videolocation', display_id,
+                        'Downloading video URL for profile %s' % profile_name,
+                        query=query, headers=player_header)
+                except ExtractorError as e:
+                    if isinstance(e.cause, compat_HTTPError) and e.cause.code == 403:
+                        resp = self._parse_json(e.cause.read().decode(), video_id)
+                        if resp.get('code') == 'GeoBlocked':
+                            self.raise_geo_restricted()
                     continue
                     continue
 
 
                 fmt_url = fmt_url_json['url']
                 fmt_url = fmt_url_json['url']
@@ -116,27 +126,13 @@ class KakaoIE(InfoExtractor):
                 pass
                 pass
         self._sort_formats(formats)
         self._sort_formats(formats)
 
 
-        thumbs = []
-        for thumb in clip.get('clipChapterThumbnailList', []):
-            thumbs.append({
-                'url': thumb.get('thumbnailUrl'),
-                'id': compat_str(thumb.get('timeInSec')),
-                'preference': -1 if thumb.get('isDefault') else 0
-            })
-        top_thumbnail = clip.get('thumbnailUrl')
-        if top_thumbnail:
-            thumbs.append({
-                'url': top_thumbnail,
-                'preference': 10,
-            })
-
         return {
         return {
             'id': display_id,
             'id': display_id,
             'title': title,
             'title': title,
             'description': strip_or_none(clip.get('description')),
             'description': strip_or_none(clip.get('description')),
-            'uploader': clip_link.get('channel', {}).get('name'),
-            'uploader_id': clip_link.get('channelId'),
-            'thumbnails': thumbs,
+            'uploader': try_get(clip_link, lambda x: x['channel']['name']),
+            'uploader_id': str_or_none(clip_link.get('channelId')),
+            'thumbnail': clip.get('thumbnailUrl'),
             'timestamp': unified_timestamp(clip_link.get('createTime')),
             'timestamp': unified_timestamp(clip_link.get('createTime')),
             'duration': int_or_none(clip.get('duration')),
             'duration': int_or_none(clip.get('duration')),
             'view_count': int_or_none(clip.get('playCount')),
             'view_count': int_or_none(clip.get('playCount')),