[videopress] add support for video.wordpress.com

[kakao] improve info extraction and detect geo restriction(closes #26577 )
[xboxclips] fix extraction(closes #27151 )
2021-02-14 22:10:06 +01:00 · 2021-02-14 19:48:26 +01:00 · 2021-02-14 16:22:45 +01:00
3 changed files with 75 additions and 60 deletions
--- a/youtube_dl/extractor/kakao.py
+++ b/youtube_dl/extractor/kakao.py
@ -3,10 +3,13 @@
 from __future__ import unicode_literals

 from .common import InfoExtractor
-from ..compat import compat_str
+from ..compat import compat_HTTPError
 from ..utils import (
+    ExtractorError,
    int_or_none,
+    str_or_none,
    strip_or_none,
+    try_get,
    unified_timestamp,
    update_url_query,
 )
@ -23,7 +26,7 @@ class KakaoIE(InfoExtractor):
            'id': '301965083',
            'ext': 'mp4',
            'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動！顔高低差GPも！」 『乃木坂工事中』',
-            'uploader_id': 2671005,
+            'uploader_id': '2671005',
            'uploader': '그랑그랑이',
            'timestamp': 1488160199,
            'upload_date': '20170227',
@ -36,11 +39,15 @@ class KakaoIE(InfoExtractor):
            'ext': 'mp4',
            'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
            'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
-            'uploader_id': 2653210,
+            'uploader_id': '2653210',
            'uploader': '쇼! 음악중심',
            'timestamp': 1485684628,
            'upload_date': '20170129',
        }
+    }, {
+        # geo restricted
+        'url': 'https://tv.kakao.com/channel/3643855/cliplink/412069491',
+        'only_matching': True,
    }]

    def _real_extract(self, url):
@ -68,8 +75,7 @@ class KakaoIE(InfoExtractor):
            'fields': ','.join([
                '-*', 'tid', 'clipLink', 'displayTitle', 'clip', 'title',
                'description', 'channelId', 'createTime', 'duration', 'playCount',
-                'likeCount', 'commentCount', 'tagList', 'channel', 'name',
-                'clipChapterThumbnailList', 'thumbnailUrl', 'timeInSec', 'isDefault',
+                'likeCount', 'commentCount', 'tagList', 'channel', 'name', 'thumbnailUrl',
                'videoOutputList', 'width', 'height', 'kbps', 'profile', 'label'])
        }

@ -82,24 +88,28 @@ class KakaoIE(InfoExtractor):

        title = clip.get('title') or clip_link.get('displayTitle')

-        query['tid'] = impress.get('tid', '')
+        query.update({
+            'fields': '-*,code,message,url',
+            'tid': impress.get('tid') or '',
+        })

        formats = []
-        for fmt in clip.get('videoOutputList', []):
+        for fmt in (clip.get('videoOutputList') or []):
            try:
                profile_name = fmt['profile']
                if profile_name == 'AUDIO':
                    continue
-                query.update({
-                    'profile': profile_name,
-                    'fields': '-*,url',
-                })
-                fmt_url_json = self._download_json(
-                    api_base + 'raw/videolocation', display_id,
-                    'Downloading video URL for profile %s' % profile_name,
-                    query=query, headers=player_header, fatal=False)
-
-                if fmt_url_json is None:
+                query['profile'] = profile_name
+                try:
+                    fmt_url_json = self._download_json(
+                        api_base + 'raw/videolocation', display_id,
+                        'Downloading video URL for profile %s' % profile_name,
+                        query=query, headers=player_header)
+                except ExtractorError as e:
+                    if isinstance(e.cause, compat_HTTPError) and e.cause.code == 403:
+                        resp = self._parse_json(e.cause.read().decode(), video_id)
+                        if resp.get('code') == 'GeoBlocked':
+                            self.raise_geo_restricted()
                    continue

                fmt_url = fmt_url_json['url']
@ -116,27 +126,13 @@ class KakaoIE(InfoExtractor):
                pass
        self._sort_formats(formats)

-        thumbs = []
-        for thumb in clip.get('clipChapterThumbnailList', []):
-            thumbs.append({
-                'url': thumb.get('thumbnailUrl'),
-                'id': compat_str(thumb.get('timeInSec')),
-                'preference': -1 if thumb.get('isDefault') else 0
-            })
-        top_thumbnail = clip.get('thumbnailUrl')
-        if top_thumbnail:
-            thumbs.append({
-                'url': top_thumbnail,
-                'preference': 10,
-            })
-
        return {
            'id': display_id,
            'title': title,
            'description': strip_or_none(clip.get('description')),
-            'uploader': clip_link.get('channel', {}).get('name'),
-            'uploader_id': clip_link.get('channelId'),
-            'thumbnails': thumbs,
+            'uploader': try_get(clip_link, lambda x: x['channel']['name']),
+            'uploader_id': str_or_none(clip_link.get('channelId')),
+            'thumbnail': clip.get('thumbnailUrl'),
            'timestamp': unified_timestamp(clip_link.get('createTime')),
            'duration': int_or_none(clip.get('duration')),
            'view_count': int_or_none(clip.get('playCount')),
--- a/youtube_dl/extractor/videopress.py
+++ b/youtube_dl/extractor/videopress.py
@ -4,21 +4,22 @@ from __future__ import unicode_literals
 import re

 from .common import InfoExtractor
-from ..compat import compat_str
 from ..utils import (
    determine_ext,
    float_or_none,
+    int_or_none,
    parse_age_limit,
    qualities,
    random_birthday,
-    try_get,
    unified_timestamp,
    urljoin,
 )


 class VideoPressIE(InfoExtractor):
-    _VALID_URL = r'https?://videopress\.com/embed/(?P<id>[\da-zA-Z]+)'
+    _ID_REGEX = r'[\da-zA-Z]{8}'
+    _PATH_REGEX = r'video(?:\.word)?press\.com/embed/'
+    _VALID_URL = r'https?://%s(?P<id>%s)' % (_PATH_REGEX, _ID_REGEX)
    _TESTS = [{
        'url': 'https://videopress.com/embed/kUJmAcSf',
        'md5': '706956a6c875873d51010921310e4bc6',
@ -36,35 +37,36 @@ class VideoPressIE(InfoExtractor):
        # 17+, requires birth_* params
        'url': 'https://videopress.com/embed/iH3gstfZ',
        'only_matching': True,
+    }, {
+        'url': 'https://video.wordpress.com/embed/kUJmAcSf',
+        'only_matching': True,
    }]

    @staticmethod
    def _extract_urls(webpage):
        return re.findall(
-            r'<iframe[^>]+src=["\']((?:https?://)?videopress\.com/embed/[\da-zA-Z]+)',
+            r'<iframe[^>]+src=["\']((?:https?://)?%s%s)' % (VideoPressIE._PATH_REGEX, VideoPressIE._ID_REGEX),
            webpage)

    def _real_extract(self, url):
        video_id = self._match_id(url)

        query = random_birthday('birth_year', 'birth_month', 'birth_day')
+        query['fields'] = 'description,duration,file_url_base,files,height,original,poster,rating,title,upload_date,width'
        video = self._download_json(
            'https://public-api.wordpress.com/rest/v1.1/videos/%s' % video_id,
            video_id, query=query)

        title = video['title']

-        def base_url(scheme):
-            return try_get(
-                video, lambda x: x['file_url_base'][scheme], compat_str)
-
-        base_url = base_url('https') or base_url('http')
+        file_url_base = video.get('file_url_base') or {}
+        base_url = file_url_base.get('https') or file_url_base.get('http')

        QUALITIES = ('std', 'dvd', 'hd')
        quality = qualities(QUALITIES)

        formats = []
-        for format_id, f in video['files'].items():
+        for format_id, f in (video.get('files') or {}).items():
            if not isinstance(f, dict):
                continue
            for ext, path in f.items():
@ -75,12 +77,14 @@ class VideoPressIE(InfoExtractor):
                        'ext': determine_ext(path, ext),
                        'quality': quality(format_id),
                    })
-        original_url = try_get(video, lambda x: x['original'], compat_str)
+        original_url = video.get('original')
        if original_url:
            formats.append({
                'url': original_url,
                'format_id': 'original',
                'quality': len(QUALITIES),
+                'width': int_or_none(video.get('width')),
+                'height': int_or_none(video.get('height')),
            })
        self._sort_formats(formats)

--- a/youtube_dl/extractor/xboxclips.py
+++ b/youtube_dl/extractor/xboxclips.py
@ -1,40 +1,55 @@
 # coding: utf-8
 from __future__ import unicode_literals

+import re
+
 from .common import InfoExtractor
+from ..compat import (
+    compat_parse_qs,
+    compat_urllib_parse_urlparse,
+)
 from ..utils import (
    int_or_none,
+    month_by_abbreviation,
    parse_filesize,
-    unified_strdate,
 )


 class XboxClipsIE(InfoExtractor):
-    _VALID_URL = r'https?://(?:www\.)?xboxclips\.com/(?:video\.php\?.*vid=|[^/]+/)(?P<id>[\w-]{36})'
-    _TEST = {
+    _VALID_URL = r'https?://(?:www\.)?(?:xboxclips\.com|gameclips\.io)/(?:video\.php\?.*vid=|[^/]+/)(?P<id>[\da-f]{8}-(?:[\da-f]{4}-){3}[\da-f]{12})'
+    _TESTS = [{
        'url': 'http://xboxclips.com/video.php?uid=2533274823424419&gamertag=Iabdulelah&vid=074a69a9-5faf-46aa-b93b-9909c1720325',
        'md5': 'fbe1ec805e920aeb8eced3c3e657df5d',
        'info_dict': {
            'id': '074a69a9-5faf-46aa-b93b-9909c1720325',
            'ext': 'mp4',
-            'title': 'Iabdulelah playing Titanfall',
+            'title': 'iAbdulElah playing Titanfall',
            'filesize_approx': 26800000,
            'upload_date': '20140807',
            'duration': 56,
        }
-    }
+    }, {
+        'url': 'https://gameclips.io/iAbdulElah/074a69a9-5faf-46aa-b93b-9909c1720325',
+        'only_matching': True,
+    }]

    def _real_extract(self, url):
        video_id = self._match_id(url)

-        webpage = self._download_webpage(url, video_id)
+        if '/video.php' in url:
+            qs = compat_parse_qs(compat_urllib_parse_urlparse(url).query)
+            url = 'https://gameclips.io/%s/%s' % (qs['gamertag'][0], qs['vid'][0])

-        video_url = self._html_search_regex(
-            r'>(?:Link|Download): <a[^>]+href="([^"]+)"', webpage, 'video URL')
-        title = self._html_search_regex(
-            r'<title>XboxClips \| ([^<]+)</title>', webpage, 'title')
-        upload_date = unified_strdate(self._html_search_regex(
-            r'>Recorded: ([^<]+)<', webpage, 'upload date', fatal=False))
+        webpage = self._download_webpage(url, video_id)
+        info = self._parse_html5_media_entries(url, webpage, video_id)[0]
+
+        title = self._html_search_meta(['og:title', 'twitter:title'], webpage)
+        upload_date = None
+        mobj = re.search(
+            r'>Recorded: (\d{2})-(Jan|Feb|Mar|Apr|May|Ju[nl]|Aug|Sep|Oct|Nov|Dec)-(\d{4})',
+            webpage)
+        if mobj:
+            upload_date = '%s%.2d%s' % (mobj.group(3), month_by_abbreviation(mobj.group(2)), mobj.group(1))
        filesize = parse_filesize(self._html_search_regex(
            r'>Size: ([^<]+)<', webpage, 'file size', fatal=False))
        duration = int_or_none(self._html_search_regex(
@ -42,12 +57,12 @@ class XboxClipsIE(InfoExtractor):
        view_count = int_or_none(self._html_search_regex(
            r'>Views: (\d+)<', webpage, 'view count', fatal=False))

-        return {
+        info.update({
            'id': video_id,
-            'url': video_url,
            'title': title,
            'upload_date': upload_date,
            'filesize_approx': filesize,
            'duration': duration,
            'view_count': view_count,
-        }
+        })
+        return info
Author	SHA1	Message	Date
Remita Amine	be2e9b76ee	[videopress] add support for video.wordpress.com	2021-02-14 22:10:06 +01:00
Remita Amine	d8085580f6	[kakao] improve info extraction and detect geo restriction(closes #26577 )	2021-02-14 19:48:26 +01:00
Remita Amine	6d32c6c6d3	[xboxclips] fix extraction(closes #27151 )	2021-02-14 16:22:45 +01:00