yt-dlp/yt_dlp/extractor/vk.py

# coding: utf-8
from __future__ import unicode_literals

import collections
import functools
import re

from .common import InfoExtractor
from ..compat import compat_urlparse
from ..utils import (
    clean_html,
    ExtractorError,
    get_element_by_class,
    int_or_none,
    OnDemandPagedList,
    orderedSet,
    str_or_none,
    str_to_int,
    unescapeHTML,
    unified_timestamp,
    url_or_none,
    urlencode_postdata,
)
from .dailymotion import DailymotionIE
from .odnoklassniki import OdnoklassnikiIE
from .pladform import PladformIE
from .vimeo import VimeoIE
from .youtube import YoutubeIE


class VKBaseIE(InfoExtractor):
    _NETRC_MACHINE = 'vk'

    def _login(self):
        username, password = self._get_login_info()
        if username is None:
            return

        login_page, url_handle = self._download_webpage_handle(
            'https://vk.com', None, 'Downloading login page')

        login_form = self._hidden_inputs(login_page)

        login_form.update({
            'email': username.encode('cp1251'),
            'pass': password.encode('cp1251'),
        })

        # vk serves two same remixlhk cookies in Set-Cookie header and expects
        # first one to be actually set
        self._apply_first_set_cookie_header(url_handle, 'remixlhk')

        login_page = self._download_webpage(
            'https://login.vk.com/?act=login', None,
            note='Logging in',
            data=urlencode_postdata(login_form))

        if re.search(r'onLoginFailed', login_page):
            raise ExtractorError(
                'Unable to login, incorrect username and/or password', expected=True)

    def _real_initialize(self):
        self._login()

    def _download_payload(self, path, video_id, data, fatal=True):
        data['al'] = 1
        code, payload = self._download_json(
            'https://vk.com/%s.php' % path, video_id,
            data=urlencode_postdata(data), fatal=fatal,
            headers={'X-Requested-With': 'XMLHttpRequest'})['payload']
        if code == '3':
            self.raise_login_required()
        elif code == '8':
            raise ExtractorError(clean_html(payload[0][1:-1]), expected=True)
        return payload


class VKIE(VKBaseIE):
    IE_NAME = 'vk'
    IE_DESC = 'VK'
    _VALID_URL = r'''(?x)
                    https?://
                        (?:
                            (?:
                                (?:(?:m|new)\.)?vk\.com/video_|
                                (?:www\.)?daxab.com/
                            )
                            ext\.php\?(?P<embed_query>.*?\boid=(?P<oid>-?\d+).*?\bid=(?P<id>\d+).*)|
                            (?:
                                (?:(?:m|new)\.)?vk\.com/(?:.+?\?.*?z=)?video|
                                (?:www\.)?daxab.com/embed/
                            )
                            (?P<videoid>-?\d+_\d+)(?:.*\blist=(?P<list_id>[\da-f]+))?
                        )
                    '''
    _TESTS = [
        {
            'url': 'http://vk.com/videos-77521?z=video-77521_162222515%2Fclub77521',
            'md5': '7babad3b85ea2e91948005b1b8b0cb84',
            'info_dict': {
                'id': '-77521_162222515',
                'ext': 'mp4',
                'title': 'ProtivoGunz - Хуёвая песня',
                'uploader': 're:(?:Noize MC|Alexander Ilyashenko).*',
                'uploader_id': '-77521',
                'duration': 195,
                'timestamp': 1329049880,
                'upload_date': '20120212',
            },
        },
        {
            'url': 'http://vk.com/video205387401_165548505',
            'info_dict': {
                'id': '205387401_165548505',
                'ext': 'mp4',
                'title': 'No name',
                'uploader': 'Tom Cruise',
                'uploader_id': '205387401',
                'duration': 9,
                'timestamp': 1374364108,
                'upload_date': '20130720',
            }
        },
        {
            'note': 'Embedded video',
            'url': 'https://vk.com/video_ext.php?oid=-77521&id=162222515&hash=87b046504ccd8bfa',
            'md5': '7babad3b85ea2e91948005b1b8b0cb84',
            'info_dict': {
                'id': '-77521_162222515',
                'ext': 'mp4',
                'uploader': 're:(?:Noize MC|Alexander Ilyashenko).*',
                'title': 'ProtivoGunz - Хуёвая песня',
                'duration': 195,
                'upload_date': '20120212',
                'timestamp': 1329049880,
                'uploader_id': '-77521',
            },
        },
        {
            # VIDEO NOW REMOVED
            # please update if you find a video whose URL follows the same pattern
            'url': 'http://vk.com/video-8871596_164049491',
            'md5': 'a590bcaf3d543576c9bd162812387666',
            'note': 'Only available for registered users',
            'info_dict': {
                'id': '-8871596_164049491',
                'ext': 'mp4',
                'uploader': 'Триллеры',
                'title': '► Бойцовский клуб / Fight Club 1999 [HD 720]',
                'duration': 8352,
                'upload_date': '20121218',
                'view_count': int,
            },
            'skip': 'Removed',
        },
        {
            'url': 'http://vk.com/hd_kino_mania?z=video-43215063_168067957%2F15c66b9b533119788d',
            'info_dict': {
                'id': '-43215063_168067957',
                'ext': 'mp4',
                'uploader': 'Bro Mazter',
                'title': ' ',
                'duration': 7291,
                'upload_date': '20140328',
                'uploader_id': '223413403',
                'timestamp': 1396018030,
            },
            'skip': 'Requires vk account credentials',
        },
        {
            'url': 'http://m.vk.com/video-43215063_169084319?list=125c627d1aa1cebb83&from=wall-43215063_2566540',
            'md5': '0c45586baa71b7cb1d0784ee3f4e00a6',
            'note': 'ivi.ru embed',
            'info_dict': {
                'id': '-43215063_169084319',
                'ext': 'mp4',
                'title': 'Книга Илая',
                'duration': 6771,
                'upload_date': '20140626',
                'view_count': int,
            },
            'skip': 'Removed',
        },
        {
            # video (removed?) only available with list id
            'url': 'https://vk.com/video30481095_171201961?list=8764ae2d21f14088d4',
            'md5': '091287af5402239a1051c37ec7b92913',
            'info_dict': {
                'id': '30481095_171201961',
                'ext': 'mp4',
                'title': 'ТюменцевВВ_09.07.2015',
                'uploader': 'Anton Ivanov',
                'duration': 109,
                'upload_date': '20150709',
                'view_count': int,
            },
            'skip': 'Removed',
        },
        {
            # youtube embed
            'url': 'https://vk.com/video276849682_170681728',
            'info_dict': {
                'id': 'V3K4mi0SYkc',
                'ext': 'mp4',
                'title': "DSWD Awards 'Children's Joy Foundation, Inc.' Certificate of Registration and License to Operate",
                'description': 'md5:bf9c26cfa4acdfb146362682edd3827a',
                'duration': 178,
                'upload_date': '20130116',
                'uploader': "Children's Joy Foundation Inc.",
                'uploader_id': 'thecjf',
                'view_count': int,
            },
        },
        {
            # dailymotion embed
            'url': 'https://vk.com/video-37468416_456239855',
            'info_dict': {
                'id': 'k3lz2cmXyRuJQSjGHUv',
                'ext': 'mp4',
                'title': 'md5:d52606645c20b0ddbb21655adaa4f56f',
                'description': 'md5:424b8e88cc873217f520e582ba28bb36',
                'uploader': 'AniLibria.Tv',
                'upload_date': '20160914',
                'uploader_id': 'x1p5vl5',
                'timestamp': 1473877246,
            },
            'params': {
                'skip_download': True,
            },
        },
        {
            # video key is extra_data not url\d+
            'url': 'http://vk.com/video-110305615_171782105',
            'md5': 'e13fcda136f99764872e739d13fac1d1',
            'info_dict': {
                'id': '-110305615_171782105',
                'ext': 'mp4',
                'title': 'S-Dance, репетиции к The way show',
                'uploader': 'THE WAY SHOW | 17 апреля',
                'uploader_id': '-110305615',
                'timestamp': 1454859345,
                'upload_date': '20160207',
            },
            'params': {
                'skip_download': True,
            },
        },
        {
            # finished live stream, postlive_mp4
            'url': 'https://vk.com/videos-387766?z=video-387766_456242764%2Fpl_-387766_-2',
            'info_dict': {
                'id': '-387766_456242764',
                'ext': 'mp4',
                'title': 'ИгроМир 2016 День 1 — Игромания Утром',
                'uploader': 'Игромания',
                'duration': 5239,
                # TODO: use act=show to extract view_count
                # 'view_count': int,
                'upload_date': '20160929',
                'uploader_id': '-387766',
                'timestamp': 1475137527,
            },
            'params': {
                'skip_download': True,
            },
        },
        {
            # live stream, hls and rtmp links, most likely already finished live
            # stream by the time you are reading this comment
            'url': 'https://vk.com/video-140332_456239111',
            'only_matching': True,
        },
        {
            # removed video, just testing that we match the pattern
            'url': 'http://vk.com/feed?z=video-43215063_166094326%2Fbb50cacd3177146d7a',
            'only_matching': True,
        },
        {
            # age restricted video, requires vk account credentials
            'url': 'https://vk.com/video205387401_164765225',
            'only_matching': True,
        },
        {
            # pladform embed
            'url': 'https://vk.com/video-76116461_171554880',
            'only_matching': True,
        },
        {
            'url': 'http://new.vk.com/video205387401_165548505',
            'only_matching': True,
        },
        {
            # This video is no longer available, because its author has been blocked.
            'url': 'https://vk.com/video-10639516_456240611',
            'only_matching': True,
        },
        {
            # The video is not available in your region.
            'url': 'https://vk.com/video-51812607_171445436',
            'only_matching': True,
        }]

    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
        video_id = mobj.group('videoid')

        mv_data = {}
        if video_id:
            data = {
                'act': 'show_inline',
                'video': video_id,
            }
            # Some videos (removed?) can only be downloaded with list id specified
            list_id = mobj.group('list_id')
            if list_id:
                data['list'] = list_id

            payload = self._download_payload('al_video', video_id, data)
            info_page = payload[1]
            opts = payload[-1]
            mv_data = opts.get('mvData') or {}
            player = opts.get('player') or {}
        else:
            video_id = '%s_%s' % (mobj.group('oid'), mobj.group('id'))

            info_page = self._download_webpage(
                'http://vk.com/video_ext.php?' + mobj.group('embed_query'), video_id)

            error_message = self._html_search_regex(
                [r'(?s)<!><div[^>]+class="video_layer_message"[^>]*>(.+?)</div>',
                    r'(?s)<div[^>]+id="video_ext_msg"[^>]*>(.+?)</div>'],
                info_page, 'error message', default=None)
            if error_message:
                raise ExtractorError(error_message, expected=True)

            if re.search(r'<!>/login\.php\?.*\bact=security_check', info_page):
                raise ExtractorError(
                    'You are trying to log in from an unusual location. You should confirm ownership at vk.com to log in with this IP.',
                    expected=True)

            ERROR_COPYRIGHT = 'Video %s has been removed from public access due to rightholder complaint.'

            ERRORS = {
                r'>Видеозапись .*? была изъята из публичного доступа в связи с обращением правообладателя.<':
                ERROR_COPYRIGHT,

                r'>The video .*? was removed from public access by request of the copyright holder.<':
                ERROR_COPYRIGHT,

                r'<!>Please log in or <':
                'Video %s is only available for registered users, '
                'use --username and --password options to provide account credentials.',

                r'<!>Unknown error':
                'Video %s does not exist.',

                r'<!>Видео временно недоступно':
                'Video %s is temporarily unavailable.',

                r'<!>Access denied':
                'Access denied to video %s.',

                r'<!>Видеозапись недоступна, так как её автор был заблокирован.':
                'Video %s is no longer available, because its author has been blocked.',

                r'<!>This video is no longer available, because its author has been blocked.':
                'Video %s is no longer available, because its author has been blocked.',

                r'<!>This video is no longer available, because it has been deleted.':
                'Video %s is no longer available, because it has been deleted.',

                r'<!>The video .+? is not available in your region.':
                'Video %s is not available in your region.',
            }

            for error_re, error_msg in ERRORS.items():
                if re.search(error_re, info_page):
                    raise ExtractorError(error_msg % video_id, expected=True)

            player = self._parse_json(self._search_regex(
                r'var\s+playerParams\s*=\s*({.+?})\s*;\s*\n',
                info_page, 'player params'), video_id)

        youtube_url = YoutubeIE._extract_url(info_page)
        if youtube_url:
            return self.url_result(youtube_url, YoutubeIE.ie_key())

        vimeo_url = VimeoIE._extract_url(url, info_page)
        if vimeo_url is not None:
            return self.url_result(vimeo_url, VimeoIE.ie_key())

        pladform_url = PladformIE._extract_url(info_page)
        if pladform_url:
            return self.url_result(pladform_url, PladformIE.ie_key())

        m_rutube = re.search(
            r'\ssrc="((?:https?:)?//rutube\.ru\\?/(?:video|play)\\?/embed(?:.*?))\\?"', info_page)
        if m_rutube is not None:
            rutube_url = self._proto_relative_url(
                m_rutube.group(1).replace('\\', ''))
            return self.url_result(rutube_url)

        dailymotion_urls = DailymotionIE._extract_urls(info_page)
        if dailymotion_urls:
            return self.url_result(dailymotion_urls[0], DailymotionIE.ie_key())

        odnoklassniki_url = OdnoklassnikiIE._extract_url(info_page)
        if odnoklassniki_url:
            return self.url_result(odnoklassniki_url, OdnoklassnikiIE.ie_key())

        m_opts = re.search(r'(?s)var\s+opts\s*=\s*({.+?});', info_page)
        if m_opts:
            m_opts_url = re.search(r"url\s*:\s*'((?!/\b)[^']+)", m_opts.group(1))
            if m_opts_url:
                opts_url = m_opts_url.group(1)
                if opts_url.startswith('//'):
                    opts_url = 'http:' + opts_url
                return self.url_result(opts_url)

        data = player['params'][0]
        title = unescapeHTML(data['md_title'])

        # 2 = live
        # 3 = post live (finished live)
        is_live = data.get('live') == 2
        if is_live:
            title = self._live_title(title)

        timestamp = unified_timestamp(self._html_search_regex(
            r'class=["\']mv_info_date[^>]+>([^<]+)(?:<|from)', info_page,
            'upload date', default=None)) or int_or_none(data.get('date'))

        view_count = str_to_int(self._search_regex(
            r'class=["\']mv_views_count[^>]+>\s*([\d,.]+)',
            info_page, 'view count', default=None))

        formats = []
        for format_id, format_url in data.items():
            format_url = url_or_none(format_url)
            if not format_url or not format_url.startswith(('http', '//', 'rtmp')):
                continue
            if (format_id.startswith(('url', 'cache'))
                    or format_id in ('extra_data', 'live_mp4', 'postlive_mp4')):
                height = int_or_none(self._search_regex(
                    r'^(?:url|cache)(\d+)', format_id, 'height', default=None))
                formats.append({
                    'format_id': format_id,
                    'url': format_url,
                    'height': height,
                })
            elif format_id == 'hls':
                formats.extend(self._extract_m3u8_formats(
                    format_url, video_id, 'mp4', 'm3u8_native',
                    m3u8_id=format_id, fatal=False, live=is_live))
            elif format_id == 'rtmp':
                formats.append({
                    'format_id': format_id,
                    'url': format_url,
                    'ext': 'flv',
                })
        self._sort_formats(formats)

        return {
            'id': video_id,
            'formats': formats,
            'title': title,
            'thumbnail': data.get('jpg'),
            'uploader': data.get('md_author'),
            'uploader_id': str_or_none(data.get('author_id') or mv_data.get('authorId')),
            'duration': int_or_none(data.get('duration') or mv_data.get('duration')),
            'timestamp': timestamp,
            'view_count': view_count,
            'like_count': int_or_none(mv_data.get('likes')),
            'comment_count': int_or_none(mv_data.get('commcount')),
            'is_live': is_live,
        }


class VKUserVideosIE(VKBaseIE):
    IE_NAME = 'vk:uservideos'
    IE_DESC = "VK - User's Videos"
    _VALID_URL = r'https?://(?:(?:m|new)\.)?vk\.com/videos(?P<id>-?[0-9]+)(?!\?.*\bz=video)(?:[/?#&](?:.*?\bsection=(?P<section>\w+))?|$)'
    _TEMPLATE_URL = 'https://vk.com/videos'
    _TESTS = [{
        'url': 'https://vk.com/videos-767561',
        'info_dict': {
            'id': '-767561_all',
        },
        'playlist_mincount': 1150,
    }, {
        'url': 'https://vk.com/videos-767561?section=uploaded',
        'info_dict': {
            'id': '-767561_uploaded',
        },
        'playlist_mincount': 425,
    }, {
        'url': 'http://vk.com/videos205387401',
        'only_matching': True,
    }, {
        'url': 'http://vk.com/videos-77521',
        'only_matching': True,
    }, {
        'url': 'http://vk.com/videos-97664626?section=all',
        'only_matching': True,
    }, {
        'url': 'http://m.vk.com/videos205387401',
        'only_matching': True,
    }, {
        'url': 'http://new.vk.com/videos205387401',
        'only_matching': True,
    }]
    _PAGE_SIZE = 1000
    _VIDEO = collections.namedtuple('Video', ['owner_id', 'id'])

    def _fetch_page(self, page_id, section, page):
        l = self._download_payload('al_video', page_id, {
            'act': 'load_videos_silent',
            'offset': page * self._PAGE_SIZE,
            'oid': page_id,
            'section': section,
        })[0][section]['list']

        for video in l:
            v = self._VIDEO._make(video[:2])
            video_id = '%d_%d' % (v.owner_id, v.id)
            yield self.url_result(
                'http://vk.com/video' + video_id, VKIE.ie_key(), video_id)

    def _real_extract(self, url):
        page_id, section = re.match(self._VALID_URL, url).groups()
        if not section:
            section = 'all'

        entries = OnDemandPagedList(
            functools.partial(self._fetch_page, page_id, section),
            self._PAGE_SIZE)

        return self.playlist_result(entries, '%s_%s' % (page_id, section))


class VKWallPostIE(VKBaseIE):
    IE_NAME = 'vk:wallpost'
    _VALID_URL = r'https?://(?:(?:(?:(?:m|new)\.)?vk\.com/(?:[^?]+\?.*\bw=)?wall(?P<id>-?\d+_\d+)))'
    _TESTS = [{
        # public page URL, audio playlist
        'url': 'https://vk.com/bs.official?w=wall-23538238_35',
        'info_dict': {
            'id': '-23538238_35',
            'title': 'Black Shadow - Wall post -23538238_35',
            'description': 'md5:3f84b9c4f9ef499731cf1ced9998cc0c',
        },
        'playlist': [{
            'md5': '5ba93864ec5b85f7ce19a9af4af080f6',
            'info_dict': {
                'id': '135220665_111806521',
                'ext': 'mp4',
                'title': 'Black Shadow - Слепое Верование',
                'duration': 370,
                'uploader': 'Black Shadow',
                'artist': 'Black Shadow',
                'track': 'Слепое Верование',
            },
        }, {
            'md5': '4cc7e804579122b17ea95af7834c9233',
            'info_dict': {
                'id': '135220665_111802303',
                'ext': 'mp4',
                'title': 'Black Shadow - Война - Негасимое Бездны Пламя!',
                'duration': 423,
                'uploader': 'Black Shadow',
                'artist': 'Black Shadow',
                'track': 'Война - Негасимое Бездны Пламя!',
            },
        }],
        'params': {
            'skip_download': True,
            'usenetrc': True,
        },
        'skip': 'Requires vk account credentials',
    }, {
        # single YouTube embed, no leading -
        'url': 'https://vk.com/wall85155021_6319',
        'info_dict': {
            'id': '85155021_6319',
            'title': 'Сергей Горбунов - Wall post 85155021_6319',
        },
        'playlist_count': 1,
        'params': {
            'usenetrc': True,
        },
        'skip': 'Requires vk account credentials',
    }, {
        # wall page URL
        'url': 'https://vk.com/wall-23538238_35',
        'only_matching': True,
    }, {
        # mobile wall page URL
        'url': 'https://m.vk.com/wall-23538238_35',
        'only_matching': True,
    }]
    _BASE64_CHARS = 'abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMN0PQRSTUVWXYZO123456789+/='
    _AUDIO = collections.namedtuple('Audio', ['id', 'owner_id', 'url', 'title', 'performer', 'duration', 'album_id', 'unk', 'author_link', 'lyrics', 'flags', 'context', 'extra', 'hashes', 'cover_url', 'ads'])

    def _decode(self, enc):
        dec = ''
        e = n = 0
        for c in enc:
            r = self._BASE64_CHARS.index(c)
            cond = n % 4
            e = 64 * e + r if cond else r
            n += 1
            if cond:
                dec += chr(255 & e >> (-2 * n & 6))
        return dec

    def _unmask_url(self, mask_url, vk_id):
        if 'audio_api_unavailable' in mask_url:
            extra = mask_url.split('?extra=')[1].split('#')
            func, base = self._decode(extra[1]).split(chr(11))
            mask_url = list(self._decode(extra[0]))
            url_len = len(mask_url)
            indexes = [None] * url_len
            index = int(base) ^ vk_id
            for n in range(url_len - 1, -1, -1):
                index = (url_len * (n + 1) ^ index + n) % url_len
                indexes[n] = index
            for n in range(1, url_len):
                c = mask_url[n]
                index = indexes[url_len - 1 - n]
                mask_url[n] = mask_url[index]
                mask_url[index] = c
            mask_url = ''.join(mask_url)
        return mask_url

    def _real_extract(self, url):
        post_id = self._match_id(url)

        webpage = self._download_payload('wkview', post_id, {
            'act': 'show',
            'w': 'wall' + post_id,
        })[1]

        description = clean_html(get_element_by_class('wall_post_text', webpage))
        uploader = clean_html(get_element_by_class('author', webpage))

        entries = []

        for audio in re.findall(r'data-audio="([^"]+)', webpage):
            audio = self._parse_json(unescapeHTML(audio), post_id)
            a = self._AUDIO._make(audio[:16])
            if not a.url:
                continue
            title = unescapeHTML(a.title)
            performer = unescapeHTML(a.performer)
            entries.append({
                'id': '%s_%s' % (a.owner_id, a.id),
                'url': self._unmask_url(a.url, a.ads['vk_id']),
                'title': '%s - %s' % (performer, title) if performer else title,
                'thumbnails': [{'url': c_url} for c_url in a.cover_url.split(',')] if a.cover_url else None,
                'duration': int_or_none(a.duration),
                'uploader': uploader,
                'artist': performer,
                'track': title,
                'ext': 'mp4',
                'protocol': 'm3u8',
            })

        for video in re.finditer(
                r'<a[^>]+href=(["\'])(?P<url>/video(?:-?[\d_]+).*?)\1', webpage):
            entries.append(self.url_result(
                compat_urlparse.urljoin(url, video.group('url')), VKIE.ie_key()))

        title = 'Wall post %s' % post_id

        return self.playlist_result(
            orderedSet(entries), post_id,
            '%s - %s' % (uploader, title) if uploader else title,
            description)
-												Unify coding cookie

											
										
										
											2016-10-02 07:39:18 -04:00
+								# coding: utf-8
-												[vk] Use unicode_literals

											
										
										
											2014-01-21 11:32:03 -05:00
+								from __future__ import unicode_literals
-												[vk:wallpost] Fix audio extraction

											
										
										
											2016-08-17 19:14:05 -04:00
+								import collections
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								import functools
-												Add an extractor for vk.com (closes #1635)

											
										
										
											2013-11-01 17:28:51 -04:00
+								import re
 								from .common import InfoExtractor
-												[vk] Remove unused import

											
										
										
											2019-04-06 15:17:54 -04:00
+								from ..compat import compat_urlparse
-												Add an extractor for vk.com (closes #1635)

											
										
										
											2013-11-01 17:28:51 -04:00
+								from ..utils import (
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								    clean_html,
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								    ExtractorError,
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								    get_element_by_class,
-												[vk] Extract video URL from extra_data (Closes #8646)

											
										
										
											2016-02-23 07:47:13 -05:00
+								    int_or_none,
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								    OnDemandPagedList,
-												Fix imports and general cleanup

· Import from compat what comes from compat. Yes, some names are available in utils too, but that's an implementation detail.
· Use _match_id consistently whenever possible
· Fix some outdated tests
· Use consistent valid URL (always match the whole protocol, no ^ at start required)
· Use modern test definitions

											
										
										
											2014-12-13 06:24:42 -05:00
+								    orderedSet,
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
+								    str_or_none,
-												[vk] Extract view count

											
										
										
											2015-06-15 10:55:25 -04:00
+								    str_to_int,
-												Add an extractor for vk.com (closes #1635)

											
										
										
											2013-11-01 17:28:51 -04:00
+								    unescapeHTML,
-												[vk] Extract timestamp (Closes #10760)

											
										
										
											2016-09-29 12:48:21 -04:00
+								    unified_timestamp,
-												Improve URL extraction

											
										
										
											2018-07-21 08:08:28 -04:00
+								    url_or_none,
-												Use urlencode_postdata across the codebase

											
										
										
											2016-03-25 16:19:24 -04:00
+								    urlencode_postdata,
-												Fix imports and general cleanup

· Import from compat what comes from compat. Yes, some names are available in utils too, but that's an implementation detail.
· Use _match_id consistently whenever possible
· Fix some outdated tests
· Use consistent valid URL (always match the whole protocol, no ^ at start required)
· Use modern test definitions

											
										
										
											2014-12-13 06:24:42 -05:00
+								)
-												[vk] Add support for dailymotion embeds

Fixes #10661

											
										
										
											2016-09-24 23:39:29 -04:00
+								from .dailymotion import DailymotionIE
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								from .odnoklassniki import OdnoklassnikiIE
-												[vk] Add support for pladform embeds (Closes #7780)

											
										
										
											2015-12-07 11:03:52 -05:00
+								from .pladform import PladformIE
-												[vk] Add support for dailymotion embeds

Fixes #10661

											
										
										
											2016-09-24 23:39:29 -04:00
+								from .vimeo import VimeoIE
-												[abcnews,chilloutsoze,cracked,vice,vk] Use dedicated YouTube embeds extraction routines

											
										
										
											2017-09-05 13:50:25 -04:00
+								from .youtube import YoutubeIE
-												Add an extractor for vk.com (closes #1635)

											
										
										
											2013-11-01 17:28:51 -04:00
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								class VKBaseIE(InfoExtractor):
 								    _NETRC_MACHINE = 'vk'
 								    def _login(self):
-												remove unnecessary assignment parenthesis

											
										
										
											2018-05-26 11:12:44 -04:00
+								        username, password = self._get_login_info()
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								        if username is None:
 								            return
 								        login_page, url_handle = self._download_webpage_handle(
 								            'https://vk.com', None, 'Downloading login page')
 								        login_form = self._hidden_inputs(login_page)
 								        login_form.update({
 								            'email': username.encode('cp1251'),
 								            'pass': password.encode('cp1251'),
 								        })
-												[extractor/common] Move workaround for applying first Set-Cookie header into a separate method

											
										
										
											2019-05-17 16:17:15 -04:00
+								        # vk serves two same remixlhk cookies in Set-Cookie header and expects
 								        # first one to be actually set
 								        self._apply_first_set_cookie_header(url_handle, 'remixlhk')
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
 								        login_page = self._download_webpage(
 								            'https://login.vk.com/?act=login', None,
-												Remove sensitive data from logging in messages

											
										
										
											2017-11-11 08:49:03 -05:00
+								            note='Logging in',
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								            data=urlencode_postdata(login_form))
 								        if re.search(r'onLoginFailed', login_page):
 								            raise ExtractorError(
 								                'Unable to login, incorrect username and/or password', expected=True)
 								    def _real_initialize(self):
 								        self._login()
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								    def _download_payload(self, path, video_id, data, fatal=True):
 								        data['al'] = 1
 								        code, payload = self._download_json(
 								            'https://vk.com/%s.php' % path, video_id,
 								            data=urlencode_postdata(data), fatal=fatal,
 								            headers={'X-Requested-With': 'XMLHttpRequest'})['payload']
 								        if code == '3':
 								            self.raise_login_required()
 								        elif code == '8':
 								            raise ExtractorError(clean_html(payload[0][1:-1]), expected=True)
 								        return payload
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
 								class VKIE(VKBaseIE):
-												[vk] Clarify extractor names

											
										
										
											2015-07-18 07:23:33 -04:00
+								    IE_NAME = 'vk'
 								    IE_DESC = 'VK'
-												[vk] Extend _VALID_URL to handle biqle.ru (Closes #6179)

											
										
										
											2015-07-08 10:27:06 -04:00
+								    _VALID_URL = r'''(?x)
 								                    https?://
 								                        (?:
-												[vk] improve extraction(fixes #7976)

											
										
										
											2016-05-06 10:02:40 -04:00
+								                            (?:
-												[vk] Extend _VALID_URLs to support new domain (Closes #9981)

											
										
										
											2016-07-02 05:43:19 -04:00
+								                                (?:(?:m|new)\.)?vk\.com/video_|
-												[vk] improve extraction(fixes #7976)

											
										
										
											2016-05-06 10:02:40 -04:00
+								                                (?:www\.)?daxab.com/
 								                            )
 								                            ext\.php\?(?P<embed_query>.*?\boid=(?P<oid>-?\d+).*?\bid=(?P<id>\d+).*)|
-												[vk] Extend _VALID_URL to handle biqle.ru (Closes #6179)

											
										
										
											2015-07-08 10:27:06 -04:00
+								                            (?:
-												[vk] Extend _VALID_URLs to support new domain (Closes #9981)

											
										
										
											2016-07-02 05:43:19 -04:00
+								                                (?:(?:m|new)\.)?vk\.com/(?:.+?\?.*?z=)?video|
-												[vk] improve extraction(fixes #7976)

											
										
										
											2016-05-06 10:02:40 -04:00
+								                                (?:www\.)?daxab.com/embed/
-												[vk] Extend _VALID_URL to handle biqle.ru (Closes #6179)

											
										
										
											2015-07-08 10:27:06 -04:00
+								                            )
-												[vk] improve extraction(fixes #7976)

											
										
										
											2016-05-06 10:02:40 -04:00
+								                            (?P<videoid>-?\d+_\d+)(?:.*\blist=(?P<list_id>[\da-f]+))?
-												[vk] Extend _VALID_URL to handle biqle.ru (Closes #6179)

											
										
										
											2015-07-08 10:27:06 -04:00
+								                        )
 								                    '''
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								    _TESTS = [
 								        {
 								            'url': 'http://vk.com/videos-77521?z=video-77521_162222515%2Fclub77521',
-												[vk] Update test


											
										
										
											2018-02-20 10:21:10 -05:00
+								            'md5': '7babad3b85ea2e91948005b1b8b0cb84',
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								            'info_dict': {
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'id': '-77521_162222515',
-												[vk] Update test


											
										
										
											2018-02-20 10:21:10 -05:00
+								                'ext': 'mp4',
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								                'title': 'ProtivoGunz - Хуёвая песня',
-												[vk] Fix test (Closes #5100)

											
										
										
											2015-03-01 16:30:18 -05:00
+								                'uploader': 're:(?:Noize MC|Alexander Ilyashenko).*',
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
+								                'uploader_id': '-77521',
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								                'duration': 195,
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
+								                'timestamp': 1329049880,
-												[vk.com] Added upload_date variable to the test cases that still work.

											
										
										
											2014-11-21 17:23:39 -05:00
+								                'upload_date': '20120212',
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								            },
-												Add an extractor for vk.com (closes #1635)

											
										
										
											2013-11-01 17:28:51 -04:00
+								        },
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								        {
-												[vk.com] Updated a test video that has been removed, and added a comment for others to update two other test videos that are also now removed.

											
										
										
											2014-11-21 17:52:01 -05:00
+								            'url': 'http://vk.com/video205387401_165548505',
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								            'info_dict': {
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'id': '205387401_165548505',
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								                'ext': 'mp4',
-												[vk.com] Updated a test video that has been removed, and added a comment for others to update two other test videos that are also now removed.

											
										
										
											2014-11-21 17:52:01 -05:00
+								                'title': 'No name',
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
+								                'uploader': 'Tom Cruise',
 								                'uploader_id': '205387401',
-												[vk.com] Updated a test video that has been removed, and added a comment for others to update two other test videos that are also now removed.

											
										
										
											2014-11-21 17:52:01 -05:00
+								                'duration': 9,
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
+								                'timestamp': 1374364108,
 								                'upload_date': '20130720',
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								            }
 								        },
-												[vk] Add support for embedded videos (Closes #2473)
											
										
										
											2014-02-28 11:51:54 -05:00
+								        {
 								            'note': 'Embedded video',
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            'url': 'https://vk.com/video_ext.php?oid=-77521&id=162222515&hash=87b046504ccd8bfa',
 								            'md5': '7babad3b85ea2e91948005b1b8b0cb84',
-												[vk] Add support for embedded videos (Closes #2473)
											
										
										
											2014-02-28 11:51:54 -05:00
+								            'info_dict': {
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                'id': '-77521_162222515',
-												[vk] Add support for embedded videos (Closes #2473)
											
										
										
											2014-02-28 11:51:54 -05:00
+								                'ext': 'mp4',
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                'uploader': 're:(?:Noize MC|Alexander Ilyashenko).*',
 								                'title': 'ProtivoGunz - Хуёвая песня',
 								                'duration': 195,
 								                'upload_date': '20120212',
 								                'timestamp': 1329049880,
 								                'uploader_id': '-77521',
-												[vk] improve extraction(fixes #7976)

											
										
										
											2016-05-06 10:02:40 -04:00
+								            },
-												[vk] Add support for embedded videos (Closes #2473)
											
										
										
											2014-02-28 11:51:54 -05:00
+								        },
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								        {
-												[vk.com] Updated a test video that has been removed, and added a comment for others to update two other test videos that are also now removed.

											
										
										
											2014-11-21 17:52:01 -05:00
+								            # VIDEO NOW REMOVED
 								            # please update if you find a video whose URL follows the same pattern
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								            'url': 'http://vk.com/video-8871596_164049491',
 								            'md5': 'a590bcaf3d543576c9bd162812387666',
 								            'note': 'Only available for registered users',
 								            'info_dict': {
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'id': '-8871596_164049491',
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								                'ext': 'mp4',
 								                'uploader': 'Триллеры',
-												[vk] Add support for more URL formats (#3172)

											
										
										
											2014-06-29 08:33:39 -04:00
+								                'title': '► Бойцовский клуб / Fight Club 1999 [HD 720]',
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								                'duration': 8352,
-												[vk] Extract view count

											
										
										
											2015-06-15 10:55:25 -04:00
+								                'upload_date': '20121218',
 								                'view_count': int,
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
+								            },
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            'skip': 'Removed',
-												[vk] Add support for embedded videos (Closes #2473)
											
										
										
											2014-02-28 11:51:54 -05:00
+								        },
-												[vk] Add support for more URL formats (#3172)

											
										
										
											2014-06-29 08:33:39 -04:00
+								        {
 								            'url': 'http://vk.com/hd_kino_mania?z=video-43215063_168067957%2F15c66b9b533119788d',
 								            'info_dict': {
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'id': '-43215063_168067957',
-												[vk] Add support for more URL formats (#3172)

											
										
										
											2014-06-29 08:33:39 -04:00
+								                'ext': 'mp4',
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                'uploader': 'Bro Mazter',
-												[vk] Add support for more URL formats (#3172)

											
										
										
											2014-06-29 08:33:39 -04:00
+								                'title': ' ',
 								                'duration': 7291,
-												[vk.com] Added upload_date variable to the test cases that still work.

											
										
										
											2014-11-21 17:23:39 -05:00
+								                'upload_date': '20140328',
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                'uploader_id': '223413403',
 								                'timestamp': 1396018030,
-												[vk] Add support for more URL formats (#3172)

											
										
										
											2014-06-29 08:33:39 -04:00
+								            },
 								            'skip': 'Requires vk account credentials',
 								        },
-												[vk] Better support for embeds

											
										
										
											2014-06-29 09:07:59 -04:00
+								        {
 								            'url': 'http://m.vk.com/video-43215063_169084319?list=125c627d1aa1cebb83&from=wall-43215063_2566540',
 								            'md5': '0c45586baa71b7cb1d0784ee3f4e00a6',
 								            'note': 'ivi.ru embed',
 								            'info_dict': {
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'id': '-43215063_169084319',
-												[vk] Better support for embeds

											
										
										
											2014-06-29 09:07:59 -04:00
+								                'ext': 'mp4',
 								                'title': 'Книга Илая',
 								                'duration': 6771,
-												[vk.com] Added upload_date variable to the test cases that still work.

											
										
										
											2014-11-21 17:23:39 -05:00
+								                'upload_date': '20140626',
-												[vk] Extract view count

											
										
										
											2015-06-15 10:55:25 -04:00
+								                'view_count': int,
-												[vk] Better support for embeds

											
										
										
											2014-06-29 09:07:59 -04:00
+								            },
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            'skip': 'Removed',
-												[vk] Better support for embeds

											
										
										
											2014-06-29 09:07:59 -04:00
+								        },
-												[vk] Add list id to info_url

											
										
										
											2015-07-11 11:23:49 -04:00
+								        {
 								            # video (removed?) only available with list id
 								            'url': 'https://vk.com/video30481095_171201961?list=8764ae2d21f14088d4',
 								            'md5': '091287af5402239a1051c37ec7b92913',
 								            'info_dict': {
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'id': '30481095_171201961',
-												[vk] Add list id to info_url

											
										
										
											2015-07-11 11:23:49 -04:00
+								                'ext': 'mp4',
 								                'title': 'ТюменцевВВ_09.07.2015',
 								                'uploader': 'Anton Ivanov',
 								                'duration': 109,
 								                'upload_date': '20150709',
 								                'view_count': int,
 								            },
-												[vk] Extract timestamp (Closes #10760)

											
										
										
											2016-09-29 12:48:21 -04:00
+								            'skip': 'Removed',
-												[vk] Add list id to info_url

											
										
										
											2015-07-11 11:23:49 -04:00
+								        },
-												[vk] Add test for youtube embed

											
										
										
											2015-07-08 10:41:08 -04:00
+								        {
 								            # youtube embed
 								            'url': 'https://vk.com/video276849682_170681728',
 								            'info_dict': {
 								                'id': 'V3K4mi0SYkc',
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'ext': 'mp4',
-												[vk] Add test for youtube embed

											
										
										
											2015-07-08 10:41:08 -04:00
+								                'title': "DSWD Awards 'Children's Joy Foundation, Inc.' Certificate of Registration and License to Operate",
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
+								                'description': 'md5:bf9c26cfa4acdfb146362682edd3827a',
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'duration': 178,
-												[vk] Add test for youtube embed

											
										
										
											2015-07-08 10:41:08 -04:00
+								                'upload_date': '20130116',
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
+								                'uploader': "Children's Joy Foundation Inc.",
-												[vk] Add test for youtube embed

											
										
										
											2015-07-08 10:41:08 -04:00
+								                'uploader_id': 'thecjf',
 								                'view_count': int,
 								            },
 								        },
-												[vk] Add support for dailymotion embeds

Fixes #10661

											
										
										
											2016-09-24 23:39:29 -04:00
+								        {
 								            # dailymotion embed
 								            'url': 'https://vk.com/video-37468416_456239855',
 								            'info_dict': {
 								                'id': 'k3lz2cmXyRuJQSjGHUv',
 								                'ext': 'mp4',
 								                'title': 'md5:d52606645c20b0ddbb21655adaa4f56f',
-												[dailymotion] improve extraction

- extract http formats included in m3u8 manifest
- fix user extraction(closes #3553)(closes #21415)
- add suport for User Authentication(closes #11491)
- fix password protected videos extraction(closes #23176)
- respect age limit option and family filter cookie value(closes #18437)
- handle video url playlist query param
- report alowed countries for geo-restricted videos

											
										
										
											2019-11-26 16:01:34 -05:00
+								                'description': 'md5:424b8e88cc873217f520e582ba28bb36',
-												[vk] Add support for dailymotion embeds

Fixes #10661

											
										
										
											2016-09-24 23:39:29 -04:00
+								                'uploader': 'AniLibria.Tv',
 								                'upload_date': '20160914',
 								                'uploader_id': 'x1p5vl5',
 								                'timestamp': 1473877246,
 								            },
 								            'params': {
 								                'skip_download': True,
-												[vk] Add support for finished live streams (#10799)

											
										
										
											2016-09-29 12:04:10 -04:00
+								            },
-												[vk] Add support for dailymotion embeds

Fixes #10661

											
										
										
											2016-09-24 23:39:29 -04:00
+								        },
-												[vk] Extract video URL from extra_data (Closes #8646)

											
										
										
											2016-02-23 07:47:13 -05:00
+								        {
 								            # video key is extra_data not url\d+
 								            'url': 'http://vk.com/video-110305615_171782105',
 								            'md5': 'e13fcda136f99764872e739d13fac1d1',
 								            'info_dict': {
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'id': '-110305615_171782105',
-												[vk] Extract video URL from extra_data (Closes #8646)

											
										
										
											2016-02-23 07:47:13 -05:00
+								                'ext': 'mp4',
 								                'title': 'S-Dance, репетиции к The way show',
 								                'uploader': 'THE WAY SHOW | 17 апреля',
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
+								                'uploader_id': '-110305615',
 								                'timestamp': 1454859345,
-												[vk] Extract video URL from extra_data (Closes #8646)

											
										
										
											2016-02-23 07:47:13 -05:00
+								                'upload_date': '20160207',
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
+								            },
 								            'params': {
 								                'skip_download': True,
-												[vk] Extract video URL from extra_data (Closes #8646)

											
										
										
											2016-02-23 07:47:13 -05:00
+								            },
 								        },
-												[vk] Add support for finished live streams (#10799)

											
										
										
											2016-09-29 12:04:10 -04:00
+								        {
-												[vk] Fix postlive videos extraction

											
										
										
											2016-12-29 16:31:19 -05:00
+								            # finished live stream, postlive_mp4
-												[vk] Add support for finished live streams (#10799)

											
										
										
											2016-09-29 12:04:10 -04:00
+								            'url': 'https://vk.com/videos-387766?z=video-387766_456242764%2Fpl_-387766_-2',
 								            'info_dict': {
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'id': '-387766_456242764',
-												[vk] Add support for finished live streams (#10799)

											
										
										
											2016-09-29 12:04:10 -04:00
+								                'ext': 'mp4',
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                'title': 'ИгроМир 2016 День 1 — Игромания Утром',
-												[vk] Add support for finished live streams (#10799)

											
										
										
											2016-09-29 12:04:10 -04:00
+								                'uploader': 'Игромания',
 								                'duration': 5239,
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								                # TODO: use act=show to extract view_count
 								                # 'view_count': int,
 								                'upload_date': '20160929',
 								                'uploader_id': '-387766',
 								                'timestamp': 1475137527,
-												[vk] Add support for finished live streams (#10799)

											
										
										
											2016-09-29 12:04:10 -04:00
+								            },
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            'params': {
 								                'skip_download': True,
 								            },
-												[vk] Add support for finished live streams (#10799)

											
										
										
											2016-09-29 12:04:10 -04:00
+								        },
-												[vk] Add support for running live streams (Closes #10799)

											
										
										
											2016-09-29 12:21:39 -04:00
+								        {
-												[vk] Fix postlive videos extraction

											
										
										
											2016-12-29 16:31:19 -05:00
+								            # live stream, hls and rtmp links, most likely already finished live
-												[vk] Add support for running live streams (Closes #10799)

											
										
										
											2016-09-29 12:21:39 -04:00
+								            # stream by the time you are reading this comment
 								            'url': 'https://vk.com/video-140332_456239111',
 								            'only_matching': True,
 								        },
-												[vk] Clarify test

											
										
										
											2014-11-23 04:11:04 -05:00
+								        {
 								            # removed video, just testing that we match the pattern
 								            'url': 'http://vk.com/feed?z=video-43215063_166094326%2Fbb50cacd3177146d7a',
 								            'only_matching': True,
 								        },
-												[vk] Add age restricted video test for reference

											
										
										
											2015-07-18 09:25:06 -04:00
+								        {
 								            # age restricted video, requires vk account credentials
 								            'url': 'https://vk.com/video205387401_164765225',
 								            'only_matching': True,
 								        },
-												[vk] Add test for pladform embed

											
										
										
											2015-12-07 11:05:54 -05:00
+								        {
 								            # pladform embed
 								            'url': 'https://vk.com/video-76116461_171554880',
 								            'only_matching': True,
-												[vk] Extend _VALID_URLs to support new domain (Closes #9981)

											
										
										
											2016-07-02 05:43:19 -04:00
+								        },
 								        {
 								            'url': 'http://new.vk.com/video205387401_165548505',
 								            'only_matching': True,
-												[vk] Catch author blocked error message

Example link (video in blocked group):
https://vk.com/search?c%5Bq%5D=%D0%9F%D1%80%D1%8B%D0%B6%D0%BE%D0%BA%20c%20%D0%BA%D1%80%D0%B0%D0%BD%D0%B0%20%D0%B2%20%D1%81%D1%82%D0%B8%D0%BB%D0%B5%20%D0%A7%D0%B5%D0%BB%D0%BE%D0%B2%D0%B5%D0%BA%D0%B0-%D0%BF%D0%B0%D1%83%D0%BA%D0%B0&c%5Bsection%5D=video&c%5Bsort%5D=2&z=video-10639516_456240611

											
										
										
											2017-02-04 03:23:35 -05:00
+								        },
 								        {
 								            # This video is no longer available, because its author has been blocked.
 								            'url': 'https://vk.com/video-10639516_456240611',
 								            'only_matching': True,
-												[vk] Detect geo restriction


											
										
										
											2018-11-17 08:59:13 -05:00
+								        },
 								        {
 								            # The video is not available in your region.
 								            'url': 'https://vk.com/video-51812607_171445436',
 								            'only_matching': True,
 								        }]
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
-												Add an extractor for vk.com (closes #1635)

											
										
										
											2013-11-01 17:28:51 -04:00
+								    def _real_extract(self, url):
 								        mobj = re.match(self._VALID_URL, url)
-												[vk] Add support for embedded videos (Closes #2473)
											
										
										
											2014-02-28 11:51:54 -05:00
+								        video_id = mobj.group('videoid')
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								        mv_data = {}
-												[vk] improve extraction(fixes #7976)

											
										
										
											2016-05-06 10:02:40 -04:00
+								        if video_id:
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            data = {
 								                'act': 'show_inline',
 								                'video': video_id,
 								            }
-												[vk] improve extraction(fixes #7976)

											
										
										
											2016-05-06 10:02:40 -04:00
+								            # Some videos (removed?) can only be downloaded with list id specified
 								            list_id = mobj.group('list_id')
 								            if list_id:
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                data['list'] = list_id
 								            payload = self._download_payload('al_video', video_id, data)
 								            info_page = payload[1]
 								            opts = payload[-1]
 								            mv_data = opts.get('mvData') or {}
 								            player = opts.get('player') or {}
-												[vk] improve extraction(fixes #7976)

											
										
										
											2016-05-06 10:02:40 -04:00
+								        else:
-												[vk] Add support for embedded videos (Closes #2473)
											
										
										
											2014-02-28 11:51:54 -05:00
+								            video_id = '%s_%s' % (mobj.group('oid'), mobj.group('id'))
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            info_page = self._download_webpage(
 								                'http://vk.com/video_ext.php?' + mobj.group('embed_query'), video_id)
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            error_message = self._html_search_regex(
 								                [r'(?s)<!><div[^>]+class="video_layer_message"[^>]*>(.+?)</div>',
 								                    r'(?s)<div[^>]+id="video_ext_msg"[^>]*>(.+?)</div>'],
 								                info_page, 'error message', default=None)
 								            if error_message:
 								                raise ExtractorError(error_message, expected=True)
-												[vk] Capture error message

											
										
										
											2015-07-18 09:15:20 -04:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            if re.search(r'<!>/login\.php\?.*\bact=security_check', info_page):
 								                raise ExtractorError(
 								                    'You are trying to log in from an unusual location. You should confirm ownership at vk.com to log in with this IP.',
 								                    expected=True)
-												[vk] Catch ownership confirmation request

											
										
										
											2015-07-06 14:04:19 -04:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            ERROR_COPYRIGHT = 'Video %s has been removed from public access due to rightholder complaint.'
-												[vk] Detect more errors due to copyright complaints (#15259)

											
										
										
											2018-01-15 10:56:45 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            ERRORS = {
 								                r'>Видеозапись .*? была изъята из публичного доступа в связи с обращением правообладателя.<':
 								                ERROR_COPYRIGHT,
-												[vk] Detect more errors due to copyright complaints (#15259)

											
										
										
											2018-01-15 10:56:45 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                r'>The video .*? was removed from public access by request of the copyright holder.<':
 								                ERROR_COPYRIGHT,
-												[vk] PEP8

											
										
										
											2014-11-23 16:14:27 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                r'<!>Please log in or <':
 								                'Video %s is only available for registered users, '
 								                'use --username and --password options to provide account credentials.',
-												[vk] PEP8

											
										
										
											2014-11-23 16:14:27 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                r'<!>Unknown error':
 								                'Video %s does not exist.',
-												[vk] Catch temporarily unavailable video error message

											
										
										
											2015-03-01 10:55:43 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                r'<!>Видео временно недоступно':
 								                'Video %s is temporarily unavailable.',
-												[vk] Handle access denied error

											
										
										
											2015-07-11 11:26:03 -04:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                r'<!>Access denied':
 								                'Access denied to video %s.',
-												[vk] Catch author blocked error message

Example link (video in blocked group):
https://vk.com/search?c%5Bq%5D=%D0%9F%D1%80%D1%8B%D0%B6%D0%BE%D0%BA%20c%20%D0%BA%D1%80%D0%B0%D0%BD%D0%B0%20%D0%B2%20%D1%81%D1%82%D0%B8%D0%BB%D0%B5%20%D0%A7%D0%B5%D0%BB%D0%BE%D0%B2%D0%B5%D0%BA%D0%B0-%D0%BF%D0%B0%D1%83%D0%BA%D0%B0&c%5Bsection%5D=video&c%5Bsort%5D=2&z=video-10639516_456240611

											
										
										
											2017-02-04 03:23:35 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                r'<!>Видеозапись недоступна, так как её автор был заблокирован.':
 								                'Video %s is no longer available, because its author has been blocked.',
-												[vk] Catch author blocked error message

Example link (video in blocked group):
https://vk.com/search?c%5Bq%5D=%D0%9F%D1%80%D1%8B%D0%B6%D0%BE%D0%BA%20c%20%D0%BA%D1%80%D0%B0%D0%BD%D0%B0%20%D0%B2%20%D1%81%D1%82%D0%B8%D0%BB%D0%B5%20%D0%A7%D0%B5%D0%BB%D0%BE%D0%B2%D0%B5%D0%BA%D0%B0-%D0%BF%D0%B0%D1%83%D0%BA%D0%B0&c%5Bsection%5D=video&c%5Bsort%5D=2&z=video-10639516_456240611

											
										
										
											2017-02-04 03:23:35 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                r'<!>This video is no longer available, because its author has been blocked.':
 								                'Video %s is no longer available, because its author has been blocked.',
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                r'<!>This video is no longer available, because it has been deleted.':
 								                'Video %s is no longer available, because it has been deleted.',
-												[vk] Detect geo restriction


											
										
										
											2018-11-17 08:59:13 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                r'<!>The video .+? is not available in your region.':
 								                'Video %s is not available in your region.',
 								            }
 								            for error_re, error_msg in ERRORS.items():
 								                if re.search(error_re, info_page):
 								                    raise ExtractorError(error_msg % video_id, expected=True)
-												[vk] Add login feature (Closes #2206)
											
										
										
											2014-02-16 14:05:15 -05:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            player = self._parse_json(self._search_regex(
 								                r'var\s+playerParams\s*=\s*({.+?})\s*;\s*\n',
 								                info_page, 'player params'), video_id)
-												[vk] Handle deleted videos

											
										
										
											2014-10-28 10:06:07 -04:00
-												[abcnews,chilloutsoze,cracked,vice,vk] Use dedicated YouTube embeds extraction routines

											
										
										
											2017-09-05 13:50:25 -04:00
+								        youtube_url = YoutubeIE._extract_url(info_page)
-												[vk] Fix youtube extraction

											
										
										
											2015-07-08 10:34:50 -04:00
+								        if youtube_url:
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            return self.url_result(youtube_url, YoutubeIE.ie_key())
-												[vk] Better support for embeds

											
										
										
											2014-06-29 09:07:59 -04:00
-												[generic] Add support for multiple vimeo embeds (Closes #10862)

											
										
										
											2016-10-06 12:22:52 -04:00
+								        vimeo_url = VimeoIE._extract_url(url, info_page)
-												[vk] Detect vimeo embeds (Closes #7021)

											
										
										
											2015-09-30 12:12:52 -04:00
+								        if vimeo_url is not None:
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            return self.url_result(vimeo_url, VimeoIE.ie_key())
-												[vk] Detect vimeo embeds (Closes #7021)

											
										
										
											2015-09-30 12:12:52 -04:00
-												[vk] Add support for pladform embeds (Closes #7780)

											
										
										
											2015-12-07 11:03:52 -05:00
+								        pladform_url = PladformIE._extract_url(info_page)
 								        if pladform_url:
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            return self.url_result(pladform_url, PladformIE.ie_key())
-												[vk] Add support for pladform embeds (Closes #7780)

											
										
										
											2015-12-07 11:03:52 -05:00
-												[vk] Add support for rutube embeds (Fixes #4514)

											
										
										
											2015-01-03 21:15:27 -05:00
+								        m_rutube = re.search(
-												[vk] Improve rutube embeds detection (Closes #8461)

											
										
										
											2016-02-08 10:30:23 -05:00
+								            r'\ssrc="((?:https?:)?//rutube\.ru\\?/(?:video|play)\\?/embed(?:.*?))\\?"', info_page)
-												[vk] Add support for rutube embeds (Fixes #4514)

											
										
										
											2015-01-03 21:15:27 -05:00
+								        if m_rutube is not None:
 								            rutube_url = self._proto_relative_url(
 								                m_rutube.group(1).replace('\\', ''))
 								            return self.url_result(rutube_url)
-												[vk] Add support for dailymotion embeds

Fixes #10661

											
										
										
											2016-09-24 23:39:29 -04:00
+								        dailymotion_urls = DailymotionIE._extract_urls(info_page)
 								        if dailymotion_urls:
 								            return self.url_result(dailymotion_urls[0], DailymotionIE.ie_key())
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								        odnoklassniki_url = OdnoklassnikiIE._extract_url(info_page)
 								        if odnoklassniki_url:
 								            return self.url_result(odnoklassniki_url, OdnoklassnikiIE.ie_key())
-												[vk] Fix extraction (Closes #5987)

											
										
										
											2015-06-15 10:46:10 -04:00
+								        m_opts = re.search(r'(?s)var\s+opts\s*=\s*({.+?});', info_page)
-												[vk] Better support for embeds

											
										
										
											2014-06-29 09:07:59 -04:00
+								        if m_opts:
-												[vk] Fix extraction (Closes #5987)

											
										
										
											2015-06-15 10:46:10 -04:00
+								            m_opts_url = re.search(r"url\s*:\s*'((?!/\b)[^']+)", m_opts.group(1))
-												[vk] Better support for embeds

											
										
										
											2014-06-29 09:07:59 -04:00
+								            if m_opts_url:
 								                opts_url = m_opts_url.group(1)
 								                if opts_url.startswith('//'):
 								                    opts_url = 'http:' + opts_url
 								                return self.url_result(opts_url)
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								        data = player['params'][0]
-												[vk] Add support for running live streams (Closes #10799)

											
										
										
											2016-09-29 12:21:39 -04:00
+								        title = unescapeHTML(data['md_title'])
-												[vk] Fix postlive videos extraction

											
										
										
											2016-12-29 16:31:19 -05:00
+								        # 2 = live
 								        # 3 = post live (finished live)
-												[vk] Extract from playerParams (closes #11555)

											
										
										
											2016-12-29 16:21:49 -05:00
+								        is_live = data.get('live') == 2
 								        if is_live:
-												[vk] Add support for running live streams (Closes #10799)

											
										
										
											2016-09-29 12:21:39 -04:00
+								            title = self._live_title(title)
-												[vk] Extract timestamp (Closes #10760)

											
										
										
											2016-09-29 12:48:21 -04:00
+								        timestamp = unified_timestamp(self._html_search_regex(
-												[vk] Improve view count extraction

											
										
										
											2016-09-29 12:51:52 -04:00
+								            r'class=["\']mv_info_date[^>]+>([^<]+)(?:<|from)', info_page,
-												[vk] fix extraction for inline only videos(fixes #16923)

											
										
										
											2018-07-26 02:24:46 -04:00
+								            'upload date', default=None)) or int_or_none(data.get('date'))
-												[vk] Fix date and view count extraction.

											
										
										
											2016-09-25 14:26:58 -04:00
-												[vk] Improve view count extraction

											
										
										
											2016-09-29 12:51:52 -04:00
+								        view_count = str_to_int(self._search_regex(
 								            r'class=["\']mv_views_count[^>]+>\s*([\d,.]+)',
-												[vk] Make view count optional (closes #14979)

											
										
										
											2017-12-15 10:53:56 -05:00
+								            info_page, 'view count', default=None))
-												[vk] Extract view count

											
										
										
											2015-06-15 10:55:25 -04:00
-												[vk] Extract video URL from extra_data (Closes #8646)

											
										
										
											2016-02-23 07:47:13 -05:00
+								        formats = []
-												[vk] Add support for running live streams (Closes #10799)

											
										
										
											2016-09-29 12:21:39 -04:00
+								        for format_id, format_url in data.items():
-												Improve URL extraction

											
										
										
											2018-07-21 08:08:28 -04:00
+								            format_url = url_or_none(format_url)
 								            if not format_url or not format_url.startswith(('http', '//', 'rtmp')):
-												[vk] Extract video URL from extra_data (Closes #8646)

											
										
										
											2016-02-23 07:47:13 -05:00
+								                continue
-												Fix W504 and disable W503 (closes #20863)

											
										
										
											2019-05-10 16:56:22 -04:00
+								            if (format_id.startswith(('url', 'cache'))
 								                    or format_id in ('extra_data', 'live_mp4', 'postlive_mp4')):
-												[vk] Add support for running live streams (Closes #10799)

											
										
										
											2016-09-29 12:21:39 -04:00
+								                height = int_or_none(self._search_regex(
 								                    r'^(?:url|cache)(\d+)', format_id, 'height', default=None))
 								                formats.append({
 								                    'format_id': format_id,
 								                    'url': format_url,
 								                    'height': height,
 								                })
 								            elif format_id == 'hls':
 								                formats.extend(self._extract_m3u8_formats(
-												[downloader/hls] immediately delegate downloading to ffmpeg in case live stream

											
										
										
											2017-03-25 14:37:54 -04:00
+								                    format_url, video_id, 'mp4', 'm3u8_native',
-												[vk] Extract from playerParams (closes #11555)

											
										
										
											2016-12-29 16:21:49 -05:00
+								                    m3u8_id=format_id, fatal=False, live=is_live))
-												[vk] Add support for running live streams (Closes #10799)

											
										
										
											2016-09-29 12:21:39 -04:00
+								            elif format_id == 'rtmp':
 								                formats.append({
 								                    'format_id': format_id,
 								                    'url': format_url,
 								                    'ext': 'flv',
 								                })
-												[vk] Add support for HQ videos (Fixes #2187)

											
										
										
											2014-01-21 12:21:44 -05:00
+								        self._sort_formats(formats)
-												Add an extractor for vk.com (closes #1635)

											
										
										
											2013-11-01 17:28:51 -04:00
+								        return {
-												[vk] use a more unique video id(closes #17848)

											
										
										
											2019-04-03 06:08:42 -04:00
+								            'id': video_id,
-												[vk] Add support for HQ videos (Fixes #2187)

											
										
										
											2014-01-21 12:21:44 -05:00
+								            'formats': formats,
-												[vk] Add support for running live streams (Closes #10799)

											
										
										
											2016-09-29 12:21:39 -04:00
+								            'title': title,
-												[vk] Add support for HQ videos (Fixes #2187)

											
										
										
											2014-01-21 12:21:44 -05:00
+								            'thumbnail': data.get('jpg'),
 								            'uploader': data.get('md_author'),
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            'uploader_id': str_or_none(data.get('author_id') or mv_data.get('authorId')),
 								            'duration': int_or_none(data.get('duration') or mv_data.get('duration')),
-												[vk] Extract timestamp (Closes #10760)

											
										
										
											2016-09-29 12:48:21 -04:00
+								            'timestamp': timestamp,
-												[vk] Extract view count

											
										
										
											2015-06-15 10:55:25 -04:00
+								            'view_count': view_count,
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            'like_count': int_or_none(mv_data.get('likes')),
 								            'comment_count': int_or_none(mv_data.get('commcount')),
-												[vk] Extract from playerParams (closes #11555)

											
										
										
											2016-12-29 16:21:49 -05:00
+								            'is_live': is_live,
-												Add an extractor for vk.com (closes #1635)

											
										
										
											2013-11-01 17:28:51 -04:00
+								        }
-												[vk] Added a new information extractor for pages that are a list of a user\'s videos on vk.com. It works in a same way to playlist style pages for the YT information extractors.

											
										
										
											2014-11-17 17:52:00 -05:00
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								class VKUserVideosIE(VKBaseIE):
-												[vk] Clarify extractor names

											
										
										
											2015-07-18 07:23:33 -04:00
+								    IE_NAME = 'vk:uservideos'
 								    IE_DESC = "VK - User's Videos"
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								    _VALID_URL = r'https?://(?:(?:m|new)\.)?vk\.com/videos(?P<id>-?[0-9]+)(?!\?.*\bz=video)(?:[/?#&](?:.*?\bsection=(?P<section>\w+))?|$)'
-												[vk] Added a new information extractor for pages that are a list of a user\'s videos on vk.com. It works in a same way to playlist style pages for the YT information extractors.

											
										
										
											2014-11-17 17:52:00 -05:00
+								    _TEMPLATE_URL = 'https://vk.com/videos'
-												[vk:uservideos] Improve extraction

											
										
										
											2015-07-18 07:22:25 -04:00
+								    _TESTS = [{
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								        'url': 'https://vk.com/videos-767561',
 								        'info_dict': {
 								            'id': '-767561_all',
 								        },
 								        'playlist_mincount': 1150,
 								    }, {
 								        'url': 'https://vk.com/videos-767561?section=uploaded',
-												[vk] Amend playlist test

											
										
										
											2015-02-17 18:33:41 -05:00
+								        'info_dict': {
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								            'id': '-767561_uploaded',
-												[vk] Amend playlist test

											
										
										
											2015-02-17 18:33:41 -05:00
+								        },
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								        'playlist_mincount': 425,
 								    }, {
 								        'url': 'http://vk.com/videos205387401',
 								        'only_matching': True,
-												[vk:uservideos] Improve extraction

											
										
										
											2015-07-18 07:22:25 -04:00
+								    }, {
 								        'url': 'http://vk.com/videos-77521',
 								        'only_matching': True,
-												[vk:uservideos] Improve _VALID_URL (Closes #8389)

											
										
										
											2016-02-01 13:52:37 -05:00
+								    }, {
 								        'url': 'http://vk.com/videos-97664626?section=all',
 								        'only_matching': True,
-												[vk] Extend _VALID_URLs to support new domain (Closes #9981)

											
										
										
											2016-07-02 05:43:19 -04:00
+								    }, {
 								        'url': 'http://m.vk.com/videos205387401',
 								        'only_matching': True,
 								    }, {
 								        'url': 'http://new.vk.com/videos205387401',
 								        'only_matching': True,
-												[vk:uservideos] Improve extraction

											
										
										
											2015-07-18 07:22:25 -04:00
+								    }]
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								    _PAGE_SIZE = 1000
 								    _VIDEO = collections.namedtuple('Video', ['owner_id', 'id'])
-												[vk:uservideos] Improve extraction

											
										
										
											2015-07-18 07:22:25 -04:00
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								    def _fetch_page(self, page_id, section, page):
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								        l = self._download_payload('al_video', page_id, {
 								            'act': 'load_videos_silent',
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								            'offset': page * self._PAGE_SIZE,
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            'oid': page_id,
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								            'section': section,
 								        })[0][section]['list']
-												[vk:uservideos] Improve extraction

											
										
										
											2015-07-18 07:22:25 -04:00
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								        for video in l:
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								            v = self._VIDEO._make(video[:2])
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            video_id = '%d_%d' % (v.owner_id, v.id)
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								            yield self.url_result(
 								                'http://vk.com/video' + video_id, VKIE.ie_key(), video_id)
 								    def _real_extract(self, url):
 								        page_id, section = re.match(self._VALID_URL, url).groups()
 								        if not section:
 								            section = 'all'
 								        entries = OnDemandPagedList(
 								            functools.partial(self._fetch_page, page_id, section),
 								            self._PAGE_SIZE)
-												[vk:uservideos] Improve extraction

											
										
										
											2015-07-18 07:22:25 -04:00
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								        return self.playlist_result(entries, '%s_%s' % (page_id, section))
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
 								class VKWallPostIE(VKBaseIE):
 								    IE_NAME = 'vk:wallpost'
 								    _VALID_URL = r'https?://(?:(?:(?:(?:m|new)\.)?vk\.com/(?:[^?]+\?.*\bw=)?wall(?P<id>-?\d+_\d+)))'
 								    _TESTS = [{
 								        # public page URL, audio playlist
 								        'url': 'https://vk.com/bs.official?w=wall-23538238_35',
 								        'info_dict': {
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            'id': '-23538238_35',
 								            'title': 'Black Shadow - Wall post -23538238_35',
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								            'description': 'md5:3f84b9c4f9ef499731cf1ced9998cc0c',
 								        },
 								        'playlist': [{
 								            'md5': '5ba93864ec5b85f7ce19a9af4af080f6',
 								            'info_dict': {
 								                'id': '135220665_111806521',
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                'ext': 'mp4',
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								                'title': 'Black Shadow - Слепое Верование',
 								                'duration': 370,
 								                'uploader': 'Black Shadow',
 								                'artist': 'Black Shadow',
 								                'track': 'Слепое Верование',
 								            },
 								        }, {
 								            'md5': '4cc7e804579122b17ea95af7834c9233',
 								            'info_dict': {
 								                'id': '135220665_111802303',
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                'ext': 'mp4',
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								                'title': 'Black Shadow - Война - Негасимое Бездны Пламя!',
 								                'duration': 423,
 								                'uploader': 'Black Shadow',
 								                'artist': 'Black Shadow',
 								                'track': 'Война - Негасимое Бездны Пламя!',
 								            },
 								        }],
-												[vk:wallpost] Fix audio extraction

											
										
										
											2016-08-17 19:14:05 -04:00
+								        'params': {
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            'skip_download': True,
-												[vk:wallpost] Fix audio extraction

											
										
										
											2016-08-17 19:14:05 -04:00
+								            'usenetrc': True,
 								        },
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								        'skip': 'Requires vk account credentials',
 								    }, {
 								        # single YouTube embed, no leading -
 								        'url': 'https://vk.com/wall85155021_6319',
 								        'info_dict': {
 								            'id': '85155021_6319',
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            'title': 'Сергей Горбунов - Wall post 85155021_6319',
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								        },
 								        'playlist_count': 1,
-												[vk:wallpost] Fix audio extraction

											
										
										
											2016-08-17 19:14:05 -04:00
+								        'params': {
 								            'usenetrc': True,
 								        },
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
+								        'skip': 'Requires vk account credentials',
 								    }, {
 								        # wall page URL
 								        'url': 'https://vk.com/wall-23538238_35',
 								        'only_matching': True,
 								    }, {
 								        # mobile wall page URL
 								        'url': 'https://m.vk.com/wall-23538238_35',
 								        'only_matching': True,
 								    }]
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								    _BASE64_CHARS = 'abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMN0PQRSTUVWXYZO123456789+/='
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								    _AUDIO = collections.namedtuple('Audio', ['id', 'owner_id', 'url', 'title', 'performer', 'duration', 'album_id', 'unk', 'author_link', 'lyrics', 'flags', 'context', 'extra', 'hashes', 'cover_url', 'ads'])
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
 								    def _decode(self, enc):
 								        dec = ''
 								        e = n = 0
 								        for c in enc:
 								            r = self._BASE64_CHARS.index(c)
 								            cond = n % 4
 								            e = 64 * e + r if cond else r
 								            n += 1
 								            if cond:
 								                dec += chr(255 & e >> (-2 * n & 6))
 								        return dec
 								    def _unmask_url(self, mask_url, vk_id):
 								        if 'audio_api_unavailable' in mask_url:
 								            extra = mask_url.split('?extra=')[1].split('#')
 								            func, base = self._decode(extra[1]).split(chr(11))
 								            mask_url = list(self._decode(extra[0]))
 								            url_len = len(mask_url)
 								            indexes = [None] * url_len
 								            index = int(base) ^ vk_id
 								            for n in range(url_len - 1, -1, -1):
 								                index = (url_len * (n + 1) ^ index + n) % url_len
 								                indexes[n] = index
 								            for n in range(1, url_len):
 								                c = mask_url[n]
 								                index = indexes[url_len - 1 - n]
 								                mask_url[n] = mask_url[index]
 								                mask_url[index] = c
 								            mask_url = ''.join(mask_url)
 								        return mask_url
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
 								    def _real_extract(self, url):
 								        post_id = self._match_id(url)
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								        webpage = self._download_payload('wkview', post_id, {
 								            'act': 'show',
 								            'w': 'wall' + post_id,
 								        })[1]
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
 								        description = clean_html(get_element_by_class('wall_post_text', webpage))
-												[vk:wallpost] Fix audio extraction

											
										
										
											2016-08-17 19:14:05 -04:00
+								        uploader = clean_html(get_element_by_class('author', webpage))
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
 								        entries = []
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								        for audio in re.findall(r'data-audio="([^"]+)', webpage):
 								            audio = self._parse_json(unescapeHTML(audio), post_id)
-												[vk] improve extraction

- fix User Videos extraction(closes #23356)
- extract all videos for lists with more than 1000 videos(#23356)
- add support for video albums(closes #14327)(closes #14492)

											
										
										
											2019-12-09 03:13:02 -05:00
+								            a = self._AUDIO._make(audio[:16])
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            if not a.url:
 								                continue
 								            title = unescapeHTML(a.title)
-												[vk] fix wall audio thumbnails extraction(closes #23135)

											
										
										
											2019-11-18 06:51:25 -05:00
+								            performer = unescapeHTML(a.performer)
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								            entries.append({
 								                'id': '%s_%s' % (a.owner_id, a.id),
 								                'url': self._unmask_url(a.url, a.ads['vk_id']),
-												[vk] fix wall audio thumbnails extraction(closes #23135)

											
										
										
											2019-11-18 06:51:25 -05:00
+								                'title': '%s - %s' % (performer, title) if performer else title,
 								                'thumbnails': [{'url': c_url} for c_url in a.cover_url.split(',')] if a.cover_url else None,
 								                'duration': int_or_none(a.duration),
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                'uploader': uploader,
-												[vk] fix wall audio thumbnails extraction(closes #23135)

											
										
										
											2019-11-18 06:51:25 -05:00
+								                'artist': performer,
-												[vk] improve extraction

- add support for Odnoklassniki embeds
- update tests
- extract more video from user lists(closes #4470)
- fix wall post audio extraction(closes #18332)
- improve error detection(closes #22568)

											
										
										
											2019-10-25 14:35:07 -04:00
+								                'track': title,
 								                'ext': 'mp4',
 								                'protocol': 'm3u8',
 								            })
-												[vk:wallpost] Add extractor

											
										
										
											2016-07-13 10:51:44 -04:00
 								        for video in re.finditer(
 								                r'<a[^>]+href=(["\'])(?P<url>/video(?:-?[\d_]+).*?)\1', webpage):
 								            entries.append(self.url_result(
 								                compat_urlparse.urljoin(url, video.group('url')), VKIE.ie_key()))
 								        title = 'Wall post %s' % post_id
 								        return self.playlist_result(
 								            orderedSet(entries), post_id,
 								            '%s - %s' % (uploader, title) if uploader else title,
 								            description)