[bilibili] Extract multipart videos (closes #3250)

This commit is contained in:
Yen Chi Hsuan 2015-04-30 18:23:35 +08:00
parent 621ffe7bf4
commit c4a21bc9db

View file

@ -2,6 +2,7 @@
from __future__ import unicode_literals from __future__ import unicode_literals
import re import re
import itertools
from .common import InfoExtractor from .common import InfoExtractor
from ..utils import ( from ..utils import (
@ -14,18 +15,25 @@ from ..utils import (
class BiliBiliIE(InfoExtractor): class BiliBiliIE(InfoExtractor):
_VALID_URL = r'http://www\.bilibili\.(?:tv|com)/video/av(?P<id>[0-9]+)/' _VALID_URL = r'http://www\.bilibili\.(?:tv|com)/video/av(?P<id>[0-9]+)/'
_TEST = { _TESTS = [{
'url': 'http://www.bilibili.tv/video/av1074402/', 'url': 'http://www.bilibili.tv/video/av1074402/',
'md5': '2c301e4dab317596e837c3e7633e7d86', 'md5': '2c301e4dab317596e837c3e7633e7d86',
'info_dict': { 'info_dict': {
'id': '1074402', 'id': '1074402_part1',
'ext': 'flv', 'ext': 'flv',
'title': '【金坷垃】金泡沫', 'title': '【金坷垃】金泡沫',
'duration': 308, 'duration': 308,
'upload_date': '20140420', 'upload_date': '20140420',
'thumbnail': 're:^https?://.+\.jpg', 'thumbnail': 're:^https?://.+\.jpg',
}, },
} }, {
'url': 'http://www.bilibili.com/video/av1041170/',
'info_dict': {
'id': '1041170',
'title': '【BD1080P】刀语【诸神&异域】',
},
'playlist_count': 9,
}]
def _real_extract(self, url): def _real_extract(self, url):
video_id = self._match_id(url) video_id = self._match_id(url)
@ -57,19 +65,14 @@ class BiliBiliIE(InfoExtractor):
cid = self._search_regex(r'cid=(\d+)', webpage, 'cid') cid = self._search_regex(r'cid=(\d+)', webpage, 'cid')
entries = []
lq_doc = self._download_xml( lq_doc = self._download_xml(
'http://interface.bilibili.com/v_cdn_play?appkey=1&cid=%s' % cid, 'http://interface.bilibili.com/v_cdn_play?appkey=1&cid=%s' % cid,
video_id, video_id,
note='Downloading LQ video info' note='Downloading LQ video info'
) )
lq_durl = lq_doc.find('./durl') lq_durls = lq_doc.findall('./durl')
formats = [{
'format_id': 'lq',
'quality': 1,
'url': lq_durl.find('./url').text,
'filesize': int_or_none(
lq_durl.find('./size'), get_attr='text'),
}]
hq_doc = self._download_xml( hq_doc = self._download_xml(
'http://interface.bilibili.com/playurl?appkey=1&cid=%s' % cid, 'http://interface.bilibili.com/playurl?appkey=1&cid=%s' % cid,
@ -77,8 +80,20 @@ class BiliBiliIE(InfoExtractor):
note='Downloading HQ video info', note='Downloading HQ video info',
fatal=False, fatal=False,
) )
if hq_doc is not False: hq_durls = hq_doc.findall('./durl') if hq_doc is not False else itertools.repeat(None)
hq_durl = hq_doc.find('./durl')
assert len(lq_durls) == len(hq_durls)
i = 1
for lq_durl, hq_durl in zip(lq_durls, hq_durls):
formats = [{
'format_id': 'lq',
'quality': 1,
'url': lq_durl.find('./url').text,
'filesize': int_or_none(
lq_durl.find('./size'), get_attr='text'),
}]
if hq_durl:
formats.append({ formats.append({
'format_id': 'hq', 'format_id': 'hq',
'quality': 2, 'quality': 2,
@ -87,13 +102,22 @@ class BiliBiliIE(InfoExtractor):
'filesize': int_or_none( 'filesize': int_or_none(
hq_durl.find('./size'), get_attr='text'), hq_durl.find('./size'), get_attr='text'),
}) })
self._sort_formats(formats) self._sort_formats(formats)
return {
'id': video_id, entries.append({
'id': '%s_part%d' % (video_id, i),
'title': title, 'title': title,
'formats': formats, 'formats': formats,
'duration': duration, 'duration': duration,
'upload_date': upload_date, 'upload_date': upload_date,
'thumbnail': thumbnail, 'thumbnail': thumbnail,
})
i += 1
return {
'_type': 'multi_video',
'entries': entries,
'id': video_id,
'title': title
} }