diff options
Diffstat (limited to 'youtube_dlc/extractor/bilibili.py')
-rw-r--r-- | youtube_dlc/extractor/bilibili.py | 209 |
1 files changed, 199 insertions, 10 deletions
diff --git a/youtube_dlc/extractor/bilibili.py b/youtube_dlc/extractor/bilibili.py index d39ee8ffe..d8a4a224f 100644 --- a/youtube_dlc/extractor/bilibili.py +++ b/youtube_dlc/extractor/bilibili.py @@ -2,9 +2,10 @@ from __future__ import unicode_literals import hashlib +import json import re -from .common import InfoExtractor +from .common import InfoExtractor, SearchInfoExtractor from ..compat import ( compat_parse_qs, compat_urlparse, @@ -32,13 +33,14 @@ class BiliBiliIE(InfoExtractor): (?: video/[aA][vV]| anime/(?P<anime_id>\d+)/play\# - )(?P<id_bv>\d+)| - video/[bB][vV](?P<id>[^/?#&]+) + )(?P<id>\d+)| + video/[bB][vV](?P<id_bv>[^/?#&]+) ) + (?:/?\?p=(?P<page>\d+))? ''' _TESTS = [{ - 'url': 'http://www.bilibili.tv/video/av1074402/', + 'url': 'http://www.bilibili.com/video/av1074402/', 'md5': '5f7d29e1a2872f3df0cf76b1f87d3788', 'info_dict': { 'id': '1074402', @@ -57,6 +59,10 @@ class BiliBiliIE(InfoExtractor): 'url': 'http://bangumi.bilibili.com/anime/1869/play#40062', 'only_matching': True, }, { + # bilibili.tv + 'url': 'http://www.bilibili.tv/video/av1074402/', + 'only_matching': True, + }, { 'url': 'http://bangumi.bilibili.com/anime/5802/play#100643', 'md5': '3f721ad1e75030cc06faf73587cfec57', 'info_dict': { @@ -124,12 +130,20 @@ class BiliBiliIE(InfoExtractor): url, smuggled_data = unsmuggle_url(url, {}) mobj = re.match(self._VALID_URL, url) - video_id = mobj.group('id') or mobj.group('id_bv') + video_id = mobj.group('id_bv') or mobj.group('id') + + av_id, bv_id = self._get_video_id_set(video_id, mobj.group('id_bv') is not None) + video_id = av_id + anime_id = mobj.group('anime_id') + page_id = mobj.group('page') webpage = self._download_webpage(url, video_id) if 'anime/' not in url: cid = self._search_regex( + r'\bcid(?:["\']:|=)(\d+),["\']page(?:["\']:|=)' + str(page_id), webpage, 'cid', + default=None + ) or self._search_regex( r'\bcid(?:["\']:|=)(\d+)', webpage, 'cid', default=None ) or compat_parse_qs(self._search_regex( @@ -207,9 +221,9 @@ class BiliBiliIE(InfoExtractor): break title = self._html_search_regex( - ('<h1[^>]+\btitle=(["\'])(?P<title>(?:(?!\1).)+)\1', - '(?s)<h1[^>]*>(?P<title>.+?)</h1>'), webpage, 'title', - group='title') + (r'<h1[^>]+\btitle=(["\'])(?P<title>(?:(?!\1).)+)\1', + r'(?s)<h1[^>]*>(?P<title>.+?)</h1>'), webpage, 'title', + group='title') + ('_p' + str(page_id) if page_id is not None else '') description = self._html_search_meta('description', webpage) timestamp = unified_timestamp(self._html_search_regex( r'<time[^>]+datetime="([^"]+)"', webpage, 'upload time', @@ -219,7 +233,8 @@ class BiliBiliIE(InfoExtractor): # TODO 'view_count' requires deobfuscating Javascript info = { - 'id': video_id, + 'id': str(video_id) if page_id is None else '%s_p%s' % (video_id, page_id), + 'cid': cid, 'title': title, 'description': description, 'timestamp': timestamp, @@ -235,27 +250,134 @@ class BiliBiliIE(InfoExtractor): 'uploader': uploader_mobj.group('name'), 'uploader_id': uploader_mobj.group('id'), }) + if not info.get('uploader'): info['uploader'] = self._html_search_meta( 'author', webpage, 'uploader', default=None) + comments = None + if self._downloader.params.get('getcomments', False): + comments = self._get_all_comment_pages(video_id) + + raw_danmaku = self._get_raw_danmaku(video_id, cid) + + raw_tags = self._get_tags(video_id) + tags = list(map(lambda x: x['tag_name'], raw_tags)) + + top_level_info = { + 'raw_danmaku': raw_danmaku, + 'comments': comments, + 'comment_count': len(comments) if comments is not None else None, + 'tags': tags, + 'raw_tags': raw_tags, + } + + ''' + # Requires https://github.com/m13253/danmaku2ass which is licenced under GPL3 + # See https://github.com/animelover1984/youtube-dl + danmaku = NiconicoIE.CreateDanmaku(raw_danmaku, commentType='Bilibili', x=1024, y=576) + entries[0]['subtitles'] = { + 'danmaku': [{ + 'ext': 'ass', + 'data': danmaku + }] + } + ''' + for entry in entries: entry.update(info) if len(entries) == 1: + entries[0].update(top_level_info) return entries[0] else: for idx, entry in enumerate(entries): entry['id'] = '%s_part%d' % (video_id, (idx + 1)) - return { + global_info = { '_type': 'multi_video', 'id': video_id, + 'bv_id': bv_id, 'title': title, 'description': description, 'entries': entries, } + global_info.update(info) + global_info.update(top_level_info) + + return global_info + + def _get_video_id_set(self, id, is_bv): + query = {'bvid': id} if is_bv else {'aid': id} + response = self._download_json( + "http://api.bilibili.cn/x/web-interface/view", + id, query=query, + note='Grabbing original ID via API') + + if response['code'] == -400: + raise ExtractorError('Video ID does not exist', expected=True, video_id=id) + elif response['code'] != 0: + raise ExtractorError('Unknown error occurred during API check (code %s)' % response['code'], expected=True, video_id=id) + return (response['data']['aid'], response['data']['bvid']) + + # recursive solution to getting every page of comments for the video + # we can stop when we reach a page without any comments + def _get_all_comment_pages(self, video_id, commentPageNumber=0): + comment_url = "https://api.bilibili.com/x/v2/reply?jsonp=jsonp&pn=%s&type=1&oid=%s&sort=2&_=1567227301685" % (commentPageNumber, video_id) + json_str = self._download_webpage( + comment_url, video_id, + note='Extracting comments from page %s' % (commentPageNumber)) + replies = json.loads(json_str)['data']['replies'] + if replies is None: + return [] + return self._get_all_children(replies) + self._get_all_comment_pages(video_id, commentPageNumber + 1) + + # extracts all comments in the tree + def _get_all_children(self, replies): + if replies is None: + return [] + + ret = [] + for reply in replies: + author = reply['member']['uname'] + author_id = reply['member']['mid'] + id = reply['rpid'] + text = reply['content']['message'] + timestamp = reply['ctime'] + parent = reply['parent'] if reply['parent'] != 0 else 'root' + + comment = { + "author": author, + "author_id": author_id, + "id": id, + "text": text, + "timestamp": timestamp, + "parent": parent, + } + ret.append(comment) + + # from the JSON, the comment structure seems arbitrarily deep, but I could be wrong. + # Regardless, this should work. + ret += self._get_all_children(reply['replies']) + + return ret + + def _get_raw_danmaku(self, video_id, cid): + # This will be useful if I decide to scrape all pages instead of doing them individually + # cid_url = "https://www.bilibili.com/widget/getPageList?aid=%s" % (video_id) + # cid_str = self._download_webpage(cid_url, video_id, note=False) + # cid = json.loads(cid_str)[0]['cid'] + + danmaku_url = "https://comment.bilibili.com/%s.xml" % (cid) + danmaku = self._download_webpage(danmaku_url, video_id, note='Downloading danmaku comments') + return danmaku + + def _get_tags(self, video_id): + tags_url = "https://api.bilibili.com/x/tag/archive/tags?aid=%s" % (video_id) + tags_json = self._download_json(tags_url, video_id, note='Downloading tags') + return tags_json['data'] + class BiliBiliBangumiIE(InfoExtractor): _VALID_URL = r'https?://bangumi\.bilibili\.com/anime/(?P<id>\d+)' @@ -324,6 +446,73 @@ class BiliBiliBangumiIE(InfoExtractor): season_info.get('bangumi_title'), season_info.get('evaluate')) +class BilibiliChannelIE(InfoExtractor): + _VALID_URL = r'https?://space.bilibili\.com/(?P<id>\d+)' + # May need to add support for pagination? Need to find a user with many video uploads to test + _API_URL = "https://api.bilibili.com/x/space/arc/search?mid=%s&pn=1&ps=25&jsonp=jsonp" + _TEST = {} # TODO: Add tests + + def _real_extract(self, url): + list_id = self._match_id(url) + json_str = self._download_webpage(self._API_URL % list_id, "None") + + json_parsed = json.loads(json_str) + entries = [{ + '_type': 'url', + 'ie_key': BiliBiliIE.ie_key(), + 'url': ('https://www.bilibili.com/video/%s' % + entry['bvid']), + 'id': entry['bvid'], + } for entry in json_parsed['data']['list']['vlist']] + + return { + '_type': 'playlist', + 'id': list_id, + 'entries': entries + } + + +class BiliBiliSearchIE(SearchInfoExtractor): + IE_DESC = 'Bilibili video search, "bilisearch" keyword' + _MAX_RESULTS = 100000 + _SEARCH_KEY = 'bilisearch' + MAX_NUMBER_OF_RESULTS = 1000 + + def _get_n_results(self, query, n): + """Get a specified number of results for a query""" + + entries = [] + pageNumber = 0 + while True: + pageNumber += 1 + # FIXME + api_url = "https://api.bilibili.com/x/web-interface/search/type?context=&page=%s&order=pubdate&keyword=%s&duration=0&tids_2=&__refresh__=true&search_type=video&tids=0&highlight=1" % (pageNumber, query) + json_str = self._download_webpage( + api_url, "None", query={"Search_key": query}, + note='Extracting results from page %s' % pageNumber) + data = json.loads(json_str)['data'] + + # FIXME: this is hideous + if "result" not in data: + return { + '_type': 'playlist', + 'id': query, + 'entries': entries[:n] + } + + videos = data['result'] + for video in videos: + e = self.url_result(video['arcurl'], 'BiliBili', str(video['aid'])) + entries.append(e) + + if(len(entries) >= n or len(videos) >= BiliBiliSearchIE.MAX_NUMBER_OF_RESULTS): + return { + '_type': 'playlist', + 'id': query, + 'entries': entries[:n] + } + + class BilibiliAudioBaseIE(InfoExtractor): def _call_api(self, path, sid, query=None): if not query: |