[xtube] Fix extraction add more metafields

This commit is contained in:
Sergey M․ 2014-03-04 16:12:11 +07:00
parent 17b75c0de1
commit 607dbbad76

View File

@ -7,19 +7,24 @@
from ..utils import ( from ..utils import (
compat_urllib_parse_urlparse, compat_urllib_parse_urlparse,
compat_urllib_request, compat_urllib_request,
parse_duration,
str_to_int,
) )
class XTubeIE(InfoExtractor): class XTubeIE(InfoExtractor):
_VALID_URL = r'^(?:https?://)?(?:www\.)?(?P<url>xtube\.com/watch\.php\?v=(?P<videoid>[^/?&]+))' _VALID_URL = r'https?://(?:www\.)?(?P<url>xtube\.com/watch\.php\?v=(?P<videoid>[^/?&]+))'
_TEST = { _TEST = {
'url': 'http://www.xtube.com/watch.php?v=kVTUy_G222_', 'url': 'http://www.xtube.com/watch.php?v=kVTUy_G222_',
'file': 'kVTUy_G222_.mp4',
'md5': '092fbdd3cbe292c920ef6fc6a8a9cdab', 'md5': '092fbdd3cbe292c920ef6fc6a8a9cdab',
'info_dict': { 'info_dict': {
"title": "strange erotica", 'id': 'kVTUy_G222_',
"description": "surreal gay themed erotica...almost an ET kind of thing", 'ext': 'mp4',
"uploader": "greenshowers", 'title': 'strange erotica',
"age_limit": 18, 'description': 'surreal gay themed erotica...almost an ET kind of thing',
'uploader': 'greenshowers',
'duration': 450,
'age_limit': 18,
} }
} }
@ -32,10 +37,23 @@ def _real_extract(self, url):
req.add_header('Cookie', 'age_verified=1') req.add_header('Cookie', 'age_verified=1')
webpage = self._download_webpage(req, video_id) webpage = self._download_webpage(req, video_id)
video_title = self._html_search_regex(r'<div class="p_5px[^>]*>([^<]+)', webpage, 'title') video_title = self._html_search_regex(r'<p class="title">([^<]+)', webpage, 'title')
video_uploader = self._html_search_regex(r'so_s\.addVariable\("owner_u", "([^"]+)', webpage, 'uploader', fatal=False) video_uploader = self._html_search_regex(
video_description = self._html_search_regex(r'<p class="video_description">([^<]+)', webpage, 'description', fatal=False) r'so_s\.addVariable\("owner_u", "([^"]+)', webpage, 'uploader', fatal=False)
video_url= self._html_search_regex(r'var videoMp4 = "([^"]+)', webpage, 'video_url').replace('\\/', '/') video_description = self._html_search_regex(
r'<p class="fieldsDesc">([^<]+)', webpage, 'description', fatal=False)
video_url = self._html_search_regex(r'var videoMp4 = "([^"]+)', webpage, 'video_url').replace('\\/', '/')
duration = parse_duration(self._html_search_regex(
r'<span class="bold">Runtime:</span> ([^<]+)</p>', webpage, 'duration', fatal=False))
view_count = self._html_search_regex(
r'<span class="bold">Views:</span> ([\d,\.]+)</p>', webpage, 'view count', fatal=False)
if view_count:
view_count = str_to_int(view_count)
comment_count = self._html_search_regex(
r'<div id="commentBar">([\d,\.]+) Comments</div>', webpage, 'comment count', fatal=False)
if comment_count:
comment_count = str_to_int(comment_count)
path = compat_urllib_parse_urlparse(video_url).path path = compat_urllib_parse_urlparse(video_url).path
extension = os.path.splitext(path)[1][1:] extension = os.path.splitext(path)[1][1:]
format = path.split('/')[5].split('_')[:2] format = path.split('/')[5].split('_')[:2]
@ -48,6 +66,9 @@ def _real_extract(self, url):
'title': video_title, 'title': video_title,
'uploader': video_uploader, 'uploader': video_uploader,
'description': video_description, 'description': video_description,
'duration': duration,
'view_count': view_count,
'comment_count': comment_count,
'url': video_url, 'url': video_url,
'ext': extension, 'ext': extension,
'format': format, 'format': format,