[ehow] improve minor bits

This commit is contained in:
Philipp Hagemeister 2013-07-11 12:11:00 +02:00
parent 3fa9550837
commit 81082e046e
2 changed files with 17 additions and 10 deletions

View file

@ -13,7 +13,7 @@
from .depositfiles import DepositFilesIE from .depositfiles import DepositFilesIE
from .dotsub import DotsubIE from .dotsub import DotsubIE
from .dreisat import DreiSatIE from .dreisat import DreiSatIE
from .ehow import EhowIE from .ehow import EHowIE
from .eighttracks import EightTracksIE from .eighttracks import EightTracksIE
from .escapist import EscapistIE from .escapist import EscapistIE
from .facebook import FacebookIE from .facebook import FacebookIE

View file

@ -1,10 +1,15 @@
import re import re
from ..utils import compat_urllib_parse
from ..utils import (
compat_urllib_parse,
determine_ext
)
from .common import InfoExtractor from .common import InfoExtractor
class EhowIE(InfoExtractor): class EHowIE(InfoExtractor):
_VALID_URL = r'(?:http://)?(?:www\.)?ehow\.com/([^/]+)' IE_NAME = u'eHow'
_VALID_URL = r'(?:https?://)?(?:www\.)?ehow\.com/[^/_?]*_(?P<id>[0-9]+)'
_TEST = { _TEST = {
u'url': u'http://www.ehow.com/video_12245069_hardwood-flooring-basics.html', u'url': u'http://www.ehow.com/video_12245069_hardwood-flooring-basics.html',
u'file': u'12245069.flv', u'file': u'12245069.flv',
@ -18,9 +23,9 @@ class EhowIE(InfoExtractor):
def _real_extract(self, url): def _real_extract(self, url):
mobj = re.match(self._VALID_URL, url) mobj = re.match(self._VALID_URL, url)
video_id = mobj.group(1).split("_")[1] video_id = mobj.group('id')
webpage = self._download_webpage(url, video_id) webpage = self._download_webpage(url, video_id)
video_url = self._search_regex(r'[^A-Za-z0-9]?(?:file|source)=(http[^\'"&]*)', video_url = self._search_regex(r'(?:file|source)=(http[^\'"&]*)',
webpage, u'video URL') webpage, u'video URL')
final_url = compat_urllib_parse.unquote(video_url) final_url = compat_urllib_parse.unquote(video_url)
thumbnail_url = self._search_regex(r'<meta property="og:image" content="(.+?)" />', thumbnail_url = self._search_regex(r'<meta property="og:image" content="(.+?)" />',
@ -28,11 +33,13 @@ def _real_extract(self, url):
uploader = self._search_regex(r'<meta name="uploader" content="(.+?)" />', uploader = self._search_regex(r'<meta name="uploader" content="(.+?)" />',
webpage, u'uploader') webpage, u'uploader')
title = self._search_regex(r'<meta property="og:title" content="(.+?)" />', title = self._search_regex(r'<meta property="og:title" content="(.+?)" />',
webpage, u'Video title').replace(' | eHow','') webpage, u'Video title').replace(' | eHow', '')
description = self._search_regex(r'<meta property="og:description" content="(.+?)" />', description = self._search_regex(r'<meta property="og:description" content="(.+?)" />',
webpage, u'video description') webpage, u'video description')
ext = final_url.split('.')[-1] ext = determine_ext(final_url)
return [{
return {
'_type': 'video',
'id': video_id, 'id': video_id,
'url': final_url, 'url': final_url,
'ext': ext, 'ext': ext,
@ -40,5 +47,5 @@ def _real_extract(self, url):
'thumbnail': thumbnail_url, 'thumbnail': thumbnail_url,
'description': description, 'description': description,
'uploader': uploader, 'uploader': uploader,
}] }