aboutsummaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorPhilipp Hagemeister <phihag@phihag.de>2014-10-26 17:28:09 +0100
committerPhilipp Hagemeister <phihag@phihag.de>2014-10-26 17:28:09 +0100
commit09e5d6a6e564ecaa44a174456882bd1998eacbf8 (patch)
treeff7441ca3ca98e338d7264753f30355308de8ef8
parent274b12b5a8229242cd750fa95205ab63621c2c40 (diff)
downloadyoutube-dl-09e5d6a6e564ecaa44a174456882bd1998eacbf8.tar.xz
[crunchyroll:playlist] Simplify (#3988)
-rw-r--r--youtube_dl/extractor/crunchyroll.py50
1 files changed, 26 insertions, 24 deletions
diff --git a/youtube_dl/extractor/crunchyroll.py b/youtube_dl/extractor/crunchyroll.py
index 2dca52660..05b21e872 100644
--- a/youtube_dl/extractor/crunchyroll.py
+++ b/youtube_dl/extractor/crunchyroll.py
@@ -293,34 +293,36 @@ Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
class CrunchyrollShowPlaylistIE(InfoExtractor):
IE_NAME = "crunchyroll:playlist"
- _VALID_URL = r'https?://(?:(?P<prefix>www|m)\.)?(?P<url>crunchyroll\.com/(?!(?:news|anime-news|library|forum|launchcalendar|lineup|store|comics|freetrial|login))(?P<show>[\w\-]+))/?$'
- _TITLE_EXTR = r'<span\s+itemprop="name">\s*(?P<showtitle>[\w\s]+)'
+ _VALID_URL = r'https?://(?:(?P<prefix>www|m)\.)?(?P<url>crunchyroll\.com/(?!(?:news|anime-news|library|forum|launchcalendar|lineup|store|comics|freetrial|login))(?P<id>[\w\-]+))/?$'
_TESTS = [{
- 'url' : 'http://www.crunchyroll.com/attack-on-titan',
- 'info_dict' : {
- 'title' : 'Attack on Titan'
+ 'url': 'http://www.crunchyroll.com/a-bridge-to-the-starry-skies-hoshizora-e-kakaru-hashi',
+ 'info_dict': {
+ 'id': 'a-bridge-to-the-starry-skies-hoshizora-e-kakaru-hashi',
+ 'title': 'A Bridge to the Starry Skies - Hoshizora e Kakaru Hashi'
},
- 'playlist_count' : 15
+ 'playlist_count': 13,
}]
- def _extract_title_entries(self,id,webpage):
- _EPISODE_ID_EXTR = r'id="showview_videos_media_(?P<vidid>\d+)".*?href="/{0}/(?P<vidurl>[\w\-]+-(?P=vidid))"'.format(id)
- title = self._html_search_regex(self._TITLE_EXTR,webpage,"title",flags=re.UNICODE|re.MULTILINE)
- episode_urls = [self.url_result('http://www.crunchyroll.com/{0}/{1}'.format(id, showmatch[1])) for
- showmatch in re.findall(_EPISODE_ID_EXTR, webpage,re.UNICODE|re.MULTILINE|re.DOTALL)]
- episode_urls.reverse()
- return title, episode_urls
-
-
def _real_extract(self, url):
- url_match = re.match(self._VALID_URL,url)
- show_id = url_match.group('show')
- webpage = self._download_webpage(url,show_id)
- (title,entries) = self._extract_title_entries(show_id,webpage)
+ show_id = self._match_id(url)
+
+ webpage = self._download_webpage(url, show_id)
+ title = self._html_search_regex(
+ r'(?s)<h1[^>]*>\s*<span itemprop="name">(.*?)</span>',
+ webpage, 'title')
+ episode_paths = re.findall(
+ r'(?s)<li id="showview_videos_media_[0-9]+"[^>]+>.*?<a href="([^"]+)"',
+ webpage)
+ entries = [
+ self.url_result('http://www.crunchyroll.com' + ep, 'Crunchyroll')
+ for ep in episode_paths
+ ]
+ entries.reverse()
+
return {
- '_type' : 'playlist',
- 'id' : show_id,
- 'title' : title,
- 'entries' : entries
- } \ No newline at end of file
+ '_type': 'playlist',
+ 'id': show_id,
+ 'title': title,
+ 'entries': entries,
+ }