diff options
author | Philipp Hagemeister <phihag@phihag.de> | 2014-10-26 17:28:09 +0100 |
---|---|---|
committer | Philipp Hagemeister <phihag@phihag.de> | 2014-10-26 17:28:09 +0100 |
commit | 09e5d6a6e564ecaa44a174456882bd1998eacbf8 (patch) | |
tree | ff7441ca3ca98e338d7264753f30355308de8ef8 /youtube_dl/extractor | |
parent | 274b12b5a8229242cd750fa95205ab63621c2c40 (diff) |
[crunchyroll:playlist] Simplify (#3988)
Diffstat (limited to 'youtube_dl/extractor')
-rw-r--r-- | youtube_dl/extractor/crunchyroll.py | 50 |
1 files changed, 26 insertions, 24 deletions
diff --git a/youtube_dl/extractor/crunchyroll.py b/youtube_dl/extractor/crunchyroll.py index 2dca52660..05b21e872 100644 --- a/youtube_dl/extractor/crunchyroll.py +++ b/youtube_dl/extractor/crunchyroll.py @@ -293,34 +293,36 @@ Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text class CrunchyrollShowPlaylistIE(InfoExtractor): IE_NAME = "crunchyroll:playlist" - _VALID_URL = r'https?://(?:(?P<prefix>www|m)\.)?(?P<url>crunchyroll\.com/(?!(?:news|anime-news|library|forum|launchcalendar|lineup|store|comics|freetrial|login))(?P<show>[\w\-]+))/?$' - _TITLE_EXTR = r'<span\s+itemprop="name">\s*(?P<showtitle>[\w\s]+)' + _VALID_URL = r'https?://(?:(?P<prefix>www|m)\.)?(?P<url>crunchyroll\.com/(?!(?:news|anime-news|library|forum|launchcalendar|lineup|store|comics|freetrial|login))(?P<id>[\w\-]+))/?$' _TESTS = [{ - 'url' : 'http://www.crunchyroll.com/attack-on-titan', - 'info_dict' : { - 'title' : 'Attack on Titan' + 'url': 'http://www.crunchyroll.com/a-bridge-to-the-starry-skies-hoshizora-e-kakaru-hashi', + 'info_dict': { + 'id': 'a-bridge-to-the-starry-skies-hoshizora-e-kakaru-hashi', + 'title': 'A Bridge to the Starry Skies - Hoshizora e Kakaru Hashi' }, - 'playlist_count' : 15 + 'playlist_count': 13, }] - def _extract_title_entries(self,id,webpage): - _EPISODE_ID_EXTR = r'id="showview_videos_media_(?P<vidid>\d+)".*?href="/{0}/(?P<vidurl>[\w\-]+-(?P=vidid))"'.format(id) - title = self._html_search_regex(self._TITLE_EXTR,webpage,"title",flags=re.UNICODE|re.MULTILINE) - episode_urls = [self.url_result('http://www.crunchyroll.com/{0}/{1}'.format(id, showmatch[1])) for - showmatch in re.findall(_EPISODE_ID_EXTR, webpage,re.UNICODE|re.MULTILINE|re.DOTALL)] - episode_urls.reverse() - return title, episode_urls - - def _real_extract(self, url): - url_match = re.match(self._VALID_URL,url) - show_id = url_match.group('show') - webpage = self._download_webpage(url,show_id) - (title,entries) = self._extract_title_entries(show_id,webpage) + show_id = self._match_id(url) + + webpage = self._download_webpage(url, show_id) + title = self._html_search_regex( + r'(?s)<h1[^>]*>\s*<span itemprop="name">(.*?)</span>', + webpage, 'title') + episode_paths = re.findall( + r'(?s)<li id="showview_videos_media_[0-9]+"[^>]+>.*?<a href="([^"]+)"', + webpage) + entries = [ + self.url_result('http://www.crunchyroll.com' + ep, 'Crunchyroll') + for ep in episode_paths + ] + entries.reverse() + return { - '_type' : 'playlist', - 'id' : show_id, - 'title' : title, - 'entries' : entries - }
\ No newline at end of file + '_type': 'playlist', + 'id': show_id, + 'title': title, + 'entries': entries, + } |