diff options
| author | Gabriel Schubiner <gabeos@cs.washington.edu> | 2014-10-19 22:47:05 -0700 | 
|---|---|---|
| committer | Gabriel Schubiner <gabeos@cs.washington.edu> | 2014-10-19 22:47:05 -0700 | 
| commit | 8230018c20595a22e636b834ebb522a6a85d0d8b (patch) | |
| tree | 5b2bb8cd19bbc8f2ca2830fc93912742eca5c339 | |
| parent | cc98a3f09651423f4405933265c6dcf2ae51011f (diff) | |
Added extractor for crunchyroll 'playlists' i.e. series. so that one can, e.g. download all episodes of a series
| -rw-r--r-- | youtube_dl/extractor/__init__.py | 5 | ||||
| -rw-r--r-- | youtube_dl/extractor/crunchyroll.py | 35 | 
2 files changed, 39 insertions, 1 deletions
diff --git a/youtube_dl/extractor/__init__.py b/youtube_dl/extractor/__init__.py index 070f9ff19..0dd763006 100644 --- a/youtube_dl/extractor/__init__.py +++ b/youtube_dl/extractor/__init__.py @@ -60,7 +60,10 @@ from .comedycentral import ComedyCentralIE, ComedyCentralShowsIE  from .condenast import CondeNastIE  from .cracked import CrackedIE  from .criterion import CriterionIE -from .crunchyroll import CrunchyrollIE +from .crunchyroll import ( +    CrunchyrollIE, +    CrunchyrollShowPlaylistIE +)  from .cspan import CSpanIE  from .d8 import D8IE  from .dailymotion import ( diff --git a/youtube_dl/extractor/crunchyroll.py b/youtube_dl/extractor/crunchyroll.py index f99888ecc..414c46b0d 100644 --- a/youtube_dl/extractor/crunchyroll.py +++ b/youtube_dl/extractor/crunchyroll.py @@ -24,6 +24,7 @@ from ..aes import (      aes_cbc_decrypt,      inc,  ) +from .common import InfoExtractor  class CrunchyrollIE(SubtitlesInfoExtractor): @@ -285,3 +286,37 @@ Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text              'subtitles':   subtitles,              'formats':     formats,          } + + +class CrunchyrollShowPlaylistIE(InfoExtractor): +    IE_NAME = "crunchyroll:playlist" +    _VALID_URL = r'https?://(?:(?P<prefix>www|m)\.)?(?P<url>crunchyroll\.com/(?!(?:news|anime-news|library|forum|launchcalendar|lineup|store|comics|freetrial|login))(?P<show>[\w\-]+))/?$' +    _TITLE_EXTR = r'<span\s+itemprop="name">\s*(?P<showtitle>[\w\s]+)' + +    _TESTS = [{ +        'url' : 'http://www.crunchyroll.com/attack-on-titan', +        'info_dict' : { +            'title' : 'Attack on Titan' +        }, +        'playlist_count' : 15 +    }] + +    def _extract_title_entries(self,id,webpage): +        _EPISODE_ID_EXTR = r'id="showview_videos_media_(?P<vidid>\d+)".*?href="/{0}/(?P<vidurl>[\w\-]+-(?P=vidid))"'.format(id) +        title = self._html_search_regex(self._TITLE_EXTR,webpage,"title",flags=re.UNICODE|re.MULTILINE) +        episode_urls = [self.url_result('http://www.crunchyroll.com/{0}/{1}'.format(id, showmatch[1])) for +                    showmatch in re.findall(_EPISODE_ID_EXTR, webpage,re.UNICODE|re.MULTILINE|re.DOTALL)] +        return title, episode_urls + + +    def _real_extract(self, url): +        url_match = re.match(self._VALID_URL,url) +        show_id = url_match.group('show') +        webpage = self._download_webpage(url,show_id) +        (title,entries) = self._extract_title_entries(show_id,webpage) +        return { +            '_type' : 'playlist', +            'id' : show_id, +            'title' : title, +            'entries' : entries +        }
\ No newline at end of file  | 
