diff options
| author | Ismael Mejia <iemejia@gmail.com> | 2013-09-11 14:26:48 +0200 | 
|---|---|---|
| committer | Ismael Mejia <iemejia@gmail.com> | 2013-09-11 14:26:48 +0200 | 
| commit | cf1dd0c59e8ff4c66130bc47954c415be5db8cc3 (patch) | |
| tree | a39741b945c88b9a7b3bc14c94efdc6fe7b3302f /youtube_dl/extractor/slideshare.py | |
| parent | 72836fcee453386f4f16325c5b8fa4c1ba1bb442 (diff) | |
| parent | 22c8b525458c8203c0709a523c646b8d152f03b0 (diff) | |
Merge branch 'master' into subtitles_rework
Diffstat (limited to 'youtube_dl/extractor/slideshare.py')
| -rw-r--r-- | youtube_dl/extractor/slideshare.py | 47 | 
1 files changed, 47 insertions, 0 deletions
| diff --git a/youtube_dl/extractor/slideshare.py b/youtube_dl/extractor/slideshare.py new file mode 100644 index 000000000..afc3001b5 --- /dev/null +++ b/youtube_dl/extractor/slideshare.py @@ -0,0 +1,47 @@ +import re +import json + +from .common import InfoExtractor +from ..utils import ( +    compat_urlparse, +    ExtractorError, +) + + +class SlideshareIE(InfoExtractor): +    _VALID_URL = r'https?://www\.slideshare\.net/[^/]+?/(?P<title>.+?)($|\?)' + +    _TEST = { +        u'url': u'http://www.slideshare.net/Dataversity/keynote-presentation-managing-scale-and-complexity', +        u'file': u'25665706.mp4', +        u'info_dict': { +            u'title': u'Managing Scale and Complexity', +            u'description': u'This was a keynote presentation at the NoSQL Now! 2013 Conference & Expo (http://www.nosqlnow.com). This presentation was given by Adrian Cockcroft from Netflix', +        }, +    } + +    def _real_extract(self, url): +        mobj = re.match(self._VALID_URL, url) +        page_title = mobj.group('title') +        webpage = self._download_webpage(url, page_title) +        slideshare_obj = self._search_regex( +            r'var slideshare_object =  ({.*?}); var user_info =', +            webpage, u'slideshare object') +        info = json.loads(slideshare_obj) +        if info['slideshow']['type'] != u'video': +            raise ExtractorError(u'Webpage type is "%s": only video extraction is supported for Slideshare' % info['slideshow']['type'], expected=True) + +        doc = info['doc'] +        bucket = info['jsplayer']['video_bucket'] +        ext = info['jsplayer']['video_extension'] +        video_url = compat_urlparse.urljoin(bucket, doc + '-SD.' + ext) + +        return { +            '_type': 'video', +            'id': info['slideshow']['id'], +            'title': info['slideshow']['title'], +            'ext': ext, +            'url': video_url, +            'thumbnail': info['slideshow']['pin_image_url'], +            'description': self._og_search_description(webpage), +        } | 
