diff options
| author | Camillo Dell'mour <cdellmour@gmail.com> | 2013-12-08 14:02:14 +0100 | 
|---|---|---|
| committer | Camillo Dell'mour <cdellmour@gmail.com> | 2013-12-08 14:43:15 +0100 | 
| commit | 56a8ab7d6073493856d7e0ab8c5d1f128e3d475b (patch) | |
| tree | 6b2a3bcf545e4847c0d01df91f929d852ef9e97c | |
| parent | 22686b91f0883e3686d039a5ddc104abd6cf3b26 (diff) | |
added arte.tv extractor support for subdomain ddc - Mit offenen Karten(german) Le Dessous des Cartes(france)
| -rw-r--r-- | youtube_dl/extractor/__init__.py | 1 | ||||
| -rw-r--r-- | youtube_dl/extractor/arte.py | 28 | 
2 files changed, 29 insertions, 0 deletions
| diff --git a/youtube_dl/extractor/__init__.py b/youtube_dl/extractor/__init__.py index d7cb6e463..7ecafb104 100644 --- a/youtube_dl/extractor/__init__.py +++ b/youtube_dl/extractor/__init__.py @@ -8,6 +8,7 @@ from .arte import (      ArteTVPlus7IE,      ArteTVCreativeIE,      ArteTVFutureIE, +    ArteTVDDCIE,  )  from .auengine import AUEngineIE  from .bambuser import BambuserIE, BambuserChannelIE diff --git a/youtube_dl/extractor/arte.py b/youtube_dl/extractor/arte.py index 56a5d009f..986643612 100644 --- a/youtube_dl/extractor/arte.py +++ b/youtube_dl/extractor/arte.py @@ -10,6 +10,7 @@ from ..utils import (      determine_ext,      get_element_by_id,      compat_str, +    get_element_by_attribute,  )  # There are different sources of video in arte.tv, the extraction process  @@ -142,7 +143,9 @@ class ArteTVPlus7IE(InfoExtractor):      def _extract_from_webpage(self, webpage, video_id, lang):          json_url = self._html_search_regex(r'arte_vp_url="(.*?)"', webpage, 'json url') +        return self._extract_from_json_url(json_url, video_id, lang) +    def _extract_from_json_url(self, json_url, video_id, lang):          json_info = self._download_webpage(json_url, video_id, 'Downloading info json')          self.report_extraction(video_id)          info = json.loads(json_info) @@ -257,3 +260,28 @@ class ArteTVFutureIE(ArteTVPlus7IE):          webpage = self._download_webpage(url, anchor_id)          row = get_element_by_id(anchor_id, webpage)          return self._extract_from_webpage(row, anchor_id, lang) + +class ArteTVDDCIE(ArteTVPlus7IE): +    IE_NAME = u'arte.tv:ddc' +    _VALID_URL = r'http?://ddc\.arte\.tv/(?P<lang>emission|folge)/(?P<id>.+)' + +    _TEST = { +        u'url': u'http://ddc.arte.tv/folge/neues-aus-mauretanien', +        u'file': u'Mit offenen Karten-049881-009_PLUS7-D.flv', +        u'info_dict': { +            u'title': u'neues-aus-mauretanien', +        }, +    } + +    def _real_extract(self, url): +        video_id, lang = self._extract_url_info(url) +        if lang == 'folge': +            lang = 'de' +        elif lang == 'emission': +            lang = 'fr' +        webpage = self._download_webpage(url, video_id) +        scriptElement = get_element_by_attribute('class', 'visu_video_block', webpage) +        script_url = self._html_search_regex(r'src="(.*?)"', scriptElement, 'script url') +        javascriptPlayerGenerator = self._download_webpage(script_url, video_id, 'Download javascript player generator') +        json_url = self._search_regex(r"json_url=(.*)&rendering_place.*", javascriptPlayerGenerator, 'json url') +        return self._extract_from_json_url(json_url, video_id, lang) | 
