[faz] fix extraction and add support for Perform Group embeds(fixes #14714)

author: Remita Amine <remitamine@gmail.com> 2017-11-24 18:42:41 +0100
committer: Remita Amine <remitamine@gmail.com> 2017-11-24 18:42:41 +0100
commit: e0a8686f48d10ed86f7be92132dd37481981adf3 (patch)
tree: 2c3f90569d5d4cf85514d32b5135c66f6db32f86 /youtube_dl/extractor/faz.py
parent: 6049176471b4c567c1e9bd8f77b298e323d5b8e7 (diff)
1 files changed, 27 insertions, 7 deletions
diff --git a/youtube_dl/extractor/faz.py b/youtube_dl/extractor/faz.py
index 4bc8fc512..312ee2aee 100644
--- a/youtube_dl/extractor/faz.py
+++ b/youtube_dl/extractor/faz.py
@@ -1,7 +1,10 @@
 # coding: utf-8
 from __future__ import unicode_literals
 
+import re
+
 from .common import InfoExtractor
+from ..compat import compat_etree_fromstring
 from ..utils import (
     xpath_element,
     xpath_text,
@@ -43,10 +46,15 @@ class FazIE(InfoExtractor):
 
         webpage = self._download_webpage(url, video_id)
         description = self._og_search_description(webpage)
-        config_xml_url = self._search_regex(
-            r'videoXMLURL\s*=\s*"([^"]+)', webpage, 'config xml url')
-        config = self._download_xml(
-            config_xml_url, video_id, 'Downloading config xml')
+        media = self._html_search_regex(
+            r"data-videojs-media='([^']+)",
+            webpage, 'media')
+        if media == 'extern':
+            perform_url = self._search_regex(
+                r"<iframe[^>]+?src='((?:http:)?//player\.performgroup\.com/eplayer/eplayer\.html#/?[0-9a-f]{26}\.[0-9a-z]{26})",
+                webpage, 'perform url')
+            return self.url_result(perform_url)
+        config = compat_etree_fromstring(media)
 
         encodings = xpath_element(config, 'ENCODINGS', 'encodings', True)
         formats = []
@@ -55,12 +63,24 @@ class FazIE(InfoExtractor):
             if encoding is not None:
                 encoding_url = xpath_text(encoding, 'FILENAME')
                 if encoding_url:
-                    formats.append({
+                    tbr = xpath_text(encoding, 'AVERAGEBITRATE', 1000)
+                    if tbr:
+                        tbr = int_or_none(tbr.replace(',', '.'))
+                    f = {
                         'url': encoding_url,
                         'format_id': code.lower(),
                         'quality': pref,
-                        'tbr': int_or_none(xpath_text(encoding, 'AVERAGEBITRATE')),
-                    })
+                        'tbr': tbr,
+                        'vcodec': xpath_text(encoding, 'CODEC'),
+                    }
+                    mobj = re.search(r'(\d+)x(\d+)_(\d+)\.mp4', encoding_url)
+                    if mobj:
+                        f.update({
+                            'width': int(mobj.group(1)),
+                            'height': int(mobj.group(2)),
+                            'tbr': tbr or int(mobj.group(3)),
+                        })
+                    formats.append(f)
         self._sort_formats(formats)
 
         return {
author	Remita Amine <remitamine@gmail.com>	2017-11-24 18:42:41 +0100
committer	Remita Amine <remitamine@gmail.com>	2017-11-24 18:42:41 +0100
commit	e0a8686f48d10ed86f7be92132dd37481981adf3 (patch)
tree	2c3f90569d5d4cf85514d32b5135c66f6db32f86 /youtube_dl/extractor/faz.py
parent	6049176471b4c567c1e9bd8f77b298e323d5b8e7 (diff)