mirror of
				https://github.com/ytdl-org/youtube-dl.git
				synced 2025-10-29 09:26:20 -07:00 
			
		
		
		
	[3sat,phoenix] Fix extraction (closes #11619)
This commit is contained in:
		| @@ -2,10 +2,19 @@ from __future__ import unicode_literals | ||||
|  | ||||
| import re | ||||
|  | ||||
| from .zdf import ZDFIE | ||||
| from .common import InfoExtractor | ||||
| from ..utils import ( | ||||
|     int_or_none, | ||||
|     unified_strdate, | ||||
|     xpath_text, | ||||
|     determine_ext, | ||||
|     qualities, | ||||
|     float_or_none, | ||||
|     ExtractorError, | ||||
| ) | ||||
|  | ||||
|  | ||||
| class DreiSatIE(ZDFIE): | ||||
| class DreiSatIE(InfoExtractor): | ||||
|     IE_NAME = '3sat' | ||||
|     _VALID_URL = r'(?:https?://)?(?:www\.)?3sat\.de/mediathek/(?:index\.php|mediathek\.php)?\?(?:(?:mode|display)=[^&]+&)*obj=(?P<id>[0-9]+)$' | ||||
|     _TESTS = [ | ||||
| @@ -31,6 +40,163 @@ class DreiSatIE(ZDFIE): | ||||
|         }, | ||||
|     ] | ||||
|  | ||||
|     def _parse_smil_formats(self, smil, smil_url, video_id, namespace=None, f4m_params=None, transform_rtmp_url=None): | ||||
|         param_groups = {} | ||||
|         for param_group in smil.findall(self._xpath_ns('./head/paramGroup', namespace)): | ||||
|             group_id = param_group.attrib.get(self._xpath_ns('id', 'http://www.w3.org/XML/1998/namespace')) | ||||
|             params = {} | ||||
|             for param in param_group: | ||||
|                 params[param.get('name')] = param.get('value') | ||||
|             param_groups[group_id] = params | ||||
|  | ||||
|         formats = [] | ||||
|         for video in smil.findall(self._xpath_ns('.//video', namespace)): | ||||
|             src = video.get('src') | ||||
|             if not src: | ||||
|                 continue | ||||
|             bitrate = float_or_none(video.get('system-bitrate') or video.get('systemBitrate'), 1000) | ||||
|             group_id = video.get('paramGroup') | ||||
|             param_group = param_groups[group_id] | ||||
|             for proto in param_group['protocols'].split(','): | ||||
|                 formats.append({ | ||||
|                     'url': '%s://%s' % (proto, param_group['host']), | ||||
|                     'app': param_group['app'], | ||||
|                     'play_path': src, | ||||
|                     'ext': 'flv', | ||||
|                     'format_id': '%s-%d' % (proto, bitrate), | ||||
|                     'tbr': bitrate, | ||||
|                 }) | ||||
|         self._sort_formats(formats) | ||||
|         return formats | ||||
|  | ||||
|     def extract_from_xml_url(self, video_id, xml_url): | ||||
|         doc = self._download_xml( | ||||
|             xml_url, video_id, | ||||
|             note='Downloading video info', | ||||
|             errnote='Failed to download video info') | ||||
|  | ||||
|         status_code = doc.find('./status/statuscode') | ||||
|         if status_code is not None and status_code.text != 'ok': | ||||
|             code = status_code.text | ||||
|             if code == 'notVisibleAnymore': | ||||
|                 message = 'Video %s is not available' % video_id | ||||
|             else: | ||||
|                 message = '%s returned error: %s' % (self.IE_NAME, code) | ||||
|             raise ExtractorError(message, expected=True) | ||||
|  | ||||
|         title = doc.find('.//information/title').text | ||||
|         description = xpath_text(doc, './/information/detail', 'description') | ||||
|         duration = int_or_none(xpath_text(doc, './/details/lengthSec', 'duration')) | ||||
|         uploader = xpath_text(doc, './/details/originChannelTitle', 'uploader') | ||||
|         uploader_id = xpath_text(doc, './/details/originChannelId', 'uploader id') | ||||
|         upload_date = unified_strdate(xpath_text(doc, './/details/airtime', 'upload date')) | ||||
|  | ||||
|         def xml_to_thumbnails(fnode): | ||||
|             thumbnails = [] | ||||
|             for node in fnode: | ||||
|                 thumbnail_url = node.text | ||||
|                 if not thumbnail_url: | ||||
|                     continue | ||||
|                 thumbnail = { | ||||
|                     'url': thumbnail_url, | ||||
|                 } | ||||
|                 if 'key' in node.attrib: | ||||
|                     m = re.match('^([0-9]+)x([0-9]+)$', node.attrib['key']) | ||||
|                     if m: | ||||
|                         thumbnail['width'] = int(m.group(1)) | ||||
|                         thumbnail['height'] = int(m.group(2)) | ||||
|                 thumbnails.append(thumbnail) | ||||
|             return thumbnails | ||||
|  | ||||
|         thumbnails = xml_to_thumbnails(doc.findall('.//teaserimages/teaserimage')) | ||||
|  | ||||
|         format_nodes = doc.findall('.//formitaeten/formitaet') | ||||
|         quality = qualities(['veryhigh', 'high', 'med', 'low']) | ||||
|  | ||||
|         def get_quality(elem): | ||||
|             return quality(xpath_text(elem, 'quality')) | ||||
|         format_nodes.sort(key=get_quality) | ||||
|         format_ids = [] | ||||
|         formats = [] | ||||
|         for fnode in format_nodes: | ||||
|             video_url = fnode.find('url').text | ||||
|             is_available = 'http://www.metafilegenerator' not in video_url | ||||
|             if not is_available: | ||||
|                 continue | ||||
|             format_id = fnode.attrib['basetype'] | ||||
|             quality = xpath_text(fnode, './quality', 'quality') | ||||
|             format_m = re.match(r'''(?x) | ||||
|                 (?P<vcodec>[^_]+)_(?P<acodec>[^_]+)_(?P<container>[^_]+)_ | ||||
|                 (?P<proto>[^_]+)_(?P<index>[^_]+)_(?P<indexproto>[^_]+) | ||||
|             ''', format_id) | ||||
|  | ||||
|             ext = determine_ext(video_url, None) or format_m.group('container') | ||||
|             if ext not in ('smil', 'f4m', 'm3u8'): | ||||
|                 format_id = format_id + '-' + quality | ||||
|             if format_id in format_ids: | ||||
|                 continue | ||||
|  | ||||
|             if ext == 'meta': | ||||
|                 continue | ||||
|             elif ext == 'smil': | ||||
|                 formats.extend(self._extract_smil_formats( | ||||
|                     video_url, video_id, fatal=False)) | ||||
|             elif ext == 'm3u8': | ||||
|                 # the certificates are misconfigured (see | ||||
|                 # https://github.com/rg3/youtube-dl/issues/8665) | ||||
|                 if video_url.startswith('https://'): | ||||
|                     continue | ||||
|                 formats.extend(self._extract_m3u8_formats( | ||||
|                     video_url, video_id, 'mp4', m3u8_id=format_id, fatal=False)) | ||||
|             elif ext == 'f4m': | ||||
|                 formats.extend(self._extract_f4m_formats( | ||||
|                     video_url, video_id, f4m_id=format_id, fatal=False)) | ||||
|             else: | ||||
|                 proto = format_m.group('proto').lower() | ||||
|  | ||||
|                 abr = int_or_none(xpath_text(fnode, './audioBitrate', 'abr'), 1000) | ||||
|                 vbr = int_or_none(xpath_text(fnode, './videoBitrate', 'vbr'), 1000) | ||||
|  | ||||
|                 width = int_or_none(xpath_text(fnode, './width', 'width')) | ||||
|                 height = int_or_none(xpath_text(fnode, './height', 'height')) | ||||
|  | ||||
|                 filesize = int_or_none(xpath_text(fnode, './filesize', 'filesize')) | ||||
|  | ||||
|                 format_note = '' | ||||
|                 if not format_note: | ||||
|                     format_note = None | ||||
|  | ||||
|                 formats.append({ | ||||
|                     'format_id': format_id, | ||||
|                     'url': video_url, | ||||
|                     'ext': ext, | ||||
|                     'acodec': format_m.group('acodec'), | ||||
|                     'vcodec': format_m.group('vcodec'), | ||||
|                     'abr': abr, | ||||
|                     'vbr': vbr, | ||||
|                     'width': width, | ||||
|                     'height': height, | ||||
|                     'filesize': filesize, | ||||
|                     'format_note': format_note, | ||||
|                     'protocol': proto, | ||||
|                     '_available': is_available, | ||||
|                 }) | ||||
|             format_ids.append(format_id) | ||||
|  | ||||
|         self._sort_formats(formats) | ||||
|  | ||||
|         return { | ||||
|             'id': video_id, | ||||
|             'title': title, | ||||
|             'description': description, | ||||
|             'duration': duration, | ||||
|             'thumbnails': thumbnails, | ||||
|             'uploader': uploader, | ||||
|             'uploader_id': uploader_id, | ||||
|             'upload_date': upload_date, | ||||
|             'formats': formats, | ||||
|         } | ||||
|  | ||||
|     def _real_extract(self, url): | ||||
|         mobj = re.match(self._VALID_URL, url) | ||||
|         video_id = mobj.group('id') | ||||
|   | ||||
| @@ -1,9 +1,9 @@ | ||||
| from __future__ import unicode_literals | ||||
|  | ||||
| from .zdf import ZDFIE | ||||
| from .dreisat import DreiSatIE | ||||
|  | ||||
|  | ||||
| class PhoenixIE(ZDFIE): | ||||
| class PhoenixIE(DreiSatIE): | ||||
|     IE_NAME = 'phoenix.de' | ||||
|     _VALID_URL = r'''(?x)https?://(?:www\.)?phoenix\.de/content/ | ||||
|         (?: | ||||
|   | ||||
		Reference in New Issue
	
	Block a user