| 
									
										
										
										
											2016-10-02 13:39:18 +02:00
										 |  |  | # coding: utf-8 | 
					
						
							| 
									
										
										
										
											2014-01-07 10:04:48 +01:00
										 |  |  | from __future__ import unicode_literals | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-06-23 20:24:07 +02:00
										 |  |  | import re | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | from .common import InfoExtractor | 
					
						
							| 
									
										
										
										
											2015-09-20 11:45:19 +06:00
										 |  |  | from ..compat import ( | 
					
						
							|  |  |  |     compat_parse_qs, | 
					
						
							| 
									
										
										
										
											2017-10-15 22:12:34 +07:00
										 |  |  |     compat_str, | 
					
						
							| 
									
										
										
										
											2015-09-20 11:45:19 +06:00
										 |  |  |     compat_urllib_parse_urlparse, | 
					
						
							|  |  |  | ) | 
					
						
							| 
									
										
										
										
											2013-06-23 20:24:07 +02:00
										 |  |  | from ..utils import ( | 
					
						
							| 
									
										
										
										
											2017-08-18 00:58:23 +07:00
										 |  |  |     ExtractorError, | 
					
						
							| 
									
										
										
										
											2013-07-11 16:12:16 +02:00
										 |  |  |     find_xpath_attr, | 
					
						
							| 
									
										
										
										
											2013-12-08 14:02:14 +01:00
										 |  |  |     get_element_by_attribute, | 
					
						
							| 
									
										
										
										
											2014-10-20 20:27:59 +07:00
										 |  |  |     int_or_none, | 
					
						
							| 
									
										
										
										
											2016-02-04 20:16:47 +01:00
										 |  |  |     NO_DEFAULT, | 
					
						
							| 
									
										
										
										
											2014-11-20 12:06:33 +01:00
										 |  |  |     qualities, | 
					
						
							| 
									
										
										
										
											2017-10-15 22:12:34 +07:00
										 |  |  |     try_get, | 
					
						
							| 
									
										
										
										
											2017-08-18 00:58:23 +07:00
										 |  |  |     unified_strdate, | 
					
						
							| 
									
										
										
										
											2013-06-23 20:24:07 +02:00
										 |  |  | ) | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2014-11-23 20:41:03 +01:00
										 |  |  | # There are different sources of video in arte.tv, the extraction process | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  | # is different for each one. The videos usually expire in 7 days, so we can't | 
					
						
							|  |  |  | # add tests. | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-06-23 20:24:07 +02:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2014-03-24 21:30:50 +01:00
										 |  |  | class ArteTvIE(InfoExtractor): | 
					
						
							| 
									
										
										
										
											2016-03-21 21:36:32 +06:00
										 |  |  |     _VALID_URL = r'https?://videos\.arte\.tv/(?P<lang>fr|de|en|es)/.*-(?P<id>.*?)\.html' | 
					
						
							| 
									
										
										
										
											2014-01-07 10:04:48 +01:00
										 |  |  |     IE_NAME = 'arte.tv' | 
					
						
							| 
									
										
										
										
											2013-06-23 20:24:07 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  |     def _real_extract(self, url): | 
					
						
							| 
									
										
										
										
											2014-03-24 22:34:03 +01:00
										 |  |  |         mobj = re.match(self._VALID_URL, url) | 
					
						
							|  |  |  |         lang = mobj.group('lang') | 
					
						
							|  |  |  |         video_id = mobj.group('id') | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-07-04 18:06:47 +02:00
										 |  |  |         ref_xml_url = url.replace('/videos/', '/do_delegate/videos/') | 
					
						
							|  |  |  |         ref_xml_url = ref_xml_url.replace('.html', ',view,asPlayerXml.xml') | 
					
						
							| 
									
										
										
										
											2014-03-08 16:04:03 +01:00
										 |  |  |         ref_xml_doc = self._download_xml( | 
					
						
							|  |  |  |             ref_xml_url, video_id, note='Downloading metadata') | 
					
						
							| 
									
										
										
										
											2013-07-11 16:12:16 +02:00
										 |  |  |         config_node = find_xpath_attr(ref_xml_doc, './/video', 'lang', lang) | 
					
						
							| 
									
										
										
										
											2013-07-04 18:06:47 +02:00
										 |  |  |         config_xml_url = config_node.attrib['ref'] | 
					
						
							| 
									
										
										
										
											2014-03-24 21:36:26 +01:00
										 |  |  |         config = self._download_xml( | 
					
						
							| 
									
										
										
										
											2014-03-08 16:04:03 +01:00
										 |  |  |             config_xml_url, video_id, note='Downloading configuration') | 
					
						
							| 
									
										
										
										
											2013-06-30 13:38:22 +02:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2014-03-24 21:36:26 +01:00
										 |  |  |         formats = [{ | 
					
						
							| 
									
										
										
										
											2014-12-28 15:42:29 +06:00
										 |  |  |             'format_id': q.attrib['quality'], | 
					
						
							| 
									
										
										
										
											2014-07-07 14:10:57 +02:00
										 |  |  |             # The playpath starts at 'mp4:', if we don't manually | 
					
						
							|  |  |  |             # split the url, rtmpdump will incorrectly parse them | 
					
						
							|  |  |  |             'url': q.text.split('mp4:', 1)[0], | 
					
						
							|  |  |  |             'play_path': 'mp4:' + q.text.split('mp4:', 1)[1], | 
					
						
							| 
									
										
										
										
											2014-03-24 22:38:51 +01:00
										 |  |  |             'ext': 'flv', | 
					
						
							| 
									
										
										
										
											2014-03-24 21:36:26 +01:00
										 |  |  |             'quality': 2 if q.attrib['quality'] == 'hd' else 1, | 
					
						
							| 
									
										
										
										
											2014-03-24 22:38:51 +01:00
										 |  |  |         } for q in config.findall('./urls/url')] | 
					
						
							| 
									
										
										
										
											2014-03-24 21:36:26 +01:00
										 |  |  |         self._sort_formats(formats) | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  |         title = config.find('.//name').text | 
					
						
							|  |  |  |         thumbnail = config.find('.//firstThumbnailUrl').text | 
					
						
							|  |  |  |         return { | 
					
						
							|  |  |  |             'id': video_id, | 
					
						
							|  |  |  |             'title': title, | 
					
						
							|  |  |  |             'thumbnail': thumbnail, | 
					
						
							| 
									
										
										
										
											2014-03-24 22:34:03 +01:00
										 |  |  |             'formats': formats, | 
					
						
							| 
									
										
										
										
											2014-03-24 21:30:50 +01:00
										 |  |  |         } | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  | class ArteTVBaseIE(InfoExtractor): | 
					
						
							| 
									
										
										
										
											2013-10-13 14:21:13 +02:00
										 |  |  |     @classmethod | 
					
						
							|  |  |  |     def _extract_url_info(cls, url): | 
					
						
							|  |  |  |         mobj = re.match(cls._VALID_URL, url) | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  |         lang = mobj.group('lang') | 
					
						
							| 
									
										
										
										
											2015-12-23 17:55:58 +01:00
										 |  |  |         query = compat_parse_qs(compat_urllib_parse_urlparse(url).query) | 
					
						
							|  |  |  |         if 'vid' in query: | 
					
						
							|  |  |  |             video_id = query['vid'][0] | 
					
						
							|  |  |  |         else: | 
					
						
							|  |  |  |             # This is not a real id, it can be for example AJT for the news | 
					
						
							|  |  |  |             # http://www.arte.tv/guide/fr/emissions/AJT/arte-journal | 
					
						
							|  |  |  |             video_id = mobj.group('id') | 
					
						
							| 
									
										
										
										
											2013-10-13 14:21:13 +02:00
										 |  |  |         return video_id, lang | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-03-07 02:19:54 +06:00
										 |  |  |     def _extract_from_json_url(self, json_url, video_id, lang, title=None): | 
					
						
							| 
									
										
										
										
											2014-03-24 22:01:47 +01:00
										 |  |  |         info = self._download_json(json_url, video_id) | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  |         player_info = info['videoJsonPlayer'] | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2017-10-15 22:12:34 +07:00
										 |  |  |         vsr = try_get(player_info, lambda x: x['VSR'], dict) | 
					
						
							| 
									
										
										
										
											2017-09-04 23:08:07 +07:00
										 |  |  |         if not vsr: | 
					
						
							| 
									
										
										
										
											2017-10-15 22:12:34 +07:00
										 |  |  |             error = None | 
					
						
							|  |  |  |             if try_get(player_info, lambda x: x['custom_msg']['type']) == 'error': | 
					
						
							|  |  |  |                 error = try_get( | 
					
						
							|  |  |  |                     player_info, lambda x: x['custom_msg']['msg'], compat_str) | 
					
						
							|  |  |  |             if not error: | 
					
						
							|  |  |  |                 error = 'Video %s is not available' % player_info.get('VID') or video_id | 
					
						
							|  |  |  |             raise ExtractorError(error, expected=True) | 
					
						
							| 
									
										
										
										
											2017-08-18 00:58:23 +07:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2014-09-29 12:45:18 +02:00
										 |  |  |         upload_date_str = player_info.get('shootingDate') | 
					
						
							|  |  |  |         if not upload_date_str: | 
					
						
							| 
									
										
										
										
											2016-02-17 22:51:08 +06:00
										 |  |  |             upload_date_str = (player_info.get('VRA') or player_info.get('VDA') or '').split(' ')[0] | 
					
						
							| 
									
										
										
										
											2014-09-29 12:45:18 +02:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-03-07 02:19:54 +06:00
										 |  |  |         title = (player_info.get('VTI') or title or player_info['VID']).strip() | 
					
						
							| 
									
										
										
										
											2014-10-21 20:08:20 +07:00
										 |  |  |         subtitle = player_info.get('VSU', '').strip() | 
					
						
							|  |  |  |         if subtitle: | 
					
						
							|  |  |  |             title += ' - %s' % subtitle | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  |         info_dict = { | 
					
						
							|  |  |  |             'id': player_info['VID'], | 
					
						
							| 
									
										
										
										
											2014-10-21 20:08:20 +07:00
										 |  |  |             'title': title, | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  |             'description': player_info.get('VDE'), | 
					
						
							| 
									
										
										
										
											2014-09-29 12:45:18 +02:00
										 |  |  |             'upload_date': unified_strdate(upload_date_str), | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  |             'thumbnail': player_info.get('programImage') or player_info.get('VTU', {}).get('IUR'), | 
					
						
							|  |  |  |         } | 
					
						
							| 
									
										
										
										
											2014-11-20 12:06:33 +01:00
										 |  |  |         qfunc = qualities(['HQ', 'MQ', 'EQ', 'SQ']) | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-02-17 22:37:05 +06:00
										 |  |  |         LANGS = { | 
					
						
							|  |  |  |             'fr': 'F', | 
					
						
							|  |  |  |             'de': 'A', | 
					
						
							|  |  |  |             'en': 'E[ANG]', | 
					
						
							|  |  |  |             'es': 'E[ESP]', | 
					
						
							|  |  |  |         } | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-05-08 06:52:42 +06:00
										 |  |  |         langcode = LANGS.get(lang, lang) | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2014-11-20 12:06:33 +01:00
										 |  |  |         formats = [] | 
					
						
							| 
									
										
										
										
											2017-08-18 00:58:23 +07:00
										 |  |  |         for format_id, format_dict in vsr.items(): | 
					
						
							| 
									
										
										
										
											2014-11-20 12:06:33 +01:00
										 |  |  |             f = dict(format_dict) | 
					
						
							|  |  |  |             versionCode = f.get('versionCode') | 
					
						
							| 
									
										
										
										
											2016-05-08 06:52:42 +06:00
										 |  |  |             l = re.escape(langcode) | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  |             # Language preference from most to least priority | 
					
						
							|  |  |  |             # Reference: section 5.6.3 of | 
					
						
							|  |  |  |             # http://www.arte.tv/sites/en/corporate/files/complete-technical-guidelines-arte-geie-v1-05.pdf | 
					
						
							|  |  |  |             PREFERENCES = ( | 
					
						
							|  |  |  |                 # original version in requested language, without subtitles | 
					
						
							|  |  |  |                 r'VO{0}$'.format(l), | 
					
						
							|  |  |  |                 # original version in requested language, with partial subtitles in requested language | 
					
						
							|  |  |  |                 r'VO{0}-ST{0}$'.format(l), | 
					
						
							|  |  |  |                 # original version in requested language, with subtitles for the deaf and hard-of-hearing in requested language | 
					
						
							|  |  |  |                 r'VO{0}-STM{0}$'.format(l), | 
					
						
							|  |  |  |                 # non-original (dubbed) version in requested language, without subtitles | 
					
						
							|  |  |  |                 r'V{0}$'.format(l), | 
					
						
							|  |  |  |                 # non-original (dubbed) version in requested language, with subtitles partial subtitles in requested language | 
					
						
							|  |  |  |                 r'V{0}-ST{0}$'.format(l), | 
					
						
							|  |  |  |                 # non-original (dubbed) version in requested language, with subtitles for the deaf and hard-of-hearing in requested language | 
					
						
							|  |  |  |                 r'V{0}-STM{0}$'.format(l), | 
					
						
							|  |  |  |                 # original version in requested language, with partial subtitles in different language | 
					
						
							|  |  |  |                 r'VO{0}-ST(?!{0}).+?$'.format(l), | 
					
						
							|  |  |  |                 # original version in requested language, with subtitles for the deaf and hard-of-hearing in different language | 
					
						
							|  |  |  |                 r'VO{0}-STM(?!{0}).+?$'.format(l), | 
					
						
							|  |  |  |                 # original version in different language, with partial subtitles in requested language | 
					
						
							|  |  |  |                 r'VO(?:(?!{0}).+?)?-ST{0}$'.format(l), | 
					
						
							|  |  |  |                 # original version in different language, with subtitles for the deaf and hard-of-hearing in requested language | 
					
						
							|  |  |  |                 r'VO(?:(?!{0}).+?)?-STM{0}$'.format(l), | 
					
						
							|  |  |  |                 # original version in different language, without subtitles | 
					
						
							|  |  |  |                 r'VO(?:(?!{0}))?$'.format(l), | 
					
						
							|  |  |  |                 # original version in different language, with partial subtitles in different language | 
					
						
							|  |  |  |                 r'VO(?:(?!{0}).+?)?-ST(?!{0}).+?$'.format(l), | 
					
						
							|  |  |  |                 # original version in different language, with subtitles for the deaf and hard-of-hearing in different language | 
					
						
							|  |  |  |                 r'VO(?:(?!{0}).+?)?-STM(?!{0}).+?$'.format(l), | 
					
						
							|  |  |  |             ) | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  |             for pref, p in enumerate(PREFERENCES): | 
					
						
							|  |  |  |                 if re.match(p, versionCode): | 
					
						
							|  |  |  |                     lang_pref = len(PREFERENCES) - pref | 
					
						
							|  |  |  |                     break | 
					
						
							|  |  |  |             else: | 
					
						
							|  |  |  |                 lang_pref = -1 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2014-11-20 12:06:33 +01:00
										 |  |  |             format = { | 
					
						
							|  |  |  |                 'format_id': format_id, | 
					
						
							|  |  |  |                 'preference': -10 if f.get('videoFormat') == 'M3U8' else None, | 
					
						
							|  |  |  |                 'language_preference': lang_pref, | 
					
						
							|  |  |  |                 'format_note': '%s, %s' % (f.get('versionCode'), f.get('versionLibelle')), | 
					
						
							|  |  |  |                 'width': int_or_none(f.get('width')), | 
					
						
							|  |  |  |                 'height': int_or_none(f.get('height')), | 
					
						
							|  |  |  |                 'tbr': int_or_none(f.get('bitrate')), | 
					
						
							| 
									
										
										
										
											2014-12-28 15:41:52 +06:00
										 |  |  |                 'quality': qfunc(f.get('quality')), | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  |             } | 
					
						
							| 
									
										
										
										
											2014-11-20 12:06:33 +01:00
										 |  |  | 
 | 
					
						
							|  |  |  |             if f.get('mediaType') == 'rtmp': | 
					
						
							|  |  |  |                 format['url'] = f['streamer'] | 
					
						
							|  |  |  |                 format['play_path'] = 'mp4:' + f['url'] | 
					
						
							|  |  |  |                 format['ext'] = 'flv' | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  |             else: | 
					
						
							| 
									
										
										
										
											2014-11-20 12:06:33 +01:00
										 |  |  |                 format['url'] = f['url'] | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  |             formats.append(format) | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2015-03-17 19:42:50 +06:00
										 |  |  |         self._check_formats(formats, video_id) | 
					
						
							| 
									
										
										
										
											2014-11-20 12:06:33 +01:00
										 |  |  |         self._sort_formats(formats) | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2014-11-20 12:06:33 +01:00
										 |  |  |         info_dict['formats'] = formats | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  |         return info_dict | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  | class ArteTVPlus7IE(ArteTVBaseIE): | 
					
						
							|  |  |  |     IE_NAME = 'arte.tv:+7' | 
					
						
							| 
									
										
										
										
											2017-04-26 01:55:29 +07:00
										 |  |  |     _VALID_URL = r'https?://(?:(?:www|sites)\.)?arte\.tv/(?:[^/]+/)?(?P<lang>fr|de|en|es)/(?:videos/)?(?:[^/]+/)*(?P<id>[^/?#&]+)' | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  | 
 | 
					
						
							|  |  |  |     _TESTS = [{ | 
					
						
							|  |  |  |         'url': 'http://www.arte.tv/guide/de/sendungen/XEN/xenius/?vid=055918-015_PLUS7-D', | 
					
						
							|  |  |  |         'only_matching': True, | 
					
						
							| 
									
										
										
										
											2016-06-18 22:23:48 +07:00
										 |  |  |     }, { | 
					
						
							| 
									
										
										
										
											2016-06-18 21:42:17 +07:00
										 |  |  |         'url': 'http://sites.arte.tv/karambolage/de/video/karambolage-22', | 
					
						
							|  |  |  |         'only_matching': True, | 
					
						
							| 
									
										
										
										
											2017-04-26 01:55:29 +07:00
										 |  |  |     }, { | 
					
						
							|  |  |  |         'url': 'http://www.arte.tv/de/videos/048696-000-A/der-kluge-bauch-unser-zweites-gehirn', | 
					
						
							|  |  |  |         'only_matching': True, | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  |     }] | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  |     @classmethod | 
					
						
							|  |  |  |     def suitable(cls, url): | 
					
						
							|  |  |  |         return False if ArteTVPlaylistIE.suitable(url) else super(ArteTVPlus7IE, cls).suitable(url) | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  |     def _real_extract(self, url): | 
					
						
							|  |  |  |         video_id, lang = self._extract_url_info(url) | 
					
						
							|  |  |  |         webpage = self._download_webpage(url, video_id) | 
					
						
							|  |  |  |         return self._extract_from_webpage(webpage, video_id, lang) | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  |     def _extract_from_webpage(self, webpage, video_id, lang): | 
					
						
							|  |  |  |         patterns_templates = (r'arte_vp_url=["\'](.*?%s.*?)["\']', r'data-url=["\']([^"]+%s[^"]+)["\']') | 
					
						
							|  |  |  |         ids = (video_id, '') | 
					
						
							|  |  |  |         # some pages contain multiple videos (like | 
					
						
							|  |  |  |         # http://www.arte.tv/guide/de/sendungen/XEN/xenius/?vid=055918-015_PLUS7-D), | 
					
						
							|  |  |  |         # so we first try to look for json URLs that contain the video id from | 
					
						
							|  |  |  |         # the 'vid' parameter. | 
					
						
							|  |  |  |         patterns = [t % re.escape(_id) for _id in ids for t in patterns_templates] | 
					
						
							|  |  |  |         json_url = self._html_search_regex( | 
					
						
							|  |  |  |             patterns, webpage, 'json vp url', default=None) | 
					
						
							|  |  |  |         if not json_url: | 
					
						
							|  |  |  |             def find_iframe_url(webpage, default=NO_DEFAULT): | 
					
						
							|  |  |  |                 return self._html_search_regex( | 
					
						
							|  |  |  |                     r'<iframe[^>]+src=(["\'])(?P<url>.+\bjson_url=.+?)\1', | 
					
						
							|  |  |  |                     webpage, 'iframe url', group='url', default=default) | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  |             iframe_url = find_iframe_url(webpage, None) | 
					
						
							|  |  |  |             if not iframe_url: | 
					
						
							|  |  |  |                 embed_url = self._html_search_regex( | 
					
						
							|  |  |  |                     r'arte_vp_url_oembed=\'([^\']+?)\'', webpage, 'embed url', default=None) | 
					
						
							|  |  |  |                 if embed_url: | 
					
						
							|  |  |  |                     player = self._download_json( | 
					
						
							|  |  |  |                         embed_url, video_id, 'Downloading player page') | 
					
						
							|  |  |  |                     iframe_url = find_iframe_url(player['html']) | 
					
						
							|  |  |  |             # en and es URLs produce react-based pages with different layout (e.g. | 
					
						
							|  |  |  |             # http://www.arte.tv/guide/en/053330-002-A/carnival-italy?zone=world) | 
					
						
							|  |  |  |             if not iframe_url: | 
					
						
							|  |  |  |                 program = self._search_regex( | 
					
						
							|  |  |  |                     r'program\s*:\s*({.+?["\']embed_html["\'].+?}),?\s*\n', | 
					
						
							|  |  |  |                     webpage, 'program', default=None) | 
					
						
							|  |  |  |                 if program: | 
					
						
							|  |  |  |                     embed_html = self._parse_json(program, video_id) | 
					
						
							|  |  |  |                     if embed_html: | 
					
						
							|  |  |  |                         iframe_url = find_iframe_url(embed_html['embed_html']) | 
					
						
							|  |  |  |             if iframe_url: | 
					
						
							|  |  |  |                 json_url = compat_parse_qs( | 
					
						
							|  |  |  |                     compat_urllib_parse_urlparse(iframe_url).query)['json_url'][0] | 
					
						
							|  |  |  |         if json_url: | 
					
						
							|  |  |  |             title = self._search_regex( | 
					
						
							|  |  |  |                 r'<h3[^>]+title=(["\'])(?P<title>.+?)\1', | 
					
						
							|  |  |  |                 webpage, 'title', default=None, group='title') | 
					
						
							|  |  |  |             return self._extract_from_json_url(json_url, video_id, lang, title=title) | 
					
						
							|  |  |  |         # Different kind of embed URL (e.g. | 
					
						
							|  |  |  |         # http://www.arte.tv/magazine/trepalium/fr/episode-0406-replay-trepalium) | 
					
						
							| 
									
										
										
										
											2016-06-18 12:34:58 +08:00
										 |  |  |         entries = [ | 
					
						
							|  |  |  |             self.url_result(url) | 
					
						
							|  |  |  |             for _, url in re.findall(r'<iframe[^>]+src=(["\'])(?P<url>.+?)\1', webpage)] | 
					
						
							|  |  |  |         return self.playlist_result(entries) | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  | # It also uses the arte_vp_url url from the webpage to extract the information | 
					
						
							|  |  |  | class ArteTVCreativeIE(ArteTVPlus7IE): | 
					
						
							| 
									
										
										
										
											2014-01-07 10:04:48 +01:00
										 |  |  |     IE_NAME = 'arte.tv:creative' | 
					
						
							| 
									
										
										
										
											2016-04-14 21:54:41 +06:00
										 |  |  |     _VALID_URL = r'https?://creative\.arte\.tv/(?P<lang>fr|de|en|es)/(?:[^/]+/)*(?P<id>[^/?#&]+)' | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2014-08-24 02:57:31 +02:00
										 |  |  |     _TESTS = [{ | 
					
						
							| 
									
										
										
										
											2016-06-18 12:34:58 +08:00
										 |  |  |         'url': 'http://creative.arte.tv/fr/episode/osmosis-episode-1', | 
					
						
							| 
									
										
										
										
											2014-01-07 10:04:48 +01:00
										 |  |  |         'info_dict': { | 
					
						
							| 
									
										
										
										
											2016-06-18 12:34:58 +08:00
										 |  |  |             'id': '057405-001-A', | 
					
						
							| 
									
										
										
										
											2014-03-20 09:14:43 +01:00
										 |  |  |             'ext': 'mp4', | 
					
						
							| 
									
										
										
										
											2016-06-18 12:34:58 +08:00
										 |  |  |             'title': 'OSMOSIS - N\'AYEZ PLUS PEUR D\'AIMER (1)', | 
					
						
							|  |  |  |             'upload_date': '20150716', | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  |         }, | 
					
						
							| 
									
										
										
										
											2014-08-24 02:57:31 +02:00
										 |  |  |     }, { | 
					
						
							|  |  |  |         'url': 'http://creative.arte.tv/fr/Monty-Python-Reunion', | 
					
						
							| 
									
										
										
										
											2016-06-18 12:34:58 +08:00
										 |  |  |         'playlist_count': 11, | 
					
						
							|  |  |  |         'add_ie': ['Youtube'], | 
					
						
							| 
									
										
										
										
											2016-04-14 21:54:41 +06:00
										 |  |  |     }, { | 
					
						
							|  |  |  |         'url': 'http://creative.arte.tv/de/episode/agentur-amateur-4-der-erste-kunde', | 
					
						
							|  |  |  |         'only_matching': True, | 
					
						
							| 
									
										
										
										
											2014-08-24 02:57:31 +02:00
										 |  |  |     }] | 
					
						
							| 
									
										
										
										
											2013-10-13 13:54:31 +02:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-10-13 14:21:13 +02:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-04-14 21:52:05 +06:00
										 |  |  | class ArteTVInfoIE(ArteTVPlus7IE): | 
					
						
							|  |  |  |     IE_NAME = 'arte.tv:info' | 
					
						
							|  |  |  |     _VALID_URL = r'https?://info\.arte\.tv/(?P<lang>fr|de|en|es)/(?:[^/]+/)*(?P<id>[^/?#&]+)' | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  |     _TESTS = [{ | 
					
						
							| 
									
										
										
										
											2016-04-14 21:52:05 +06:00
										 |  |  |         'url': 'http://info.arte.tv/fr/service-civique-un-cache-misere', | 
					
						
							|  |  |  |         'info_dict': { | 
					
						
							|  |  |  |             'id': '067528-000-A', | 
					
						
							|  |  |  |             'ext': 'mp4', | 
					
						
							|  |  |  |             'title': 'Service civique, un cache misère ?', | 
					
						
							|  |  |  |             'upload_date': '20160403', | 
					
						
							|  |  |  |         }, | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  |     }] | 
					
						
							| 
									
										
										
										
											2016-04-14 21:52:05 +06:00
										 |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-10-13 14:21:13 +02:00
										 |  |  | class ArteTVFutureIE(ArteTVPlus7IE): | 
					
						
							| 
									
										
										
										
											2014-01-07 10:04:48 +01:00
										 |  |  |     IE_NAME = 'arte.tv:future' | 
					
						
							| 
									
										
										
										
											2016-02-18 21:29:52 +06:00
										 |  |  |     _VALID_URL = r'https?://future\.arte\.tv/(?P<lang>fr|de|en|es)/(?P<id>[^/?#&]+)' | 
					
						
							| 
									
										
										
										
											2016-01-21 18:47:43 +01:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-01-22 23:00:05 +06:00
										 |  |  |     _TESTS = [{ | 
					
						
							|  |  |  |         'url': 'http://future.arte.tv/fr/info-sciences/les-ecrevisses-aussi-sont-anxieuses', | 
					
						
							|  |  |  |         'info_dict': { | 
					
						
							|  |  |  |             'id': '050940-028-A', | 
					
						
							|  |  |  |             'ext': 'mp4', | 
					
						
							|  |  |  |             'title': 'Les écrevisses aussi peuvent être anxieuses', | 
					
						
							| 
									
										
										
										
											2016-02-21 14:23:58 +06:00
										 |  |  |             'upload_date': '20140902', | 
					
						
							| 
									
										
										
										
											2013-10-13 14:21:13 +02:00
										 |  |  |         }, | 
					
						
							| 
									
										
										
										
											2016-01-22 23:00:05 +06:00
										 |  |  |     }, { | 
					
						
							|  |  |  |         'url': 'http://future.arte.tv/fr/la-science-est-elle-responsable', | 
					
						
							|  |  |  |         'only_matching': True, | 
					
						
							|  |  |  |     }] | 
					
						
							| 
									
										
										
										
											2013-12-08 14:02:14 +01:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-12-08 16:34:57 +01:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-12-08 14:02:14 +01:00
										 |  |  | class ArteTVDDCIE(ArteTVPlus7IE): | 
					
						
							| 
									
										
										
										
											2014-01-07 10:04:48 +01:00
										 |  |  |     IE_NAME = 'arte.tv:ddc' | 
					
						
							| 
									
										
										
										
											2016-02-18 21:29:52 +06:00
										 |  |  |     _VALID_URL = r'https?://ddc\.arte\.tv/(?P<lang>emission|folge)/(?P<id>[^/?#&]+)' | 
					
						
							| 
									
										
										
										
											2013-12-08 14:02:14 +01:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  |     _TESTS = [] | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2013-12-08 14:02:14 +01:00
										 |  |  |     def _real_extract(self, url): | 
					
						
							|  |  |  |         video_id, lang = self._extract_url_info(url) | 
					
						
							|  |  |  |         if lang == 'folge': | 
					
						
							|  |  |  |             lang = 'de' | 
					
						
							|  |  |  |         elif lang == 'emission': | 
					
						
							|  |  |  |             lang = 'fr' | 
					
						
							|  |  |  |         webpage = self._download_webpage(url, video_id) | 
					
						
							|  |  |  |         scriptElement = get_element_by_attribute('class', 'visu_video_block', webpage) | 
					
						
							|  |  |  |         script_url = self._html_search_regex(r'src="(.*?)"', scriptElement, 'script url') | 
					
						
							|  |  |  |         javascriptPlayerGenerator = self._download_webpage(script_url, video_id, 'Download javascript player generator') | 
					
						
							|  |  |  |         json_url = self._search_regex(r"json_url=(.*)&rendering_place.*", javascriptPlayerGenerator, 'json url') | 
					
						
							|  |  |  |         return self._extract_from_json_url(json_url, video_id, lang) | 
					
						
							| 
									
										
										
										
											2014-03-20 09:09:56 +01:00
										 |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | class ArteTVConcertIE(ArteTVPlus7IE): | 
					
						
							|  |  |  |     IE_NAME = 'arte.tv:concert' | 
					
						
							| 
									
										
										
										
											2016-02-18 21:29:52 +06:00
										 |  |  |     _VALID_URL = r'https?://concert\.arte\.tv/(?P<lang>fr|de|en|es)/(?P<id>[^/?#&]+)' | 
					
						
							| 
									
										
										
										
											2014-03-20 09:09:56 +01:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  |     _TESTS = [{ | 
					
						
							| 
									
										
										
										
											2014-03-20 09:09:56 +01:00
										 |  |  |         'url': 'http://concert.arte.tv/de/notwist-im-pariser-konzertclub-divan-du-monde', | 
					
						
							|  |  |  |         'md5': '9ea035b7bd69696b67aa2ccaaa218161', | 
					
						
							|  |  |  |         'info_dict': { | 
					
						
							|  |  |  |             'id': '186', | 
					
						
							|  |  |  |             'ext': 'mp4', | 
					
						
							|  |  |  |             'title': 'The Notwist im Pariser Konzertclub "Divan du Monde"', | 
					
						
							|  |  |  |             'upload_date': '20140128', | 
					
						
							| 
									
										
										
										
											2014-03-22 14:20:05 +01:00
										 |  |  |             'description': 'md5:486eb08f991552ade77439fe6d82c305', | 
					
						
							| 
									
										
										
										
											2014-03-20 09:09:56 +01:00
										 |  |  |         }, | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  |     }] | 
					
						
							| 
									
										
										
										
											2014-03-24 22:01:47 +01:00
										 |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-01-22 23:00:50 +06:00
										 |  |  | class ArteTVCinemaIE(ArteTVPlus7IE): | 
					
						
							|  |  |  |     IE_NAME = 'arte.tv:cinema' | 
					
						
							| 
									
										
										
										
											2016-02-17 21:53:53 +06:00
										 |  |  |     _VALID_URL = r'https?://cinema\.arte\.tv/(?P<lang>fr|de|en|es)/(?P<id>.+)' | 
					
						
							| 
									
										
										
										
											2016-01-22 23:00:50 +06:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  |     _TESTS = [{ | 
					
						
							| 
									
										
										
										
											2016-06-18 12:34:58 +08:00
										 |  |  |         'url': 'http://cinema.arte.tv/fr/article/les-ailes-du-desir-de-julia-reck', | 
					
						
							|  |  |  |         'md5': 'a5b9dd5575a11d93daf0e3f404f45438', | 
					
						
							| 
									
										
										
										
											2016-01-22 23:00:50 +06:00
										 |  |  |         'info_dict': { | 
					
						
							| 
									
										
										
										
											2016-06-18 12:34:58 +08:00
										 |  |  |             'id': '062494-000-A', | 
					
						
							| 
									
										
										
										
											2016-01-22 23:00:50 +06:00
										 |  |  |             'ext': 'mp4', | 
					
						
							| 
									
										
										
										
											2016-06-18 12:34:58 +08:00
										 |  |  |             'title': 'Film lauréat du concours web - "Les ailes du désir" de Julia Reck', | 
					
						
							|  |  |  |             'upload_date': '20150807', | 
					
						
							| 
									
										
										
										
											2016-01-22 23:00:50 +06:00
										 |  |  |         }, | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  |     }] | 
					
						
							| 
									
										
										
										
											2016-01-22 23:00:50 +06:00
										 |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-02-08 22:17:21 +01:00
										 |  |  | class ArteTVMagazineIE(ArteTVPlus7IE): | 
					
						
							|  |  |  |     IE_NAME = 'arte.tv:magazine' | 
					
						
							| 
									
										
										
										
											2016-02-18 21:29:07 +06:00
										 |  |  |     _VALID_URL = r'https?://(?:www\.)?arte\.tv/magazine/[^/]+/(?P<lang>fr|de|en|es)/(?P<id>[^/?#&]+)' | 
					
						
							| 
									
										
										
										
											2016-02-08 22:17:21 +01:00
										 |  |  | 
 | 
					
						
							|  |  |  |     _TESTS = [{ | 
					
						
							| 
									
										
										
										
											2016-02-21 13:55:25 +06:00
										 |  |  |         # Embedded via <iframe src="http://www.arte.tv/arte_vp/index.php?json_url=..." | 
					
						
							| 
									
										
										
										
											2016-02-08 22:17:21 +01:00
										 |  |  |         'url': 'http://www.arte.tv/magazine/trepalium/fr/entretien-avec-le-realisateur-vincent-lannoo-trepalium', | 
					
						
							| 
									
										
										
										
											2016-02-21 13:57:30 +06:00
										 |  |  |         'md5': '2a9369bcccf847d1c741e51416299f25', | 
					
						
							| 
									
										
										
										
											2016-02-08 22:17:21 +01:00
										 |  |  |         'info_dict': { | 
					
						
							|  |  |  |             'id': '065965-000-A', | 
					
						
							|  |  |  |             'ext': 'mp4', | 
					
						
							|  |  |  |             'title': 'Trepalium - Extrait Ep.01', | 
					
						
							| 
									
										
										
										
											2016-02-21 13:57:30 +06:00
										 |  |  |             'upload_date': '20160121', | 
					
						
							| 
									
										
										
										
											2016-02-08 22:17:21 +01:00
										 |  |  |         }, | 
					
						
							| 
									
										
										
										
											2016-02-21 13:55:25 +06:00
										 |  |  |     }, { | 
					
						
							|  |  |  |         # Embedded via <iframe src="http://www.arte.tv/guide/fr/embed/054813-004-A/medium" | 
					
						
							|  |  |  |         'url': 'http://www.arte.tv/magazine/trepalium/fr/episode-0406-replay-trepalium', | 
					
						
							|  |  |  |         'md5': 'fedc64fc7a946110fe311634e79782ca', | 
					
						
							|  |  |  |         'info_dict': { | 
					
						
							|  |  |  |             'id': '054813-004_PLUS7-F', | 
					
						
							|  |  |  |             'ext': 'mp4', | 
					
						
							|  |  |  |             'title': 'Trepalium (4/6)', | 
					
						
							|  |  |  |             'description': 'md5:10057003c34d54e95350be4f9b05cb40', | 
					
						
							|  |  |  |             'upload_date': '20160218', | 
					
						
							|  |  |  |         }, | 
					
						
							| 
									
										
										
										
											2016-02-08 22:17:21 +01:00
										 |  |  |     }, { | 
					
						
							|  |  |  |         'url': 'http://www.arte.tv/magazine/metropolis/de/frank-woeste-german-paris-metropolis', | 
					
						
							| 
									
										
										
										
											2016-02-18 21:29:07 +06:00
										 |  |  |         'only_matching': True, | 
					
						
							| 
									
										
										
										
											2016-02-08 22:17:21 +01:00
										 |  |  |     }] | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2014-03-24 22:01:47 +01:00
										 |  |  | class ArteTVEmbedIE(ArteTVPlus7IE): | 
					
						
							|  |  |  |     IE_NAME = 'arte.tv:embed' | 
					
						
							|  |  |  |     _VALID_URL = r'''(?x)
 | 
					
						
							|  |  |  |         http://www\.arte\.tv | 
					
						
							| 
									
										
										
										
											2016-04-11 19:17:11 +08:00
										 |  |  |         /(?:playerv2/embed|arte_vp/index)\.php\?json_url= | 
					
						
							| 
									
										
										
										
											2014-03-24 22:01:47 +01:00
										 |  |  |         (?P<json_url> | 
					
						
							|  |  |  |             http://arte\.tv/papi/tvguide/videos/stream/player/ | 
					
						
							|  |  |  |             (?P<lang>[^/]+)/(?P<id>[^/]+)[^&]* | 
					
						
							|  |  |  |         ) | 
					
						
							|  |  |  |     '''
 | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  |     _TESTS = [] | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2014-03-24 22:01:47 +01:00
										 |  |  |     def _real_extract(self, url): | 
					
						
							|  |  |  |         mobj = re.match(self._VALID_URL, url) | 
					
						
							|  |  |  |         video_id = mobj.group('id') | 
					
						
							|  |  |  |         lang = mobj.group('lang') | 
					
						
							|  |  |  |         json_url = mobj.group('json_url') | 
					
						
							|  |  |  |         return self._extract_from_json_url(json_url, video_id, lang) | 
					
						
							| 
									
										
										
										
											2016-10-16 00:24:06 +07:00
										 |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | class TheOperaPlatformIE(ArteTVPlus7IE): | 
					
						
							|  |  |  |     IE_NAME = 'theoperaplatform' | 
					
						
							|  |  |  |     _VALID_URL = r'https?://(?:www\.)?theoperaplatform\.eu/(?P<lang>fr|de|en|es)/(?P<id>[^/?#&]+)' | 
					
						
							| 
									
										
										
										
											2016-10-13 20:13:54 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  |     _TESTS = [{ | 
					
						
							|  |  |  |         'url': 'http://www.theoperaplatform.eu/de/opera/verdi-otello', | 
					
						
							| 
									
										
										
										
											2016-10-16 00:24:06 +07:00
										 |  |  |         'md5': '970655901fa2e82e04c00b955e9afe7b', | 
					
						
							| 
									
										
										
										
											2016-10-13 20:13:54 +02:00
										 |  |  |         'info_dict': { | 
					
						
							|  |  |  |             'id': '060338-009-A', | 
					
						
							|  |  |  |             'ext': 'mp4', | 
					
						
							|  |  |  |             'title': 'Verdi - OTELLO', | 
					
						
							|  |  |  |             'upload_date': '20160927', | 
					
						
							|  |  |  |         }, | 
					
						
							|  |  |  |     }] | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | class ArteTVPlaylistIE(ArteTVBaseIE): | 
					
						
							|  |  |  |     IE_NAME = 'arte.tv:playlist' | 
					
						
							|  |  |  |     _VALID_URL = r'https?://(?:www\.)?arte\.tv/guide/(?P<lang>fr|de|en|es)/[^#]*#collection/(?P<id>PL-\d+)' | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  |     _TESTS = [{ | 
					
						
							|  |  |  |         'url': 'http://www.arte.tv/guide/de/plus7/?country=DE#collection/PL-013263/ARTETV', | 
					
						
							|  |  |  |         'info_dict': { | 
					
						
							|  |  |  |             'id': 'PL-013263', | 
					
						
							|  |  |  |             'title': 'Areva & Uramin', | 
					
						
							| 
									
										
										
										
											2016-06-28 22:39:53 +07:00
										 |  |  |             'description': 'md5:a1dc0312ce357c262259139cfd48c9bf', | 
					
						
							| 
									
										
										
										
											2016-06-02 01:10:23 +07:00
										 |  |  |         }, | 
					
						
							|  |  |  |         'playlist_mincount': 6, | 
					
						
							|  |  |  |     }, { | 
					
						
							|  |  |  |         'url': 'http://www.arte.tv/guide/de/playlists?country=DE#collection/PL-013190/ARTETV', | 
					
						
							|  |  |  |         'only_matching': True, | 
					
						
							|  |  |  |     }] | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  |     def _real_extract(self, url): | 
					
						
							|  |  |  |         playlist_id, lang = self._extract_url_info(url) | 
					
						
							|  |  |  |         collection = self._download_json( | 
					
						
							|  |  |  |             'https://api.arte.tv/api/player/v1/collectionData/%s/%s?source=videos' | 
					
						
							|  |  |  |             % (lang, playlist_id), playlist_id) | 
					
						
							|  |  |  |         title = collection.get('title') | 
					
						
							|  |  |  |         description = collection.get('shortDescription') or collection.get('teaserText') | 
					
						
							|  |  |  |         entries = [ | 
					
						
							|  |  |  |             self._extract_from_json_url( | 
					
						
							|  |  |  |                 video['jsonUrl'], video.get('programId') or playlist_id, lang) | 
					
						
							|  |  |  |             for video in collection['videos'] if video.get('jsonUrl')] | 
					
						
							|  |  |  |         return self.playlist_result(entries, playlist_id, title, description) |