l1ving_youtube-dl/youtube_dl/extractor/soundgasm.py

# coding: utf-8
from __future__ import unicode_literals

import re

from .common import InfoExtractor


class SoundgasmIE(InfoExtractor):
    IE_NAME = 'soundgasm'
    _VALID_URL = r'(?P<archive>https?://web\.archive\.org/web/\d+(?:if_)?/)?' + \
                 r'https?://(?:www\.)?soundgasm\.net(?::80)?/u/' + \
                 r'(?P<user>[0-9a-zA-Z_-]+)/(?P<display_id>[0-9a-zA-Z_-]+)'
    _TESTS = [{
        'url': 'http://soundgasm.net/u/ytdl/Piano-sample',
        'md5': '010082a2c802c5275bb00030743e75ad',
        'info_dict': {
            'id': '88abd86ea000cafe98f96321b23cc1206cbcbcc9',
            'ext': 'm4a',
            'title': 'Piano sample',
            'description': 'Royalty Free Sample Music',
            'uploader': 'ytdl'
        }
    }, {
        'url': 'http://web.archive.org/web/20181218221507/https://soundgasm.net/u/ytdl/Piano-sample',
        'md5': '010082a2c802c5275bb00030743e75ad',
        'info_dict': {
            'id': '88abd86ea000cafe98f96321b23cc1206cbcbcc9',
            'ext': 'm4a',
            'title': 'Piano sample',
            'description': 'Royalty Free Sample Music',
            'uploader': 'ytdl'
        }
    }]

    def _real_extract(self, url):
        mobj = re.match(self._VALID_URL, url)
        display_id = mobj.group('display_id')

        webpage = self._download_webpage(url, display_id)

        audio_url = self._html_search_regex(
            r'(?s)m4a\s*:\s*(["\'])(?P<url>(?:(?!\1).)+)\1', webpage,
            'audio URL', group='url')

        if mobj.group('archive'):
            pos = mobj.span('archive')[1] - 1
            audio_url = audio_url[:pos] + "if_" + audio_url[pos:]

        title = self._search_regex(
            r'<div[^>]+\bclass=["\']jp-title[^>]+>([^<]+)',
            webpage, 'title', default=display_id)

        description = self._html_search_regex(
            (r'(?s)<div[^>]+\bclass=["\']jp-description[^>]+>(.+?)</div>',
             r'(?s)<li>Description:\s(.*?)<\/li>'),
            webpage, 'description', fatal=False)

        audio_id = self._search_regex(
            r'/([^/]+)\.m4a', audio_url, 'audio id', default=display_id)

        return {
            'id': audio_id,
            'display_id': display_id,
            'url': audio_url,
            'vcodec': 'none',
            'title': title,
            'description': description,
            'uploader': mobj.group('user'),
        }


class SoundgasmProfileIE(InfoExtractor):
    IE_NAME = 'soundgasm:profile'
    _VALID_URL = r'(?P<archive>https?://web\.archive\.org/web/\d+(?:if_)?/)?' + \
                 r'https?://(?:www\.)?soundgasm\.net/u/' + \
                 r'(?P<id>[^/]+)/?(?:\#.*)?$'
    _TESTS = [{
        'url': 'http://soundgasm.net/u/ytdl',
        'info_dict': {
            'id': 'ytdl',
        },
        'playlist_count': 1
    }, {
        'url': 'http://web.archive.org/web/20181218222843/https://soundgasm.net/u/ytdl',
        'info_dict': {
            'id': 'ytdl'
        },
        'playlist_count': 1
    }]

    def _real_extract(self, url):
        profile_id = self._match_id(url)

        webpage = self._download_webpage(url, profile_id)

        entries = [
            self.url_result(audio_url, 'Soundgasm')
            for audio_url in re.findall(r'href="([^"]+/u/%s/[^"]+)' % profile_id, webpage)]

        return self.playlist_result(entries, profile_id)
[Soundgasm] Add new extractor 2014-06-25 18:07:23 +02:00			`# coding: utf-8`
			`from __future__ import unicode_literals`

			`import re`

			`from .common import InfoExtractor`

[soundgasm] PEP8 and add a display_id (#3155) 2014-06-25 23:47:38 +02:00
[Soundgasm] Add new extractor 2014-06-25 18:07:23 +02:00			`class SoundgasmIE(InfoExtractor):`
[soundgasm] Clarify extractors' IE_NAMEs 2015-02-23 21:27:56 +06:00			`IE_NAME = 'soundgasm'`
[soundgasm] adjust _VALID_URL 2018-12-19 02:39:12 +01:00			`_VALID_URL = r'(?P<archive>https?://web\.archive\.org/web/\d+(?:if_)?/)?' + \`
[soundgasm] add internet archive support 2018-12-19 00:04:03 +01:00			`r'https?://(?:www\.)?soundgasm\.net(?::80)?/u/' + \`
			`r'(?P<user>[0-9a-zA-Z_-]+)/(?P<display_id>[0-9a-zA-Z_-]+)'`
			`_TESTS = [{`
[soundgasm] reformat code 2020-04-02 09:03:25 +02:00			`'url': 'http://soundgasm.net/u/ytdl/Piano-sample',`
			`'md5': '010082a2c802c5275bb00030743e75ad',`
			`'info_dict': {`
			`'id': '88abd86ea000cafe98f96321b23cc1206cbcbcc9',`
			`'ext': 'm4a',`
			`'title': 'Piano sample',`
			`'description': 'Royalty Free Sample Music',`
			`'uploader': 'ytdl'`
[Soundgasm] Add new extractor 2014-06-25 18:07:23 +02:00			`}`
[soundgasm] reformat code 2020-04-02 09:03:25 +02:00			`}, {`
			`'url': 'http://web.archive.org/web/20181218221507/https://soundgasm.net/u/ytdl/Piano-sample',`
			`'md5': '010082a2c802c5275bb00030743e75ad',`
			`'info_dict': {`
			`'id': '88abd86ea000cafe98f96321b23cc1206cbcbcc9',`
			`'ext': 'm4a',`
			`'title': 'Piano sample',`
			`'description': 'Royalty Free Sample Music',`
			`'uploader': 'ytdl'`
			`}`
			`}]`
[Soundgasm] Add new extractor 2014-06-25 18:07:23 +02:00
			`def _real_extract(self, url):`
			`mobj = re.match(self._VALID_URL, url)`
[soundgasm] Improve extraction (closes #14588) 2017-10-26 23:16:16 +07:00			`display_id = mobj.group('display_id')`

[soundgasm] PEP8 and add a display_id (#3155) 2014-06-25 23:47:38 +02:00			`webpage = self._download_webpage(url, display_id)`
[soundgasm] Improve extraction (closes #14588) 2017-10-26 23:16:16 +07:00
[soundgasm] PEP8 and add a display_id (#3155) 2014-06-25 23:47:38 +02:00			`audio_url = self._html_search_regex(`
[soundgasm] Improve extraction (closes #14588) 2017-10-26 23:16:16 +07:00			`r'(?s)m4a\s:\s(["\'])(?P<url>(?:(?!\1).)+)\1', webpage,`
			`'audio URL', group='url')`

[soundgasm] add internet archive support 2018-12-19 00:04:03 +01:00			`if mobj.group('archive'):`
[soundgasm] reformat code 2020-04-02 09:03:25 +02:00			`pos = mobj.span('archive')[1] - 1`
			`audio_url = audio_url[:pos] + "if_" + audio_url[pos:]`
[soundgasm] add internet archive support 2018-12-19 00:04:03 +01:00
[soundgasm] Improve extraction (closes #14588) 2017-10-26 23:16:16 +07:00			`title = self._search_regex(`
			`r'<div[^>]+\bclass=["\']jp-title[^>]+>([^<]+)',`
			`webpage, 'title', default=display_id)`

[soundgasm] PEP8 and add a display_id (#3155) 2014-06-25 23:47:38 +02:00			`description = self._html_search_regex(`
[soundgasm] Improve extraction (closes #14588) 2017-10-26 23:16:16 +07:00			`(r'(?s)<div[^>]+\bclass=["\']jp-description[^>]+>(.+?)</div>',`
			`r'(?s)<li>Description:\s(.*?)<\/li>'),`
			`webpage, 'description', fatal=False)`

			`audio_id = self._search_regex(`
			`r'/([^/]+)\.m4a', audio_url, 'audio id', default=display_id)`
[Soundgasm] Add new extractor 2014-06-25 18:07:23 +02:00
			`return {`
			`'id': audio_id,`
[soundgasm] PEP8 and add a display_id (#3155) 2014-06-25 23:47:38 +02:00			`'display_id': display_id,`
[Soundgasm] Add new extractor 2014-06-25 18:07:23 +02:00			`'url': audio_url,`
[soundgasm] Improve extraction (closes #14588) 2017-10-26 23:16:16 +07:00			`'vcodec': 'none',`
			`'title': title,`
			`'description': description,`
			`'uploader': mobj.group('user'),`
[Soundgasm] Add new extractor 2014-06-25 18:07:23 +02:00			`}`
[soundgasm] add profile IE. 2015-02-23 12:11:19 +01:00
[soundgasm] PEP8 2015-02-23 16:51:21 +01:00
[soundgasm] add profile IE. 2015-02-23 12:11:19 +01:00			`class SoundgasmProfileIE(InfoExtractor):`
[soundgasm] Clarify extractors' IE_NAMEs 2015-02-23 21:27:56 +06:00			`IE_NAME = 'soundgasm:profile'`
[soundgasm] adjust _VALID_URL 2018-12-19 02:39:12 +01:00			`_VALID_URL = r'(?P<archive>https?://web\.archive\.org/web/\d+(?:if_)?/)?' + \`
[soundgasm] add internet archive support 2018-12-19 00:04:03 +01:00			`r'https?://(?:www\.)?soundgasm\.net/u/' + \`
			`r'(?P<id>[^/]+)/?(?:\#.*)?$'`
			`_TESTS = [{`
[soundgasm] add profile IE. 2015-02-23 12:11:19 +01:00			`'url': 'http://soundgasm.net/u/ytdl',`
			`'info_dict': {`
			`'id': 'ytdl',`
[soundgasm:profile] Simplify 2015-02-23 21:27:24 +06:00			`},`
[soundgasm] reformat code 2020-04-02 09:03:25 +02:00			`'playlist_count': 1`
[soundgasm] properly format tests 2020-04-02 09:06:41 +02:00			`}, {`
			`'url': 'http://web.archive.org/web/20181218222843/https://soundgasm.net/u/ytdl',`
			`'info_dict': {`
			`'id': 'ytdl'`
			`},`
			`'playlist_count': 1`
[soundgasm] reformat code 2020-04-02 09:03:25 +02:00			`}]`
[soundgasm] add profile IE. 2015-02-23 12:11:19 +01:00
			`def _real_extract(self, url):`
			`profile_id = self._match_id(url)`

[soundgasm:profile] Simplify 2015-02-23 21:27:24 +06:00			`webpage = self._download_webpage(url, profile_id)`
[soundgasm] add profile IE. 2015-02-23 12:11:19 +01:00
[soundgasm:profile] Simplify 2015-02-23 21:27:24 +06:00			`entries = [`
			`self.url_result(audio_url, 'Soundgasm')`
			`for audio_url in re.findall(r'href="([^"]+/u/%s/[^"]+)' % profile_id, webpage)]`
[soundgasm] add profile IE. 2015-02-23 12:11:19 +01:00
[soundgasm:profile] Simplify 2015-02-23 21:27:24 +06:00			`return self.playlist_result(entries, profile_id)`