[soundgasm] Improve extraction (closes #14588)

This commit is contained in:
Sergey M․ 2017-10-26 23:16:16 +07:00
parent dc24a7d4a2
commit 30e6161799
No known key found for this signature in database
GPG key ID: 2C393E0F18A9236D

View file

@ -8,36 +8,49 @@
class SoundgasmIE(InfoExtractor): class SoundgasmIE(InfoExtractor):
IE_NAME = 'soundgasm' IE_NAME = 'soundgasm'
_VALID_URL = r'https?://(?:www\.)?soundgasm\.net/u/(?P<user>[0-9a-zA-Z_\-]+)/(?P<title>[0-9a-zA-Z_\-]+)' _VALID_URL = r'https?://(?:www\.)?soundgasm\.net/u/(?P<user>[0-9a-zA-Z_-]+)/(?P<display_id>[0-9a-zA-Z_-]+)'
_TEST = { _TEST = {
'url': 'http://soundgasm.net/u/ytdl/Piano-sample', 'url': 'http://soundgasm.net/u/ytdl/Piano-sample',
'md5': '010082a2c802c5275bb00030743e75ad', 'md5': '010082a2c802c5275bb00030743e75ad',
'info_dict': { 'info_dict': {
'id': '88abd86ea000cafe98f96321b23cc1206cbcbcc9', 'id': '88abd86ea000cafe98f96321b23cc1206cbcbcc9',
'ext': 'm4a', 'ext': 'm4a',
'title': 'ytdl_Piano-sample', 'title': 'Piano sample',
'description': 'Royalty Free Sample Music' 'description': 'Royalty Free Sample Music',
'uploader': 'ytdl',
} }
} }
def _real_extract(self, url): def _real_extract(self, url):
mobj = re.match(self._VALID_URL, url) mobj = re.match(self._VALID_URL, url)
display_id = mobj.group('title') display_id = mobj.group('display_id')
audio_title = mobj.group('user') + '_' + mobj.group('title')
webpage = self._download_webpage(url, display_id) webpage = self._download_webpage(url, display_id)
audio_url = self._html_search_regex( audio_url = self._html_search_regex(
r'(?s)m4a\:\s"([^"]+)"', webpage, 'audio URL') r'(?s)m4a\s*:\s*(["\'])(?P<url>(?:(?!\1).)+)\1', webpage,
audio_id = re.split(r'\/|\.', audio_url)[-2] 'audio URL', group='url')
title = self._search_regex(
r'<div[^>]+\bclass=["\']jp-title[^>]+>([^<]+)',
webpage, 'title', default=display_id)
description = self._html_search_regex( description = self._html_search_regex(
r'(?s)<li>Description:\s(.*?)<\/li>', webpage, 'description', (r'(?s)<div[^>]+\bclass=["\']jp-description[^>]+>(.+?)</div>',
fatal=False) r'(?s)<li>Description:\s(.*?)<\/li>'),
webpage, 'description', fatal=False)
audio_id = self._search_regex(
r'/([^/]+)\.m4a', audio_url, 'audio id', default=display_id)
return { return {
'id': audio_id, 'id': audio_id,
'display_id': display_id, 'display_id': display_id,
'url': audio_url, 'url': audio_url,
'title': audio_title, 'vcodec': 'none',
'description': description 'title': title,
'description': description,
'uploader': mobj.group('user'),
} }