Finished audiomack extractor

This commit is contained in:
xavier 2014-10-23 23:54:59 -05:00
parent 5c565ac9e7
commit 9e9bc793f3

View File

@ -1,43 +1,67 @@
# Xavier Beynon 2014
# coding: utf-8 # coding: utf-8
from __future__ import unicode_literals from __future__ import unicode_literals
from .common import InfoExtractor from .common import InfoExtractor
from .soundcloud import SoundcloudIE
import datetime import datetime
import time import time
import urllib.request
import json
class AudiomackIE(InfoExtractor): class AudiomackIE(InfoExtractor):
_VALID_URL = r'https?://(?:www\.)?audiomack\.com/song/(?P<id>[\w/-]+)' _VALID_URL = r'https?://(?:www\.)?audiomack\.com/song/(?P<id>[\w/-]+)'
_TEST = { IE_NAME = 'audiomack'
'url': 'https://www.audiomack.com/song/crewneckkramer/story-i-tell', _TESTS = [
'info_dict': { #hosted on audiomack
'id': 'story-i-tell', {
'ext': 'mp3', 'url': 'http://www.audiomack.com/song/roosh-williams/extraordinary',
'title': 'story-i-tell' 'file': 'Roosh Williams - Extraordinary.mp3',
'info_dict':
{
'ext': 'mp3',
'title': 'Roosh Williams - Extraordinary'
}
},
#hosted on soundcloud via audiomack
{
'url': 'http://www.audiomack.com/song/xclusiveszone/take-kare',
'file': '172419696.mp3',
'info_dict':
{
'ext': 'mp3',
'title': 'Young Thug ft Lil Wayne - Take Kare',
"upload_date": "20141016",
"description": "New track produced by London On Da Track called “Take Kare\"\n\nhttp://instagram.com/theyoungthugworld\nhttps://www.facebook.com/ThuggerThuggerCashMoney\n",
"uploader": "Young Thug World"
}
} }
} ]
def _real_extract(self, url): def _real_extract(self, url):
# TODO more code goes here, for example ... #id is what follows /song/ in url, usually the uploader name + title
#webpage = self._download_webpage(url, video_id) id = url[url.index("/song/")+5:]
#title = self._html_search_regex(r'<h1>(.*?)</h1>', webpage, 'title')
assert("/song/" in url) #Call the api, which gives us a json doc with the real url inside
songurl = url[url.index("/song/")+5:] rightnow = int(time.mktime(datetime.datetime.now().timetuple()))
title = songurl[songurl.rindex("/")+1:] apiresponse = self._download_json("http://www.audiomack.com/api/music/url/song"+id+"?_="+str(rightnow), id)
video_id = title if not url in apiresponse:
t = int(time.mktime(datetime.datetime.now().timetuple())) raise Exception("Unable to deduce api url of song")
s = "http://www.audiomack.com/api/music/url/song"+songurl+"?_="+str(t) realurl = apiresponse["url"]
f = urllib.request.urlopen(s)
j = f.read(1000).decode("utf-8")
data = json.loads(j)
return { #Audiomack wraps a lot of soundcloud tracks in their branded wrapper
'id': video_id, # - if so, pass the work off to the soundcloud extractor
'title': title, if SoundcloudIE.suitable(realurl):
'url' : data["url"], sc = SoundcloudIE(downloader=self._downloader)
'ext' : 'mp3' return sc._real_extract(realurl)
# TODO more properties (see youtube_dl/extractor/common.py) else:
} #Pull out metadata
page = self._download_webpage(url, id)
artist = self._html_search_regex(r'<span class="artist">(.*)</span>', page, "artist")
songtitle = self._html_search_regex(r'<h1 class="profile-title song-title"><span class="artist">.*</span>(.*)</h1>', page, "title")
title = artist+" - "+songtitle
return {
'id': title, # ignore id, which is not useful in song name
'title': title,
'url': realurl,
'ext': 'mp3'
}