grooveshark.py 7.3 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205
  1. # coding: utf-8
  2. from __future__ import unicode_literals
  3. import time
  4. import math
  5. import re
  6. from urllib import quote, urlencode
  7. from os.path import basename
  8. from .common import InfoExtractor
  9. from ..utils import ExtractorError, compat_urllib_request, compat_html_parser
  10. from ..utils import compat_urlparse
  11. urlparse = compat_urlparse.urlparse
  12. urlunparse = compat_urlparse.urlunparse
  13. urldefrag = compat_urlparse.urldefrag
  14. class GroovesharkHtmlParser(compat_html_parser.HTMLParser):
  15. def __init__(self):
  16. self._current_object = None
  17. self.objects = []
  18. compat_html_parser.HTMLParser.__init__(self)
  19. def handle_starttag(self, tag, attrs):
  20. attrs = dict((k, v) for k, v in attrs)
  21. if tag == 'object':
  22. self._current_object = {'attrs': attrs, 'params': []}
  23. elif tag == 'param':
  24. self._current_object['params'].append(attrs)
  25. def handle_endtag(self, tag):
  26. if tag == 'object':
  27. self.objects.append(self._current_object)
  28. self._current_object = None
  29. @classmethod
  30. def extract_object_tags(cls, html):
  31. p = cls()
  32. p.feed(html)
  33. p.close()
  34. return p.objects
  35. class GroovesharkIE(InfoExtractor):
  36. _VALID_URL = r'https?://(www\.)?grooveshark\.com/#!/s/([^/]+)/([^/]+)'
  37. _TEST = {
  38. 'url': 'http://grooveshark.com/#!/s/Jolene+Tenth+Key+Remix+Ft+Will+Sessions/6SS1DW?src=5',
  39. 'md5': 'bbccc50b19daca23b8f961152c1dc95b',
  40. 'info_dict': {
  41. 'id': '6SS1DW',
  42. 'title': 'Jolene (Tenth Key Remix ft. Will Sessions)',
  43. 'ext': 'mp3',
  44. 'duration': 227
  45. }
  46. }
  47. do_playerpage_request = True
  48. do_bootstrap_request = True
  49. def _parse_target(self, target):
  50. uri = urlparse(target)
  51. hash = uri.fragment[1:].split('?')[0]
  52. token = basename(hash.rstrip('/'))
  53. return (uri, hash, token)
  54. def _build_bootstrap_url(self, target):
  55. (uri, hash, token) = self._parse_target(target)
  56. query = 'getCommunicationToken=1&hash=%s&%d' % (quote(hash, safe=''), self.ts)
  57. return (urlunparse((uri.scheme, uri.netloc, '/preload.php', None, query, None)), token)
  58. def _build_meta_url(self, target):
  59. (uri, hash, token) = self._parse_target(target)
  60. query = 'hash=%s&%d' % (quote(hash, safe=''), self.ts)
  61. return (urlunparse((uri.scheme, uri.netloc, '/preload.php', None, query, None)), token)
  62. def _build_stream_url(self, meta):
  63. return urlunparse(('http', meta['streamKey']['ip'], '/stream.php', None, None, None))
  64. def _build_swf_referer(self, target, obj):
  65. (uri, _, _) = self._parse_target(target)
  66. return urlunparse((uri.scheme, uri.netloc, obj['attrs']['data'], None, None, None))
  67. def _transform_bootstrap(self, js):
  68. return re.split('(?m)^\s*try\s*{', js)[0] \
  69. .split(' = ', 1)[1].strip().rstrip(';')
  70. def _transform_meta(self, js):
  71. return js.split('\n')[0].split('=')[1].rstrip(';')
  72. def _get_meta(self, target):
  73. (meta_url, token) = self._build_meta_url(target)
  74. self.to_screen('Metadata URL: %s' % meta_url)
  75. headers = {'Referer': urldefrag(target)[0]}
  76. req = compat_urllib_request.Request(meta_url, headers=headers)
  77. res = self._download_json(req, token,
  78. transform_source=self._transform_meta)
  79. if 'getStreamKeyWithSong' not in res:
  80. raise ExtractorError(
  81. 'Metadata not found. URL may be malformed, or Grooveshark API may have changed.')
  82. if res['getStreamKeyWithSong'] is None:
  83. raise ExtractorError(
  84. 'Metadata download failed, probably due to Grooveshark anti-abuse throttling. Wait at least an hour before retrying from this IP.',
  85. expected=True)
  86. return res['getStreamKeyWithSong']
  87. def _get_bootstrap(self, target):
  88. (bootstrap_url, token) = self._build_bootstrap_url(target)
  89. headers = {'Referer': urldefrag(target)[0]}
  90. req = compat_urllib_request.Request(bootstrap_url, headers=headers)
  91. res = self._download_json(req, token, fatal=False,
  92. note='Downloading player bootstrap data',
  93. errnote='Unable to download player bootstrap data',
  94. transform_source=self._transform_bootstrap)
  95. return res
  96. def _get_playerpage(self, target):
  97. (_, _, token) = self._parse_target(target)
  98. res = self._download_webpage(
  99. target, token,
  100. note='Downloading player page',
  101. errnote='Unable to download player page',
  102. fatal=False)
  103. if res is not None:
  104. o = GroovesharkHtmlParser.extract_object_tags(res)
  105. return (res, [x for x in o if x['attrs']['id'] == 'jsPlayerEmbed'])
  106. return (res, None)
  107. def _real_extract(self, url):
  108. (target_uri, _, token) = self._parse_target(url)
  109. # 1. Fill cookiejar by making a request to the player page
  110. if self.do_playerpage_request:
  111. (_, player_objs) = self._get_playerpage(url)
  112. if player_objs is not None:
  113. swf_referer = self._build_swf_referer(url, player_objs[0])
  114. self.to_screen('SWF Referer: %s' % swf_referer)
  115. # 2. Ask preload.php for swf bootstrap data to better mimic webapp
  116. if self.do_bootstrap_request:
  117. bootstrap = self._get_bootstrap(url)
  118. self.to_screen('CommunicationToken: %s' % bootstrap['getCommunicationToken'])
  119. # 3. Ask preload.php for track metadata.
  120. meta = self._get_meta(url)
  121. # 4. Construct stream request for track.
  122. stream_url = self._build_stream_url(meta)
  123. duration = int(math.ceil(float(meta['streamKey']['uSecs']) / 1000000))
  124. post_dict = {'streamKey': meta['streamKey']['streamKey']}
  125. post_data = urlencode(post_dict).encode('utf-8')
  126. headers = {
  127. 'Content-Length': len(post_data),
  128. 'Content-Type': 'application/x-www-form-urlencoded'
  129. }
  130. if 'swf_referer' in locals():
  131. headers['Referer'] = swf_referer
  132. info_dict = {
  133. 'id': token,
  134. 'title': meta['song']['Name'],
  135. 'http_method': 'POST',
  136. 'url': stream_url,
  137. 'ext': 'mp3',
  138. 'format': 'mp3 audio',
  139. 'duration': duration,
  140. # various ways of supporting the download request.
  141. # remove keys unnecessary to the eventual post implementation
  142. 'post_data': post_data,
  143. 'post_dict': post_dict,
  144. 'headers': headers
  145. }
  146. if 'swf_referer' in locals():
  147. info_dict['http_referer'] = swf_referer
  148. return info_dict
  149. def _real_initialize(self):
  150. self.ts = int(time.time() * 1000) # timestamp in millis
  151. def _download_json(self, url_or_request, video_id,
  152. note=u'Downloading JSON metadata',
  153. errnote=u'Unable to download JSON metadata',
  154. fatal=True,
  155. transform_source=None):
  156. try:
  157. out = super(GroovesharkIE, self)._download_json(
  158. url_or_request, video_id, note, errnote, transform_source)
  159. return out
  160. except ExtractorError as ee:
  161. if fatal:
  162. raise ee
  163. return None