espn.py 2.7 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374
  1. from __future__ import unicode_literals
  2. from .common import InfoExtractor
  3. from ..utils import remove_end
  4. class ESPNIE(InfoExtractor):
  5. _VALID_URL = r'https?://espn\.go\.com/(?:[^/]+/)*(?P<id>[^/]+)'
  6. _TESTS = [{
  7. 'url': 'http://espn.go.com/video/clip?id=10365079',
  8. 'md5': '60e5d097a523e767d06479335d1bdc58',
  9. 'info_dict': {
  10. 'id': 'FkYWtmazr6Ed8xmvILvKLWjd4QvYZpzG',
  11. 'ext': 'mp4',
  12. 'title': '30 for 30 Shorts: Judging Jewell',
  13. 'description': None,
  14. },
  15. 'add_ie': ['OoyalaExternal'],
  16. }, {
  17. # intl video, from http://www.espnfc.us/video/mls-highlights/150/video/2743663/must-see-moments-best-of-the-mls-season
  18. 'url': 'http://espn.go.com/video/clip?id=2743663',
  19. 'md5': 'f4ac89b59afc7e2d7dbb049523df6768',
  20. 'info_dict': {
  21. 'id': '50NDFkeTqRHB0nXBOK-RGdSG5YQPuxHg',
  22. 'ext': 'mp4',
  23. 'title': 'Must-See Moments: Best of the MLS season',
  24. },
  25. 'add_ie': ['OoyalaExternal'],
  26. }, {
  27. 'url': 'https://espn.go.com/video/iframe/twitter/?cms=espn&id=10365079',
  28. 'only_matching': True,
  29. }, {
  30. 'url': 'http://espn.go.com/nba/recap?gameId=400793786',
  31. 'only_matching': True,
  32. }, {
  33. 'url': 'http://espn.go.com/blog/golden-state-warriors/post/_/id/593/how-warriors-rapidly-regained-a-winning-edge',
  34. 'only_matching': True,
  35. }, {
  36. 'url': 'http://espn.go.com/sports/endurance/story/_/id/12893522/dzhokhar-tsarnaev-sentenced-role-boston-marathon-bombings',
  37. 'only_matching': True,
  38. }, {
  39. 'url': 'http://espn.go.com/nba/playoffs/2015/story/_/id/12887571/john-wall-washington-wizards-no-swelling-left-hand-wrist-game-5-return',
  40. 'only_matching': True,
  41. }]
  42. def _real_extract(self, url):
  43. video_id = self._match_id(url)
  44. webpage = self._download_webpage(url, video_id)
  45. video_id = self._search_regex(
  46. r'class=(["\']).*?video-play-button.*?\1[^>]+data-id=["\'](?P<id>\d+)',
  47. webpage, 'video id', group='id')
  48. cms = 'espn'
  49. if 'data-source="intl"' in webpage:
  50. cms = 'intl'
  51. player_url = 'https://espn.go.com/video/iframe/twitter/?id=%s&cms=%s' % (video_id, cms)
  52. player = self._download_webpage(
  53. player_url, video_id)
  54. pcode = self._search_regex(
  55. r'["\']pcode=([^"\']+)["\']', player, 'pcode')
  56. title = remove_end(
  57. self._og_search_title(webpage),
  58. '- ESPN Video').strip()
  59. return {
  60. '_type': 'url_transparent',
  61. 'url': 'ooyalaexternal:%s:%s:%s' % (cms, video_id, pcode),
  62. 'ie_key': 'OoyalaExternal',
  63. 'title': title,
  64. }