commit: afb74964162eaee64c5c9b72990837daae945fec
parent 69e6efac1669da68c0746419657160311cde2671
Author: Remita Amine <remitamine@gmail.com>
Date: Fri, 5 Apr 2019 11:45:49 +0100
[adultswim] fix extraction(closes #18025)
Diffstat:
1 file changed, 117 insertions(+), 75 deletions(-)
diff --git a/youtube_dl/extractor/adultswim.py b/youtube_dl/extractor/adultswim.py
@@ -1,13 +1,19 @@
# coding: utf-8
from __future__ import unicode_literals
+import json
import re
from .turner import TurnerBaseIE
from ..utils import (
+ determine_ext,
+ float_or_none,
int_or_none,
+ mimetype2ext,
+ parse_age_limit,
+ parse_iso8601,
strip_or_none,
- url_or_none,
+ try_get,
)
@@ -21,8 +27,8 @@ class AdultSwimIE(TurnerBaseIE):
'ext': 'mp4',
'title': 'Rick and Morty - Pilot',
'description': 'Rick moves in with his daughter\'s family and establishes himself as a bad influence on his grandson, Morty.',
- 'timestamp': 1493267400,
- 'upload_date': '20170427',
+ 'timestamp': 1543294800,
+ 'upload_date': '20181127',
},
'params': {
# m3u8 download
@@ -43,6 +49,7 @@ class AdultSwimIE(TurnerBaseIE):
# m3u8 download
'skip_download': True,
},
+ 'skip': '404 Not Found',
}, {
'url': 'http://www.adultswim.com/videos/decker/inside-decker-a-new-hero/',
'info_dict': {
@@ -61,9 +68,9 @@ class AdultSwimIE(TurnerBaseIE):
}, {
'url': 'http://www.adultswim.com/videos/attack-on-titan',
'info_dict': {
- 'id': 'b7A69dzfRzuaXIECdxW8XQ',
+ 'id': 'attack-on-titan',
'title': 'Attack on Titan',
- 'description': 'md5:6c8e003ea0777b47013e894767f5e114',
+ 'description': 'md5:41caa9416906d90711e31dc00cb7db7e',
},
'playlist_mincount': 12,
}, {
@@ -78,83 +85,118 @@ class AdultSwimIE(TurnerBaseIE):
# m3u8 download
'skip_download': True,
},
+ 'skip': '404 Not Found',
}]
def _real_extract(self, url):
show_path, episode_path = re.match(self._VALID_URL, url).groups()
display_id = episode_path or show_path
- webpage = self._download_webpage(url, display_id)
- initial_data = self._parse_json(self._search_regex(
- r'AS_INITIAL_DATA(?:__)?\s*=\s*({.+?});',
- webpage, 'initial data'), display_id)
-
- is_stream = show_path == 'streams'
- if is_stream:
- if not episode_path:
- episode_path = 'live-stream'
+ query = '''query {
+ getShowBySlug(slug:"%s") {
+ %%s
+ }
+}''' % show_path
+ if episode_path:
+ query = query % '''title
+ getVideoBySlug(slug:"%s") {
+ _id
+ auth
+ description
+ duration
+ episodeNumber
+ launchDate
+ mediaID
+ seasonNumber
+ poster
+ title
+ tvRating
+ }''' % episode_path
+ ['getVideoBySlug']
+ else:
+ query = query % '''metaDescription
+ title
+ videos(first:1000,sort:["episode_number"]) {
+ edges {
+ node {
+ _id
+ slug
+ }
+ }
+ }'''
+ show_data = self._download_json(
+ 'https://www.adultswim.com/api/search', display_id,
+ data=json.dumps({'query': query}).encode(),
+ headers={'Content-Type': 'application/json'})['data']['getShowBySlug']
+ if episode_path:
+ video_data = show_data['getVideoBySlug']
+ video_id = video_data['_id']
+ episode_title = title = video_data['title']
+ series = show_data.get('title')
+ if series:
+ title = '%s - %s' % (series, title)
+ info = {
+ 'id': video_id,
+ 'title': title,
+ 'description': strip_or_none(video_data.get('description')),
+ 'duration': float_or_none(video_data.get('duration')),
+ 'formats': [],
+ 'subtitles': {},
+ 'age_limit': parse_age_limit(video_data.get('tvRating')),
+ 'thumbnail': video_data.get('poster'),
+ 'timestamp': parse_iso8601(video_data.get('launchDate')),
+ 'series': series,
+ 'season_number': int_or_none(video_data.get('seasonNumber')),
+ 'episode': episode_title,
+ 'episode_number': int_or_none(video_data.get('episodeNumber')),
+ }
- video_data = next(stream for stream_path, stream in initial_data['streams'].items() if stream_path == episode_path)
- video_id = video_data.get('stream')
+ auth = video_data.get('auth')
+ media_id = video_data.get('mediaID')
+ if media_id:
+ info.update(self._extract_ngtv_info(media_id, {
+ # CDN_TOKEN_APP_ID from:
+ # https://d2gg02c3xr550i.cloudfront.net/assets/asvp.e9c8bef24322d060ef87.bundle.js
+ 'appId': 'eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJhcHBJZCI6ImFzLXR2ZS1kZXNrdG9wLXB0enQ2bSIsInByb2R1Y3QiOiJ0dmUiLCJuZXR3b3JrIjoiYXMiLCJwbGF0Zm9ybSI6ImRlc2t0b3AiLCJpYXQiOjE1MzI3MDIyNzl9.BzSCk-WYOZ2GMCIaeVb8zWnzhlgnXuJTCu0jGp_VaZE',
+ }, {
+ 'url': url,
+ 'site_name': 'AdultSwim',
+ 'auth_required': auth,
+ }))
- if not video_id:
- entries = []
- for episode in video_data.get('archiveEpisodes', []):
- episode_url = url_or_none(episode.get('url'))
- if not episode_url:
+ if not auth:
+ extract_data = self._download_json(
+ 'https://www.adultswim.com/api/shows/v1/videos/' + video_id,
+ video_id, query={'fields': 'stream'}, fatal=False) or {}
+ assets = try_get(extract_data, lambda x: x['data']['video']['stream']['assets'], list) or []
+ for asset in assets:
+ asset_url = asset.get('url')
+ if not asset_url:
continue
- entries.append(self.url_result(
- episode_url, 'AdultSwim', episode.get('id')))
- return self.playlist_result(
- entries, video_data.get('id'), video_data.get('title'),
- strip_or_none(video_data.get('description')))
- else:
- show_data = initial_data['show']
-
- if not episode_path:
- entries = []
- for video in show_data.get('videos', []):
- slug = video.get('slug')
- if not slug:
+ ext = determine_ext(asset_url, mimetype2ext(asset.get('mime_type')))
+ if ext == 'm3u8':
+ info['formats'].extend(self._extract_m3u8_formats(
+ asset_url, video_id, 'mp4', m3u8_id='hls', fatal=False))
+ elif ext == 'f4m':
continue
- entries.append(self.url_result(
- 'http://adultswim.com/videos/%s/%s' % (show_path, slug),
- 'AdultSwim', video.get('id')))
- return self.playlist_result(
- entries, show_data.get('id'), show_data.get('title'),
- strip_or_none(show_data.get('metadata', {}).get('description')))
-
- video_data = show_data['sluggedVideo']
- video_id = video_data['id']
+ # info['formats'].extend(self._extract_f4m_formats(
+ # asset_url, video_id, f4m_id='hds', fatal=False))
+ elif ext in ('scc', 'ttml', 'vtt'):
+ info['subtitles'].setdefault('en', []).append({
+ 'url': asset_url,
+ })
+ self._sort_formats(info['formats'])
- info = self._extract_cvp_info(
- 'http://www.adultswim.com/videos/api/v0/assets?platform=desktop&id=' + video_id,
- video_id, {
- 'secure': {
- 'media_src': 'http://androidhls-secure.cdn.turner.com/adultswim/big',
- 'tokenizer_src': 'http://www.adultswim.com/astv/mvpd/processors/services/token_ipadAdobe.do',
- },
- }, {
- 'url': url,
- 'site_name': 'AdultSwim',
- 'auth_required': video_data.get('auth'),
- })
-
- info.update({
- 'id': video_id,
- 'display_id': display_id,
- 'description': info.get('description') or strip_or_none(video_data.get('description')),
- })
- if not is_stream:
- info.update({
- 'duration': info.get('duration') or int_or_none(video_data.get('duration')),
- 'timestamp': info.get('timestamp') or int_or_none(video_data.get('launch_date')),
- 'season_number': info.get('season_number') or int_or_none(video_data.get('season_number')),
- 'episode': info['title'],
- 'episode_number': info.get('episode_number') or int_or_none(video_data.get('episode_number')),
- })
-
- info['series'] = video_data.get('collection_title') or info.get('series')
- if info['series'] and info['series'] != info['title']:
- info['title'] = '%s - %s' % (info['series'], info['title'])
-
- return info
+ return info
+ else:
+ entries = []
+ for edge in show_data.get('videos', {}).get('edges', []):
+ video = edge.get('node') or {}
+ slug = video.get('slug')
+ if not slug:
+ continue
+ entries.append(self.url_result(
+ 'http://adultswim.com/videos/%s/%s' % (show_path, slug),
+ 'AdultSwim', video.get('_id')))
+ return self.playlist_result(
+ entries, show_path, show_data.get('title'),
+ strip_or_none(show_data.get('metaDescription')))