411fef84cac6b5309ff8b6e7f04b54191afadbf7
[ytdl] / youtube_dl / extractor / scrippsnetworks.py
1 # coding: utf-8
2 from __future__ import unicode_literals
3
4 import datetime
5 import json
6 import hashlib
7 import hmac
8 import re
9
10 from .common import InfoExtractor
11 from .anvato import AnvatoIE
12 from ..utils import (
13     urlencode_postdata,
14     xpath_text,
15 )
16
17
18 class ScrippsNetworksWatchIE(InfoExtractor):
19     IE_NAME = 'scrippsnetworks:watch'
20     _VALID_URL = r'''(?x)
21                     https?://
22                         watch\.
23                         (?P<site>hgtv|foodnetwork|travelchannel|diynetwork|cookingchanneltv|geniuskitchen)\.com/
24                         (?:
25                             player\.[A-Z0-9]+\.html\#|
26                             show/(?:[^/]+/){2}|
27                             player/
28                         )
29                         (?P<id>\d+)
30                     '''
31     _TESTS = [{
32         'url': 'http://watch.hgtv.com/show/HGTVE/Best-Ever-Treehouses/2241515/Best-Ever-Treehouses/',
33         'md5': '26545fd676d939954c6808274bdb905a',
34         'info_dict': {
35             'id': '4173834',
36             'ext': 'mp4',
37             'title': 'Best Ever Treehouses',
38             'description': "We're searching for the most over the top treehouses.",
39             'uploader': 'ANV',
40             'upload_date': '20170922',
41             'timestamp': 1506056400,
42         },
43         'params': {
44             'skip_download': True,
45         },
46         'add_ie': [AnvatoIE.ie_key()],
47     }, {
48         'url': 'http://watch.diynetwork.com/show/DSAL/Salvage-Dawgs/2656646/Covington-Church/',
49         'only_matching': True,
50     }, {
51         'url': 'http://watch.diynetwork.com/player.HNT.html#2656646',
52         'only_matching': True,
53     }, {
54         'url': 'http://watch.geniuskitchen.com/player/3787617/Ample-Hills-Ice-Cream-Bike/',
55         'only_matching': True,
56     }]
57
58     _SNI_TABLE = {
59         'hgtv': 'hgtv',
60         'diynetwork': 'diy',
61         'foodnetwork': 'food',
62         'cookingchanneltv': 'cook',
63         'travelchannel': 'trav',
64         'geniuskitchen': 'genius',
65     }
66     _SNI_HOST = 'web.api.video.snidigital.com'
67
68     _AWS_REGION = 'us-east-1'
69     _AWS_IDENTITY_ID_JSON = json.dumps({
70         'IdentityId': '%s:7655847c-0ae7-4d9b-80d6-56c062927eb3' % _AWS_REGION
71     })
72     _AWS_USER_AGENT = 'aws-sdk-js/2.80.0 callback'
73     _AWS_API_KEY = 'E7wSQmq0qK6xPrF13WmzKiHo4BQ7tip4pQcSXVl1'
74     _AWS_SERVICE = 'execute-api'
75     _AWS_REQUEST = 'aws4_request'
76     _AWS_SIGNED_HEADERS = ';'.join([
77         'host', 'x-amz-date', 'x-amz-security-token', 'x-api-key'])
78     _AWS_CANONICAL_REQUEST_TEMPLATE = '''GET
79 %(uri)s
80
81 host:%(host)s
82 x-amz-date:%(date)s
83 x-amz-security-token:%(token)s
84 x-api-key:%(key)s
85
86 %(signed_headers)s
87 %(payload_hash)s'''
88
89     def _real_extract(self, url):
90         mobj = re.match(self._VALID_URL, url)
91         site_id, video_id = mobj.group('site', 'id')
92
93         def aws_hash(s):
94             return hashlib.sha256(s.encode('utf-8')).hexdigest()
95
96         token = self._download_json(
97             'https://cognito-identity.us-east-1.amazonaws.com/', video_id,
98             data=self._AWS_IDENTITY_ID_JSON.encode('utf-8'),
99             headers={
100                 'Accept': '*/*',
101                 'Content-Type': 'application/x-amz-json-1.1',
102                 'Referer': url,
103                 'X-Amz-Content-Sha256': aws_hash(self._AWS_IDENTITY_ID_JSON),
104                 'X-Amz-Target': 'AWSCognitoIdentityService.GetOpenIdToken',
105                 'X-Amz-User-Agent': self._AWS_USER_AGENT,
106             })['Token']
107
108         sts = self._download_xml(
109             'https://sts.amazonaws.com/', video_id, data=urlencode_postdata({
110                 'Action': 'AssumeRoleWithWebIdentity',
111                 'RoleArn': 'arn:aws:iam::710330595350:role/Cognito_WebAPIUnauth_Role',
112                 'RoleSessionName': 'web-identity',
113                 'Version': '2011-06-15',
114                 'WebIdentityToken': token,
115             }), headers={
116                 'Referer': url,
117                 'X-Amz-User-Agent': self._AWS_USER_AGENT,
118                 'Content-Type': 'application/x-www-form-urlencoded; charset=utf-8',
119             })
120
121         def get(key):
122             return xpath_text(
123                 sts, './/{https://sts.amazonaws.com/doc/2011-06-15/}%s' % key,
124                 fatal=True)
125
126         access_key_id = get('AccessKeyId')
127         secret_access_key = get('SecretAccessKey')
128         session_token = get('SessionToken')
129
130         # Task 1: http://docs.aws.amazon.com/general/latest/gr/sigv4-create-canonical-request.html
131         uri = '/1/web/brands/%s/episodes/scrid/%s' % (self._SNI_TABLE[site_id], video_id)
132         datetime_now = datetime.datetime.utcnow().strftime('%Y%m%dT%H%M%SZ')
133         date = datetime_now[:8]
134         canonical_string = self._AWS_CANONICAL_REQUEST_TEMPLATE % {
135             'uri': uri,
136             'host': self._SNI_HOST,
137             'date': datetime_now,
138             'token': session_token,
139             'key': self._AWS_API_KEY,
140             'signed_headers': self._AWS_SIGNED_HEADERS,
141             'payload_hash': aws_hash(''),
142         }
143
144         # Task 2: http://docs.aws.amazon.com/general/latest/gr/sigv4-create-string-to-sign.html
145         credential_string = '/'.join([date, self._AWS_REGION, self._AWS_SERVICE, self._AWS_REQUEST])
146         string_to_sign = '\n'.join([
147             'AWS4-HMAC-SHA256', datetime_now, credential_string,
148             aws_hash(canonical_string)])
149
150         # Task 3: http://docs.aws.amazon.com/general/latest/gr/sigv4-calculate-signature.html
151         def aws_hmac(key, msg):
152             return hmac.new(key, msg.encode('utf-8'), hashlib.sha256)
153
154         def aws_hmac_digest(key, msg):
155             return aws_hmac(key, msg).digest()
156
157         def aws_hmac_hexdigest(key, msg):
158             return aws_hmac(key, msg).hexdigest()
159
160         k_secret = 'AWS4' + secret_access_key
161         k_date = aws_hmac_digest(k_secret.encode('utf-8'), date)
162         k_region = aws_hmac_digest(k_date, self._AWS_REGION)
163         k_service = aws_hmac_digest(k_region, self._AWS_SERVICE)
164         k_signing = aws_hmac_digest(k_service, self._AWS_REQUEST)
165
166         signature = aws_hmac_hexdigest(k_signing, string_to_sign)
167
168         auth_header = ', '.join([
169             'AWS4-HMAC-SHA256 Credential=%s' % '/'.join(
170                 [access_key_id, date, self._AWS_REGION, self._AWS_SERVICE, self._AWS_REQUEST]),
171             'SignedHeaders=%s' % self._AWS_SIGNED_HEADERS,
172             'Signature=%s' % signature,
173         ])
174
175         mcp_id = self._download_json(
176             'https://%s%s' % (self._SNI_HOST, uri), video_id, headers={
177                 'Accept': '*/*',
178                 'Referer': url,
179                 'Authorization': auth_header,
180                 'X-Amz-Date': datetime_now,
181                 'X-Amz-Security-Token': session_token,
182                 'X-Api-Key': self._AWS_API_KEY,
183             })['results'][0]['mcpId']
184
185         return self.url_result(
186             'anvato:anvato_scripps_app_web_prod_0837996dbe373629133857ae9eb72e740424d80a:%s' % mcp_id,
187             AnvatoIE.ie_key(), video_id=mcp_id)