irongit

An AI project that includes a scraper for NFL data, postgres database, and user interface. It uses a classification algorithm to predict winners.

261 lines9.1 KBPython
1
2import time
3from urllib.parse import urljoin
4from browsermobproxy import Server
5from selenium import webdriver
6from pathlib import Path
7from .url_utils import urlparse_json, generate_openapi_spec, har_entry_parse
8import json
9import requests
10import os
11
12class NFLClient():
13 """
14 ## Manages access to the internal nfl.com api
15
16 ### Global Variables
17 * `BMP_PATH` - Path to browsermob-proxy
18 * `AUTH_PATH` - Path to auth file
19 * `AUTH_ENDPOINT` - Endpoint used to get auth token
20 * `API_ROOT` - Root for nfl.com api endpoints
21 * `HAR_DIR` - Storage directory for har files
22 * `URL_JSON_PATH` - Path for json list of found api endpoints
23 * `OPENAPI_PATH` - Path for openapi yaml
24 * `TOKEN_EXPIRE_RATE` - How many seconds to download a new auth token
25 * `TOKEN_AUTH_TIMEOUT` - Timeout when searching har file for Authorization
26
27 ### Class Variables
28 * `har` - The current har in json
29 * `har_path` - Where to store the current har
30 * `server` - The browsermob server
31 * `proxy` - The browsermob proxy
32 * `driver` - The selenium driver
33 * `auth_token` - The current auth token
34 * `last_auth_download_time` - When the current auth token was downloaded
35 * `headers` - Headers sent through requests
36
37 ### Notes
38 * HAR stands for HTTP Access Requests. This is like the network tab on Chrome inspector.
39 """
40 BMP_PATH = os.path.abspath('browsermob-proxy-2.1.4/bin/browsermob-proxy.bat')
41 AUTH_PATH = Path('auth.json')
42 AUTH_ENDPOINT = 'https://nfl.com/scores'
43 API_ROOT = 'https://api.nfl.com'
44 HAR_DIR = Path('nfl_client_data/har_files')
45 URL_JSON_PATH = Path('nfl_client_data/urls.json')
46 OPENAPI_PATH = Path('nfl_client_data/openapi.yaml')
47 TOKEN_EXPIRE_RATE = 60*60
48 TOKEN_AUTH_TIMEOUT = 10
49
50 # Har download variables
51 har = {}
52 har_path = None
53 server = None
54 proxy = None
55 driver = None
56
57 # Auth variables
58 auth_token = None
59 last_auth_download_time = 0
60 headers = {}
61
62 def __init__(self, headless=True):
63 os.makedirs(self.HAR_DIR, exist_ok=True)
64 self.headless = headless
65
66 def load_auth_token(self, store_har=False):
67 """Loads the auth token into the client."""
68 if os.path.exists(self.AUTH_PATH):
69 with open(self.AUTH_PATH, 'r') as f:
70 auth_json = json.load(f)
71 self.auth_token = auth_json['token']
72 self.last_auth_download_time = auth_json['time']
73 if self.last_auth_download_time < time.time()-self.TOKEN_EXPIRE_RATE or not self.auth_token:
74 self.download_auth_token(store_har=store_har)
75 self.headers={
76 'Authorization': self.auth_token
77 }
78
79 def prep_proxy(self, endpoint=None):
80 """Prepares the driver and proxy to get the har"""
81 endpoint = endpoint or self.AUTH_ENDPOINT
82 # Start BrowserMob Proxy
83 self.server = Server(self.BMP_PATH)
84 self.server.start()
85 self.proxy = self.server.create_proxy()
86
87 # Configure Chrome with the proxy
88 chrome_options = webdriver.ChromeOptions()
89 chrome_options.add_argument(f'--proxy-server={self.proxy.proxy}')
90 chrome_options.add_argument('--ignore-certificate-errors')
91 if self.headless:
92 chrome_options.add_argument('--headless')
93 self.driver = webdriver.Chrome(options=chrome_options)
94 # Navigate to the website
95 self.driver.get(endpoint)
96
97 # Start capturing network traffic
98 self.proxy.new_har(f"nfl{time.time()}", options={'captureHeaders': True, 'captureContent': True})
99
100 # Create the path for storing har data
101 try:
102 raw_path = endpoint.split('.com')[1]
103 raw_path = raw_path.split('?')[0]
104 str_path = raw_path.replace('/', '__')
105 except:
106 str_path = 'nfl'
107 self.har_path = self.HAR_DIR/Path(str_path+'.har')
108
109 def close_proxy(self):
110 """Closes the driver and proxy"""
111 self.server.stop()
112 self.driver.quit()
113
114 def wait_for_auth_token(self):
115 """Downloads the auth token using the scores page of nfl.com"""
116 start_time = time.time()
117 while time.time() - start_time < self.TOKEN_AUTH_TIMEOUT:
118 # Capture the HAR once
119 self.har = self.proxy.har
120 for entry in self.har['log']['entries']:
121 request_headers = entry['request']['headers']
122 for header in request_headers:
123 if header['name'] == 'Authorization':
124 self.auth_token = header['value']
125 return self.auth_token
126 time.sleep(1)
127 return None
128
129 def store_auth_token(self):
130 """Stores the current auth token"""
131 # Store the auth token
132 self.time = time.time()
133 auth_json = {
134 'time': self.time,
135 'token': self.auth_token
136 }
137 with open(self.AUTH_PATH, 'w') as f:
138 json.dump(auth_json, f)
139
140 def store_har(self):
141 """Stores the current har"""
142 # Store the har data
143 with open(self.har_path, 'w') as f:
144 json.dump(self.har, f)
145
146 def download_auth_token(self, store_har=False):
147 """Downloads the auth token"""
148 self.prep_proxy()
149 self.wait_for_auth_token()
150 self.store_auth_token()
151 if store_har:
152 self.store_har()
153
154 def download_endpoints(self, endpoint=None, wait_time=20):
155 """Downloads the endpoints found in har, stores json, updates readme, updates Mixin for the class"""
156 if not endpoint:
157 endpoint = self.AUTH_ENDPOINT
158 self.prep_proxy(endpoint)
159 time.sleep(wait_time)
160 self.har = self.proxy.har
161 self.wait_for_auth_token()
162 self.store_auth_token()
163 self.store_har()
164 self.close_proxy()
165
166 # Load previous urls
167 url_json = {}
168 if os.path.exists(self.URL_JSON_PATH):
169 with open(self.URL_JSON_PATH, 'r') as f:
170 url_json = json.load(f)
171
172 # Gather the list of endpoints
173 for entry in self.har['log']['entries']:
174 if self.API_ROOT in entry['request']['url']:
175 url_json.update(har_entry_parse(entry))
176
177 # Store the url json
178 with open(self.URL_JSON_PATH, 'w') as f:
179 json.dump(url_json, f)
180
181 # Store the url openapi
182 with open(self.OPENAPI_PATH, 'w') as f:
183 f.write(generate_openapi_spec(url_json))
184
185 return url_json
186
187 def request(self, endpoint, params={}) -> dict:
188 """Request an endpoint"""
189 self.load_auth_token()
190 url = urljoin(self.API_ROOT, endpoint)
191 print(url)
192 return requests.get(url, headers=self.headers, params=params)
193
194 def get_week(self, season: int, week: int, seasonType='REG', withExternalIds=True) -> requests.Response:
195 """## Get summary of games during a week.
196 ### Args
197 * `season`: Year of the season (i.e. 2023)
198 * `week`: Week of the season (i.e. 3)
199 * `seasonType`: `REG` or `PRE`
200 * `withExternalIds`: boolean
201 """
202 url = f'/football/v2/games/season/{season}/seasonType/{seasonType}/week/{week}'
203 return self.request(url, {
204 'withExternalIds': withExternalIds
205 })
206
207 def get_game(self, game_id: str, withExternalIds=True) -> requests.Response:
208 """## Get game details.
209 ### Args
210 * `game_id`: Can be found iterating the week.
211 * `withExternalIds`: boolean
212 """
213 return self.request(f'/football/v2/games/{game_id}', {
214 'withExternalIds': withExternalIds
215 })
216
217 def get_standings(self, week: int, season: int, seasonType='REG', limit=100) -> requests.Response:
218 """## Get team standings.
219 ### Args
220 * `season`: Year of the standings (i.e. 2023)
221 * `week`: Week of the standings (i.e. 3)
222 * `seasonType`: `REG` or `PRE`
223 * `limit`: Max items to get
224 """
225 return self.request(f'/football/v2/standings', {
226 'season': season,
227 'week': week,
228 'seasonType': seasonType,
229 'limit': limit
230 })
231
232 def get_game_summary(self, game_summary_id) -> requests.Request:
233 """## Get game results.
234 ### Args
235 * `game_summary_id`: Lookup id, can be found in `get_week`.
236 """
237 return self.request(f'/football/v2/stats/live/game-summaries/{game_summary_id}')
238
239 def get_game_summaries(self, season: int, week: int, seasonType='REG') -> requests.Response:
240 """## Get game summaries for a week.
241 ## Args
242 * `season`: Season
243 * `week`: Week
244 * `seasonType`: `REG` or `PRE`
245 """
246 return self.request(f'/football/v2/stats/live/game-summaries', {
247 'season': season,
248 'week': week,
249 'seasonType': seasonType
250 })
251
252 def get_teams(self, season: int, limit=100) -> requests.Response:
253 """## Get teams in season.
254 ### Args
255 * `season`: Year
256 * `limit`: Query limit
257 """
258 return self.request(f'/football/v2/teams/history', {
259 'season': season,
260 'limit': limit,
261 })