-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathmain.py
More file actions
361 lines (286 loc) · 13.2 KB
/
Copy pathmain.py
File metadata and controls
361 lines (286 loc) · 13.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
# ==========================================
# NFL PREDICTOR - MASTER SCRIPT
# ==========================================
# --- BLOCK 1: IMPORTS ---
import nflreadpy as nfl
import pandas as pd
from sklearn.preprocessing import StandardScaler
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import accuracy_score, confusion_matrix
# --- BLOCK 2: LOAD DATA ---
years = list(range(2015, 2026)) # 2015 to 2025
# 1. Load Team Stats
team_stats_polars = nfl.load_team_stats(seasons=years)
team_stats = team_stats_polars.to_pandas()
# 2. Load Schedules (Crucial for Home/Away info)
sched_polars = nfl.load_schedules(seasons=years)
sched = sched_polars.to_pandas()
# --- BLOCK 3: CLEAN & MAP TEAMS ---
team_abbr_map = {
"atl": "Atlanta Falcons", "falcons": "Atlanta Falcons",
"tb": "Tampa Bay Buccaneers", "bucs": "Tampa Bay Buccaneers", "buccaneers": "Tampa Bay Buccaneers",
"nyg": "New York Giants", "giants": "New York Giants",
"dal": "Dallas Cowboys", "cowboys": "Dallas Cowboys",
"phi": "Philadelphia Eagles", "eagles": "Philadelphia Eagles",
"was": "Washington Commanders", "washington": "Washington Commanders", "commanders": "Washington Commanders",
"chi": "Chicago Bears", "bears": "Chicago Bears",
"det": "Detroit Lions", "lions": "Detroit Lions",
"gb": "Green Bay Packers", "packers": "Green Bay Packers",
"min": "Minnesota Vikings", "vikings": "Minnesota Vikings",
"car": "Carolina Panthers", "panthers": "Carolina Panthers",
"nor": "New Orleans Saints", "saints": "New Orleans Saints",
"ari": "Arizona Cardinals", "cardinals": "Arizona Cardinals",
"sf": "San Francisco 49ers", "49ers": "San Francisco 49ers",
"sea": "Seattle Seahawks", "seahawks": "Seattle Seahawks",
"lac": "Los Angeles Chargers", "chargers": "Los Angeles Chargers",
"den": "Denver Broncos", "broncos": "Denver Broncos",
"kc": "Kansas City Chiefs", "chiefs": "Kansas City Chiefs",
"lv": "Las Vegas Raiders", "raiders": "Las Vegas Raiders",
"lar": "Los Angeles Rams", "rams": "Los Angeles Rams", "stl": "Los Angeles Rams", "la": "Los Angeles Rams",
"mia": "Miami Dolphins", "dolphins": "Miami Dolphins",
"buf": "Buffalo Bills", "bills": "Buffalo Bills",
"nyj": "New York Jets", "jets": "New York Jets",
"pit": "Pittsburgh Steelers", "steelers": "Pittsburgh Steelers",
"cin": "Cincinnati Bengals", "bengals": "Cincinnati Bengals",
"bal": "Baltimore Ravens", "ravens": "Baltimore Ravens",
"cle": "Cleveland Browns", "browns": "Cleveland Browns",
"ten": "Tennessee Titans", "titans": "Tennessee Titans",
"jax": "Jacksonville Jaguars", "jaguars": "Jacksonville Jaguars",
"hou": "Houston Texans", "texans": "Houston Texans",
"ind": "Indianapolis Colts", "colts": "Indianapolis Colts",
"ne": "New England Patriots", "patriots": "New England Patriots"
}
def canonical_team(x):
x = str(x).lower()
return team_abbr_map.get(x, None)
# Apply mapping
team_stats["team_full"] = team_stats["team"].apply(canonical_team)
team_stats = team_stats[team_stats["team_full"].notna()]
sched["home_full"] = sched["home_team"].apply(canonical_team)
sched["away_full"] = sched["away_team"].apply(canonical_team)
sched = sched[(sched["home_full"].notna()) | (sched["away_full"].notna())]
# --- BLOCK 4: MERGE STATS & SCHEDULES ---
# Build Win/Loss DataFrame
records = []
for _, r in sched.iterrows():
home, away = r["home_full"], r["away_full"]
hs, ascore = r["home_score"], r["away_score"]
if pd.notna(hs) and pd.notna(ascore):
if home: records.append([r["season"], r["week"], home, home, away, int(hs > ascore)])
if away: records.append([r["season"], r["week"], away, home, away, int(ascore > hs)])
win_df = pd.DataFrame(records, columns=["season", "week", "team_full", "home_full", "away_full", "win"])
# Merge with stats
df = pd.merge(team_stats, win_df, on=["season","week","team_full"], how="inner")
# Drop unnecessary columns
drop_cols = ["team","season_type","opponent_team","fg_made_list","fg_missed_list","fg_blocked_list"]
df = df.drop(columns=[c for c in drop_cols if c in df.columns])
df = df.fillna(0)
# --- BLOCK 5: FEATURE ENGINEERING (ANCHOR & SPARK) ---
# 1. Filter Garbage Time
df = df[df["week"] <= 18]
# 2. Sort Data
df = df.sort_values(by=["team_full", "season", "week"])
# 3. Create is_home
df["is_home"] = (df["team_full"] == df["home_full"]).astype(int)
# 4. Define Base Features
exclude_cols = ["team_full", "win", "season", "week", "home_full", "away_full", "opponent", "is_home"]
base_features = [c for c in df.columns if c not in exclude_cols and df[c].dtype != "object"]
# --- STRATEGY A: THE ANCHOR (Lifetime Expanding Mean) ---
# Stability: Uses all history to judge true team quality
df_anchor = df.groupby("team_full")[base_features].transform(
lambda x: x.expanding().mean().shift(1)
)
# --- STRATEGY B: THE SPARK (Weighted Last 5 Games) ---
# Recent Form: Uses last 5 games to catch hot/cold streaks
# We use EWM (Exponential Moving Average) to prioritize the very last game
df_spark = df.groupby("team_full")[base_features].transform(
lambda x: x.ewm(span=5, min_periods=1).mean().shift(1)
)
# Rename Spark columns so they don't clash
df_spark.columns = [f"recent_{c}" for c in df_spark.columns]
# 5. Merge Both Strategies
# We combine the "Anchor" stats and the "Spark" stats into one row
cols_to_keep = ["season", "week", "team_full", "home_full", "away_full", "win", "is_home"]
df_model = pd.concat([df[cols_to_keep], df_anchor, df_spark], axis=1)
# 6. Add Opponent Stats (For both Anchor and Spark)
opponent_stats = df_model.copy()
# We need to grab ALL stat columns (Anchor + Spark)
stat_cols = [c for c in df_model.columns if c not in cols_to_keep]
# Rename everything to 'opp_'
opponent_stats = opponent_stats.rename(columns={c: f"opp_{c}" for c in stat_cols})
# Create joining key
df_model["opponent_name"] = df_model.apply(lambda x: x["away_full"] if x["team_full"] == x["home_full"] else x["home_full"], axis=1)
# Final Merge
df_final = pd.merge(
df_model,
opponent_stats[["season", "week", "team_full"] + [f"opp_{c}" for c in stat_cols]],
left_on=["season", "week", "opponent_name"],
right_on=["season", "week", "team_full"],
how="left",
suffixes=("", "_dup")
)
df_final = df_final.drop(columns=["team_full_dup", "opponent_name"])
df_final = df_final.dropna()
# Update the 'features' list to include everything
features = stat_cols + [f"opp_{c}" for c in stat_cols] + ["is_home"]
print(f"Feature Engineering Complete. Model will now learn from {len(features)} variables.")
#"""
# --- BLOCK 6: SEASON-LONG WALK-FORWARD BACKTEST ---
print("\n--- STARTING SEASON SIMULATION ---")
target_season = 2025
# Find the latest week available in the data for this season
max_week = int(df_final[df_final["season"] == target_season]["week"].max())
weekly_results = []
all_predictions = []
print(f"Simulating Season {target_season} (Weeks 1 to {max_week})...")
print("-" * 60)
print(f"{'Week':<6} | {'Accuracy':<10} | {'Correct/Total':<15} | {'Note'}")
print("-" * 60)
for w in range(1, max_week + 1):
# 1. Dynamic Split
# Train on everything BEFORE this specific week
train_mask = (
(df_final["season"] < target_season) |
((df_final["season"] == target_season) & (df_final["week"] < w))
)
test_mask = (df_final["season"] == target_season) & (df_final["week"] == w)
train_df = df_final[train_mask]
test_df = df_final[test_mask]
# Skip if data missing (e.g., bye weeks or weird data gaps)
if test_df.empty:
continue
# 2. Train
X_train = train_df[features]
y_train = train_df["win"]
# Scale
scaler = StandardScaler()
X_train_scaled = scaler.fit_transform(X_train)
# Fit
model = LogisticRegression(max_iter=3000)
model.fit(X_train_scaled, y_train)
# 3. Predict
# Filter for home games only for counting (since 1 row = 1 team, 2 rows = 1 game)
test_games = test_df[test_df["is_home"] == 1].copy()
if test_games.empty: continue
X_test = test_games[features]
X_test_scaled = scaler.transform(X_test)
preds = model.predict(X_test_scaled)
# 4. Score
actuals = test_games["win"].values
correct = (preds == actuals).sum()
total = len(test_games)
accuracy = correct / total
weekly_results.append({
"week": w,
"accuracy": accuracy,
"correct": correct,
"total": total
})
# Add a visual note for good/bad weeks
note = "🔥" if accuracy >= 0.70 else ("⚠️" if accuracy < 0.50 else "")
print(f"Wk {w:<3} | {accuracy:.1%} | {correct}/{total:<13} | {note}")
# --- SUMMARY ---
total_correct = sum(r["correct"] for r in weekly_results)
total_games = sum(r["total"] for r in weekly_results)
total_acc = total_correct / total_games if total_games > 0 else 0
print("-" * 60)
print(f"SEASON TOTAL: {total_acc:.1%} ({total_correct}/{total_games})")
print("=" * 60)
#"""
# --- BLOCK 7: INTERACTIVE PREDICTION TOOL ---
print("\n--- RETRAINING MODEL ON FULL HISTORY ---")
# 1. Train on EVERYTHING (No test split)
X_full = df_final[features]
y_full = df_final["win"]
# Scale & Fit
full_scaler = StandardScaler()
X_full_scaled = full_scaler.fit_transform(X_full)
full_model = LogisticRegression(max_iter=3000)
full_model.fit(X_full_scaled, y_full)
print(f"Model updated. Knowledge cutoff: Season {df_final['season'].max()}, Week {df_final['week'].max()}")
# 2. Prep Latest Stats
# FIX: We need ALL team stats (Anchor + Spark), not just base_features.
# We take the global 'features' list and remove opponent stats and 'is_home'
feature_cols = [c for c in features if not c.startswith("opp_") and c != "is_home"]
latest_stats = df_final.sort_values(["season", "week"]).groupby("team_full").tail(1)
latest_team_stats = latest_stats.set_index("team_full")[feature_cols]
def matchup_prob(home_input, away_input):
home_team = team_abbr_map.get(home_input.lower().strip())
away_team = team_abbr_map.get(away_input.lower().strip())
if not home_team or not away_team:
print(f"Error: Could not find team(s): {home_input}, {away_input}")
return
try:
# We need to copy() to avoid SettingWithCopy warnings on the slices
h_stats = latest_team_stats.loc[home_team].copy()
a_stats = latest_team_stats.loc[away_team].copy()
except KeyError:
print("Error: Stats not found. (Check if teams have played recently)")
return
# Build Matchup Row
# 1. Set Home Team Identity
h_stats["is_home"] = 1
# 2. Rename Away Stats to Opponent Stats
# We must explicitly match the column names expected by the model
# The model expects "opp_passing_yards", "opp_recent_passing_yards", etc.
a_stats_opp = a_stats.rename(lambda x: f"opp_{x}")
# 3. Combine
matchup_row = pd.concat([h_stats, a_stats_opp])
# 4. Reorder to match training data exactly
# (This was the line causing the crash previously)
matchup_row = matchup_row[features].to_frame().T
# Scale & Predict
matchup_scaled = full_scaler.transform(matchup_row)
prob_home = full_model.predict_proba(matchup_scaled)[0][1]
print(f"\nMatchup: {home_team} (Home) vs {away_team} (Away)")
print(f"{home_team} Win Probability: {prob_home:.1%}")
print(f"{away_team} Win Probability: {1 - prob_home:.1%}")
# --- USER INPUT ---
print("\n--- PREDICTION TOOL ---")
h = input("Enter home team: ").strip()
a = input("Enter away team: ").strip()
matchup_prob(h, a)
#"""
# --- BLOCK 8: VISUALIZE THE LEARNING CURVE ---
import matplotlib.pyplot as plt
import matplotlib.ticker as mtick
from matplotlib.ticker import MultipleLocator
print("\nGenerating Accuracy Chart...")
# 1. Extract Data from the Backtest Results
if not weekly_results:
print("No backtest results to plot. (Did you run the season simulation?)")
else:
weeks = [r["week"] for r in weekly_results]
accuracies = [r["accuracy"] for r in weekly_results]
# Calculate Cumulative Accuracy
running_correct = 0
running_total = 0
cumulative_acc = []
for r in weekly_results:
running_correct += r["correct"]
running_total += r["total"]
cumulative_acc.append(running_correct / running_total)
# 2. Setup Plot
plt.figure(figsize=(12, 6))
# Plot 1: Weekly Volatility (The "Spark")
plt.plot(weeks, accuracies, marker='o', linestyle='--', color='gray', alpha=0.5, label='Weekly Accuracy')
# Plot 2: Cumulative Trend (The "Anchor")
plt.plot(weeks, cumulative_acc, marker='o', linewidth=3, color='#007acc', label='Season Total Accuracy')
# 3. Styling
plt.title(f'Model Learning Curve: {target_season} Season', fontsize=16, fontweight='bold')
plt.xlabel('Week', fontsize=12)
plt.ylabel('Accuracy', fontsize=12)
plt.ylim(0, 1.0) # Scale from 0% to 100%
# Add Reference Lines
plt.axhline(y=0.50, color='red', linestyle=':', label='Coin Flip (50%)')
plt.axhline(y=0.60, color='green', linestyle=':', label='Pro Target (60%)')
# Format Y-Axis as Percentages
plt.gca().yaxis.set_major_formatter(mtick.PercentFormatter(1.0))
# Set X-Axis intervals to 2
plt.gca().xaxis.set_major_locator(MultipleLocator(2))
plt.legend(loc='lower right')
plt.grid(True, alpha=0.3)
# 4. Show
print("Chart generated.")
plt.show()
#"""