{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":17484,"databundleVersionId":828454}],"dockerImageVersionId":31328,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:35:41.747233Z","iopub.execute_input":"2026-04-24T14:35:41.747637Z","iopub.status.idle":"2026-04-24T14:35:41.755567Z","shell.execute_reply.started":"2026-04-24T14:35:41.747605Z","shell.execute_reply":"2026-04-24T14:35:41.754723Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## IMPORTS\n#### LogisticRegression : Classification model used for making probabilitic predictions \n#### train_test_split : Splitting the dataset into 2 parts testing and training set, so that we have some part of the original dataset to test things on\n#### StandardScaler : Used to normalize attributes to a suitable scale so as to avoid model biasing\n\n","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:35:44.726696Z","iopub.execute_input":"2026-04-24T14:35:44.727079Z","iopub.status.idle":"2026-04-24T14:35:44.731789Z","shell.execute_reply.started":"2026-04-24T14:35:44.727047Z","shell.execute_reply":"2026-04-24T14:35:44.730732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pathRecord = \"/kaggle/input/competitions/nfl-playing-surface-analytics/InjuryRecord.csv\"\ndfRecord = pd.read_csv(pathRecord)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:35:47.377744Z","iopub.execute_input":"2026-04-24T14:35:47.378553Z","iopub.status.idle":"2026-04-24T14:35:47.408793Z","shell.execute_reply.started":"2026-04-24T14:35:47.378516Z","shell.execute_reply":"2026-04-24T14:35:47.407692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dfRecord.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:35:48.204854Z","iopub.execute_input":"2026-04-24T14:35:48.205253Z","iopub.status.idle":"2026-04-24T14:35:48.242848Z","shell.execute_reply.started":"2026-04-24T14:35:48.205224Z","shell.execute_reply":"2026-04-24T14:35:48.241749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"playListPath = \"/kaggle/input/competitions/nfl-playing-surface-analytics/PlayList.csv\"\ndfPlayList = pd.read_csv(playListPath)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:35:49.221700Z","iopub.execute_input":"2026-04-24T14:35:49.222035Z","iopub.status.idle":"2026-04-24T14:35:49.931702Z","shell.execute_reply.started":"2026-04-24T14:35:49.222006Z","shell.execute_reply":"2026-04-24T14:35:49.930695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dfPlayList.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:35:50.865271Z","iopub.execute_input":"2026-04-24T14:35:50.866065Z","iopub.status.idle":"2026-04-24T14:35:50.879698Z","shell.execute_reply.started":"2026-04-24T14:35:50.866028Z","shell.execute_reply":"2026-04-24T14:35:50.878770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"playTrackPath = \"/kaggle/input/competitions/nfl-playing-surface-analytics/PlayerTrackData.csv\"\ndfPlayTrack = pd.read_csv(playTrackPath)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:35:51.369109Z","iopub.execute_input":"2026-04-24T14:35:51.369406Z","iopub.status.idle":"2026-04-24T14:37:28.267470Z","shell.execute_reply.started":"2026-04-24T14:35:51.369381Z","shell.execute_reply":"2026-04-24T14:37:28.265682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dfPlayTrack.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:37:28.270080Z","iopub.execute_input":"2026-04-24T14:37:28.270551Z","iopub.status.idle":"2026-04-24T14:37:28.290622Z","shell.execute_reply.started":"2026-04-24T14:37:28.270515Z","shell.execute_reply":"2026-04-24T14:37:28.289771Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Engineeering\nCalculated acceleration to try and see the correlation between sudden burst of speed and injury\nGrouped and aggregated the dataset with maximum speed, maximum and minimum acceleration and the total distance travelled by the players.. since these are the most key features often associated with fatigue","metadata":{}},{"cell_type":"code","source":"dfPlayTrack.sort_values(['PlayKey', 'time'], inplace = True)\ndfPlayTrack['accel'] = dfPlayTrack['s'].diff()/0.1\nmask = dfPlayTrack['PlayKey'] != dfPlayTrack['PlayKey'].shift(1)\ndfPlayTrack.loc[mask, 'accel'] = 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:37:28.292068Z","iopub.execute_input":"2026-04-24T14:37:28.292494Z","iopub.status.idle":"2026-04-24T14:38:01.689797Z","shell.execute_reply.started":"2026-04-24T14:37:28.292417Z","shell.execute_reply":"2026-04-24T14:38:01.688720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"playSummary = dfPlayTrack.groupby('PlayKey').agg({\n    's' : 'max',\n    'accel' : ['max', 'min'],\n    'dis' : 'sum'\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:01.692231Z","iopub.execute_input":"2026-04-24T14:38:01.692638Z","iopub.status.idle":"2026-04-24T14:38:09.837806Z","shell.execute_reply.started":"2026-04-24T14:38:01.692609Z","shell.execute_reply":"2026-04-24T14:38:09.836819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"playSummary.sample(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:09.839171Z","iopub.execute_input":"2026-04-24T14:38:09.839523Z","iopub.status.idle":"2026-04-24T14:38:09.870899Z","shell.execute_reply.started":"2026-04-24T14:38:09.839484Z","shell.execute_reply":"2026-04-24T14:38:09.869968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"playSummary.columns = ['_'.join(col).strip() for col in playSummary.columns.values]\nplaySummary.reset_index(inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:09.872193Z","iopub.execute_input":"2026-04-24T14:38:09.872506Z","iopub.status.idle":"2026-04-24T14:38:09.884284Z","shell.execute_reply.started":"2026-04-24T14:38:09.872479Z","shell.execute_reply":"2026-04-24T14:38:09.883269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"playSummary.sample(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:09.885377Z","iopub.execute_input":"2026-04-24T14:38:09.885686Z","iopub.status.idle":"2026-04-24T14:38:09.917647Z","shell.execute_reply.started":"2026-04-24T14:38:09.885659Z","shell.execute_reply":"2026-04-24T14:38:09.916666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"injuryKeys = dfRecord['PlayKey'].dropna().unique()\n\nplaySummary['is_injured'] = playSummary['PlayKey'].isin(injuryKeys).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:09.918833Z","iopub.execute_input":"2026-04-24T14:38:09.919198Z","iopub.status.idle":"2026-04-24T14:38:09.946062Z","shell.execute_reply.started":"2026-04-24T14:38:09.919158Z","shell.execute_reply":"2026-04-24T14:38:09.945145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"playSummary.groupby('is_injured')[['s_max', 'accel_max', 'accel_min', 'dis_sum']].mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:09.947318Z","iopub.execute_input":"2026-04-24T14:38:09.947725Z","iopub.status.idle":"2026-04-24T14:38:09.990119Z","shell.execute_reply.started":"2026-04-24T14:38:09.947694Z","shell.execute_reply":"2026-04-24T14:38:09.989275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"playSummary['PlayerKey'] = playSummary['PlayKey'].apply(lambda x: int(x.split('-')[0]))\nplaySummary = pd.merge(playSummary, dfPlayList[['PlayKey', 'PlayerDay']], on='PlayKey', how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:21.122789Z","iopub.execute_input":"2026-04-24T14:38:21.123125Z","iopub.status.idle":"2026-04-24T14:38:21.573995Z","shell.execute_reply.started":"2026-04-24T14:38:21.123095Z","shell.execute_reply":"2026-04-24T14:38:21.572986Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ACWR(Acute Chronic Work Rate) : \nThe Acute:Chronic Workload Ratio (ACWR) is a metric used in sports science to monitor training load and predict injury risk by comparing an athlete's short-term workload (acute) to their longer-term average (chronic).","metadata":{}},{"cell_type":"code","source":"daily_load = playSummary.groupby(['PlayerKey', 'PlayerDay'])['dis_sum'].sum().reset_index()\n\n\ndaily_load['acute'] = daily_load.groupby('PlayerKey')['dis_sum'].transform(lambda x: x.rolling(window=7, min_periods=1).mean())\ndaily_load['chronic'] = daily_load.groupby('PlayerKey')['dis_sum'].transform(lambda x: x.rolling(window=28, min_periods=1).mean())\n\ndaily_load['acwr'] = daily_load['acute'] / daily_load['chronic']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:26.789607Z","iopub.execute_input":"2026-04-24T14:38:26.790275Z","iopub.status.idle":"2026-04-24T14:38:26.907195Z","shell.execute_reply.started":"2026-04-24T14:38:26.790228Z","shell.execute_reply":"2026-04-24T14:38:26.906211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"daily_load['acwr'].max()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:27.846582Z","iopub.execute_input":"2026-04-24T14:38:27.847169Z","iopub.status.idle":"2026-04-24T14:38:27.853152Z","shell.execute_reply.started":"2026-04-24T14:38:27.847134Z","shell.execute_reply":"2026-04-24T14:38:27.852325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"daily_load['acwr'].min()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:29.221616Z","iopub.execute_input":"2026-04-24T14:38:29.222054Z","iopub.status.idle":"2026-04-24T14:38:29.228569Z","shell.execute_reply.started":"2026-04-24T14:38:29.222020Z","shell.execute_reply":"2026-04-24T14:38:29.227788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"injuredPlayerKeys = dfRecord['PlayerKey'].dropna().unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:30.148389Z","iopub.execute_input":"2026-04-24T14:38:30.149132Z","iopub.status.idle":"2026-04-24T14:38:30.153726Z","shell.execute_reply.started":"2026-04-24T14:38:30.149098Z","shell.execute_reply":"2026-04-24T14:38:30.152898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"daily_load['is_injured'] = daily_load['PlayerKey'].isin(injuredPlayerKeys).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:32.835066Z","iopub.execute_input":"2026-04-24T14:38:32.835381Z","iopub.status.idle":"2026-04-24T14:38:32.841161Z","shell.execute_reply.started":"2026-04-24T14:38:32.835354Z","shell.execute_reply":"2026-04-24T14:38:32.839996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"daily_load.groupby('is_injured')[['acwr']].max()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:33.838452Z","iopub.execute_input":"2026-04-24T14:38:33.838769Z","iopub.status.idle":"2026-04-24T14:38:33.850158Z","shell.execute_reply.started":"2026-04-24T14:38:33.838741Z","shell.execute_reply":"2026-04-24T14:38:33.849312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"injury_event_days = pd.merge(dfRecord, dfPlayList[['PlayKey', 'PlayerDay']], on='PlayKey', how='inner')\n\ninjury_day_acwr = pd.merge(injury_event_days[['PlayerKey', 'PlayerDay']], daily_load, on=['PlayerKey', 'PlayerDay'], how='inner')\n\nprint(f\"Mean ACWR on Injury Day: {injury_day_acwr['acwr'].mean()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:35.629298Z","iopub.execute_input":"2026-04-24T14:38:35.630233Z","iopub.status.idle":"2026-04-24T14:38:35.736204Z","shell.execute_reply.started":"2026-04-24T14:38:35.630197Z","shell.execute_reply":"2026-04-24T14:38:35.735361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df = pd.merge(\n    playSummary, \n    daily_load[['PlayerKey', 'PlayerDay', 'acwr']], \n    on=['PlayerKey', 'PlayerDay'], \n    how='left'\n)\n\n# 2. Join Surface data from InjuryRecord\nfinal_df = pd.merge(\n    final_df, \n    dfPlayList[['PlayKey', 'FieldType']], \n    on='PlayKey', \n    how='left'\n)\n\n# 3. Handle categorical data and NaNs\nfinal_df['is_synthetic'] = (final_df['FieldType'] == 'Synthetic').astype(int)\nfinal_df['acwr'] = final_df['acwr'].fillna(1.0)\n\n# 4. Define features and target\nfeatures = [\n    's_max', \n    'accel_max', \n    'accel_min', \n    'dis_sum', \n    'acwr', \n    'is_synthetic'\n]\nX = final_df[features]\ny = final_df['is_injured']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:37.341584Z","iopub.execute_input":"2026-04-24T14:38:37.341946Z","iopub.status.idle":"2026-04-24T14:38:37.654028Z","shell.execute_reply.started":"2026-04-24T14:38:37.341901Z","shell.execute_reply":"2026-04-24T14:38:37.653221Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Scaling the features to a suitable scale to avoid bias","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\nX_scaled_df = pd.DataFrame(X_scaled, columns=features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:41.596268Z","iopub.execute_input":"2026-04-24T14:38:41.596791Z","iopub.status.idle":"2026-04-24T14:38:41.632767Z","shell.execute_reply.started":"2026-04-24T14:38:41.596756Z","shell.execute_reply":"2026-04-24T14:38:41.631886Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Modelling:\nCreated a logistic regression model and trained it on the features extracted above to make predictions","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X_scaled_df, y, test_size=0.2, random_state=42)\n\nmodel = LogisticRegression(class_weight='balanced')\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:45.514566Z","iopub.execute_input":"2026-04-24T14:38:45.515354Z","iopub.status.idle":"2026-04-24T14:38:45.920655Z","shell.execute_reply.started":"2026-04-24T14:38:45.515300Z","shell.execute_reply":"2026-04-24T14:38:45.919723Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Mining:\nLooked at the patterns affecting injuries which werent obvious when we looked at the dataset. For eg : High ACWE is often associated with injuries but here the correlation was very less for the particular dataset","metadata":{}},{"cell_type":"code","source":"coef_df = pd.DataFrame({\n    'Feature': features,\n    'Coefficient': model.coef_[0]\n})\n\n# Sort by importance (absolute value)\ncoef_df['Abs_Coef'] = coef_df['Coefficient'].abs()\ncoef_df = coef_df.sort_values(by='Abs_Coef', ascending=False)\n\nprint(coef_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:49.110423Z","iopub.execute_input":"2026-04-24T14:38:49.111156Z","iopub.status.idle":"2026-04-24T14:38:49.120211Z","shell.execute_reply.started":"2026-04-24T14:38:49.111119Z","shell.execute_reply":"2026-04-24T14:38:49.119195Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Evaluation:\nBuild confusion matrix for the above shown classification model and found out the key metrics.. Note that the model for the given use case should have high recall since missing out an injury is more catastrophic than being precautionary","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix, ConfusionMatrixDisplay\nimport matplotlib.pyplot as plt\n\n# Get predictions\ny_pred = model.predict(X_test)\n\n# Print the report\nprint(classification_report(y_test, y_pred))\n\n# Plot the Confusion Matrix\ncm = confusion_matrix(y_test, y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:38:56.970699Z","iopub.execute_input":"2026-04-24T14:38:56.971027Z","iopub.status.idle":"2026-04-24T14:38:57.353713Z","shell.execute_reply.started":"2026-04-24T14:38:56.970999Z","shell.execute_reply":"2026-04-24T14:38:57.352902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get probabilities for the test set\nprobs = model.predict_proba(X_test)[:, 1]\n\n# Create a DataFrame to visualize the \"Risk Zone\"\nrisk_df = pd.DataFrame({'Risk_Prob': probs, 'Actual_Injury': y_test})\n\n# See how many players are in the \"High Risk\" ( > 80%) category\nprint(risk_df[risk_df['Risk_Prob'] > 0.8].shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:39:02.794456Z","iopub.execute_input":"2026-04-24T14:39:02.795172Z","iopub.status.idle":"2026-04-24T14:39:02.806181Z","shell.execute_reply.started":"2026-04-24T14:39:02.795134Z","shell.execute_reply":"2026-04-24T14:39:02.805179Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Downloading scaler and the Logistic Regression model for Reference ","metadata":{}},{"cell_type":"code","source":"import joblib\n\n# Exporting the trained model\njoblib.dump(model, 'injury_model.pkl')\n\n# Exporting the scaler (Crucial for consistent inference)\njoblib.dump(scaler, 'scaler.pkl')\n\nprint(\"Files saved to /kaggle/working/\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:39:07.688169Z","iopub.execute_input":"2026-04-24T14:39:07.688497Z","iopub.status.idle":"2026-04-24T14:39:07.697042Z","shell.execute_reply.started":"2026-04-24T14:39:07.688464Z","shell.execute_reply":"2026-04-24T14:39:07.696066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df.to_csv('processed_football_data.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:39:08.591763Z","iopub.execute_input":"2026-04-24T14:39:08.592660Z","iopub.status.idle":"2026-04-24T14:39:10.751530Z","shell.execute_reply.started":"2026-04-24T14:39:08.592622Z","shell.execute_reply":"2026-04-24T14:39:10.750670Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Reciever Operating Characteristic Curve (ROC Curve) and the Area Under the Curve(AUC):\nWhile a standard classification report only shows performance at one threshold, the ROC Curve evaluates our model across all possible risk levels. Our current AUC of 0.6040 outperforms a random guess by roughly 10%. Practically, this means our model has a 60% chance of correctly identifying a high-risk player over a low-risk one—providing a vital 'early warning' that could prevent injuries and save a club's season","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, roc_auc_score, precision_recall_curve\nimport matplotlib.pyplot as plt\n\n# 1. Get the probabilities (not just 0 or 1)\ny_probs = model.predict_proba(X_test)[:, 1]\n\n# 2. Calculate AUC\nauc_score = roc_auc_score(y_test, y_probs)\n\n# 3. Plot the ROC Curve\nfpr, tpr, thresholds = roc_curve(y_test, y_probs)\n\nplt.figure(figsize=(8, 6))\nplt.plot(fpr, tpr, label=f'Logistic Regression (AUC = {auc_score:.2f})')\nplt.plot([0, 1], [0, 1], 'k--', label='Random Guess (AUC = 0.50)')\nplt.xlabel('False Positive Rate (Risk of False Alarms)')\nplt.ylabel('True Positive Rate (Recall / Catching Injuries)')\nplt.title('ROC Curve: Injury Prediction Performance')\nplt.legend()\nplt.show()\n\nprint(f\"Your Model's AUC Score: {auc_score:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:39:16.155624Z","iopub.execute_input":"2026-04-24T14:39:16.156010Z","iopub.status.idle":"2026-04-24T14:39:16.356088Z","shell.execute_reply.started":"2026-04-24T14:39:16.155980Z","shell.execute_reply":"2026-04-24T14:39:16.354988Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Precision Recall Curve:\nGiven the extreme class imbalance in NFL injury data, we prioritized Recall over Precision. In a high-stakes environment like the NFL, a False Negative (missing an injury) is significantly more costly than a False Positive (extra rest for a healthy player). While the low Precision reflects a high rate of 'cautionary flags,' the model serves its primary purpose: ensuring that high-risk movement patterns are not overlooked","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import precision_recall_curve, average_precision_score\nimport matplotlib.pyplot as plt\n\n# 1. Get probabilities\ny_probs = model.predict_proba(X_test)[:, 1]\n\n# 2. Calculate Precision, Recall, and Thresholds\nprecision, recall, thresholds = precision_recall_curve(y_test, y_probs)\n\n# 3. Calculate Average Precision (The 'AUC' of this curve)\nap_score = average_precision_score(y_test, y_probs)\n\n# 4. Plot the Curve\nplt.figure(figsize=(8, 6))\nplt.plot(recall, precision, label=f'PR Curve (AP = {ap_score:.4f})', color='purple')\nplt.axhline(y=len(y_test[y_test==1])/len(y_test), color='r', linestyle='--', label='Baseline (No Skill)')\nplt.xlabel('Recall (Percentage of Injuries Caught)')\nplt.ylabel('Precision (Accuracy of Injury Alarms)')\nplt.title('Precision-Recall Curve: Injury Detection')\nplt.legend()\nplt.grid(True, alpha=0.3)\nplt.show()\n\nprint(f\"Average Precision (AP) Score: {ap_score:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:41:45.362569Z","iopub.execute_input":"2026-04-24T14:41:45.363106Z","iopub.status.idle":"2026-04-24T14:41:45.579966Z","shell.execute_reply.started":"2026-04-24T14:41:45.363049Z","shell.execute_reply":"2026-04-24T14:41:45.579024Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Making the Best Fit Playing XI For a Manager:\nUsed linear programming and traditional 0/1 Knapsack Dynamic Programming Approach to build a Lineup that has the highest rating and has an injury risk score less than the threshold","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom scipy.optimize import linprog\n\n# 1. Use .iloc for positional indexing on DataFrames\n# and .values to ensure we're working with a clean NumPy array\ntest_probs = model.predict_proba(X_test)[:, 1]\nperformance_scores = X_test.iloc[:, 0].values  # s_max is our first column\n\n# 2. Define Optimization Parameters\nn_players = len(performance_scores)\nlineup_size = 11\nrisk_budget = 1.5 \n\n# 3. Setup ILP\nc = -performance_scores \n# A_ub handles the 'Less than' constraints\nA = [test_probs, np.ones(n_players)] \nb = [risk_budget, lineup_size] \n\nres = linprog(c, A_ub=A, b_ub=b, bounds=(0, 1), method='highs')\n\n# 4. Extract Results\nselected_indices = np.where(res.x > 0.5)[0]\nfinal_lineup_risk = test_probs[selected_indices].sum()\nfinal_lineup_perf = performance_scores[selected_indices].sum()\n\nprint(f\"Lineup successfully optimized.\")\nprint(f\"Total Team Risk: {final_lineup_risk:.2f}\")\nprint(f\"Total Performance: {final_lineup_perf:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:46:33.526144Z","iopub.execute_input":"2026-04-24T14:46:33.527283Z","iopub.status.idle":"2026-04-24T14:46:34.416308Z","shell.execute_reply.started":"2026-04-24T14:46:33.527243Z","shell.execute_reply":"2026-04-24T14:46:34.415383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Extract the metadata for the test set\n# We need the original index to find the PlayerKeys\ntest_metadata = final_df.iloc[y_test.index].copy()\ntest_metadata['Risk_Probability'] = test_probs\n\n# 2. Filter for only the players chosen by the ILP solver\nrecommended_lineup = test_metadata.iloc[selected_indices]\n\n# 3. Clean up the display\nlineup_display = recommended_lineup[[\n    'PlayerKey', 's_max', 'is_synthetic', 'acwr', 'Risk_Probability'\n]].sort_values(by='Risk_Probability')\n\nprint(\"--- OPTIMIZED STARTING LINEUP ---\")\nprint(lineup_display)\n\n# 4. Identify the \"Benched\" High-Risk Stars (for context)\nbenched = test_metadata.iloc[np.setdiff1d(np.arange(n_players), selected_indices)]\nhigh_risk_benched = benched[benched['Risk_Probability'] > 0.2].sort_values(by='s_max', ascending=False)\n\nif not high_risk_benched.empty:\n    print(\"\\n--- SAFETY ALERTS (Benched Players) ---\")\n    print(high_risk_benched[['PlayerKey', 's_max', 'Risk_Probability', 'is_synthetic']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-24T14:46:39.955352Z","iopub.execute_input":"2026-04-24T14:46:39.955864Z","iopub.status.idle":"2026-04-24T14:46:40.017166Z","shell.execute_reply.started":"2026-04-24T14:46:39.955830Z","shell.execute_reply":"2026-04-24T14:46:40.016278Z"}},"outputs":[],"execution_count":null}]}