{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom  sklearn.model_selection  import train_test_split\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom xgboost import XGBRegressor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import cross_val_score\npd.plotting.register_matplotlib_converters()\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n         pass\n\n#data_path_test_hemlet='/kaggle/input/nfl-player-contact-detection/test_baseline_helmets.csv'\ndata_path_train_hemlet='/kaggle/input/nfl-player-contact-detection/train_baseline_helmets.csv'\n#data_path_test_tracking='/kaggle/input/nfl-player-contact-detection/test_player_tracking.csv'\ndata_path_train_tracking='/kaggle/input/nfl-player-contact-detection/train_player_tracking.csv'\n#data_path_test_vedio='/kaggle/input/nfl-player-contact-detection/train_video_metadata.csv'\ndata_path_train_vedio='/kaggle/input/nfl-player-contact-detection/train_video_metadata.csv'\ndata_path_train_label='/kaggle/input/nfl-player-contact-detection/train_labels.csv'\ndata_path_sample='/kaggle/input/nfl-player-contact-detection/sample_submission.csv'\n\n# reading csv file data 1 step\ntraining_data_hemlet=pd.read_csv(data_path_train_hemlet)\n#test_data_hemlet=pd.read_csv(data_path_test_hemlet)\n\n#test_data_tracking=pd.read_csv(data_path_test_tracking)\ntrain_data_tracking=pd.read_csv(data_path_train_tracking)\n\n#test_data_vedio=pd.read_csv(data_path_test_vedio)\ntrain_data_vedio=pd.read_csv(data_path_train_vedio)\n\ntrain_data_label=pd.read_csv(data_path_train_label)\nsimple_data=pd.read_csv(data_path_sample)\n\n# printing summary of each csv file for  clear target selection\n# print(test_data_hemlet.head())\n#print(training_data_hemlet.head())\n\n# print(test_data_tracking.head())\n#print(train_data_tracking.head())\n\n# print(test_data_vedio.head())\n#print(train_data_vedio.head())\n\n#print(train_data_label.head())\nprint('=========================')\nprint(train_data_label.describe())\n#print(simple_data.head())\nprint('=========================')\n# cheacking rows and columns of csv file\nprint('No. of Rows & Columns:\\n',train_data_label.shape)\nX=train_data_label.drop(['game_play','datetime','contact'],axis=1)\ny=train_data_label.contact\n#print(X.head())\n\nX_train_full,X_valid_full,y_train,y_valid=train_test_split(X,y,random_state=0)\n\ncols_with_missing=[col for col in X_train_full.columns\n                   if X_train_full[col].isnull().any()]\n\n# print(cols_with_missing)C_W_M is empty\n# as the ans suggest no messing values columns we don't have to handle missing data\n#checking categorical data columns\n#not adding categorical data columns because this data need feature engeenring\n\ncat_cols=[C_col for C_col in X_train_full.columns\n                if X_train_full[C_col].dtype=='object' and\n                   X_train_full[C_col].nunique()>15]\n\n#print(categorical_cols)\n\n#checking numerical  data columns\nnum_cols=[N_col for N_col in X_train_full.columns\n                if X_train_full[N_col].dtype in ['int64','float64']]\n\n#print(numerical_cols)\n\nmy_cols=num_cols\nX_train = X_train_full[my_cols].copy()\nX_valid = X_valid_full[my_cols].copy()  \n\n# Data visualization works\n#---------------------------------\n\nplt.figure(figsize=(10,10))\nsns.lineplot(data=training_data_hemlet[0:100])\n\n\n# Preprocessing for numerical data\nnum_transf = SimpleImputer(strategy='constant')\n\n# Preprocessing for categorical data\ncat_transf = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('OrdiEnco', OrdinalEncoder())])\n                         \npreprocessor = ColumnTransformer(transformers=[\n        ('num', num_transf, num_cols)])\n\nprint('_________________')\nprint('-----------------')\n#model_1 = RandomForestRegressor(n_estimators=10000,random_state=0,min_samples_split=20, max_depth=10,criterion='absolute_error')\n#model_1.fit(X_train,y_train)\n#predic_1=model_1.predict(X_valid)\n\n#print('RFR model checking:',predic_1.mean())\n\nmy_model = XGBRegressor(n_estimators=200,learning_rate=0.4,n_jobs=20,early_stopping_rounds=10) \nmy_model.fit(X_train,y_train,eval_set=[(X_valid, y_valid)],verbose=False)\npredic_2=my_model.predict(X_valid)\n\nprint('XGB model checking:',predic_2.mean())\n\n#print('------------------------')\n#my_pipeline = Pipeline(steps=[('preprocessor', preprocessor),\n#                              ('model', model_1)])\n\nmy_pipeline2 = Pipeline(steps=[('preprocessor', preprocessor),\n                              ('model', my_model)])\n# Preprocessing of training data, fit model \n#my_pipeline.fit(X_train, y_train)\n#my_pipeline2.fit(X_train, y_train,early_stopping_rounds=5, \n             #eval_set=[(X_valid, y_valid)],\n             #verbose=False)\n\n# Preprocessing of validation data, get predictions\n#preds = my_pipeline.predict(X_valid)\n#print('RFR Pipeline Predition:\\n',preds.mean())\n      \n#preds2 = my_pipeline2.predict(X_valid)\n#print('XGB Pipeline Prediction:\\n',preds2.mean())\n\n# Evaluate the model\nscore = mean_absolute_error(y_valid, predic_2)\nprint('XGB  MAE:\\n',score)\n#score2 = mean_absolute_error(y_valid, preds2)\n#print('XGB Pipeline MAE:\\n',score2)\n\n#scores = -1 * cross_val_score(my_pipeline, X, y,\n                          #    cv=5,\n                           #   scoring='neg_mean_absolute_error')\n\n#print(\"MAE cross_validation scores:\\n\", scores.mean())\n\noutput = pd.DataFrame({'Id':train_data_label.contact_id[0:len(predic_2)],'XGB Predition':predic_2})\noutput.to_csv('submission.csv', index=False)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-27T07:09:50.757042Z","iopub.execute_input":"2023-02-27T07:09:50.757498Z","iopub.status.idle":"2023-02-27T07:14:48.838138Z","shell.execute_reply.started":"2023-02-27T07:09:50.757464Z","shell.execute_reply":"2023-02-27T07:14:48.837073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom  sklearn.model_selection  import train_test_split\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom xgboost import XGBRegressor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.model_selection import cross_val_score\npd.plotting.register_matplotlib_converters()\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\n\ndata_path_train_hemlet='/kaggle/input/nfl-player-contact-detection/train_baseline_helmets.csv'\ntraining_data_hemlet=pd.read_csv(data_path_train_hemlet)\nprint(training_data_hemlet.head())\nplt.figure(figsize=(10,6))\n\nsns.barplot(x=training_data_hemlet[0:1000].player_label, y=training_data_hemlet[0:1000]['nfl_player_id'])\n\nplt.figure(figsize=(8,6))\nsns.lineplot(data=training_data_hemlet[0:1000].width ,label=\"width\")\nsns.lineplot(data=training_data_hemlet[0:1000].height ,label=\"height\")\nsns.lineplot(data=training_data_hemlet[0:1000].top ,label=\"top\")\nsns.lineplot(data=training_data_hemlet[0:1000].left ,label=\"left\")\nplt.xlabel(\"game_play\")","metadata":{"execution":{"iopub.status.busy":"2023-02-27T07:04:30.620449Z","iopub.status.idle":"2023-02-27T07:04:30.621126Z","shell.execute_reply.started":"2023-02-27T07:04:30.620915Z","shell.execute_reply":"2023-02-27T07:04:30.620937Z"},"trusted":true},"execution_count":null,"outputs":[]}]}