{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.cm as cm\nfrom sklearn.decomposition import PCA\nfrom PIL import Image\nimport glob\nimport warnings\nimport shap\nimport lightgbm as lgb\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import KFold\nimport random\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-31T02:19:28.662560Z","iopub.execute_input":"2021-12-31T02:19:28.665645Z","iopub.status.idle":"2021-12-31T02:19:28.679694Z","shell.execute_reply.started":"2021-12-31T02:19:28.665509Z","shell.execute_reply":"2021-12-31T02:19:28.678835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How to make labels of tracking datasets","metadata":{}},{"cell_type":"markdown","source":"I tried to make labels of tracking data because I think these have some features.\n\nI changed the tracking datasets to png datasets.\nAnd I made labels of these png datasets eith PCA and KMeans.\n\nI made the png datasets of tracking data only 2020.\nBecause using all datasets needs too big memories.\n\nIf you think that this is useful ,please upvote.\nYour upvote makes me so encouraged.","metadata":{}},{"cell_type":"code","source":"filepath1='../input/nfl-make-png-2020-1-w/'\nlist_2020_1=os.listdir(filepath1)\n\nfilepath2='../input/nfl-make-png-2020-2-w/'\nlist_2020_2=os.listdir(filepath2)\n\nfilepath3='../input/nfl-make-png-2020-3-w/'\nlist_2020_3=os.listdir(filepath3)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:19:28.681528Z","iopub.execute_input":"2021-12-31T02:19:28.682673Z","iopub.status.idle":"2021-12-31T02:19:28.709829Z","shell.execute_reply.started":"2021-12-31T02:19:28.682632Z","shell.execute_reply":"2021-12-31T02:19:28.709063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_2020_1_=[i for i in list_2020_1 if 'png' in i]\nlist_2020_2_=[i for i in list_2020_2 if 'png' in i]\nlist_2020_3_=[i for i in list_2020_3 if 'png' in i]","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:19:28.711375Z","iopub.execute_input":"2021-12-31T02:19:28.711907Z","iopub.status.idle":"2021-12-31T02:19:28.717782Z","shell.execute_reply.started":"2021-12-31T02:19:28.711848Z","shell.execute_reply":"2021-12-31T02:19:28.717170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load png data\nimage_size = 64\n\nx = []\npng_data=[]\nfor i in list_2020_1_:\n    image = Image.open(filepath1+i)\n    #image = image.convert('L')  \n    image = image.resize((image_size, image_size)) \n    data = np.asarray(image)\n    x.append(data)\n    png_data.append(i)\n    \nfor i in list_2020_2_:\n    image = Image.open(filepath2+i)\n    #image = image.convert('L')  \n    image = image.resize((image_size, image_size)) \n    data = np.asarray(image)\n    x.append(data)\n    png_data.append(i)\n    \nfor i in list_2020_3_:\n    image = Image.open(filepath3+i)\n    #image = image.convert('L')  \n    image = image.resize((image_size, image_size)) \n    data = np.asarray(image)\n    x.append(data)\n    png_data.append(i)\n    \nX = np.array(x)\nX = X.reshape(X.shape[0], image_size * image_size,4) \nX = X / 255.0  \n\n\nprint('X.shape =', X.shape)\nrows, cols = 5, 10  \nfig, aX_invs = plt.subplots(ncols=cols, nrows=rows, figsize=(18, 10)) \nfor i in range(50):\n    r = i // cols\n    c = i % cols\n    aX_invs[r, c].imshow(X[i].reshape(image_size,image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[r, c].set_title('data %d' % (i+1))\n    aX_invs[r, c].get_xaxis().set_visible(False)\n    aX_invs[r, c].get_yaxis().set_visible(False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:19:28.719257Z","iopub.execute_input":"2021-12-31T02:19:28.719661Z","iopub.status.idle":"2021-12-31T02:20:27.476995Z","shell.execute_reply.started":"2021-12-31T02:19:28.719628Z","shell.execute_reply":"2021-12-31T02:20:27.475959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X2=X.reshape(X.shape[0],4096*4)\nX2.shape","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:20:27.479667Z","iopub.execute_input":"2021-12-31T02:20:27.480255Z","iopub.status.idle":"2021-12-31T02:20:27.491203Z","shell.execute_reply.started":"2021-12-31T02:20:27.480205Z","shell.execute_reply":"2021-12-31T02:20:27.490272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"# PCA(Principal component analysis)\nN = 400  #400 component\npca = PCA(n_components=N)\npca.fit(X2)\nprint('n_components = '+str(N))\nprint('explained_variance_ratio = ', pca.explained_variance_ratio_.sum())\n","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:20:27.492477Z","iopub.execute_input":"2021-12-31T02:20:27.492801Z","iopub.status.idle":"2021-12-31T02:20:54.700700Z","shell.execute_reply.started":"2021-12-31T02:20:27.492768Z","shell.execute_reply":"2021-12-31T02:20:54.699881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Dimensionality reduction and dimension restoration coefficients\nX_trans = pca.transform(X2)\nX_inv = pca.inverse_transform(X_trans)\nprint ('X.shape =', X.shape)\nprint ('X_trans.shape =', X_trans.shape)\nprint ('X_inv.shape =', X_inv.shape)\n\n\nrows, cols = 2, 8  # 2行8列\nfig, aX_invs = plt.subplots(ncols=cols, nrows=rows, figsize=(18,4))\nfor i in range(8):\n    aX_invs[0, i].imshow(X[i,:].reshape(image_size, image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[0, i].set_title('original %d' % (i+1))\n    aX_invs[0, i].get_xaxis().set_visible(False)\n    aX_invs[0, i].get_yaxis().set_visible(False)\n    aX_invs[1, i].imshow(X_inv[i,:].reshape(image_size, image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[1, i].set_title('restore %d' % (i+1))\n    aX_invs[1, i].get_xaxis().set_visible(False)\n    aX_invs[1, i].get_yaxis().set_visible(False)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:20:54.702955Z","iopub.execute_input":"2021-12-31T02:20:54.703266Z","iopub.status.idle":"2021-12-31T02:20:58.689043Z","shell.execute_reply.started":"2021-12-31T02:20:54.703222Z","shell.execute_reply":"2021-12-31T02:20:58.688180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KMeans","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport itertools \n\n#Kmeans\nclass KMeans:\n    def __init__(self, n_clusters, max_iter = 1000, random_seed = 0):\n        self.n_clusters = n_clusters\n        self.max_iter = max_iter\n        self.random_state = np.random.RandomState(random_seed)\n\n    def fit(self, X):\n        cycle = itertools.cycle(range(self.n_clusters))\n        self.labels_ = np.fromiter(itertools.islice(cycle, X.shape[0]), dtype = np.int)\n        self.random_state.shuffle(self.labels_)\n        labels_prev = np.zeros(X.shape[0])\n        count = 0\n        self.cluster_centers_ = np.zeros((self.n_clusters, X.shape[1]))\n\n        \n        while (not (self.labels_ == labels_prev).all() and count < self.max_iter):\n            for i in range(self.n_clusters):\n                XX = X[self.labels_ == i, :]\n                self.cluster_centers_[i, :] = XX.mean(axis = 0)\n            dist = ((X[:, :, np.newaxis] - self.cluster_centers_.T[np.newaxis, :, :]) ** 2).sum(axis = 1)\n            labels_prev = self.labels_\n            self.labels_ = dist.argmin(axis = 1)\n            count += 1\n\n    def predict(self, X):\n        dist = ((X[:, :, np.newaxis] - self.cluster_centers_.T[np.newaxis, :, :]) ** 2).sum(axis = 1)\n        labels = dist.argmin(axis = 1)\n        return labels","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:20:58.690327Z","iopub.execute_input":"2021-12-31T02:20:58.690758Z","iopub.status.idle":"2021-12-31T02:20:58.701052Z","shell.execute_reply.started":"2021-12-31T02:20:58.690727Z","shell.execute_reply":"2021-12-31T02:20:58.700297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#make 8 labels\nmodel =  KMeans(8)\nmodel.fit(X_trans)\n\nprint(model.labels_)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:20:58.702178Z","iopub.execute_input":"2021-12-31T02:20:58.702400Z","iopub.status.idle":"2021-12-31T02:21:01.004082Z","shell.execute_reply.started":"2021-12-31T02:20:58.702374Z","shell.execute_reply":"2021-12-31T02:21:01.003272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p_0=X_inv[model.labels_ == 0, :]#label0 no gazou(*4096)\np_1=X_inv[model.labels_ == 1, :]\np_2=X_inv[model.labels_ == 2, :]\np_3=X_inv[model.labels_ == 3, :]\np_4=X_inv[model.labels_ == 4, :]\np_5=X_inv[model.labels_ == 5, :]\np_6=X_inv[model.labels_ == 6, :]\np_7=X_inv[model.labels_ == 7, :]","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:01.005535Z","iopub.execute_input":"2021-12-31T02:21:01.005753Z","iopub.status.idle":"2021-12-31T02:21:01.289245Z","shell.execute_reply.started":"2021-12-31T02:21:01.005725Z","shell.execute_reply":"2021-12-31T02:21:01.288353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(model.labels_)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:01.290596Z","iopub.execute_input":"2021-12-31T02:21:01.290816Z","iopub.status.idle":"2021-12-31T02:21:01.296612Z","shell.execute_reply.started":"2021-12-31T02:21:01.290789Z","shell.execute_reply":"2021-12-31T02:21:01.295741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"confirm how png datasets are labeled.","metadata":{}},{"cell_type":"code","source":"rows, cols = 8, 8\nfig, aX_invs = plt.subplots(ncols=cols, nrows=rows, figsize=(18,25))\nfor i in range(8):\n    aX_invs[0, i].imshow(p_0[i,:].reshape(image_size, image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[0, i].set_title('label0 %d' % (i+1))\n    aX_invs[0, i].get_xaxis().set_visible(False)\n    aX_invs[0, i].get_yaxis().set_visible(False)\n    aX_invs[1, i].imshow(p_1[i,:].reshape(image_size, image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[1, i].set_title('label1 %d' % (i+1))\n    aX_invs[1, i].get_xaxis().set_visible(False)\n    aX_invs[1, i].get_yaxis().set_visible(False)\n    aX_invs[2, i].imshow(p_2[i,:].reshape(image_size, image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[2, i].set_title('label2 %d' % (i+1))\n    aX_invs[2, i].get_xaxis().set_visible(False)\n    aX_invs[2, i].get_yaxis().set_visible(False)\n    aX_invs[3, i].imshow(p_3[i,:].reshape(image_size, image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[3, i].set_title('label3 %d' % (i+1))\n    aX_invs[3, i].get_xaxis().set_visible(False)\n    aX_invs[3, i].get_yaxis().set_visible(False)\n    aX_invs[4, i].imshow(p_4[i,:].reshape(image_size, image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[4, i].set_title('label4 %d' % (i+1))\n    aX_invs[4, i].get_xaxis().set_visible(False)\n    aX_invs[4, i].get_yaxis().set_visible(False)\n    aX_invs[5, i].imshow(p_5[i,:].reshape(image_size, image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[5, i].set_title('label5 %d' % (i+1))\n    aX_invs[5, i].get_xaxis().set_visible(False)\n    aX_invs[5, i].get_yaxis().set_visible(False)\n    aX_invs[6, i].imshow(p_6[i,:].reshape(image_size, image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[6, i].set_title('label6 %d' % (i+1))\n    aX_invs[6, i].get_xaxis().set_visible(False)\n    aX_invs[6, i].get_yaxis().set_visible(False)\n    aX_invs[7, i].imshow(p_7[i,:].reshape(image_size, image_size,4),vmin=0.0,vmax=1.0, cmap = cm.Greys_r)\n    aX_invs[7, i].set_title('label7 %d' % (i+1))\n    aX_invs[7, i].get_xaxis().set_visible(False)\n    aX_invs[7, i].get_yaxis().set_visible(False)\n\n","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:01.298515Z","iopub.execute_input":"2021-12-31T02:21:01.298823Z","iopub.status.idle":"2021-12-31T02:21:04.341287Z","shell.execute_reply.started":"2021-12-31T02:21:01.298779Z","shell.execute_reply":"2021-12-31T02:21:04.340322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's almost working!","metadata":{}},{"cell_type":"markdown","source":"### concat labels that I made and other data\n.","metadata":{}},{"cell_type":"code","source":"label_data=pd.DataFrame(np.arange(len(X)),columns={'number'})\nlabel_data['labels']=model.labels_\nlabel_data['png_data']=png_data\nlabel_data","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:04.342737Z","iopub.execute_input":"2021-12-31T02:21:04.343004Z","iopub.status.idle":"2021-12-31T02:21:04.361666Z","shell.execute_reply.started":"2021-12-31T02:21:04.342972Z","shell.execute_reply":"2021-12-31T02:21:04.360910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df=pd.read_csv('../input/nfl-big-data-bowl-2022/tracking2020.csv')\n#df","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:04.364600Z","iopub.execute_input":"2021-12-31T02:21:04.364833Z","iopub.status.idle":"2021-12-31T02:21:04.369188Z","shell.execute_reply.started":"2021-12-31T02:21:04.364805Z","shell.execute_reply":"2021-12-31T02:21:04.368477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a_a=label_data['png_data'].str.split('.', expand=True)\na_a.rename(columns={0: 'ID', 1: 'png'}, inplace=True)\nlabel_data['ID']=a_a['ID']","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:04.370569Z","iopub.execute_input":"2021-12-31T02:21:04.371101Z","iopub.status.idle":"2021-12-31T02:21:04.394052Z","shell.execute_reply.started":"2021-12-31T02:21:04.371067Z","shell.execute_reply":"2021-12-31T02:21:04.393087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"play=pd.read_csv('../input/nfl-big-data-bowl-2022/plays.csv')\nplay['gameId']=play['gameId'].astype(str)\nplay['playId']=play['playId'].astype(str)\nplay['ID']=play['gameId'].str.cat(play['playId'], sep='_')","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:04.395388Z","iopub.execute_input":"2021-12-31T02:21:04.395627Z","iopub.status.idle":"2021-12-31T02:21:04.517182Z","shell.execute_reply.started":"2021-12-31T02:21:04.395597Z","shell.execute_reply":"2021-12-31T02:21:04.516101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', 50)\ndata_concat=pd.merge(play,label_data, on='ID')\ndata_concat","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:04.518600Z","iopub.execute_input":"2021-12-31T02:21:04.518891Z","iopub.status.idle":"2021-12-31T02:21:04.573551Z","shell.execute_reply.started":"2021-12-31T02:21:04.518830Z","shell.execute_reply":"2021-12-31T02:21:04.572720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## playResult vs labels of tracking data separate with specialTeamsPlayType ****","metadata":{}},{"cell_type":"code","source":"data_concat['specialTeamsPlayType'].unique()","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:04.574941Z","iopub.execute_input":"2021-12-31T02:21:04.575180Z","iopub.status.idle":"2021-12-31T02:21:04.581132Z","shell.execute_reply.started":"2021-12-31T02:21:04.575149Z","shell.execute_reply":"2021-12-31T02:21:04.580138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('fivethirtyeight')\nplt.figure(figsize=(16,10))\nplt.grid(False)\ndata_ko=data_concat[data_concat['specialTeamsPlayType']=='Kickoff']\ndata_ep=data_concat[data_concat['specialTeamsPlayType']=='Extra Point']\ndata_punt=data_concat[data_concat['specialTeamsPlayType']=='Punt']\ndata_fg=data_concat[data_concat['specialTeamsPlayType']=='Field Goal']\nplt.scatter(x=data_ko['labels'],y=data_ko['playResult'],c='r',label='Kickoff',alpha=0.3)\nplt.scatter(x=data_ep['labels'],y=data_ep['playResult'],c='g',label='Extra Point',alpha=0.3)\nplt.scatter(x=data_punt['labels'],y=data_punt['playResult'],c='y',label='Punt',alpha=0.3)\nplt.scatter(x=data_fg['labels'],y=data_fg['playResult'],c='b',label='Field Goal',alpha=0.3)\nplt.title(\"playResult vs labels of tracking data separate with specialTeamsPlayType\")\nplt.xlabel(\"labels of tracking data\")\nplt.ylabel(\"playResult\")\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:04.582302Z","iopub.execute_input":"2021-12-31T02:21:04.583051Z","iopub.status.idle":"2021-12-31T02:21:05.241141Z","shell.execute_reply.started":"2021-12-31T02:21:04.583015Z","shell.execute_reply":"2021-12-31T02:21:05.240209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_concat","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:05.242631Z","iopub.execute_input":"2021-12-31T02:21:05.242965Z","iopub.status.idle":"2021-12-31T02:21:05.285765Z","shell.execute_reply.started":"2021-12-31T02:21:05.242922Z","shell.execute_reply":"2021-12-31T02:21:05.284826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ****LGBM and SHAP with labels of tracking datasets****","metadata":{}},{"cell_type":"code","source":"#Code by rossinEndrew https://www.kaggle.com/endrewrossin/fast-initial-lightgbm-model-to-detect-exam-result/comments\n\nSEED = 99\nrandom.seed(SEED)\nnp.random.seed(SEED)\n#Code by rossinEndrew https://www.kaggle.com/endrewrossin/fast-initial-lightgbm-model-to-detect-exam-result/comments\ndata_concat=data_concat.drop(['playDescription','gameClock','playId','png_data','number','ID','gameId'], axis = 1)\nplaysmodel = data_concat.copy()\n\n# read the \"object\" columns and use labelEncoder to transform to numeric\nfor col in playsmodel.columns[playsmodel.dtypes == 'object']:\n    le = LabelEncoder()\n    playsmodel[col] = playsmodel[col].astype(str)\n    le.fit(playsmodel[col])\n    playsmodel[col] = le.transform(playsmodel[col])","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:05.287356Z","iopub.execute_input":"2021-12-31T02:21:05.288041Z","iopub.status.idle":"2021-12-31T02:21:05.323213Z","shell.execute_reply.started":"2021-12-31T02:21:05.287988Z","shell.execute_reply":"2021-12-31T02:21:05.322474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"playsmodel","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:05.324230Z","iopub.execute_input":"2021-12-31T02:21:05.324772Z","iopub.status.idle":"2021-12-31T02:21:05.351293Z","shell.execute_reply.started":"2021-12-31T02:21:05.324739Z","shell.execute_reply":"2021-12-31T02:21:05.350410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = playsmodel.drop(['playResult'], axis = 1)\ny = playsmodel['playResult']\nlgb_params = {\n                    'objective':'binary',\n                    'metric':'auc',\n                    'n_jobs':-1,\n                    'learning_rate':0.005,\n                    'num_leaves': 20,\n                    'max_depth':-1,\n                    'subsample':0.9,\n                    'n_estimators':2500,\n                    'seed': SEED,\n                    'early_stopping_rounds':100, \n                }","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:05.352529Z","iopub.execute_input":"2021-12-31T02:21:05.352790Z","iopub.status.idle":"2021-12-31T02:21:05.361346Z","shell.execute_reply.started":"2021-12-31T02:21:05.352758Z","shell.execute_reply":"2021-12-31T02:21:05.360407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Code by rossinEndrew https://www.kaggle.com/endrewrossin/fast-initial-lightgbm-model-to-detect-exam-result/comments\n\n# choose the number of folds, and create a variable to store the auc values and the iteration values.\nK = 5\nfolds = KFold(K, shuffle = True, random_state = SEED)\nbest_scorecv= 0\nbest_iteration=0\n\n# Separate data in folds, create train and validation dataframes, train the model and cauculate the mean AUC.\nfor fold , (train_index,test_index) in enumerate(folds.split(X, y)):\n    print('Fold:',fold+1)\n          \n    X_traincv, X_testcv = X.iloc[train_index], X.iloc[test_index]\n    y_traincv, y_testcv = y.iloc[train_index], y.iloc[test_index]\n    \n    train_data = lgb.Dataset(X_traincv, y_traincv)\n    val_data   = lgb.Dataset(X_testcv, y_testcv)\n    \n    LGBM = lgb.train(lgb_params, train_data, valid_sets=[train_data,val_data], verbose_eval=250)\n    best_scorecv += LGBM.best_score['valid_1']['auc']\n    best_iteration += LGBM.best_iteration\n\nbest_scorecv /= K\nbest_iteration /= K\nprint('\\n Mean AUC score:', best_scorecv)\nprint('\\n Mean best iteration:', best_iteration)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:05.362811Z","iopub.execute_input":"2021-12-31T02:21:05.363132Z","iopub.status.idle":"2021-12-31T02:21:06.753532Z","shell.execute_reply.started":"2021-12-31T02:21:05.363090Z","shell.execute_reply":"2021-12-31T02:21:06.752579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#XLGB SHAP\n#Code by rossinEndrew https://www.kaggle.com/endrewrossin/fast-initial-lightgbm-model-to-detect-exam-result/comments\n\nlgb_params = {\n                    'objective':'binary',\n                    'metric':'auc',\n                    'n_jobs':-1,\n                    'learning_rate':0.05,\n                    'num_leaves': 20,\n                    'max_depth':-1,\n                    'subsample':0.9,\n                    'n_estimators':round(best_iteration),\n                    'seed': SEED,\n                    'early_stopping_rounds':None, \n                }\n\ntrain_data_final = lgb.Dataset(X, y)\nLGBM = lgb.train(lgb_params, train_data)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:06.755007Z","iopub.execute_input":"2021-12-31T02:21:06.756038Z","iopub.status.idle":"2021-12-31T02:21:06.867562Z","shell.execute_reply.started":"2021-12-31T02:21:06.755988Z","shell.execute_reply":"2021-12-31T02:21:06.866770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# telling wich model to use\nexplainer = shap.TreeExplainer(LGBM)\n# Calculating the Shap values of X features\nshap_values = explainer.shap_values(X)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:21:06.868936Z","iopub.execute_input":"2021-12-31T02:21:06.869845Z","iopub.status.idle":"2021-12-31T02:21:07.902435Z","shell.execute_reply.started":"2021-12-31T02:21:06.869800Z","shell.execute_reply":"2021-12-31T02:21:07.901660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.summary_plot(shap_values[0], X, plot_type=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:22:00.134870Z","iopub.execute_input":"2021-12-31T02:22:00.135244Z","iopub.status.idle":"2021-12-31T02:22:00.374971Z","shell.execute_reply.started":"2021-12-31T02:22:00.135194Z","shell.execute_reply":"2021-12-31T02:22:00.373993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.summary_plot(shap_values[0], X)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T02:22:05.335980Z","iopub.execute_input":"2021-12-31T02:22:05.336293Z","iopub.status.idle":"2021-12-31T02:22:07.313217Z","shell.execute_reply.started":"2021-12-31T02:22:05.336262Z","shell.execute_reply":"2021-12-31T02:22:07.312241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusion","metadata":{}},{"cell_type":"markdown","source":"'playResult' have relations with the 'specialTeamPlayType' the most.\nLabels that I made in this notebook had a little impact.\nBut they are similar to SpecialTeamPlayType.\n\nIf you apply this method,we can label the movements of one player and a ball.\nIt may lead to better results...\n\nThank you for watching this notebook till the end.\n\n","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}