{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**PUBG Finish Placement Model Creation**","metadata":{"id":"DLttDlMDwPoV"}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\ntrainpath='../input/pubg-finish-placement-prediction/train_V2.csv'\ntestpath='../input/pubg-finish-placement-prediction/test_V2.csv'\nsamplepath='../input/pubg-finish-placement-prediction/sample_submission_V2.csv'\n\ntrain = pd.read_csv(trainpath)\ntest = pd.read_csv(testpath)\nsample= pd.read_csv(samplepath)","metadata":{"id":"xlzvdk7FvRYG","execution":{"iopub.status.busy":"2022-08-13T01:29:03.826374Z","iopub.execute_input":"2022-08-13T01:29:03.827069Z","iopub.status.idle":"2022-08-13T01:29:33.845891Z","shell.execute_reply.started":"2022-08-13T01:29:03.827019Z","shell.execute_reply":"2022-08-13T01:29:33.844463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"id":"ruiDTggSxUA9","outputId":"a34dca86-6631-400f-c150-85a28a0fd611","execution":{"iopub.status.busy":"2022-08-13T01:29:33.847990Z","iopub.execute_input":"2022-08-13T01:29:33.848394Z","iopub.status.idle":"2022-08-13T01:29:33.874882Z","shell.execute_reply.started":"2022-08-13T01:29:33.848358Z","shell.execute_reply":"2022-08-13T01:29:33.873628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head(10)","metadata":{"id":"NGEi6QOQ07iL","outputId":"fc016df4-e009-4f62-f25b-dc9ee968cd22","execution":{"iopub.status.busy":"2022-08-13T01:29:33.876832Z","iopub.execute_input":"2022-08-13T01:29:33.877604Z","iopub.status.idle":"2022-08-13T01:29:33.908688Z","shell.execute_reply.started":"2022-08-13T01:29:33.877565Z","shell.execute_reply":"2022-08-13T01:29:33.907790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Okay, so looking at the head of the data. I can already spot a few rows to get rid of since they are not useful for our purposes here. I think ID, GroupID, and Match ID are irrelevant to our predictions as they are essentially just labels. Additionally, there is the Match_Type, which at a base level may be meaningless. But we could actually cut up this data to make multiple models for the multiple game-modes. For the sake of a first look implementation and competition requirements. Let's not for now. \n\nAdditionally there's quite a few anomalous values for rank points, with -1 values amongst 1500's or greater. This just seems not great so we'll have to do something about it.","metadata":{"id":"DuxSoo2D1HkT"}},{"cell_type":"code","source":"train=train.fillna(method='ffill')\ntrain.drop(['Id','groupId','matchId'],inplace=True,axis=1)\ntest.drop(['Id','groupId','matchId'],inplace=True,axis=1)","metadata":{"id":"XoVhqqNv0-l8","execution":{"iopub.status.busy":"2022-08-13T01:29:33.912121Z","iopub.execute_input":"2022-08-13T01:29:33.912767Z","iopub.status.idle":"2022-08-13T01:29:36.187339Z","shell.execute_reply.started":"2022-08-13T01:29:33.912730Z","shell.execute_reply":"2022-08-13T01:29:36.186016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sn\ncorrMatrix=train.corr()\nsn.heatmap(corrMatrix, annot=True)\nplt.savefig('heatmap.png')\nplt.show()\ncorrMatrix","metadata":{"execution":{"iopub.status.busy":"2022-08-13T01:29:36.188931Z","iopub.execute_input":"2022-08-13T01:29:36.189568Z","iopub.status.idle":"2022-08-13T01:29:48.505433Z","shell.execute_reply.started":"2022-08-13T01:29:36.189519Z","shell.execute_reply":"2022-08-13T01:29:48.504536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see from the correlation matrix, the variables that have the largest correlation to winPlacePerc are boosts, walking distance, weapons acquired. ETC, So I'll toss the other uncorrelated values from the data and train on just those. Since I find that the others will be relatively useless bloat. ","metadata":{}},{"cell_type":"code","source":"xtrain=train[['boosts','walkDistance','weaponsAcquired','damageDealt','kills']]\nytrain=train['winPlacePerc']\nxtest=test[['boosts','walkDistance','weaponsAcquired','damageDealt','kills']]","metadata":{"id":"U4RQrukwDYXW","execution":{"iopub.status.busy":"2022-08-13T01:29:48.506828Z","iopub.execute_input":"2022-08-13T01:29:48.507740Z","iopub.status.idle":"2022-08-13T01:29:48.594951Z","shell.execute_reply.started":"2022-08-13T01:29:48.507701Z","shell.execute_reply":"2022-08-13T01:29:48.593099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import tensorflow as tf\n#from tensorflow import keras\n#from tensorflow.keras import layers\n#from tensorflow.keras import models\n#from tensorflow.keras.layers import Flatten\n#def myNN(train_data):\n#  model=tf.keras.models.Sequential()\n#  model.add(layers.Dense(units=64,input_shape=(train_data.shape[1],),activation='relu',name='input'))\n#  model.add(layers.Dropout(0.15))\n#  model.add(layers.Dense(units=64,activation='relu',name='hidden1'))\n#  model.add(layers.Dropout(0.15))\n#  model.add(layers.Dense(units=64,activation='relu',name='hidden2'))\n#  model.add(layers.Dropout(0.15))\n#  model.add(layers.Dense(units=64,activation='relu',name='hidden3'))\n#\n#  model.add(layers.Dense(units=1,activation=None,name='output'))\n#\n#  model.compile(optimizer='rmsprop',loss='mse',metrics=['mae','accuracy'])\n#\n#  return model","metadata":{"id":"rg2N-X9q5kEt","execution":{"iopub.status.busy":"2022-08-13T01:29:48.596910Z","iopub.execute_input":"2022-08-13T01:29:48.597608Z","iopub.status.idle":"2022-08-13T01:29:48.604076Z","shell.execute_reply.started":"2022-08-13T01:29:48.597559Z","shell.execute_reply":"2022-08-13T01:29:48.602868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#PUBGmodel = myNN(xtrain)\n#PUBGmodel.summary()","metadata":{"id":"NmlfXOi_Jykk","outputId":"f60fe1e5-c782-4625-d680-14739de8c074","execution":{"iopub.status.busy":"2022-08-13T01:29:48.605752Z","iopub.execute_input":"2022-08-13T01:29:48.606458Z","iopub.status.idle":"2022-08-13T01:29:48.620657Z","shell.execute_reply.started":"2022-08-13T01:29:48.606420Z","shell.execute_reply":"2022-08-13T01:29:48.619332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ne = 2 #number of train loops\n\n#Train\n#callbacks=[keras.callbacks.ModelCheckpoint(\"pubg.keras\",save_best_only=True)]\n#history=PUBGmodel.fit(xtrain, ytrain, epochs=ne, batch_size=128, verbose=1,validation_split=0.25,callbacks=callbacks)\n#model = keras.models.load_model(\"pubg.keras\")","metadata":{"id":"ls1r0z627kvB","outputId":"3fdab0b6-bbfc-4c38-98a8-435c0dab847d","execution":{"iopub.status.busy":"2022-08-13T01:29:48.623419Z","iopub.execute_input":"2022-08-13T01:29:48.624053Z","iopub.status.idle":"2022-08-13T01:29:48.632848Z","shell.execute_reply.started":"2022-08-13T01:29:48.624001Z","shell.execute_reply":"2022-08-13T01:29:48.631837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#hd = history.history\n#print(hd)\n#loss_tr = hd['accuracy']\n#loss_va = hd['val_accuracy']\n#epochs = range(0, ne) #ne is number of epochs. Set it! \n\n#import matplotlib.pyplot as plt\n#import seaborn as sns\n#sns.set()\n\n#plt.plot(epochs, loss_tr, '-.o', label='Training Acc')\n#plt.plot(epochs, loss_va, 'r', label='Validation Acc')\n#plt.xlabel('Epochs')\n#plt.ylabel('Accuracy')\n#plt.legend()\n#plt.show()","metadata":{"id":"RefvAApD-JiZ","outputId":"71501b29-e1d4-44cf-a358-992253f306fe","execution":{"iopub.status.busy":"2022-08-13T01:29:48.637760Z","iopub.execute_input":"2022-08-13T01:29:48.638561Z","iopub.status.idle":"2022-08-13T01:29:48.644461Z","shell.execute_reply.started":"2022-08-13T01:29:48.638510Z","shell.execute_reply":"2022-08-13T01:29:48.643242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ytest=model.predict(xtest)","metadata":{"id":"j2NZN90TW6Lq","execution":{"iopub.status.busy":"2022-08-13T01:29:48.646218Z","iopub.execute_input":"2022-08-13T01:29:48.646658Z","iopub.status.idle":"2022-08-13T01:29:48.657538Z","shell.execute_reply.started":"2022-08-13T01:29:48.646624Z","shell.execute_reply":"2022-08-13T01:29:48.656243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So I've submitted this dense model before, and honestly it's quite bad. Pretty much as far up the leaderboard as I can be. Our other option is to try a Linear Regression as that is the kind of model we spent the most time using for this kind of math. ","metadata":{}},{"cell_type":"code","source":"#from sklearn import linear_model\n#PUBGModel2=linear_model.LinearRegression()\n#PUBGModel2.fit(xtrain,ytrain)\n#ytest=PUBGModel2.predict(xtest)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T01:29:48.658811Z","iopub.execute_input":"2022-08-13T01:29:48.659916Z","iopub.status.idle":"2022-08-13T01:29:48.669558Z","shell.execute_reply.started":"2022-08-13T01:29:48.659862Z","shell.execute_reply":"2022-08-13T01:29:48.668262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This simple linear regression model was able to climb pretty well up the leaderboard alone with no other changes and just some data preprocessing. MAE: Score: 0.12912\n\nLet's try something else","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import HistGradientBoostingRegressor\nPUBGM=HistGradientBoostingRegressor().fit(xtrain,ytrain)\nytest=PUBGM.predict(xtest)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T01:29:48.670936Z","iopub.execute_input":"2022-08-13T01:29:48.671720Z","iopub.status.idle":"2022-08-13T01:30:27.176052Z","shell.execute_reply.started":"2022-08-13T01:29:48.671683Z","shell.execute_reply":"2022-08-13T01:30:27.175025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submit = sample.copy()\ndf_submit['winPlacePerc']=ytest\ndf_submit.head()\ndf_submit.to_csv('submission.csv', index=False)","metadata":{"id":"etqYGVXkYNW0","execution":{"iopub.status.busy":"2022-08-13T01:30:27.177662Z","iopub.execute_input":"2022-08-13T01:30:27.178441Z","iopub.status.idle":"2022-08-13T01:30:31.806689Z","shell.execute_reply.started":"2022-08-13T01:30:27.178391Z","shell.execute_reply":"2022-08-13T01:30:31.805865Z"},"trusted":true},"execution_count":null,"outputs":[]}]}