{"cells":[{"metadata":{"_uuid":"a2aba12a151cdab0eb0b2481be8f2d3b3e2183a2"},"cell_type":"markdown","source":"# This is the second part of my Kernel containing only the Feature Engineering and LightGBM algorithm. The Kernel is divided into two sections due to memory and time constraints of kaggle kernel. For Exploratory Data Analysis and Base Model of my kernel, Visit the first part of the model <a href='https://www.kaggle.com/iamarjunchandra/part-1-pubg-eda-base-model'>Here!</a>"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"#Import Libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport lightgbm as lgb\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import mean_absolute_error\nimport gc\n\n#Figures Inline and Visualization style\n%matplotlib inline\nsb.set()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"635546ea5b178de15c0a03ac2414de57f8a8b747"},"cell_type":"code","source":"train = pd.read_csv('../input/train_V2.csv')\ntest = pd.read_csv('../input/test_V2.csv')\ntrain.dropna(inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"561f22e621789b732751d03f38cdc0d31b0db8c3"},"cell_type":"markdown","source":"# **3. FEATURE ENGINEERING**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"Let's Inspect the categorical colmn match type. "},{"metadata":{"trusted":true,"_uuid":"7ce6bde1502eb01e747a4afea72cd3c835f035ca"},"cell_type":"code","source":"train['matchType'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"405766d20c2dc338de84032c884b5b3ec4ef244f"},"cell_type":"markdown","source":"'groupId' and 'matchId' are available in the data.  From these, no. of players in the team and total players entered in the match can be extracted."},{"metadata":{"trusted":true,"_uuid":"a4e8af3634ec22c22f719b2f650fe97553255e5c"},"cell_type":"code","source":"train['teamPlayers']=train.groupId.map(train.groupId.value_counts())\ntest['teamPlayers']=test.groupId.map(test.groupId.value_counts())\ntrain['gamePlayers']=train.matchId.map(train.matchId.value_counts())\ntest['gamePlayers']=test.matchId.map(test.matchId.value_counts())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6e1d22f72cb14c760ffafe5fea627ab17fe96e81"},"cell_type":"markdown","source":"Let's create a new column with total enemy players . The players remaining other than the player's squad. "},{"metadata":{"trusted":true,"_uuid":"f1641684fc033bde827fe3819d60321a7a05f13d"},"cell_type":"code","source":"train['enemyPlayers']=train['gamePlayers']-train['teamPlayers']\ntest['enemyPlayers']=test['gamePlayers']-test['teamPlayers']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"eac50de5415e3db25bf0c3dd89e6ec4f28c5a147"},"cell_type":"markdown","source":"Let's create a new column representing the total distance(ride+swim+walk) covered by the player in the game. "},{"metadata":{"trusted":true,"_uuid":"a69b895774123f7359ee1f7fb7dc8fe8c1a291bb"},"cell_type":"code","source":"train['totalDistance']=train['rideDistance']+train['swimDistance']+train['walkDistance']\ntest['totalDistance']=test['rideDistance']+test['swimDistance']+test['walkDistance']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"eaaf374d061155ac41cda605774eb7f30f4455ed"},"cell_type":"markdown","source":"New column which is the sum of assists and kills."},{"metadata":{"trusted":true,"_uuid":"811d8e6fca3953d2511a02bf38352157dd8dc3bd"},"cell_type":"code","source":"train['enemyDamage']=train['assists']+train['kills']\ntest['enemyDamage']=test['assists']+test['kills']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"57e7f1ee37c1f39f9005cb311c13c4e64d187cbd"},"cell_type":"markdown","source":"New column containing total kills by the team. For this, rows are grouped based on 'matchId', 'groupId' and the sum of matching row 'kills' are taken."},{"metadata":{"trusted":true,"_uuid":"5a34c91d6bac8594c7fdb25d10636e48c2984302"},"cell_type":"code","source":"totalKills = train.groupby(['matchId','groupId']).agg({'kills': lambda x: x.sum()})\ntotalKills.rename(columns={\"kills\": \"squadKills\"}, inplace=True)\ntrain = train.join(other=totalKills, on=['matchId', 'groupId'])\ntotalKills = test.groupby(['matchId','groupId']).agg({'kills': lambda x: x.sum()})\ntotalKills.rename(columns={\"kills\": \"squadKills\"}, inplace=True)\ntest = test.join(other=totalKills, on=['matchId', 'groupId'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"20ebd2f7d8eaa162e9e03ee2b0ab264d4d878c96"},"cell_type":"markdown","source":"Lets create  new columns and find if any of them helps improve model prediction."},{"metadata":{"trusted":true,"_uuid":"84d6666e9b672c15f8f97fec33ccbcae73d9c2a3"},"cell_type":"code","source":"train['medicKits']=train['heals']+train['boosts']\ntest['medicKits']=test['heals']+test['boosts']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7bc4a3ca4bfe757a02926b293d072fd22153e0e2"},"cell_type":"code","source":"train['medicPerKill'] = train['medicKits']/train['enemyDamage']\ntest['medicPerKill'] = test['medicKits']/test['enemyDamage']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b2ef3f5ec7b4dfb8a53529467646e2e9b882d954"},"cell_type":"code","source":"train['distancePerHeals'] = train['totalDistance']/train['heals']\ntest['distancePerHeals'] = test['totalDistance']/test['heals']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d50e2b90e9425ef703e7f746598a7c1f78488556"},"cell_type":"code","source":"train['headShotKillRatio']=train['headshotKills']/train['kills']\ntest['headShotKillRatio']=test['headshotKills']/test['kills']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"02e733d7a087889989a656f594727bab33964065"},"cell_type":"code","source":"train['headshotKillRate'] = train['headshotKills'] / train['kills']\ntest['headshotKillRate'] = test['headshotKills'] / test['kills']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"93aea151b3f5998b16adb7bef0760c3d9d14d5d1"},"cell_type":"code","source":"train['killPlaceOverMaxPlace'] = train['killPlace'] / train['maxPlace']\ntest['killPlaceOverMaxPlace'] = test['killPlace'] / test['maxPlace']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0f259769c1c764e929eee21e89cc7214ac04314"},"cell_type":"code","source":"train['kills/distance']=train['kills']/train['totalDistance']\ntest['kills/distance']=test['kills']/test['totalDistance']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"849070337ed045ea5065f79af64b4a25ca6e25c6"},"cell_type":"code","source":"train['kills/walkDistance']=train['kills']/train['walkDistance']\ntest['kills/walkDistance']=test['kills']/test['walkDistance']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"56a9dd5f7bc15e69db0c689e45fa65945479e954"},"cell_type":"code","source":"train['avgKills'] = train['squadKills']/train['teamPlayers']\ntest['avgKills'] = test['squadKills']/test['teamPlayers']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"145331b6cc714bcf429fd6711aec49275ddb187b"},"cell_type":"code","source":"train['damageRatio'] = train['damageDealt']/train['enemyDamage']\ntest['damageRatio'] = test['damageDealt']/test['enemyDamage']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b8fac9c034a286462e0d592f884729ca5e441c9f"},"cell_type":"code","source":"train['distTravelledPerGame'] = train['totalDistance']/train['matchDuration']\ntest['distTravelledPerGame'] = test['totalDistance']/test['matchDuration']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb9aee533e230c23cf70a9c730b2e86f9b31396c"},"cell_type":"code","source":"train['killPlacePerc'] = train['killPlace']/train['gamePlayers']\ntest['killPlacePerc'] = test['killPlace']/test['gamePlayers']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"764488dac4f5d8aed9550be252e620993c07bedf"},"cell_type":"code","source":"train[\"playerSkill\"] = train[\"headshotKills\"]+ train[\"roadKills\"]+train[\"assists\"]-(5*train['teamKills']) \ntest[\"playerSkill\"] = test[\"headshotKills\"]+ test[\"roadKills\"]+test[\"assists\"]-(5*test['teamKills'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0e9a1900116811b51a9f5446fe4b9c2e2a29ed6"},"cell_type":"code","source":"train['gamePlacePerc'] = train['killPlace']/train['maxPlace']\ntest['gamePlacePerc'] = test['killPlace']/test['maxPlace']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"714ed1c037f0c666932b7c62dd0d45ff441c6296"},"cell_type":"markdown","source":"The newly created features contains missing values and Infinity values in it. Let's replace these with 0."},{"metadata":{"trusted":true,"_uuid":"235b6faf5fa9b7c957b548ec31c403a49f9c8a1a"},"cell_type":"code","source":"train.fillna(0,inplace=True)\ntrain.replace(np.inf, 0, inplace=True)\ntest.fillna(0,inplace=True)\ntest.replace(np.inf, 0, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff4e73384b8a15716fa1e5eecd277398cc4d0fde"},"cell_type":"code","source":"train.count()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"91c25ccb395d651b8a7793f0868d07cb4c74fefb"},"cell_type":"markdown","source":"From the heat map, killPoints, rankPoints, winPoints, maxPlace are found to be not having any significance in determining winPlacePerc. So let's remove these features from the data set. "},{"metadata":{"trusted":true,"_uuid":"5dcd318712fbccd559559c2495b1ef4ec273a999"},"cell_type":"code","source":"train.drop(columns=['killPoints','rankPoints','winPoints','maxPlace'],inplace=True)\ntest.drop(columns=['killPoints','rankPoints','winPoints','maxPlace'],inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c8800c47d12a79262f0c33f60c33ada679f4617"},"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage, took from Kaggle.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n                    \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    return df","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a7b2c4ac3a526922354d2b943d4abe2b59ab15a3"},"cell_type":"markdown","source":"In Pubg, if a player wins, his team mates are also winners. So instead on finding winPlacePerc for individual payers, let's find the winPlacePerc for each group in a match.  Let's write a function that will create new columns that are the match wise and group wise mean, max, min of all the current features and also rank them."},{"metadata":{"trusted":true,"_uuid":"df8b41fbfd4d59059d14cc9bce75d84ddd561d4b"},"cell_type":"code","source":"def feature(df):\n    features = list(df.columns)\n    features.remove(\"Id\")\n    features.remove(\"matchId\")\n    features.remove(\"groupId\")\n    features.remove(\"matchType\")\n    condition='False'\n    \n    if 'winPlacePerc' in df.columns:\n        y = np.array(df.groupby(['matchId','groupId'])['winPlacePerc'].agg('mean'), dtype=np.float64)\n        features.remove(\"winPlacePerc\")\n        condition='True'\n        \n    print(\"get group mean feature\")\n    agg = df.groupby(['matchId','groupId'])[features].agg('mean')\n    agg_rank = agg.groupby('matchId')[features].rank(pct=True).reset_index()\n    df_out = agg.reset_index()[['matchId','groupId']]\n    df_out = df_out.merge(agg.reset_index(), suffixes=[\"\", \"\"], how='left', on=['matchId', 'groupId'])\n    df_out = df_out.merge(agg_rank, suffixes=[\"_mean\", \"_mean_rank\"], how='left', on=['matchId', 'groupId'])\n        \n    print(\"get group max feature\")\n    agg = df.groupby(['matchId','groupId'])[features].agg('max')\n    agg_rank = agg.groupby('matchId')[features].rank(pct=True).reset_index()\n    df_out = df_out.merge(agg.reset_index(), suffixes=[\"\", \"\"], how='left', on=['matchId', 'groupId'])\n    df_out = df_out.merge(agg_rank, suffixes=[\"_max\", \"_max_rank\"], how='left', on=['matchId', 'groupId'])\n    \n    print(\"get group min feature\")\n    agg = df.groupby(['matchId','groupId'])[features].agg('min')\n    agg_rank = agg.groupby('matchId')[features].rank(pct=True).reset_index()\n    df_out = df_out.merge(agg.reset_index(), suffixes=[\"\", \"\"], how='left', on=['matchId', 'groupId'])\n    df_out = df_out.merge(agg_rank, suffixes=[\"_min\", \"_min_rank\"], how='left', on=['matchId', 'groupId'])\n    \n    print(\"get match mean feature\")\n    agg = df.groupby(['matchId'])[features].agg('mean').reset_index()\n    df_out = df_out.merge(agg, suffixes=[\"\", \"_match_mean\"], how='left', on=['matchId'])\n    df_id=df_out[[\"matchId\", \"groupId\"]].copy()\n    df_out.drop([\"matchId\", \"groupId\"], axis=1, inplace=True)\n    \n    del df, agg, agg_rank\n    gc.collect()\n    if condition=='True':\n        return df_out,pd.DataFrame(y),df_id\n    else:\n        return df_out,df_id","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"16ec47ca281d1cb944207e485056d3e7ad4e4980"},"cell_type":"code","source":"x,y,id_train=feature(reduce_mem_usage(train))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d13e8c58a7293810d9e43691b326ec6d21d1bedd"},"cell_type":"code","source":"x_test,id_test=feature(reduce_mem_usage(test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0d345a2d42a68c6561a3018d92a17a034b15dfe"},"cell_type":"code","source":"del train,test\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5dfdfb8a6e3956d675f9fcded5b2ea5306f8b6c6"},"cell_type":"markdown","source":"# **4. GRADIENT BOOSTING MODEL**"},{"metadata":{"_uuid":"595e9a5c2f4685ea02d32127e9ef7f00fb75c324"},"cell_type":"markdown","source":"Split the data into train and validation set."},{"metadata":{"trusted":true,"_uuid":"5336cfde7ee206d158df9a36b62732df45512d02"},"cell_type":"code","source":"x['matchId']=id_train['matchId']\nx['groupId']=id_train['groupId']\n# Train test split\nx_train,x_val,y_train,y_val=train_test_split(reduce_mem_usage(x),y,test_size=.1)\nx_test=reduce_mem_usage(x_test)\nid_val=x_val[['matchId','groupId']]\nx_val.drop(['matchId','groupId'],axis=1,inplace=True)\nx_train.drop(['matchId','groupId'],axis=1,inplace=True)\nx.drop(['matchId','groupId'],axis=1,inplace=True)\ndel y\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b14679021a68eeaf65feb7afa612a2feb3a44a83"},"cell_type":"code","source":"params = {\n        \"objective\" : \"regression\", \n        \"metric\" : \"mae\", \n        \"num_leaves\" : 149, \n        \"learning_rate\" : 0.03, \n        \"bagging_fraction\" : 0.9,\n        \"bagging_seed\" : 0, \n        \"num_threads\" : 4,\n        \"colsample_bytree\" : 0.5,\n        'min_data_in_leaf':1900, \n        'min_split_gain':0.00011,\n        'lambda_l2':9\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"84d4d3dc6d54b48b25a2e507dd2e53da9b9f17f9"},"cell_type":"code","source":"# create dataset for lightgbm\nlgb_train = lgb.Dataset(x_train, y_train,\n                       free_raw_data=False)\nlgb_eval = lgb.Dataset(x_val, y_val, reference=lgb_train,\n                      free_raw_data=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd49e0acefcdb32ad0798471ec5bfdee3c30a82b"},"cell_type":"code","source":"model = lgb.train(params,\n                lgb_train,\n                num_boost_round=22000,\n                valid_sets=lgb_eval,\n                early_stopping_rounds=10,\n                verbose_eval=1000)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"265ea911290b62a1c1994041616e5eb01c8ec8de"},"cell_type":"markdown","source":"# 6.Post Processing"},{"metadata":{"_uuid":"f1f569dff39858cac42e0a173158dac48867fef3"},"cell_type":"markdown","source":"Now that we have trained the model, let' have a look if we can make some tweaks in the predicted data so that the predicted value can be improved. First let's merge the predicted value with appropriate gamer Id in the train data."},{"metadata":{"trusted":true,"_uuid":"ed79d0b3dd2a5e4eca003a305e58aa846699526c"},"cell_type":"code","source":"y_pred_val = model.predict(x, num_iteration=model.best_iteration)\nid_train['win_pred']=y_pred_val\nid_train.set_index(['matchId','groupId'])\ntrain = reduce_mem_usage(pd.read_csv(\"../input/train_V2.csv\"))\n\ndf=pd.merge(train,id_train,on=['matchId','groupId'],how='right')\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cdfdfcf925f1b35165fb521289fc55f063e3a952"},"cell_type":"code","source":"print('The mae score is {}'.format(mean_absolute_error(df['winPlacePerc'],df['win_pred'])))\ndf = df[[\"Id\", \"matchId\", \"groupId\", \"maxPlace\", \"numGroups\",'winPlacePerc', 'win_pred']]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"81e62ce2e85d2a37476e5e379af6ea5ab7bdc35b"},"cell_type":"markdown","source":"Let's take only one row from each groupby matchId and groupId since the winPlacePerc is almost same for each player in a team. Now sort and rank each group in a match. Rank is directly proportional to winPlacePerc."},{"metadata":{"trusted":true,"_uuid":"fb2ba230bcbeacafe120f49bf00e3b8149a6e117"},"cell_type":"code","source":"df_grouped = df.groupby([\"matchId\", \"groupId\"]).first().reset_index()\ndf_grouped[\"team_place\"] = df_grouped.groupby([\"matchId\"])[\"win_pred\"].rank()\ndf_grouped","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a90a8fc8826acd61c87d0b4dc1d0e3112d4b3322"},"cell_type":"markdown","source":"It has been found out that rank of team/team_place is proportional to winPlacePerc. So team_place can be used as the most important factor judging winplacePerc. Let's try to explain winPlacePerc as the ratio of team_place to numGroups. team_place will never be equal to zero. However winPlacePerc can also be zero. So let's subtract 1 from team_place as that will return zero in cases where team_place=1."},{"metadata":{"trusted":true,"_uuid":"0e6ec9b0f6e0d99185b9da8d4e22d16a23838dd0"},"cell_type":"code","source":"df_grouped[\"win_perc\"] = (df_grouped[\"team_place\"] - 1) / (df_grouped[\"numGroups\"]-1)\ndf = df.merge(df_grouped[[\"win_perc\",\"matchId\", \"groupId\"]], on=[\"matchId\", \"groupId\"], how=\"left\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0097574a9aa85028781a84c21cbe00f3ff1baaae"},"cell_type":"markdown","source":"Let's post process the new win_perc similar. winPlacePerc shoul not exceed 1 and should not drop below 0. It should be between 1 and 0. Also maxPlace=0 is impossible in a game and maxPlace=0 means their is no team. So winPerc=0. Similarly maxPlace=0 means only one team. "},{"metadata":{"trusted":true,"_uuid":"20b3a9ed61c82e0ef8deb612f01a2a3e28e3fc1b"},"cell_type":"code","source":"df.loc[df['maxPlace'] == 0, \"win_perc\"] = 0\ndf.loc[df['maxPlace'] == 1, \"win_perc\"] = 1\ndf.loc[(df['maxPlace'] > 1) & (df['numGroups'] == 1), \"win_perc\"] = 0\ndf.loc[df['win_perc'] < 0,\"win_perc\"] = 0\ndf.loc[df['win_perc'] > 1,\"win_perc\"] = 1\ndf['win_perc'].fillna(df['win_pred'],inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"36665899761cb2ecc9a9e4ebf5f0c148f2852a43"},"cell_type":"code","source":"df_grouped[df_grouped['maxPlace']>1][['winPlacePerc','win_perc','maxPlace','numGroups','team_place']]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"78edbe559c5d4f38ec85150f6940593bbfb36e67"},"cell_type":"markdown","source":"# This idea I got while referring similar kernels published publicly during the competion time and the credit goes for <a href='https://www.kaggle.com/anycode/simple-nn-baseline-3'>Kernel Here</a>. This helps to change the predicted win by few decimal points and improve the mae score. "},{"metadata":{"trusted":true,"_uuid":"ca911ff697d0321b95247e4be36787ab87ae263f"},"cell_type":"code","source":"subset = df.loc[df['maxPlace'] > 1]\ngap = 1 / (subset['maxPlace'].values-1)\nnew_perc = np.around(subset['win_perc'].values / gap) * gap\ndf.loc[df.maxPlace > 1, \"win_perc\"] = new_perc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e91babe33fa6e50980c68f3f0dd85c001447bbe8"},"cell_type":"code","source":"print('The new mae score is {}'.format(mean_absolute_error(df['winPlacePerc'],df['win_perc'])))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6eb24b1c03a393119dde5940e1169259ab220d1b"},"cell_type":"markdown","source":"# Woahh!!! The Score has improved a lot. "},{"metadata":{"trusted":true,"_uuid":"3025cc7edb1b7f735d02bcff02f10d0863b9da9c"},"cell_type":"code","source":"del x,train,df\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7678ff1928ae9aea045aa982563bfb491dbcca47"},"cell_type":"markdown","source":"# SUBMISSION"},{"metadata":{"trusted":true,"_uuid":"b3ac8e58e0a859dcb19b6c9256a527fe195045c2"},"cell_type":"code","source":"y_pred = model.predict(x_test, num_iteration=model.best_iteration)\nid_test['win_pred']=y_pred\nid_test.set_index(['matchId','groupId'])\ndel x_train,x_val,y_train,y_val,x_test\ngc.collect()\n\ntest = reduce_mem_usage(pd.read_csv(\"../input/test_V2.csv\"))\ndf=pd.merge(test,id_test,on=['matchId','groupId'],how='right')\ndel id_test,test\ngc.collect()\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6dd0b3c6bd2a93915e922d0b4819f7850794e37b"},"cell_type":"code","source":"df = df[[\"Id\", \"matchId\", \"groupId\", \"maxPlace\", \"numGroups\",'win_pred']]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cd816ae4ef115f8f6c8b1615b1a93a6dffd6eb5b"},"cell_type":"markdown","source":"Let's take only one row from each groupby matchId and groupId since the winPlacePerc is almost same for each player in a team. Now sort and rank each group in a match. Rank is directly proportional to predicted winPerc."},{"metadata":{"trusted":true,"_uuid":"049c763e6bdfbae83e477d51eb2241625545857f"},"cell_type":"code","source":"df_grouped = df.groupby([\"matchId\", \"groupId\"]).first().reset_index()\ndf_grouped[\"team_place\"] = df_grouped.groupby([\"matchId\"])[\"win_pred\"].rank()\ndf_grouped","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"175de9db199f0e31f7a5ca844346534a867ff6ee"},"cell_type":"code","source":"df_grouped[\"win_perc\"] = (df_grouped[\"team_place\"] - 1) / (df_grouped[\"numGroups\"]-1)\ndf = df.merge(df_grouped[[\"win_perc\", \"matchId\", \"groupId\"]], on=[\"matchId\", \"groupId\"], how=\"left\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ba5d7202960ef02fc0c2f6393ac644ee6fb50b2"},"cell_type":"code","source":"df.loc[df.maxPlace == 0, \"win_perc\"] = 0\ndf.loc[df.maxPlace == 1, \"win_perc\"] = 1\ndf.loc[(df.maxPlace > 1) & (df.numGroups == 1), \"win_perc\"] = 0\ndf.loc[df['win_perc'] < 0,\"win_perc\"] = 0\ndf.loc[df['win_perc'] > 1,\"win_perc\"] = 1\ndf['win_perc'].fillna(df['win_pred'],inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"beddf815d1cdc20906d121ec94d7f99923167199"},"cell_type":"code","source":"subset = df.loc[df['maxPlace'] > 1]\ngap = 1 / (subset['maxPlace'].values-1)\nnew_perc = np.around(subset['win_perc'].values / gap) * gap\ndf.loc[df.maxPlace > 1, \"win_perc\"] = new_perc\ndf['winPlacePerc']=df['win_perc']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a902678d985e2f4dc124968993d86724c0c44a25"},"cell_type":"code","source":"df=df[['Id','winPlacePerc']]\ndf.to_csv(\"submission_final.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"04c4c0b0959dc945a6486714461ebf5248c2221c"},"cell_type":"markdown","source":"# If you liked the kernel, DO upvote! "}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}