{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fc2d6303c673d3c0c1369278796e250008885f86"},"cell_type":"code","source":"def toTapleList(list1,list2):\n    return list(itertools.product(list1,list2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2f110ed228fff7a6606a7dce0a5d270903a5551d"},"cell_type":"code","source":"# Memory saving function credit to https://www.kaggle.com/gemartin/load-data-reduce-memory-usage\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.\n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                #if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                #    df[col] = df[col].astype(np.float16)\n                #el\n                if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        #else:\n            #df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB --> {:.2f} MB (Decreased by {:.1f}%)'.format(\n        start_mem, end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport itertools\nimport gc\nimport os\nimport sys\n\nsns.set_style('darkgrid')\nsns.set_palette('bone')\n\n#pd.options.display.float_format = '{:.5g}'.format\npd.options.display.float_format = '{:,.3f}'.format\n\nprint(os.listdir(\"../input\"))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b0ab267ec1fdc282f602170cf665a7e4c6ce412"},"cell_type":"code","source":"%%time\ntrain = pd.read_csv('../input/train_V2.csv')\ntrain = reduce_mem_usage(train)\ntest = pd.read_csv('../input/test_V2.csv')\ntest = reduce_mem_usage(test)\nprint(train.shape, test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc99d69f8444b947827cb2d9ec8687d9e1b10222"},"cell_type":"code","source":"train.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ae6ba2b60127dece858946346a509bbe004ccc6"},"cell_type":"code","source":"null_cnt = train.isnull().sum().sort_values()\nprint('null count:', null_cnt[null_cnt > 0])\n# dropna\ntrain.dropna(inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c28f664f5b5013ac280d1a70c09f509012cf9dc3"},"cell_type":"code","source":"train.describe(include=np.number).drop('count').T","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0a4f2c93d0a2020365a8e12ff15451e207264342"},"cell_type":"code","source":"for c in ['Id','groupId','matchId']:\n    print(f'unique [{c}] count:', train[c].nunique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3fad36b124e48062960e1ff7ce72496812fc3194"},"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(12, 4))\n\ntrain.groupby('matchId')['matchType'].first().value_counts().plot.bar(ax=ax[0])\n\n'''\nsolo  <-- solo,solo-fpp,normal-solo,normal-solo-fpp\nduo   <-- duo,duo-fpp,normal-duo,normal-duo-fpp,crashfpp,crashtpp\nsquad <-- squad,squad-fpp,normal-squad,normal-squad-fpp,flarefpp,flaretpp\n'''\nmapper = lambda x: 'solo' if ('solo' in x) else 'duo' if ('duo' in x) or ('crash' in x) else 'squad'\ntrain['matchType'] = train['matchType'].apply(mapper)\ntrain.groupby('matchId')['matchType'].first().value_counts().plot.bar(ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a007d25139246678ee5075e7fd5401197641e310"},"cell_type":"code","source":"for q in ['numGroups == maxPlace','numGroups != maxPlace']:\n    print(q, ':', len(train.query(q)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a3093be69641a4a0972cd6a011f5317abde5140"},"cell_type":"code","source":"# describe\ncols = ['numGroups','maxPlace']\ndesc1 = train.groupby('matchType')[cols].describe()[toTapleList(cols,['min','mean','max'])]\n# groups in match\ngroup = train.groupby(['matchType','matchId','groupId']).count().groupby(['matchType','matchId']).size().to_frame('groups in match')\ndesc2 = group.groupby('matchType').describe()[toTapleList(['groups in match'],['min','mean','max'])]\n\npd.concat([desc1, desc2], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"83f8a5532d071f5bc789718b375ed0aee73c7d8b"},"cell_type":"code","source":"match = train.groupby(['matchType','matchId']).size().to_frame('players in match')\ngroup = train.groupby(['matchType','matchId','groupId']).size().to_frame('players in group')\npd.concat([match.groupby('matchType').describe()[toTapleList(['players in match'],['min','mean','max'])], \n           group.groupby('matchType').describe()[toTapleList(['players in group'],['min','mean','max'])]], axis=1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c43069a61737a3b55056dc2416dbfe195e17a098"},"cell_type":"code","source":"print(group['players in group'].nlargest(5))\ndel match,group","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5d225284a7b18c129dcf76d9f0331b132806948e"},"cell_type":"code","source":"''' ex) matchId=='41a634f62f86b7', groupId=='128b07271aa012'\n'''\nsubset = train[train['matchId']=='41a634f62f86b7']\nsub_grp = subset[subset['groupId']=='128b07271aa012']\n\nprint('matchId==\\'41a634f62f86b7\\' & groupId==\\'128b07271aa012\\'')\nprint('-'*50)\nprint('players:',len(subset))\nprint('groups:',subset['groupId'].nunique())\nprint('numGroups:',subset['numGroups'].unique())\nprint('maxPlace:',subset['maxPlace'].unique())\nprint('-'*50)\nprint('max-group players:',len(sub_grp))\nprint('max-group winPlacePerc:',sub_grp['winPlacePerc'].unique())\nprint('-'*50)\nprint('winPlacePerc:',subset['winPlacePerc'].sort_values().unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59674c4cc1ca8b7b37bd6dfc3a0a1f3803b08af3"},"cell_type":"code","source":"group = train.groupby(['matchId','groupId','matchType'])['Id'].count().to_frame('players').reset_index()\ngroup.loc[group['players'] > 4, 'players'] = '5+'\ngroup['players'] = group['players'].astype(str)\n\nfig, ax = plt.subplots(1, 3, figsize=(16, 4))\nfor mt, ax in zip(['solo','duo','squad'], ax.ravel()):\n    ax.set_xlabel(mt)\n    group[group['matchType'] == mt]['players'].value_counts().sort_index().plot.bar(ax=ax)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54b6a31ae2f46c652ec50ba7170ef6102efa768d"},"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(12, 4))\n# there are two types of maps?\ntrain['matchDuration'].hist(bins=50, ax=ax[0])\ntrain.query('matchDuration >= 1400 & matchDuration <= 1800')['matchDuration'].hist(bins=50, ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"650051c1c05bad3994f6ad55642b9be0403a9cd1"},"cell_type":"code","source":"train[train['matchDuration'] == train['matchDuration'].min()].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dcd1938f51c57d44789b9654595483078d5e152b"},"cell_type":"code","source":"train[train['matchDuration'] == train['matchDuration'].max()].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65c9069860959e2ab0b4a230354e0b1f85802783"},"cell_type":"code","source":"(train.groupby('matchId')['matchDuration'].nunique() > 1).any()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf46915a573031f84b70e3df6713d9c165c50d7e"},"cell_type":"code","source":"fig, ax = plt.subplots(2, 2, figsize=(16, 8))\n\ncols = ['boosts','heals']\nfor col, ax in zip(cols, ax):\n    sub = train[['winPlacePerc',col]].copy()\n    mv = (sub[col].max() // 5) + 1\n    sub[col] = pd.cut(sub[col], [5*x for x in range(0,mv)], right=False)\n    sub.groupby(col).mean()['winPlacePerc'].plot.bar(ax=ax[0])\n    train[col].hist(bins=20, ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6c208599df82d88432508448ad6b13013e1f92f6"},"cell_type":"code","source":"print('solo player has revives:', 'solo' in train.query('revives > 0')['matchType'].unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f358fdfe1c09bd8fd1e186e46b2e08bf579278ae"},"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(16, 4))\ncol = 'revives'\nsub = train.loc[~train['matchType'].str.contains('solo'),['winPlacePerc',col]].copy()\nsub[col] = pd.cut(sub[col], [5*x for x in range(0,8)], right=False)\nsub.groupby(col).mean()['winPlacePerc'].plot.bar(ax=ax[0])\ntrain[col].hist(bins=20, ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e5b6b060fbdfee03b4c6a7016acac3362609317"},"cell_type":"code","source":"train.groupby(['matchType'])['killPlace'].describe()[['min','mean','max']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c552c27d271238a86be81ef1e890efdef750942f"},"cell_type":"code","source":"plt.figure(figsize=(8,4))\ncol = 'killPlace'\nsub = train[['winPlacePerc',col]].copy()\nsub[col] = pd.cut(sub[col], [10*x for x in range(0,11)], right=False)\nsub.groupby(col).mean()['winPlacePerc'].plot.bar()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"18cda860a804cb0c55aa79f002215179e97a921d"},"cell_type":"code","source":"''' important \n'''\nsubMatch = train[train['matchId'] == train['matchId'].min()].sort_values(['winPlacePerc','killPlace'])\ncols = ['groupId','kills','winPlacePerc','killPlace']\nsubMatch[cols]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa527e3f6d7759d93dca3edf71ee9cdd5ab9b62f"},"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(16, 4))\n\ncol = 'kills'\nsub = train[['winPlacePerc',col]].copy()\nsub[col] = pd.cut(sub[col], [5*x for x in range(0,20)], right=False)\nsub.groupby(col).mean()['winPlacePerc'].plot.bar(ax=ax[0])\ntrain[train['kills'] < 20][col].hist(bins=20, ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9142a0f1f0f6e02cbfe4db5c730308802eaed09c"},"cell_type":"code","source":"sub = train['matchType'].str.contains('solo')\npd.concat([train.loc[sub].groupby('matchId')['kills'].sum().describe(),\n         train.loc[~sub].groupby('matchId')['kills'].sum().describe()], keys=['solo','team'], axis=1).T","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c69ab5c76c3a30e745cddeb6a103ed66cea27f7f"},"cell_type":"code","source":"fig, ax = plt.subplots(2, 2, figsize=(16, 8))\n\ncols = ['killStreaks','DBNOs']\nfor col, ax in zip(cols, ax):\n    sub = train[['winPlacePerc',col]].copy()\n    sub[col] = pd.cut(sub[col], 6)\n    sub.groupby(col).mean()['winPlacePerc'].plot.bar(ax=ax[0])\n    train[col].hist(bins=20, ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"66375737834e632405b20ed8a9ff32c7f2d76cbe"},"cell_type":"code","source":"fig, ax = plt.subplots(3, 2, figsize=(16, 12))\n\ncols = ['headshotKills','roadKills','teamKills']\nfor col, ax in zip(cols, ax):\n    sub = train[['winPlacePerc',col]].copy()\n    sub.loc[sub[col] >= 5, col] = '5+'  \n    sub[col] = sub[col].astype(str)\n    sub.groupby(col).mean()['winPlacePerc'].plot.bar(ax=ax[0])\n    train[col].hist(bins=20, ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6601f4af2816627b6f6baa996cf6f1aaee749f50"},"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(16, 4))\n\ncol = 'assists'\nsub = train[['winPlacePerc',col]].copy()\nsub.loc[sub[col] >= 5, col] = '5+'  \nsub[col] = sub[col].astype(str)\nsub.groupby(col).mean()['winPlacePerc'].plot.bar(ax=ax[0])\ntrain[col].hist(bins=20, ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ca0d7e839d1f8c1325b73aeae7753ab6e694483"},"cell_type":"code","source":"pd.concat([train[train['matchType'] == 'solo'].describe()['assists'],\n           train[train['matchType'] != 'solo'].describe()['assists']],\n          keys=['solo','team'], axis=1).T\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67c3d64f7467f7b5c256a19e319995f855c1bb6d"},"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(16, 4))\n\ncol = 'longestKill'\nsub = train[['winPlacePerc',col]].copy()\nsub[col] = pd.cut(sub[col], 6)\nsub.groupby(col).mean()['winPlacePerc'].plot.bar(ax=ax[0])\ntrain[col].hist(bins=20, ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fcfc53e8cac5c64376183b69cd5f62a7b13ec62f"},"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(16, 4))\n\ncol = 'damageDealt'\nsub = train[['winPlacePerc',col]].copy()\nsub[col] = pd.cut(sub[col], 6)\nsub.groupby(col).mean()['winPlacePerc'].plot.bar(ax=ax[0])\ntrain[col].hist(bins=20, ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd569c26c1a5174d2ae5a3799302e3255f5b4b8d"},"cell_type":"code","source":"train.query('damageDealt == 0 & (kills > 0 | DBNOs > 0)')[\n    ['damageDealt','kills','DBNOs','headshotKills','roadKills','teamKills']].head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7276d8f3b6aa3acc85b1574bcbd1684a0654237b"},"cell_type":"code","source":"fig, ax = plt.subplots(3, 2, figsize=(16, 12))\n\ncols = ['walkDistance', 'rideDistance', 'swimDistance']\nfor col, ax in zip(cols, ax):\n    sub = train[['winPlacePerc',col]].copy()\n    sub[col] = pd.cut(sub[col], 6)\n    sub.groupby(col).mean()['winPlacePerc'].plot.bar(ax=ax[0])\n    train[col].hist(bins=20, ax=ax[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"665b7d1be9a05be460e899ec3241cdb76c8f3b37"},"cell_type":"code","source":"sub = train[['walkDistance','rideDistance','swimDistance','winPlacePerc']].copy()\nwalk = train['walkDistance']\nsub['walkDistanceBin'] = pd.cut(walk, [0, 0.001, walk.quantile(.25), walk.quantile(.5), walk.quantile(.75), 99999])\nsub['rideDistanceBin'] = (train['rideDistance'] > 0).astype(int)\nsub['swimDistanceBin'] = (train['swimDistance'] > 0).astype(int)\n\nfig, ax = plt.subplots(1, 3, figsize=(16, 3), sharey=True)\nsub.groupby('walkDistanceBin').mean()['winPlacePerc'].plot.bar(ax=ax[0])\nsub.groupby('rideDistanceBin').mean()['winPlacePerc'].plot.bar(ax=ax[1])\nsub.groupby('swimDistanceBin').mean()['winPlacePerc'].plot.bar(ax=ax[2])\ndel sub, walk","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d4798791b4bf568c0b6a162859bf593df7280486"},"cell_type":"code","source":"# zombie\nsub = train.query('walkDistance == 0 & kills == 0 & weaponsAcquired == 0 & \\'solo\\' in matchType')\nprint('count:', len(sub), ' winPlacePerc:', round(sub['winPlacePerc'].mean(),3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4494731399d1a1a33335696b947edbc57007da60"},"cell_type":"code","source":"sq = 'kills > 3 & (headshotKills / kills) >= 0.8'\nsub = train.query(sq)\nprint(sq, '\\n count:', len(sub), ' winPlacePerc:', round(sub['winPlacePerc'].mean(),3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25f52a453b8eaa87aa2b251c6a414966473f1e3a"},"cell_type":"code","source":"fig, ax = plt.subplots(1, 3, figsize=(16, 4), sharey=True)\n\ncols = ['killPoints','rankPoints','winPoints']\nfor col, ax in zip(cols, ax.ravel()): \n    train.plot.scatter(x=col, y='winPlacePerc', ax=ax)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"78c007665f2d782083e63663d5e73150e85bcbff"},"cell_type":"code","source":"# rankPoint: being deprecated\n# killPoints,winPoints: If there is a value other than -1 in rankPoints, then any 0 should be treated as a “None”.\nsign = lambda x: 'p<=0' if x <= 0 else 'p>0'\npd.concat([\n    pd.crosstab(train['rankPoints'].apply(sign), train['winPoints'].apply(sign), margins=False),\n    pd.crosstab(train['rankPoints'].apply(sign), train['killPoints'].apply(sign), margins=False)\n], keys=['winPoints','killPoints'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"282da628e0c888d5ce3dbb9c9738c8b68da643da"},"cell_type":"code","source":"train['winPlacePerc'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b8c3fcbf2808762f6f9cec306795765a5f3aa2d"},"cell_type":"code","source":"# confirm unique winPlace in group\n#nuniquePlace = train.groupby(['matchId','groupId'])['winPlacePerc'].nunique()\n#print('not unique winPlace in group:', len(nuniquePlace[nuniquePlace > 1]))\n#del nuniquePlace","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"23e6ac4c7fcef303844b84ebfc7ed79b1361908d"},"cell_type":"code","source":"print('match count:', train['matchId'].nunique())\n\n# not contains 1st place\nmaxPlacePerc = train.groupby('matchId')['winPlacePerc'].max()\nprint('match [not contains 1st place]:', len(maxPlacePerc[maxPlacePerc != 1]))\ndel maxPlacePerc\n\n# edge case\nsub = train[(train['maxPlace'] > 1) & (train['numGroups'] == 1)]\nprint('match [maxPlace>1 & numGroups==1]:', len(sub.groupby('matchId')))\nprint(' - unique winPlacePerc:', sub['winPlacePerc'].unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"984c4a8d532745137e35f02e00ce0fd7a4261c4e"},"cell_type":"code","source":"pd.concat([train[train['winPlacePerc'] == 1].head(5),\n           train[train['winPlacePerc'] == 0].head(5)],\n          keys=['winPlacePerc_1', 'winPlacePerc_0'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3110872157be0de76512859a4b12da0122c6a504"},"cell_type":"code","source":"cols = ['kills','teamKills','DBNOs','revives','assists','boosts','heals','damageDealt',\n    'walkDistance','rideDistance','swimDistance','weaponsAcquired']\n\naggs = ['count','min','mean','max']\n# summary of solo-match\ngrp = train.loc[train['matchType'].str.contains('solo')].groupby('matchId')\ngrpSolo = grp[cols].sum()\n# summary of team-match\ngrp = train.loc[~train['matchType'].str.contains('solo')].groupby('matchId')\ngrpTeam = grp[cols].sum()\n\npd.concat([grpSolo.describe().T[aggs], grpTeam.describe().T[aggs]], keys=['solo', 'team'], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4f71645fb1692bd932c1d978db79e5d1199dd38c"},"cell_type":"code","source":"grpSolo.nlargest(5, 'kills')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cfa1f0c859d138fccd1315973279e33eca6ead5a"},"cell_type":"code","source":"grpTeam.nlargest(5, 'kills')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b84103dabcf0977127b66d7956166ec1b1e192d"},"cell_type":"code","source":"del grpSolo, grpTeam","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d74f72c73a4b32a104677ce181db8349ebc0acc"},"cell_type":"code","source":"cols = ['kills','teamKills','DBNOs','revives','assists','boosts','heals','damageDealt',\n    'walkDistance','rideDistance','swimDistance','weaponsAcquired']\ncols.extend(['killPlace','winPlacePerc'])\ngroup = train.groupby(['matchId','groupId'])[cols]\n\nfig, ax = plt.subplots(3, 1, figsize=(12, 18), sharey=True)\nfor df, ax in zip([group.mean(), group.min(), group.max()], ax.ravel()):\n    sns.heatmap(df.corr(), annot=True, linewidths=.6, fmt='.2f', vmax=1, vmin=-1, center=0, cmap='Blues', ax=ax)\n\ndel df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"04bcfd2360c2507c9d818afe15c18bce3b524141"},"cell_type":"code","source":"def printMatchStats(matchIds):\n    for mid in matchIds:\n        subMatch = train[train['matchId'] == mid]\n        print('matchType:', subMatch['matchType'].values[0])\n\n        grp1st = subMatch[subMatch['winPlacePerc'] == 1]\n        grpOther = subMatch[subMatch['winPlacePerc'] != 1]\n        print('players'.ljust(10), ' total:{:>3}  1st:{:>3}  other:{:>3}'.format(len(subMatch), len(grp1st), len(grpOther)))\n        for c in ['kills','teamKills','roadKills','DBNOs','revives','assists']:\n            print(c.ljust(10), ' total:{:>3}  1st:{:>3}  other:{:>3}'.format(subMatch[c].sum(), grp1st[c].sum(), grpOther[c].sum()))\n        print('-' * 30)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f91fd3be57b468d50e9265ae259d54fd1385823"},"cell_type":"code","source":"sampleMid = train['matchId'].unique()[0:5]\nprintMatchStats(sampleMid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b3af655305bc4f82e743990ad46c1304a0c63433"},"cell_type":"code","source":"match = train.groupby(['matchId'])['Id'].count()\nfullplayer = match[match == 100].reset_index()\nsampleMid = fullplayer['matchId'][0:5]\nprintMatchStats(sampleMid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f4f315fac3fec666d1274c0f7b292d81d164b60"},"cell_type":"code","source":"#print(pd.DataFrame([[val for val in dir()], [sys.getsizeof(eval(val)) for val in dir()]],\n#                   index=['name','size']).T.sort_values('size', ascending=False).reset_index(drop=True)[:10])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45c82d26247d6e5dad05f8e9948b971bafcc12be"},"cell_type":"code","source":"all_data = train.append(test, sort=False).reset_index(drop=True)\ndel train, test\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa6637187bf411d3418607b847390203772ccecf"},"cell_type":"code","source":"match = all_data.groupby('matchId')\nall_data['killPlacePerc'] = match['kills'].rank(pct=True).values\nall_data['walkDistancePerc'] = match['walkDistance'].rank(pct=True).values\n#all_data['damageDealtPerc'] = match['damageDealt'].rank(pct=True).values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"03c61c5962d0c3caec638ecc1347e805ac6a6927"},"cell_type":"code","source":"all_data['_totalDistance'] = all_data['rideDistance'] + all_data['walkDistance'] + all_data['swimDistance']\n#all_data['_rideBin'] = (all_data['rideDistance'] > 0).astype(int)\n#all_data['_swimBin'] = (all_data['swimDistance'] > 0).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e53c91bfc024c95ecbbe55678987ae1a9ca227d4"},"cell_type":"code","source":"def fillInf(df, val):\n    numcols = df.select_dtypes(include='number').columns\n    cols = numcols[numcols != 'winPlacePerc']\n    df[df == np.Inf] = np.NaN\n    df[df == np.NINF] = np.NaN\n    for c in cols: df[c].fillna(val, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c11157a914d920abcbc47669cdc557520421f6cb"},"cell_type":"code","source":"all_data['_healthItems'] = all_data['heals'] + all_data['boosts']\nall_data['_headshotKillRate'] = all_data['headshotKills'] / all_data['kills']\nall_data['_killPlaceOverMaxPlace'] = all_data['killPlace'] / all_data['maxPlace']\nall_data['_killsOverWalkDistance'] = all_data['kills'] / all_data['walkDistance']\n#all_data['_killsOverDistance'] = all_data['kills'] / all_data['_totalDistance']\n#all_data['_walkDistancePerSec'] = all_data['walkDistance'] / all_data['matchDuration']\n\nfillInf(all_data, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad1cb1daf1df9c8f1140d24e6044abb6f761e41a"},"cell_type":"code","source":"all_data.drop(['boosts','heals','killStreaks','DBNOs'], axis=1, inplace=True)\nall_data.drop(['headshotKills','roadKills','vehicleDestroys'], axis=1, inplace=True)\nall_data.drop(['rideDistance','swimDistance','matchDuration'], axis=1, inplace=True)\nall_data.drop(['rankPoints','killPoints','winPoints'], axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fbefa7908893cebd358daf62413eccc035a2ec60"},"cell_type":"code","source":"match = all_data.groupby(['matchId'])\ngroup = all_data.groupby(['matchId','groupId','matchType'])\n\n# target feature (max, min)\nagg_col = list(all_data.columns)\nexclude_agg_col = ['Id','matchId','groupId','matchType','maxPlace','numGroups','winPlacePerc']\nfor c in exclude_agg_col:\n    agg_col.remove(c)\nprint(agg_col)\n\n# target feature (sum)\nsum_col = ['kills','killPlace','damageDealt','walkDistance','_healthItems']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eb8a3b37a9e966f00c413851e222bba63e2e2fd0"},"cell_type":"code","source":"''' match sum, match max, match mean, group sum\n'''\nmatch_data = pd.concat([\n    match.size().to_frame('m.players'), \n    match[sum_col].sum().rename(columns=lambda s: 'm.sum.' + s), \n    match[sum_col].max().rename(columns=lambda s: 'm.max.' + s),\n    match[sum_col].mean().rename(columns=lambda s: 'm.mean.' + s)\n    ], axis=1).reset_index()\nmatch_data = pd.merge(match_data, \n    group[sum_col].sum().rename(columns=lambda s: 'sum.' + s).reset_index())\nmatch_data = reduce_mem_usage(match_data)\n\nprint(match_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36871c354517148837bba64424996cb4eab6607e"},"cell_type":"code","source":"''' ranking of kills and killPlace in each match\n'''\nminKills = all_data.sort_values(['matchId','groupId','kills','killPlace']).groupby(\n    ['matchId','groupId','kills']).first().reset_index().copy()\nfor n in np.arange(4):\n    c = 'kills_' + str(n) + '_Place'\n    nKills = (minKills['kills'] == n)\n    minKills.loc[nKills, c] = minKills[nKills].groupby(['matchId'])['killPlace'].rank().values\n    match_data = pd.merge(match_data, minKills[nKills][['matchId','groupId',c]], how='left')\n    #match_data[c].fillna(0, inplace=True)\nmatch_data = reduce_mem_usage(match_data)\ndel minKills, nKills\n\nprint(match_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c46a2167316f0804824665365353e9ba6a73151"},"cell_type":"code","source":"match_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40c35fe976d7bec60414bc0d9a8c8f27ef1ec02f"},"cell_type":"code","source":"''' group mean, max, min\n'''\nall_data = pd.concat([\n    group.size().to_frame('players'),\n    group.mean(),\n    group[agg_col].max().rename(columns=lambda s: 'max.' + s),\n    group[agg_col].min().rename(columns=lambda s: 'min.' + s),\n    ], axis=1).reset_index()\nall_data = reduce_mem_usage(all_data)\n\nprint(all_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b94826169a4f0a72853df30a0d397d091d63458"},"cell_type":"code","source":"# suicide: solo and teamKills > 0\n#all_data['_suicide'] = ((all_data['players'] == 1) & (all_data['teamKills'] > 0)).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"01c8929a8dcc8ccfa6f398a2cdda835d414fc115"},"cell_type":"code","source":"numcols = all_data.select_dtypes(include='number').columns.values\nnumcols = numcols[numcols != 'winPlacePerc']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"80841b5fe4907cacdaff9fc5ceb956dc29bfcf8a"},"cell_type":"code","source":"''' match summary, max\n'''\nall_data = pd.merge(all_data, match_data)\ndel match_data\ngc.collect()\n\nall_data['enemy.players'] = all_data['m.players'] - all_data['players']\nfor c in sum_col:\n    #all_data['enemy.' + c] = (all_data['m.sum.' + c] - all_data['sum.' + c]) / all_data['enemy.players']\n    #all_data['p.sum_msum.' + c] = all_data['sum.' + c] / all_data['m.sum.' + c]\n    #all_data['p.max_mmean.' + c] = all_data['max.' + c] / all_data['m.mean.' + c]\n    all_data['p.max_msum.' + c] = all_data['max.' + c] / all_data['m.sum.' + c]\n    all_data['p.max_mmax.' + c] = all_data['max.' + c] / all_data['m.max.' + c]\n    all_data.drop(['m.sum.' + c, 'm.max.' + c], axis=1, inplace=True)\n    \nfillInf(all_data, 0)\nprint(all_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c6282e7a91dda69ba19db2942e550cd23ff9564"},"cell_type":"code","source":"''' match rank\n'''\nmatch = all_data.groupby('matchId')\nmatchRank = match[numcols].rank(pct=True).rename(columns=lambda s: 'rank.' + s)\nall_data = reduce_mem_usage(pd.concat([all_data, matchRank], axis=1))\nrank_col = matchRank.columns\ndel matchRank\ngc.collect()\n\n# instead of rank(pct=True, method='dense')\nmatch = all_data.groupby('matchId')\nmatchRank = match[rank_col].max().rename(columns=lambda s: 'max.' + s).reset_index()\nall_data = pd.merge(all_data, matchRank)\nfor c in numcols:\n    all_data['rank.' + c] = all_data['rank.' + c] / all_data['max.rank.' + c]\n    all_data.drop(['max.rank.' + c], axis=1, inplace=True)\ndel matchRank\ngc.collect()\n\nprint(all_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0bc2fcdb575854a6fc4a40501fd69f9c19df6bc8"},"cell_type":"code","source":"''' TODO: incomplete\n''' \nkillMinorRank = all_data[['matchId','min.kills','max.killPlace']].copy()\ngroup = killMinorRank.groupby(['matchId','min.kills'])\nkillMinorRank['rank.minor.maxKillPlace'] = group.rank(pct=True).values\nall_data = pd.merge(all_data, killMinorRank)\n\nkillMinorRank = all_data[['matchId','max.kills','min.killPlace']].copy()\ngroup = killMinorRank.groupby(['matchId','max.kills'])\nkillMinorRank['rank.minor.minKillPlace'] = group.rank(pct=True).values\nall_data = pd.merge(all_data, killMinorRank)\n\ndel killMinorRank\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e591f066fe722ac01655e660b7594c07d0c7eb72"},"cell_type":"code","source":"# drop constant column\nconstant_column = [col for col in all_data.columns if all_data[col].nunique() == 1]\nprint('drop columns:', constant_column)\nall_data.drop(constant_column, axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f36d2d197ea10df9aae7a7ffb45c1590006a2b55"},"cell_type":"code","source":"'''\nsolo  <-- solo,solo-fpp,normal-solo,normal-solo-fpp\nduo   <-- duo,duo-fpp,normal-duo,normal-duo-fpp,crashfpp,crashtpp\nsquad <-- squad,squad-fpp,normal-squad,normal-squad-fpp,flarefpp,flaretpp\n'''\nall_data['matchType'] = all_data['matchType'].apply(mapper)\n\nall_data = pd.concat([all_data, pd.get_dummies(all_data['matchType'])], axis=1)\nall_data.drop(['matchType'], axis=1, inplace=True)\n\nall_data['matchId'] = all_data['matchId'].apply(lambda x: int(x,16))\nall_data['groupId'] = all_data['groupId'].apply(lambda x: int(x,16))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1c72b4985d9fe95d01e59274880127a1a7c366f"},"cell_type":"code","source":"null_cnt = all_data.isnull().sum().sort_values()\nprint(null_cnt[null_cnt > 0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aade8e081d0c3bda6f6bbae33e2b716f74a467e4"},"cell_type":"code","source":"#all_data.drop([],axis=1,inplace=True)\n\ncols = [col for col in all_data.columns if col not in ['Id','matchId','groupId']]\nfor i, t in all_data.loc[:, cols].dtypes.iteritems():\n    if t == object:\n        all_data[i] = pd.factorize(all_data[i])[0]\n\nall_data = reduce_mem_usage(all_data)\nall_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91962eced1ad6eeb1657c33252c5f8c9303bd3a3"},"cell_type":"code","source":"X_train = all_data[all_data['winPlacePerc'].notnull()].reset_index(drop=True)\nX_test = all_data[all_data['winPlacePerc'].isnull()].drop(['winPlacePerc'], axis=1).reset_index(drop=True)\ndel all_data\ngc.collect()\n\nY_train = X_train.pop('winPlacePerc')\nX_test_grp = X_test[['matchId','groupId']].copy()\ntrain_matchId = X_train['matchId']\n\n# drop matchId,groupId\nX_train.drop(['matchId','groupId'], axis=1, inplace=True)\nX_test.drop(['matchId','groupId'], axis=1, inplace=True)\n\nprint(X_train.shape, X_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de3f946f511f006554ec508c4bf15d4405a1bb76"},"cell_type":"code","source":"print(pd.DataFrame([[val for val in dir()], [sys.getsizeof(eval(val)) for val in dir()]],\n                   index=['name','size']).T.sort_values('size', ascending=False).reset_index(drop=True)[:10])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fdc0cb2ff1e2ad454892b54181c9b8278ec3bf90"},"cell_type":"code","source":"from sklearn.model_selection import GroupKFold\nfrom sklearn.preprocessing import minmax_scale\nimport lightgbm as lgb\n\nparams={'learning_rate': 0.1,\n        'objective':'mae',\n        'metric':'mae',\n        'num_leaves': 31,\n        'verbose': 1,\n        'random_state':42,\n        'bagging_fraction': 0.7,\n        'feature_fraction': 0.7\n       }\n\nreg = lgb.LGBMRegressor(**params, n_estimators=5000)\nreg.fit(X_train, Y_train)\npred = reg.predict(X_test, num_iteration=reg.best_iteration_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"676b7d9b385b7a569b7dc0a0185595d12ae77b51"},"cell_type":"code","source":"# Plot feature importance\nfeature_importance = reg.feature_importances_\nfeature_importance = 100.0 * (feature_importance / feature_importance.max())\nsorted_idx = np.argsort(feature_importance)\nsorted_idx = sorted_idx[len(feature_importance) - 30:]\npos = np.arange(sorted_idx.shape[0]) + .5\n\nplt.figure(figsize=(12,8))\nplt.barh(pos, feature_importance[sorted_idx], align='center')\nplt.yticks(pos, X_train.columns[sorted_idx])\nplt.xlabel('Relative Importance')\nplt.title('Variable Importance')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8aa5eb87a17caa89e84c7d22d71c8a94865e68a9"},"cell_type":"code","source":"X_train.columns[np.argsort(-feature_importance)].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"baadbd433ed5c388c5244732d3ca2c078b387b2e"},"cell_type":"code","source":"X_test_grp['_nofit.winPlacePerc'] = pred\n\ngroup = X_test_grp.groupby(['matchId'])\nX_test_grp['winPlacePerc'] = pred\nX_test_grp['_rank.winPlacePerc'] = group['winPlacePerc'].rank(method='min')\nX_test = pd.concat([X_test, X_test_grp], axis=1)\n\nsub_match = X_test_grp[['matchId','_rank.winPlacePerc']].groupby(['matchId'])\nsub_group = group.count().reset_index()['matchId'].to_frame()\n\nX_test = pd.merge(X_test, sub_group)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d27658cc196c06153bac4eed29c8ef8272e88e5b"},"cell_type":"code","source":"fullgroup = (X_test['numGroups'] == X_test['maxPlace'])\n\n# full group (201366) --> calculate from rank\nsubset = X_test.loc[fullgroup]\nX_test.loc[fullgroup, 'winPlacePerc'] = (subset['_rank.winPlacePerc'].values - 1) / (subset['maxPlace'].values - 1)\n\n# not full group (684872) --> align with maxPlace\nsubset = X_test.loc[~fullgroup]\ngap = 1.0 / (subset['maxPlace'].values - 1)\nnew_perc = np.around(subset['winPlacePerc'].values / gap) * gap  # half&up\nX_test.loc[~fullgroup, 'winPlacePerc'] = new_perc\n\nX_test['winPlacePerc'] = X_test['winPlacePerc'].clip(lower=0,upper=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d83b7cca3d610b70ea4b221ba4fad87508b7f5a"},"cell_type":"code","source":"_='''\nsubset = X_test.loc[~fullgroup].groupby(['matchId','_pred.winPlace']).filter(lambda x: len(x)>1)\n\nrank1p, rank1m = list(), list()\nfor n, df in subset.groupby(['matchId','_pred.winPlace']):\n    matchId, rank = n[0], n[1]\n    matchRanks = X_test[X_test['matchId'] == matchId]['_pred.winPlace'].values\n    df = df.sort_values(['_rank.winPlacePerc'])\n    dupCount = len(df)\n    \n    hasUpper = (rank == 1) or ((rank - 1) in matchRanks)\n    hasLower = (rank == df['maxPlace'].values[0]) or ((rank + 1) in matchRanks)\n    if hasUpper and not hasLower:\n        rank1p.append(df.index[dupCount-1])\n    elif not hasUpper and hasLower:\n        rank1m.append(df.index[0])\n    elif not hasUpper and not hasLower:\n        if (dupCount > 2):\n            rank1p.append(df.index[dupCount-1])\n            rank1m.append(df.index[0])\n        else:\n            base = 1.0 / (df['maxPlace'].values[0] - 1) * rank\n            percs = df['_nofit.winPlacePerc'].values\n            if abs(percs[0] - base) < abs(percs[dupCount-1] - base):\n                rank1p.append(df.index[dupCount-1])\n            else:\n                rank1m.append(df.index[0])\n                                \nX_test.loc[rank1p, '_pred.winPlace'] = X_test.loc[rank1p, '_pred.winPlace'] + 1\nX_test.loc[rank1m, '_pred.winPlace'] = X_test.loc[rank1m, '_pred.winPlace'] - 1\nprint(len(rank1p),len(rank1m))\n\nsubset = X_test.loc[~fullgroup]\ngap = 1.0 / (subset['maxPlace'].values - 1)\nnew_perc = (subset['_pred.winPlace'].values - 1) * gap\nX_test.loc[~fullgroup, 'winPlacePerc'] = new_perc\n'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"393a4d540ffcc8d00c112713b43922fea78f51d6"},"cell_type":"code","source":"# edge cases\nX_test.loc[X_test['maxPlace'] == 0, 'winPlacePerc'] = 0\nX_test.loc[X_test['maxPlace'] == 1, 'winPlacePerc'] = 1  # nothing\nX_test.loc[(X_test['maxPlace'] > 1) & (X_test['numGroups'] == 1), 'winPlacePerc'] = 0\nX_test['winPlacePerc'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e56a4db60010323a2ef7b09dfea8b702db7d5a28"},"cell_type":"code","source":"test = pd.read_csv('../input/test_V2.csv')\ntest['matchId'] = test['matchId'].apply(lambda x: int(x,16))\ntest['groupId'] = test['groupId'].apply(lambda x: int(x,16))\n\nsubmission = pd.merge(test, X_test[['matchId','groupId','winPlacePerc']])\nsubmission = submission[['Id','winPlacePerc']]\nsubmission.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9613d150672be16e648f3fb8a21e6919460652c4"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}