{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import time\nnotebookstart= time.time()\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Jupyter Specific Packages\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport shap\n%matplotlib inline\nfrom sklearn.metrics import mean_squared_error\nimport math\nfrom IPython.display import display\n\n# Gradient Boosting\nimport lightgbm as lgb\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\n\ntrain = pd.read_csv(\"../input/train_V2.csv\")#.sample(20000)\ntest = pd.read_csv(\"../input/test_V2.csv\")#.sample(20000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2cd82f3de386c909b180a29758319f47fb238dc8"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c3221f643544d6036309b61c5c8816520efcaa79"},"cell_type":"code","source":"for col in [\"matchId\",\"Id\",\"groupId\"]:\n    print(\"Does {} Feature Overlap Between Train/Test Set?         {}\".format(col, any(np.intersect1d(test[col].unique(), train[col].unique()))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2500e58ac32209c078872876b59075870a65139b"},"cell_type":"code","source":"matchcount = train.matchId.nunique()\nprint(\"Number of unique matches: {}\".format(train.matchId.nunique()))\nprint(\"Train Shape Before: {} Rows, {} Cols\".format(*train.shape))\ntrain = train.loc[train.matchId.isin(sorted(train.matchId.unique())[int(matchcount* 0.50):]),:]\nprint(\"Train Shape After: {} Rows, {} Cols\".format(*train.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7944b528522fb5904cf1229132106cbac2b72e68"},"cell_type":"code","source":"# Label Encoder\nfrom sklearn import preprocessing\n# Encoder:\nlbl = preprocessing.LabelEncoder()\nfor col in ['matchType']:\n    lbl.fit(train[col])\n    train[col] = lbl.transform(train[col])\n    test[col] = lbl.transform(test[col])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3745c0953d99fc20bceadb919d60c5e6c31a0fbe"},"cell_type":"code","source":"id_cols = [\"Id\",\"groupId\",\"matchId\"]\nexclude = [\"Id\",\"groupId\",\"matchId\"]\ntrainlen = train.shape[0]\n# LGBM Dataset\nmatchcount = train.matchId.nunique()\n\ntraining = train.loc[train.matchId.isin(sorted(train.matchId.unique())[:int(matchcount* 0.85)]),\n                    [x for x in train.columns if x not in exclude]]\nprint(\"Training Shape: {} Rows, {} Cols\".format(*training.shape))\nvalidating = train.loc[train.matchId.isin(sorted(train.matchId.unique())[int(matchcount* 0.85):]),\n                       [x for x in train.columns if x not in exclude]]\nprint(\"Validating Shape: {} Rows, {} Cols\".format(*validating.shape))\n\ntrain_y = training.winPlacePerc\ntraining.drop(\"winPlacePerc\", axis =1, inplace=True)\nvalid_y = validating.winPlacePerc\nvalidating.drop(\"winPlacePerc\", axis =1, inplace=True)\n                                                             \nlgb_train = lgb.Dataset(training, train_y,feature_name = \"auto\")\nlgb_valid = lgb.Dataset(validating, valid_y, feature_name = \"auto\")\ndel training, validating","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"71f43c0bc13df18e1b42041c4310f7b247f2a9df"},"cell_type":"code","source":"print(\"Light Gradient Boosting Regressor: \")\nlgbm_params =  {\n    'task': 'train',\n    'boosting_type': 'gbdt',\n    'objective': 'regression',\n    'metric': 'rmse',\n    'num_boost_round' : 5000\n#     \"learning_rate\": 0.01,\n#     \"num_leaves\": 180,\n#     \"feature_fraction\": 0.50,\n#     \"bagging_fraction\": 0.50,\n#     'bagging_freq': 4,\n#     \"max_depth\": -1,\n#     \"reg_alpha\": 0.3,\n#     \"reg_lambda\": 0.1,\n#     #\"min_split_gain\":0.2,\n#     \"min_child_weight\":10,\n#     'zero_as_missing':True\n                }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"85aa3ece5876dd0132124f6933fbf385e1cf76ec"},"cell_type":"code","source":"stage = 'model training'\ngbm = lgb.train(lgbm_params,\n                lgb_train,\n                num_boost_round=10000,\n                valid_sets=[lgb_train, lgb_valid],\n                feature_name='auto',\n                early_stopping_rounds=50,\n                verbose_eval=250\n                )\n\n# Feature Importance Plot\nf, ax = plt.subplots(figsize=[7,10])\nlgb.plot_importance(gbm, max_num_features=25, ax=ax)\nplt.title(\"Light GBM Feature Importance\\n\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0dd67b791a1899820944d31822083f4eeeb5cc88"},"cell_type":"code","source":"pred = gbm.predict(test.loc[:,[x for x in test.columns if x not in id_cols]])\ntest['winPlacePercPred'] = np.clip(pred, a_min=0, a_max=1)\n\naux = test.groupby(['matchId','groupId'])['winPlacePercPred'].agg('mean').groupby('matchId').rank(pct=True).reset_index()\naux.columns = ['matchId','groupId','winPlacePerc']\ntest_sub = test.merge(aux, how='left', on=['matchId','groupId'])\n    \nsubmission = test_sub[['Id', 'winPlacePerc']]\nsubmission.to_csv('PubGG_LGBM.csv', index=False)\ndisplay(submission.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b274872a9598fb113429b3b8da8b8838fedf04c5"},"cell_type":"code","source":"notcat = [\"assists\",\"boosts\",\"damageDealt\",\"DBNOs\",\"heals\",\"headshotKills\",\"heals\",\"killPlace\",\"killPoints\",\"kills\",\n         \"killStreaks\",\"longestKill\",\"maxPlace\",\"numGroups\",\"revives\",\"rideDistance\",\"roadKills\",\"swimDistance\",\n         \"teamKills\",\"vehicleDestroys\",\"walkDistance\",\"weaponsAcquired\",\"winPoints\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51e46ae17174d150157939f7448b25064eeebac0"},"cell_type":"code","source":"y = train.winPlacePerc.copy()\ntrain.drop(\"winPlacePerc\",axis =1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"501212a36c3ede2ec0f8a4aff5c7efec4b64448a"},"cell_type":"code","source":"def agg_dataset(dataset):\n    for id_col in id_cols:\n        agg_features = dataset.groupby(id_col).agg({k:[\"sum\",\"mean\",\"std\"] for k in\n                                [\"killPlace\",\"walkDistance\",\"numGroups\",\"maxPlace\",\"kills\",\"longestKill\",\"weaponsAcquired\"]})\n        agg_features.columns = pd.Index([\"{}_agg_\".format(id_col) + e[0] +\"_\"+ e[1] for e in agg_features.columns.tolist()])\n        dataset = pd.merge(dataset,agg_features, on = id_col, how= \"left\")\n    return dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"30f280e3c2a879e6f4cce7bc1ebebe76dfac9f3b"},"cell_type":"code","source":"train = agg_dataset(train)\ntest = agg_dataset(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ba0d864125b6eadec4b596e8b04d531c7e7e261"},"cell_type":"code","source":"# Remove Columns with 95%+ Missing\nmissing = round(train.isnull().sum()/ train.shape[0]*100).reset_index().rename({\"index\":\"columns\",0:\"missing\"}, axis =1 )\nhigh_missing_columns = missing.loc[missing.missing > 65, \"columns\"]\nprint(\"Columns to remove (65% missing Values and Over)\\n\", list(high_missing_columns))\ntrain.drop(high_missing_columns,axis =1, inplace= True)\ntest.drop(high_missing_columns,axis =1, inplace= True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f4fd1b1b587da8f1ff12390068d8387d3b62e16"},"cell_type":"code","source":"train = pd.concat([train.reset_index(drop=True), y.reset_index(drop=True)], axis=1)\n\ntraining = train.loc[train.matchId.isin(sorted(train.matchId.unique())[:int(matchcount* 0.85)]),\n                    [x for x in train.columns if x not in exclude]]\nprint(\"Training Shape: {} Rows, {} Cols\".format(*training.shape))\nvalidating = train.loc[train.matchId.isin(sorted(train.matchId.unique())[int(matchcount* 0.85):]),\n                       [x for x in train.columns if x not in exclude]]\nprint(\"Validating Shape: {} Rows, {} Cols\".format(*validating.shape))\n\ntrain_y = training.winPlacePerc\ntraining.drop(\"winPlacePerc\", axis =1, inplace=True)\nvalid_y = validating.winPlacePerc\nvalidating.drop(\"winPlacePerc\", axis =1, inplace=True)\n                                                             \nlgb_train = lgb.Dataset(training, train_y,feature_name = \"auto\")\nlgb_valid = lgb.Dataset(validating, valid_y, feature_name = \"auto\")\n# del training, validating","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ef8c2340905aa08c1a10f1826b828f3cc1e9038e"},"cell_type":"code","source":"print(\"Light Gradient Boosting Regressor: \")\nlgbm_params =  {\n    'task': 'train',\n    'boosting_type': 'gbdt',\n    'objective': 'regression',\n    'metric': 'rmse',\n    'num_boost_round' : 5000\n#     \"learning_rate\": 0.01,\n#     \"num_leaves\": 180,\n#     \"feature_fraction\": 0.50,\n#     \"bagging_fraction\": 0.50,\n#     'bagging_freq': 4,\n#     \"max_depth\": -1,\n#     \"reg_alpha\": 0.3,\n#     \"reg_lambda\": 0.1,\n#     #\"min_split_gain\":0.2,\n#     \"min_child_weight\":10,\n#     'zero_as_missing':True\n                }\n\nstage = 'model training'\ngbm = lgb.train(lgbm_params,\n                lgb_train,\n                num_boost_round=10000,\n                valid_sets=[lgb_train, lgb_valid],\n                feature_name='auto',\n                early_stopping_rounds=50,\n                verbose_eval=250\n                )\n\n# Feature Importance Plot\nf, ax = plt.subplots(figsize=[7,10])\nlgb.plot_importance(gbm, max_num_features=25, ax=ax)\nplt.title(\"Light GBM Feature Importance\\n\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b5f67128dd00aaf5ec2ea232c9955d6919011198"},"cell_type":"code","source":"pred = gbm.predict(test.loc[:,[x for x in test.columns if x not in id_cols]])\ntest['winPlacePercPred'] = np.clip(pred, a_min=0, a_max=1)\n\naux = test.groupby(['matchId','groupId'])['winPlacePercPred'].agg('mean').groupby('matchId').rank(pct=True).reset_index()\naux.columns = ['matchId','groupId','winPlacePerc']\ntest_sub = test.merge(aux, how='left', on=['matchId','groupId'])\n    \nsubmission = test_sub[['Id', 'winPlacePerc']]\nsubmission.to_csv('AGG_PubGG_LGBM.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a5e571fbaa7a47a1f5ac461b1a95749c8c970307"},"cell_type":"code","source":"print(\"Notebook Runtime: %0.2f Minutes\"%((time.time() - notebookstart)/60))\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d06ca31b0d2bb2c9d537da59c62100aa8347c15"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}