{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-25T13:19:42.953677Z","iopub.execute_input":"2023-08-25T13:19:42.954277Z","iopub.status.idle":"2023-08-25T13:19:42.968257Z","shell.execute_reply.started":"2023-08-25T13:19:42.954197Z","shell.execute_reply":"2023-08-25T13:19:42.967539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport numpy as np\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn import metrics\nimport scikitplot.plotters as skplt\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2023-08-25T13:19:42.969603Z","iopub.execute_input":"2023-08-25T13:19:42.970572Z","iopub.status.idle":"2023-08-25T13:19:44.065088Z","shell.execute_reply.started":"2023-08-25T13:19:42.970493Z","shell.execute_reply":"2023-08-25T13:19:44.064348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtype = {  'ip' : 'uint32',\n           'app' : 'uint16',\n           'device' : 'uint16',\n           'os' : 'uint16',\n           'channel' : 'uint8',\n           'is_attributed' : 'uint8'}\n\nusecol=['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']","metadata":{"_uuid":"5ea16bdc6c1294ae849ac24bb24288e9dafa1a8e","_cell_guid":"f7784f2b-5f50-4692-9c31-4cd3ec6b7bcc","execution":{"iopub.status.busy":"2023-08-25T13:19:44.066744Z","iopub.execute_input":"2023-08-25T13:19:44.067511Z","iopub.status.idle":"2023-08-25T13:19:44.076508Z","shell.execute_reply.started":"2023-08-25T13:19:44.067445Z","shell.execute_reply":"2023-08-25T13:19:44.075469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/train.csv\", dtype=dtype, infer_datetime_format=True, usecols=usecol, \n                               low_memory = True,nrows=20000000)\ndf_test = pd.read_csv(\"../input/test.csv\")","metadata":{"_uuid":"2a32ec76536e504ba18e456a98d26d981d0cf2ce","_cell_guid":"4eaadf96-ea6c-47d3-a848-e14e9038c6f7","execution":{"iopub.status.busy":"2023-08-25T13:19:44.077793Z","iopub.execute_input":"2023-08-25T13:19:44.078164Z","iopub.status.idle":"2023-08-25T13:20:38.171953Z","shell.execute_reply.started":"2023-08-25T13:19:44.078093Z","shell.execute_reply":"2023-08-25T13:20:38.170893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"_uuid":"84ee4df2b8140eca21574c6fb3d88c483a9249fc","_cell_guid":"79adf401-d328-4de9-9436-ef64de349fde","execution":{"iopub.status.busy":"2023-08-25T13:20:38.173701Z","iopub.execute_input":"2023-08-25T13:20:38.174718Z","iopub.status.idle":"2023-08-25T13:20:38.200291Z","shell.execute_reply.started":"2023-08-25T13:20:38.174632Z","shell.execute_reply":"2023-08-25T13:20:38.199167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"_uuid":"781a380e4cd1fd5cc9c6038cc542d0a66d04b9be","_cell_guid":"011cfa0b-6655-44e6-87a3-d1eeb6b438e2","execution":{"iopub.status.busy":"2023-08-25T13:20:38.201864Z","iopub.execute_input":"2023-08-25T13:20:38.202321Z","iopub.status.idle":"2023-08-25T13:20:38.217865Z","shell.execute_reply.started":"2023-08-25T13:20:38.202182Z","shell.execute_reply":"2023-08-25T13:20:38.217108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['is_attributed'].value_counts()","metadata":{"_uuid":"91d525d96925879cbf29191c14d92b7ba33c10ff","_cell_guid":"92e299ab-8584-48f2-b0c2-7b235482f7aa","scrolled":true,"execution":{"iopub.status.busy":"2023-08-25T13:20:38.219154Z","iopub.execute_input":"2023-08-25T13:20:38.219481Z","iopub.status.idle":"2023-08-25T13:20:38.583737Z","shell.execute_reply.started":"2023-08-25T13:20:38.219409Z","shell.execute_reply":"2023-08-25T13:20:38.582805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['ip', 'app', 'device', 'os', 'channel']\nuniques_train = {col :df_train[col].nunique() for col in cols}\nprint('Train : Unique Values')\nuniques_train","metadata":{"_uuid":"7257119ecec6ca3f18321a2686d9e343b72a7ac0","_cell_guid":"f7112acb-afb5-46e2-80b2-22aea3b3dc46","execution":{"iopub.status.busy":"2023-08-25T13:20:38.585290Z","iopub.execute_input":"2023-08-25T13:20:38.585621Z","iopub.status.idle":"2023-08-25T13:20:39.967202Z","shell.execute_reply.started":"2023-08-25T13:20:38.585564Z","shell.execute_reply":"2023-08-25T13:20:39.966268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['ip', 'app', 'device', 'os', 'channel']\nuniques_test = {col :df_test[col].nunique() for col in cols}\nprint('Test : Unique Values')\nuniques_test","metadata":{"_uuid":"9c7637a74f30fe58af643a7f9756f6d4812e8444","_cell_guid":"4e4e977d-dc55-4739-ad28-3945d5a91041","execution":{"iopub.status.busy":"2023-08-25T13:20:39.968739Z","iopub.execute_input":"2023-08-25T13:20:39.969109Z","iopub.status.idle":"2023-08-25T13:20:40.702962Z","shell.execute_reply.started":"2023-08-25T13:20:39.969043Z","shell.execute_reply":"2023-08-25T13:20:40.702140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mean_test_encoding(df_trn, df_tst, cols, target):    \n   \n    for col in cols:\n        df_tst[col + '_mean_encoded'] = np.nan\n        \n    for col in cols:\n        tr_mean = df_trn.groupby(col)[target].mean()\n        mean = df_tst[col].map(tr_mean)\n        df_tst[col + '_mean_encoded'] = mean\n\n    prior = df_trn[target].mean()\n\n    for col in cols:\n        df_tst[col + '_mean_encoded'].fillna(prior, inplace = True) \n        \n    return df_tst\n","metadata":{"_uuid":"78c0ef70906347aed49dfa5fa67e2061e7c5eaf1","execution":{"iopub.status.busy":"2023-08-25T13:20:40.704127Z","iopub.execute_input":"2023-08-25T13:20:40.704392Z","iopub.status.idle":"2023-08-25T13:20:40.726082Z","shell.execute_reply.started":"2023-08-25T13:20:40.704347Z","shell.execute_reply":"2023-08-25T13:20:40.725047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mean_train_encoding(df, cols, target):\n    y_tr = df[target].values\n    skf = StratifiedKFold(5, shuffle = True, random_state=123)\n\n    for col in cols:\n        df[col + '_mean_encoded'] = np.nan\n\n    for trn_ind , val_ind in skf.split(df,y_tr):\n        x_tr, x_val = df.iloc[trn_ind], df.iloc[val_ind]\n\n        for col in cols:\n            tr_mean = x_tr.groupby(col)[target].mean()\n            mean = x_val[col].map(tr_mean)\n            df[col + '_mean_encoded'].iloc[val_ind] = mean\n\n    prior = df[target].mean()\n\n    for col in cols:\n        df[col + '_mean_encoded'].fillna(prior, inplace = True) \n        \n    return df","metadata":{"_uuid":"fa94e485f0ca15bf64a6c0960d2066c4d1edf5d7","execution":{"iopub.status.busy":"2023-08-25T13:20:40.727480Z","iopub.execute_input":"2023-08-25T13:20:40.727770Z","iopub.status.idle":"2023-08-25T13:20:40.766293Z","shell.execute_reply.started":"2023-08-25T13:20:40.727711Z","shell.execute_reply":"2023-08-25T13:20:40.765321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df_train['is_attributed']\ncols = ['app', 'channel']\ntarget = 'is_attributed'\ndf_train = mean_train_encoding(df_train, cols, target)\ndf_test  = mean_test_encoding(df_train, df_test, cols, target)","metadata":{"_uuid":"10e2e6adaa45a9be35f91472dae79acc0d3cec79","execution":{"iopub.status.busy":"2023-08-25T13:20:40.767709Z","iopub.execute_input":"2023-08-25T13:20:40.768051Z","iopub.status.idle":"2023-08-25T13:21:36.324668Z","shell.execute_reply.started":"2023-08-25T13:20:40.767982Z","shell.execute_reply":"2023-08-25T13:21:36.323830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.drop(['click_time','is_attributed'], axis = 1, inplace = True)   \ndf_test.drop(['click_time','click_id'], axis = 1, inplace = True)   ","metadata":{"_uuid":"21a0e8bfa333c3592240442e51149aec52cf0cb8","_cell_guid":"6a06c475-abc3-4f96-b70e-05c97833819d","execution":{"iopub.status.busy":"2023-08-25T13:21:36.326001Z","iopub.execute_input":"2023-08-25T13:21:36.326300Z","iopub.status.idle":"2023-08-25T13:21:38.576397Z","shell.execute_reply.started":"2023-08-25T13:21:36.326240Z","shell.execute_reply":"2023-08-25T13:21:38.575546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_score(m, df, y):\n    print('Accuracy: [Train , Val]')\n    res =  [m.score(df, y)]\n    if hasattr(m, 'oob_score_'): res.append(m.oob_score_)           \n    print(res)\n    \n    print('Train Confusion Matrix')\n    df_train_proba = m.predict_proba(df)\n    df_train_pred_indices = np.argmax(df_train_proba, axis=1)\n    classes_train = np.unique(y)\n    preds_train = classes_train[df_train_pred_indices]    \n    skplt.plot_confusion_matrix(y, preds_train)      ","metadata":{"_uuid":"8e65374e8f2558a57a55fb26d17a1ef6c7d5fc02","_cell_guid":"32e4e6cb-bec9-4c05-9763-6df4bbec7d1b","execution":{"iopub.status.busy":"2023-08-25T13:21:38.577661Z","iopub.execute_input":"2023-08-25T13:21:38.577992Z","iopub.status.idle":"2023-08-25T13:21:38.594869Z","shell.execute_reply.started":"2023-08-25T13:21:38.577926Z","shell.execute_reply":"2023-08-25T13:21:38.593705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"_uuid":"0b876bc68459170c37f6d7c14a776bdaf174c335","_cell_guid":"feb83a30-e6eb-488d-b1da-83b445ad4556","execution":{"iopub.status.busy":"2023-08-25T13:21:38.596111Z","iopub.execute_input":"2023-08-25T13:21:38.596427Z","iopub.status.idle":"2023-08-25T13:21:38.618709Z","shell.execute_reply.started":"2023-08-25T13:21:38.596377Z","shell.execute_reply":"2023-08-25T13:21:38.618054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_submission = pd.read_csv(\"../input/sample_submission.csv\")\ntest_submission.head()","metadata":{"_uuid":"9b07491080d90a225c8fa0f1507d3896a4d88173","_cell_guid":"c1cce813-2913-4ca2-809d-2868dd48fc76","execution":{"iopub.status.busy":"2023-08-25T13:21:38.619873Z","iopub.execute_input":"2023-08-25T13:21:38.620362Z","iopub.status.idle":"2023-08-25T13:21:45.220987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = RandomForestClassifier(n_estimators=12, max_depth=6, min_samples_leaf=100, max_features=0.5, bootstrap=False, n_jobs=-1, random_state=123)\nclf.fit(df_train, y)\nprint_score(clf, df_train, y)","metadata":{"_uuid":"082060c259903bafc52d2aa61bb10fdcb401d18e","_cell_guid":"048bb3f5-1c95-4aa0-a614-6a410f02fff5","execution":{"iopub.status.busy":"2023-08-25T13:21:45.222413Z","iopub.execute_input":"2023-08-25T13:21:45.222947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = df_train.columns\nImp = clf.feature_importances_\nfeature_imp_dict = {}\nfor i in range(len(cols)):\n    feature_imp_dict[cols[i]] = Imp[i]\nprint(feature_imp_dict)\n","metadata":{"_uuid":"daa99c4573e6d2a07d66e11748075577111a1f7e","_cell_guid":"a4924c1a-13b6-4f3a-8f43-6dfce0777f7c","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = clf.predict_proba(df_test)\ntest_submission['is_attributed'] = y_pred[:,1]\ntest_submission.head()","metadata":{"_uuid":"41e2df759c148e4fd5062485818dd758b7be556c","_cell_guid":"ccbc50ba-b9b8-4da1-b6e8-f7a3c5fccf54","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_submission.to_csv('submission_rf_.csv', index=False)","metadata":{"_uuid":"23520e86d648cc8f764f9d4acf585f260f92e277","_cell_guid":"8056f216-848b-4258-afc4-e19fe7ddc111","execution":{"iopub.status.idle":"2023-08-25T13:29:25.762539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"f0a49a03cffc877fbd6c5080a301921dc0927e13","_cell_guid":"d76870d3-da2e-4b5d-a0c3-e7dc7a027531"},"execution_count":null,"outputs":[]}]}