{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-05-24T07:04:19.199780Z","iopub.execute_input":"2021-05-24T07:04:19.200185Z","iopub.status.idle":"2021-05-24T07:04:19.209313Z","shell.execute_reply.started":"2021-05-24T07:04:19.200085Z","shell.execute_reply":"2021-05-24T07:04:19.207711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split, PredefinedSplit, GridSearchCV, cross_validate\nimport category_encoders as ce\n\nfrom sklearn import tree\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import roc_auc_score, classification_report, confusion_matrix\n\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:10:03.806209Z","iopub.execute_input":"2021-05-24T07:10:03.806728Z","iopub.status.idle":"2021-05-24T07:10:05.456023Z","shell.execute_reply.started":"2021-05-24T07:10:03.806623Z","shell.execute_reply":"2021-05-24T07:10:05.455209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load, clean and visualize data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/talkingdata-adtracking-fraud-detection/train_sample.csv')\nprint('This data frame has %d rows and %d columns.' % (df.shape[0], df.shape[1]))","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:10:09.592237Z","iopub.execute_input":"2021-05-24T07:10:09.592692Z","iopub.status.idle":"2021-05-24T07:10:09.797094Z","shell.execute_reply.started":"2021-05-24T07:10:09.592656Z","shell.execute_reply":"2021-05-24T07:10:09.795791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:10:18.505236Z","iopub.execute_input":"2021-05-24T07:10:18.505862Z","iopub.status.idle":"2021-05-24T07:10:18.535958Z","shell.execute_reply.started":"2021-05-24T07:10:18.505799Z","shell.execute_reply":"2021-05-24T07:10:18.534810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('precision', 2)\ndf.describe()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:10:23.059878Z","iopub.execute_input":"2021-05-24T07:10:23.060288Z","iopub.status.idle":"2021-05-24T07:10:23.111876Z","shell.execute_reply.started":"2021-05-24T07:10:23.060250Z","shell.execute_reply":"2021-05-24T07:10:23.110610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counts = df['is_attributed'].value_counts()\nfraud = counts[0]\nclick = counts[1]\ntot = click+fraud\nprint('There are %d fraudulent clicks (%.2f%%) and %d normal clicks (%.2f%%)' % (fraud, fraud/tot*100, click, click/tot*100))\n\ncat_features = ['ip', 'app', 'device', 'os', 'channel']\navg_count = dict()\nfor col in cat_features:\n  n = len(df[col].value_counts())\n  avg_count[col] = tot // n\n  print('There are %d %s among %d examples, average count : %d.' % (n, col, tot, avg_count[col]))","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:10:40.416332Z","iopub.execute_input":"2021-05-24T07:10:40.416718Z","iopub.status.idle":"2021-05-24T07:10:40.445433Z","shell.execute_reply.started":"2021-05-24T07:10:40.416684Z","shell.execute_reply":"2021-05-24T07:10:40.444172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:10:45.155173Z","iopub.execute_input":"2021-05-24T07:10:45.155586Z","iopub.status.idle":"2021-05-24T07:10:45.181174Z","shell.execute_reply.started":"2021-05-24T07:10:45.155550Z","shell.execute_reply":"2021-05-24T07:10:45.180044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# only normal clicks has a valid attributed_time, this feature is dropped\ndf.drop(columns=['attributed_time'], inplace=True)\n# encode click time\ndf['click_time'] = pd.to_datetime(df['click_time'])\ndf['click_day'] = df['click_time'].dt.day.astype('uint8')\ndf['click_hr'] = df['click_time'].dt.hour.astype('uint8')\ndf['click_min'] = df['click_time'].dt.minute.astype('uint8')\ndf['click_sec'] = df['click_time'].dt.second.astype('uint8')\n\ndf.drop(columns=['click_time'], inplace=True)\n# train validate split\nx = df.drop('is_attributed', axis=1)\ny = df['is_attributed']\nx_train, x_val, y_train, y_val = train_test_split(x, y, test_size=0.2, stratify=y, random_state=2021)\ntrain = pd.concat([x_train, y_train], axis=1)\nval = pd.concat([x_val, y_val], axis=1)\ntrain.hist(figsize=(10,10))\nplt.show()\nsns.heatmap(train.corr(), vmin=-1, vmax=1, center= 0, cmap= 'coolwarm')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:11:33.441302Z","iopub.execute_input":"2021-05-24T07:11:33.441968Z","iopub.status.idle":"2021-05-24T07:11:35.618912Z","shell.execute_reply.started":"2021-05-24T07:11:33.441928Z","shell.execute_reply":"2021-05-24T07:11:35.617841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load test\ntest = pd.read_csv('/kaggle/input/talkingdata-adtracking-fraud-detection/test.csv')\ntest['click_time'] = pd.to_datetime(test['click_time'])\ntest['click_day'] = test['click_time'].dt.day.astype('uint8')\ntest['click_hr'] = test['click_time'].dt.hour.astype('uint8')\ntest['click_min'] = test['click_time'].dt.minute.astype('uint8')\ntest['click_sec'] = test['click_time'].dt.second.astype('uint8')\n\ntest.drop(columns=['click_time'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:12:30.406397Z","iopub.execute_input":"2021-05-24T07:12:30.406777Z","iopub.status.idle":"2021-05-24T07:13:04.801008Z","shell.execute_reply.started":"2021-05-24T07:12:30.406738Z","shell.execute_reply":"2021-05-24T07:13:04.799947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## encode categorical features","metadata":{}},{"cell_type":"code","source":"# count encoder\ncount_encode = ce.CountEncoder(cols=cat_features, handle_unknown=avg_count)\ncount_encode.fit(train[cat_features])\ntrain = train.join(count_encode.transform(train[cat_features]).add_suffix('_cnt'))\nval = val.join(count_encode.transform(val[cat_features]).add_suffix('_cnt'))\ntest = test.join(count_encode.transform(test[cat_features]).add_suffix('_cnt'))\n# target encoder\ntarget_encode = ce.TargetEncoder(cols=cat_features, handle_unknown='value')\ntarget_encode.fit(train[cat_features], train['is_attributed'])\ntrain = train.join(target_encode.transform(train[cat_features]).add_suffix('_target'))\nval = val.join(target_encode.transform(val[cat_features]).add_suffix('_target'))\ntest = test.join(target_encode.transform(test[cat_features]).add_suffix('_target'))","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:13:31.746307Z","iopub.execute_input":"2021-05-24T07:13:31.746775Z","iopub.status.idle":"2021-05-24T07:14:29.158394Z","shell.execute_reply.started":"2021-05-24T07:13:31.746738Z","shell.execute_reply":"2021-05-24T07:14:29.157118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['ip_cnt', 'app_cnt','device_cnt', 'os_cnt', 'channel_cnt', 'ip_target', 'app_target',\n       'device_target', 'os_target', 'channel_target', 'click_day', 'click_hr', 'click_min', 'click_sec']\nfig, axs = plt.subplots(2, 7, figsize=(21,6))\nfor i, ax1 in enumerate(axs):\n    for j, ax in enumerate(ax1):\n        f = features[i*7+j]\n        train.groupby('is_attributed')[f].plot(kind='hist', alpha=0.3, legend=True, ax=ax)\n        ax.set_xlabel(f)\nfig.tight_layout(pad=2)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:14:29.160455Z","iopub.execute_input":"2021-05-24T07:14:29.160774Z","iopub.status.idle":"2021-05-24T07:14:32.765931Z","shell.execute_reply.started":"2021-05-24T07:14:29.160735Z","shell.execute_reply":"2021-05-24T07:14:32.764754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2 = train[features+['is_attributed']]\nsns.heatmap(train2.corr(), vmin=-1, vmax=1, center= 0, cmap= 'coolwarm')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:14:32.767288Z","iopub.execute_input":"2021-05-24T07:14:32.767581Z","iopub.status.idle":"2021-05-24T07:14:33.263668Z","shell.execute_reply.started":"2021-05-24T07:14:32.767553Z","shell.execute_reply":"2021-05-24T07:14:33.262514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train with weighted decision tree","metadata":{}},{"cell_type":"code","source":"clf = DecisionTreeClassifier(class_weight={0:1, 1:400}, max_depth=2, min_samples_split=2000)\nfeat = ['ip_cnt', 'app_cnt', 'device_cnt', 'os_cnt', 'channel_cnt', 'ip_target', 'app_target','device_target', 'os_target', 'channel_target', 'click_day', 'click_hr','click_min', 'click_sec']\nx_train  = train[feat]\ny_train = train['is_attributed']\nclf.fit(x_train, y_train)\ntree.plot_tree(clf, feature_names=feat)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:15:08.879356Z","iopub.execute_input":"2021-05-24T07:15:08.879759Z","iopub.status.idle":"2021-05-24T07:15:09.277890Z","shell.execute_reply.started":"2021-05-24T07:15:08.879729Z","shell.execute_reply":"2021-05-24T07:15:09.276888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = test[feat]\nresult = test.loc[:, ['click_id']]\nresult['is_attributed'] = clf.predict(x_test)\nresult.to_csv('sample_wdt_ce.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:16:35.719634Z","iopub.execute_input":"2021-05-24T07:16:35.720361Z","iopub.status.idle":"2021-05-24T07:17:15.940648Z","shell.execute_reply.started":"2021-05-24T07:16:35.720308Z","shell.execute_reply":"2021-05-24T07:17:15.939466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle competitions submit -c talkingdata-adtracking-fraud-detection -f sample_wdt_ce.csv -m \"train wdt with sample data\"","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:23:37.547488Z","iopub.execute_input":"2021-05-24T07:23:37.548123Z","iopub.status.idle":"2021-05-24T07:23:46.653611Z","shell.execute_reply.started":"2021-05-24T07:23:37.548084Z","shell.execute_reply":"2021-05-24T07:23:46.652372Z"},"trusted":true},"execution_count":null,"outputs":[]}]}