{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.tree import DecisionTreeClassifier\n\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn import metrics\n\nimport xgboost as xgb\nfrom xgboost import XGBClassifier\nfrom xgboost import plot_importance\nimport gc\n\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dtypes={\n    'ip': 'uint16', 'app': 'uint16','device':'uint16','os':'uint16','channel':'uint16','ips_attributed':'uint16','click_id':'uint32'\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"testing=True\nif testing:\n    train_path=\"../input/talkingdata-adtracking-fraud-detection/train_sample.csv\"\n    skiprows=None\n    nrows=None\n    colnames=['ip','app','device','os', 'channel', 'click_time', 'is_attributed']\nelse:\n    train_path=\"../input/talkingdata-adtracking-fraud-detection/train.csv\"\n    skiprows=None\n    nrows=None\n    colnames=['ip','app','device','os', 'channel', 'click_time', 'is_attributed']\n\ntrain_sample = pd.read_csv(train_path, skiprows=skiprows, nrows=nrows, dtype=dtypes, usecols=colnames)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_sample.index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_sample.memory_usage())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Training dataset uses {0} MB'.format(train_sample.memory_usage().sum()/1024**2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_sample.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_sample.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def fraction_unique(x):\n    return len(train_sample[x].unique())\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"number_unique_vals={x:fraction_unique(x) for x in train_sample.columns}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"number_unique_vals","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_sample.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(14,10))\nsns.countplot(x='app',data=train_sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(14, 8))\nsns.countplot(x=\"device\", data=train_sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(14, 8))\nsns.countplot(x=\"channel\", data=train_sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(14, 8))\nsns.countplot(x=\"os\", data=train_sample)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_sample['is_attributed'].astype('object').value_counts()/len(train_sample.index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"app_target=train_sample.groupby('app').is_attributed.agg(['mean','count'])\napp_target","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"frequent_apps=train_sample.groupby('app').size().reset_index(name='count')\nfrequent_apps=frequent_apps[frequent_apps['count']>frequent_apps['count'].quantile(0.80)]\nfrequent_apps=frequent_apps.merge(train_sample,on='app',how='inner')\nfrequent_apps.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,10))\nsns.countplot(y=\"app\", hue=\"is_attributed\", data=frequent_apps);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def time_features(df):\n    df['datetime']=pd.to_datetime(df['click_time'])\n    df['day_of_week']=df['datetime'].dt.dayofweek\n    df['day_of_year']=df['datetime'].dt.dayofyear\n    df['month']=df['datetime'].dt.month\n    df['hour']=df['datetime'].dt.hour\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_sample=time_features(train_sample)\ntrain_sample.drop(['click_time','datetime'],axis=1,inplace=True)\ntrain_sample.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_sample.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"int_vars = ['app', 'device', 'os', 'channel', 'day_of_week','day_of_year', 'month', 'hour']\ntrain_sample[int_vars]=train_sample[int_vars].astype('uint16')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_sample.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ip_count = train_sample.groupby('ip').size().reset_index(name='ip_count').astype('int16')\nip_count.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# creates groupings of IP addresses with other features and appends the new features to the df\ndef grouped_features(df):\n    # ip_count\n    ip_count = df.groupby('ip').size().reset_index(name='ip_count').astype('uint16')\n    ip_day_hour = df.groupby(['ip', 'day_of_week', 'hour']).size().reset_index(name='ip_day_hour').astype('uint16')\n    ip_hour_channel = df[['ip', 'hour', 'channel']].groupby(['ip', 'hour', 'channel']).size().reset_index(name='ip_hour_channel').astype('uint16')\n    ip_hour_os = df.groupby(['ip', 'hour', 'os']).channel.count().reset_index(name='ip_hour_os').astype('uint16')\n    ip_hour_app = df.groupby(['ip', 'hour', 'app']).channel.count().reset_index(name='ip_hour_app').astype('uint16')\n    ip_hour_device = df.groupby(['ip', 'hour', 'device']).channel.count().reset_index(name='ip_hour_device').astype('uint16')\n    \n    # merge the new aggregated features with the df\n    df = pd.merge(df, ip_count, on='ip', how='left')\n    del ip_count\n    df = pd.merge(df, ip_day_hour, on=['ip', 'day_of_week', 'hour'], how='left')\n    del ip_day_hour\n    df = pd.merge(df, ip_hour_channel, on=['ip', 'hour', 'channel'], how='left')\n    del ip_hour_channel\n    df = pd.merge(df, ip_hour_os, on=['ip', 'hour', 'os'], how='left')\n    del ip_hour_os\n    df = pd.merge(df, ip_hour_app, on=['ip', 'hour', 'app'], how='left')\n    del ip_hour_app\n    df = pd.merge(df, ip_hour_device, on=['ip', 'hour', 'device'], how='left')\n    del ip_hour_device\n    \n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_sample = grouped_features(train_sample)\ntrain_sample.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = train_sample.drop('is_attributed', axis=1)\ny = train_sample[['is_attributed']]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.20, random_state=101)\nprint(X_train.shape)\nprint(y_train.shape)\nprint(X_test.shape)\nprint(y_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(y_train.mean())\nprint(y_test.mean())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tree=DecisionTreeClassifier(max_depth=2)\n\nadaboost_model_1=AdaBoostClassifier(\n    base_estimator=tree,\n    n_estimators=600,\n    learning_rate=1.54,\n    algorithm=\"SAMME\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"adaboost_model_1.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions=adaboost_model_1.predict_proba(X_test)\npredictions[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"metrics.roc_auc_score(y_test,predictions[:,1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"param_grid={\"base_estimator__max_depth\":[2,5],\n           \"n_estimators\":[200,400,600]\n           }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tree=DecisionTreeClassifier()\nABC=AdaBoostClassifier(\n    base_estimator=tree,\n    learning_rate=0.6,\n    algorithm=\"SAMME\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"folds=3\ngrid_search_ABC=GridSearchCV(ABC,\n                            cv=folds,\n                            param_grid=param_grid,\n                            scoring='roc_auc',\n                            return_train_score=True,\n                            verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"grid_search_ABC.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cv_results = pd.DataFrame(grid_search_ABC.cv_results_)\ncv_results","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plotting AUC with hyperparameter combinations\n\nplt.figure(figsize=(16,6))\nfor n, depth in enumerate(param_grid['base_estimator__max_depth']):\n    \n\n    # subplot 1/n\n    plt.subplot(1,3, n+1)\n    depth_df = cv_results[cv_results['param_base_estimator__max_depth']==depth]\n\n    plt.plot(depth_df[\"param_n_estimators\"], depth_df[\"mean_test_score\"])\n    plt.plot(depth_df[\"param_n_estimators\"], depth_df[\"mean_train_score\"])\n    plt.xlabel('n_estimators')\n    plt.ylabel('AUC')\n    plt.title(\"max_depth={0}\".format(depth))\n    plt.ylim([0.60, 1])\n    plt.legend(['test score', 'train score'], loc='lower left')\n    plt.xscale('log')\n\n    \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tree = DecisionTreeClassifier(max_depth=2)\nABC = AdaBoostClassifier(\n    base_estimator=tree,\n    learning_rate=0.6,\n    n_estimators=200,\n    algorithm=\"SAMME\")\n\nABC.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = ABC.predict_proba(X_test)\npredictions[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"metrics.roc_auc_score(y_test, predictions[:, 1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"param_grid = {\"learning_rate\": [0.2, 0.6, 0.9],\n              \"subsample\": [0.3, 0.6, 0.9]\n             }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"GBC = GradientBoostingClassifier(max_depth=2, n_estimators=200)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"folds = 3\ngrid_search_GBC = GridSearchCV(GBC, \n                               cv = folds,\n                               param_grid=param_grid, \n                               scoring = 'roc_auc', \n                               return_train_score=True,                         \n                               verbose = 1)\n\ngrid_search_GBC.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cv_results = pd.DataFrame(grid_search_GBC.cv_results_)\ncv_results.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16,6))\n\n\nfor n, subsample in enumerate(param_grid['subsample']):\n    \n\n    # subplot 1/n\n    plt.subplot(1,len(param_grid['subsample']), n+1)\n    df = cv_results[cv_results['param_subsample']==subsample]\n\n    plt.plot(df[\"param_learning_rate\"], df[\"mean_test_score\"])\n    plt.plot(df[\"param_learning_rate\"], df[\"mean_train_score\"])\n    plt.xlabel('learning_rate')\n    plt.ylabel('AUC')\n    plt.title(\"subsample={0}\".format(subsample))\n    plt.ylim([0.60, 1])\n    plt.legend(['test score', 'train score'], loc='upper left')\n    plt.xscale('log')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = XGBClassifier()\nmodel.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred = model.predict_proba(X_test)\ny_pred[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"roc = metrics.roc_auc_score(y_test, y_pred[:, 1])\nprint(\"AUC: %.2f%%\" % (roc * 100.0))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"folds = 3\n\n# specify range of hyperparameters\nparam_grid = {'learning_rate': [0.2, 0.6], \n             'subsample': [0.3, 0.6, 0.9]}          \n\n\n# specify model\nxgb_model = XGBClassifier(max_depth=2, n_estimators=200)\n\n# set up GridSearchCV()\nmodel_cv = GridSearchCV(estimator = xgb_model, \n                        param_grid = param_grid, \n                        scoring= 'roc_auc', \n                        cv = folds, \n                        verbose = 1,\n                        return_train_score=True)      \n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_cv.fit(X_train, y_train)       ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cv_results = pd.DataFrame(model_cv.cv_results_)\ncv_results","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cv_results['param_learning_rate'] = cv_results['param_learning_rate'].astype('float')\ncv_results.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16,6))\n\nparam_grid = {'learning_rate': [0.2, 0.6], \n             'subsample': [0.3, 0.6, 0.9]} \n\n\nfor n, subsample in enumerate(param_grid['subsample']):\n    \n\n    # subplot 1/n\n    plt.subplot(1,len(param_grid['subsample']), n+1)\n    df = cv_results[cv_results['param_subsample']==subsample]\n\n    plt.plot(df[\"param_learning_rate\"], df[\"mean_test_score\"])\n    plt.plot(df[\"param_learning_rate\"], df[\"mean_train_score\"])\n    plt.xlabel('learning_rate')\n    plt.ylabel('AUC')\n    plt.title(\"subsample={0}\".format(subsample))\n    plt.ylim([0.60, 1])\n    plt.legend(['test score', 'train score'], loc='upper left')\n    plt.xscale('log')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"params = {'learning_rate': 0.2,\n          'max_depth': 2, \n          'n_estimators':200,\n          'subsample':0.6,\n         'objective':'binary:logistic'}\n\n# fit model on training data\nmodel = XGBClassifier(params = params)\nmodel.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred = model.predict_proba(X_test)\ny_pred[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"auc = sklearn.metrics.roc_auc_score(y_test, y_pred[:, 1])\nauc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results = pd.concat([ pd.Series(y_pred[:, 1], name=\"is_attributed\")], axis=1)\nresults.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_path=\"../input/talkingdata-adtracking-fraud-detection/test.csv\"\nskiprows = None\nnrows = None\ncolnames=['click_id']\ntest_sample = pd.read_csv(test_path,usecols=colnames)\n\ntest_sample.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results = pd.concat([test_sample, pd.Series(y_pred[:, 1], name=\"is_attributed\")], axis=1)\nresults.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results = results.sort_values(by=\"click_id\", axis=0).reset_index().drop(\"index\", axis=1)\nresults.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results.to_csv(\"submission_file.csv\", sep=',', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results.isna()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results=results.dropna()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"results.to_csv(\"submission_file.csv\", sep=',', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}