{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-07T16:53:13.314489Z","iopub.execute_input":"2022-01-07T16:53:13.315007Z","iopub.status.idle":"2022-01-07T16:53:13.33534Z","shell.execute_reply.started":"2022-01-07T16:53:13.314947Z","shell.execute_reply":"2022-01-07T16:53:13.334351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:13.337242Z","iopub.execute_input":"2022-01-07T16:53:13.338594Z","iopub.status.idle":"2022-01-07T16:53:13.343509Z","shell.execute_reply.started":"2022-01-07T16:53:13.338549Z","shell.execute_reply":"2022-01-07T16:53:13.342385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nfrom sklearn import preprocessing, metrics, ensemble\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold, GridSearchCV, cross_val_score\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn import metrics\n\nimport xgboost as xgb\nfrom xgboost import XGBClassifier, plot_importance","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:13.345123Z","iopub.execute_input":"2022-01-07T16:53:13.346002Z","iopub.status.idle":"2022-01-07T16:53:13.365478Z","shell.execute_reply.started":"2022-01-07T16:53:13.345958Z","shell.execute_reply":"2022-01-07T16:53:13.364316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= pd.read_csv(\"../input/talkingdata-adtracking-fraud-detection/train_sample.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:13.367239Z","iopub.execute_input":"2022-01-07T16:53:13.368048Z","iopub.status.idle":"2022-01-07T16:53:13.527411Z","shell.execute_reply.started":"2022-01-07T16:53:13.368003Z","shell.execute_reply":"2022-01-07T16:53:13.526343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Count of rows and column are: \" , df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:13.529922Z","iopub.execute_input":"2022-01-07T16:53:13.530256Z","iopub.status.idle":"2022-01-07T16:53:13.536856Z","shell.execute_reply.started":"2022-01-07T16:53:13.530219Z","shell.execute_reply":"2022-01-07T16:53:13.535865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:13.538263Z","iopub.execute_input":"2022-01-07T16:53:13.539178Z","iopub.status.idle":"2022-01-07T16:53:13.570733Z","shell.execute_reply.started":"2022-01-07T16:53:13.539134Z","shell.execute_reply":"2022-01-07T16:53:13.569896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring the Data - Univariate Analysis","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:13.572093Z","iopub.execute_input":"2022-01-07T16:53:13.572631Z","iopub.status.idle":"2022-01-07T16:53:13.631876Z","shell.execute_reply.started":"2022-01-07T16:53:13.572587Z","shell.execute_reply":"2022-01-07T16:53:13.630815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in df.columns:\n    cnt = len(df[i].unique())\n    print(i,\":\",cnt)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:13.633334Z","iopub.execute_input":"2022-01-07T16:53:13.633616Z","iopub.status.idle":"2022-01-07T16:53:13.692259Z","shell.execute_reply.started":"2022-01-07T16:53:13.633569Z","shell.execute_reply":"2022-01-07T16:53:13.691232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = ['ip','app','device','os','channel','is_attributed']\nfor i in col:\n    df[i]=df[i].astype('category')","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:13.695456Z","iopub.execute_input":"2022-01-07T16:53:13.695962Z","iopub.status.idle":"2022-01-07T16:53:13.734179Z","shell.execute_reply.started":"2022-01-07T16:53:13.695905Z","shell.execute_reply":"2022-01-07T16:53:13.733111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['click_time']=pd.to_datetime(df['click_time'])\ndf['attributed_time']=pd.to_datetime(df['attributed_time'])\n\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:13.735487Z","iopub.execute_input":"2022-01-07T16:53:13.7359Z","iopub.status.idle":"2022-01-07T16:53:13.82334Z","shell.execute_reply.started":"2022-01-07T16:53:13.735861Z","shell.execute_reply":"2022-01-07T16:53:13.822262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = ['ip','app','device','os','channel']\ncnt = [len(df[i].unique()) for i in col]\n\n## Plotting on Barchart !\n\nplt.figure(figsize=(12,7))\nax=sns.barplot(x=col, y=cnt, log= True)\nax.set(xlabel='Feature', ylabel='log of unique count',title=\"Count within each feature\")\nfor p, uni in zip(ax.patches, cnt):\n    height = p.get_height()\n    ax.text(p.get_x()+p.get_width()/2.,\n            height + 10,\n            uni,\n            ha=\"center\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:13.826166Z","iopub.execute_input":"2022-01-07T16:53:13.827226Z","iopub.status.idle":"2022-01-07T16:53:14.32469Z","shell.execute_reply.started":"2022-01-07T16:53:13.827167Z","shell.execute_reply":"2022-01-07T16:53:14.323925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.pie(df['is_attributed'].value_counts(normalize=True)*100,autopct='%1.2f%%')\nplt.title(\"Plot of App Downloaded vs Not Downloaded\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:14.326146Z","iopub.execute_input":"2022-01-07T16:53:14.32724Z","iopub.status.idle":"2022-01-07T16:53:14.432986Z","shell.execute_reply.started":"2022-01-07T16:53:14.327194Z","shell.execute_reply":"2022-01-07T16:53:14.431853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(25,10))\nsns.barplot(x=df['device'].value_counts().index,y=df['device'].value_counts(), log=True)\nplt.xticks(rotation=45)\nplt.title(\"Device Type for Click\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:14.43408Z","iopub.execute_input":"2022-01-07T16:53:14.434368Z","iopub.status.idle":"2022-01-07T16:53:16.618592Z","shell.execute_reply.started":"2022-01-07T16:53:14.434339Z","shell.execute_reply":"2022-01-07T16:53:16.617794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['is_attributed']=df['is_attributed'].astype(int)\nprop = df[['ip', 'is_attributed']].groupby('ip', as_index=False).median().sort_values('is_attributed', ascending=False) \ncounts = df[['ip', 'is_attributed']].groupby('ip', as_index=False).count().sort_values('is_attributed', ascending=False)\n\nmerge = counts.merge(prop, on='ip', how='left')\nmerge.columns = ['ip', 'click_count', 'prop_downloaded']\n\nax = merge[:300].plot(secondary_y='prop_downloaded')\nplt.title('Conversion Rates over Counts of 300 Most Popular IPs')\nax.set(ylabel='Count of clicks')\nplt.ylabel('Proportion Downloaded')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:16.622912Z","iopub.execute_input":"2022-01-07T16:53:16.623533Z","iopub.status.idle":"2022-01-07T16:53:17.221507Z","shell.execute_reply.started":"2022-01-07T16:53:16.62349Z","shell.execute_reply":"2022-01-07T16:53:17.220688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"proportion = df[['channel', 'is_attributed']].groupby('channel', as_index=False).mean().sort_values('is_attributed', ascending=False)\ncounts = df[['channel', 'is_attributed']].groupby('channel', as_index=False).count().sort_values('is_attributed', ascending=False)\nmerge = counts.merge(proportion, on='channel', how='left')\nmerge.columns = ['channel', 'click_count', 'prop_downloaded']\nax = merge[:100].plot(secondary_y='prop_downloaded')\nplt.title('Conversion Rates over Counts of 100 Most Popular Apps')\nax.set(ylabel='Count of clicks')\nplt.ylabel('Proportion Downloaded')\nplt.plot()\n","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:17.223031Z","iopub.execute_input":"2022-01-07T16:53:17.223921Z","iopub.status.idle":"2022-01-07T16:53:17.643553Z","shell.execute_reply.started":"2022-01-07T16:53:17.223852Z","shell.execute_reply":"2022-01-07T16:53:17.64248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"app_target = df.groupby('app').is_attributed.agg(['mean', 'count'])\nax = app_target.plot(secondary_y='mean')\nplt.title('Conversion Rates over Counts of Most Popular Apps')\nax.set(ylabel='Count of clicks')\nplt.ylabel('Proportion Downloaded')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:17.645054Z","iopub.execute_input":"2022-01-07T16:53:17.645356Z","iopub.status.idle":"2022-01-07T16:53:18.052718Z","shell.execute_reply.started":"2022-01-07T16:53:17.645323Z","shell.execute_reply":"2022-01-07T16:53:18.051205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"df[\"datetime\"]=pd.to_datetime(df[\"click_time\"])\ndf[\"day_of_week\"]=df[\"datetime\"].dt.dayofweek\ndf[\"day_of_year\"]=df[\"datetime\"].dt.dayofyear\ndf[\"month\"]=df[\"datetime\"].dt.month\ndf[\"hour\"]=df[\"datetime\"].dt.hour","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:18.054664Z","iopub.execute_input":"2022-01-07T16:53:18.055385Z","iopub.status.idle":"2022-01-07T16:53:18.127194Z","shell.execute_reply.started":"2022-01-07T16:53:18.055333Z","shell.execute_reply":"2022-01-07T16:53:18.126261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['ip']=df['ip'].astype(int)\ndf['app']=df['app'].astype(int)\ndf['device']=df['device'].astype(int)\ndf['os']=df['os'].astype(int)\ndf['channel']=df['channel'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:18.128967Z","iopub.execute_input":"2022-01-07T16:53:18.129581Z","iopub.status.idle":"2022-01-07T16:53:18.148126Z","shell.execute_reply.started":"2022-01-07T16:53:18.12953Z","shell.execute_reply":"2022-01-07T16:53:18.146606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=df.drop([\"click_time\",\"datetime\",\"attributed_time\"], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:18.150333Z","iopub.execute_input":"2022-01-07T16:53:18.150756Z","iopub.status.idle":"2022-01-07T16:53:18.168259Z","shell.execute_reply.started":"2022-01-07T16:53:18.150692Z","shell.execute_reply":"2022-01-07T16:53:18.167121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=df.drop(\"is_attributed\",axis=1)\nY=df[[\"is_attributed\"]]","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:18.170372Z","iopub.execute_input":"2022-01-07T16:53:18.170942Z","iopub.status.idle":"2022-01-07T16:53:18.181564Z","shell.execute_reply.started":"2022-01-07T16:53:18.170898Z","shell.execute_reply":"2022-01-07T16:53:18.180792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1,x2,y1,y2=train_test_split(X,Y,\n                             test_size=0.25,\n                             stratify=Y,\n                             random_state=100)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:18.183362Z","iopub.execute_input":"2022-01-07T16:53:18.183677Z","iopub.status.idle":"2022-01-07T16:53:18.86448Z","shell.execute_reply.started":"2022-01-07T16:53:18.183633Z","shell.execute_reply":"2022-01-07T16:53:18.86354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y1.mean())\nprint(y2.mean())","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:18.868355Z","iopub.execute_input":"2022-01-07T16:53:18.868664Z","iopub.status.idle":"2022-01-07T16:53:18.87696Z","shell.execute_reply.started":"2022-01-07T16:53:18.868628Z","shell.execute_reply":"2022-01-07T16:53:18.876033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## AdaBoost","metadata":{}},{"cell_type":"code","source":"#BAse Estimator\ntree = DecisionTreeClassifier(max_depth=2)\n\n#Adaboost using base estimator - tree\n\nada_model =  AdaBoostClassifier(\n    base_estimator=tree,\n    n_estimators=600,\n    learning_rate=1.5,\n    algorithm=\"SAMME\")\n\nada_model.fit(x1,y1)\n\ny_pred = ada_model.predict_proba(x2)\n\ny_pred[:10]\n","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:53:18.878543Z","iopub.execute_input":"2022-01-07T16:53:18.879085Z","iopub.status.idle":"2022-01-07T16:54:02.156909Z","shell.execute_reply.started":"2022-01-07T16:53:18.879041Z","shell.execute_reply":"2022-01-07T16:54:02.155655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROC_Score = metrics.roc_auc_score(y2,y_pred[:,1])\nprint(\"ROC Score of AdaBoost Model: \", ROC_Score)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:54:02.158693Z","iopub.execute_input":"2022-01-07T16:54:02.159133Z","iopub.status.idle":"2022-01-07T16:54:02.184408Z","shell.execute_reply.started":"2022-01-07T16:54:02.159084Z","shell.execute_reply":"2022-01-07T16:54:02.183374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parameter = {\"base_estimator__max_depth\":[2,3],\n            \"n_estimators\":[100,300,500]}\n\ntree= DecisionTreeClassifier()\n\nadaboostmodel = AdaBoostClassifier(base_estimator=tree,\n                               learning_rate=0.9,\n                                  algorithm=\"SAMME\")\n\nfold=3\n\ngrid_search_cv = GridSearchCV(adaboostmodel,\n                             cv=fold,\n                             param_grid=parameter,\n                             scoring='roc_auc',\n                             return_train_score=True,\n                             verbose=1)\ngrid_search_cv.fit(x1,y1)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:54:02.185907Z","iopub.execute_input":"2022-01-07T16:54:02.186984Z","iopub.status.idle":"2022-01-07T16:59:34.08873Z","shell.execute_reply.started":"2022-01-07T16:54:02.18693Z","shell.execute_reply":"2022-01-07T16:59:34.087785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ada_cv_result = pd.DataFrame(grid_search_cv.cv_results_)\nada_cv_result","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:59:34.090539Z","iopub.execute_input":"2022-01-07T16:59:34.091322Z","iopub.status.idle":"2022-01-07T16:59:34.119617Z","shell.execute_reply.started":"2022-01-07T16:59:34.091279Z","shell.execute_reply":"2022-01-07T16:59:34.118756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tree = DecisionTreeClassifier(max_depth=2)\n\nada_model1 = AdaBoostClassifier(base_estimator=tree,learning_rate=0.5,n_estimators=100,algorithm=\"SAMME\")\n\nada_model1.fit(x1,y1)\ny_pred1 = ada_model1.predict_proba(x2)\n\nROC_Score=metrics.roc_auc_score(y2,y_pred1[:,1])\nprint(\"ROC Score of Hyperparameter Tunned AdaBoost Model: \", ROC_Score)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:59:34.12066Z","iopub.execute_input":"2022-01-07T16:59:34.120885Z","iopub.status.idle":"2022-01-07T16:59:40.33799Z","shell.execute_reply.started":"2022-01-07T16:59:34.120861Z","shell.execute_reply":"2022-01-07T16:59:40.337008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost","metadata":{}},{"cell_type":"code","source":"XGB_model = XGBClassifier()\nXGB_model.fit(x1,y1)\n\ny_pred3 =XGB_model.predict_proba(x2)\ny_pred3[:10]","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:59:40.339262Z","iopub.execute_input":"2022-01-07T16:59:40.339482Z","iopub.status.idle":"2022-01-07T16:59:43.283943Z","shell.execute_reply.started":"2022-01-07T16:59:40.33945Z","shell.execute_reply":"2022-01-07T16:59:43.283316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROC_Score=metrics.roc_auc_score(y2,y_pred3[:,1])\nprint(\"ROC Score of XGBoost Model :%.2f%%\" % (ROC_Score * 100.0) )","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:59:43.28506Z","iopub.execute_input":"2022-01-07T16:59:43.285748Z","iopub.status.idle":"2022-01-07T16:59:43.300503Z","shell.execute_reply.started":"2022-01-07T16:59:43.285696Z","shell.execute_reply":"2022-01-07T16:59:43.299868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fold = 3\n\nparameter = {\"learning_rate\":[0.1,0.3,0.5],\n            \"subsample\":[0.3,0.6,0.8],\n            \"n_estimators\":[100,200,300,500],\n            \"max_depth\":[2,3,4]}\n\nxgb_model = XGBClassifier()\n\ngrid_xgb_model = GridSearchCV(xgb_model,\n                             param_grid=parameter,\n                             cv=fold,\n                             scoring=\"roc_auc\",return_train_score=True,\n                             verbose=0)\n\ngrid_xgb_model.fit(x1,y1)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T16:59:43.301754Z","iopub.execute_input":"2022-01-07T16:59:43.302172Z","iopub.status.idle":"2022-01-07T17:16:56.595182Z","shell.execute_reply.started":"2022-01-07T16:59:43.302135Z","shell.execute_reply":"2022-01-07T17:16:56.594208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_results = pd.DataFrame(grid_xgb_model.cv_results_)\ncv_results","metadata":{"execution":{"iopub.status.busy":"2022-01-07T17:16:56.596605Z","iopub.execute_input":"2022-01-07T17:16:56.596854Z","iopub.status.idle":"2022-01-07T17:16:56.638383Z","shell.execute_reply.started":"2022-01-07T17:16:56.596826Z","shell.execute_reply":"2022-01-07T17:16:56.637417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"XGBC_model = XGBClassifier(max_depth=2,\n                                       n_estimators=100,\n                                       learning_rate=0.1,\n                                       subsample=0.6)\nXGBC_model.fit(x1,y1)\ny_pred4=XGBC_model.predict_proba(x2)\ny_pred4[:10]","metadata":{"execution":{"iopub.status.busy":"2022-01-07T17:16:56.639678Z","iopub.execute_input":"2022-01-07T17:16:56.640058Z","iopub.status.idle":"2022-01-07T17:16:58.062341Z","shell.execute_reply.started":"2022-01-07T17:16:56.640027Z","shell.execute_reply":"2022-01-07T17:16:58.061473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROC_Score=metrics.roc_auc_score(y2,y_pred4[:,1])\nprint(\"ROC Score of Hyperparameter Tunned XGBoost Model :%.2f%%\" % (ROC_Score * 100.0) )","metadata":{"execution":{"iopub.status.busy":"2022-01-07T17:16:58.063741Z","iopub.execute_input":"2022-01-07T17:16:58.066579Z","iopub.status.idle":"2022-01-07T17:16:58.084662Z","shell.execute_reply.started":"2022-01-07T17:16:58.066525Z","shell.execute_reply":"2022-01-07T17:16:58.083829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics.plot_roc_curve(XGBC_model,x2,y2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T17:16:58.085981Z","iopub.execute_input":"2022-01-07T17:16:58.086434Z","iopub.status.idle":"2022-01-07T17:16:58.2812Z","shell.execute_reply.started":"2022-01-07T17:16:58.086394Z","shell.execute_reply":"2022-01-07T17:16:58.280397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(range(len(XGBC_model.feature_importances_)), XGBC_model.feature_importances_)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T17:16:58.282401Z","iopub.execute_input":"2022-01-07T17:16:58.282661Z","iopub.status.idle":"2022-01-07T17:16:58.442665Z","shell.execute_reply.started":"2022-01-07T17:16:58.282634Z","shell.execute_reply":"2022-01-07T17:16:58.441773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature importance\nimportance = dict(zip(x1.columns, XGBC_model.feature_importances_))\nimportance","metadata":{"execution":{"iopub.status.busy":"2022-01-07T17:16:58.44406Z","iopub.execute_input":"2022-01-07T17:16:58.445466Z","iopub.status.idle":"2022-01-07T17:16:58.460101Z","shell.execute_reply.started":"2022-01-07T17:16:58.44542Z","shell.execute_reply":"2022-01-07T17:16:58.459281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"../input/talkingdata-adtracking-fraud-detection/test.csv\")\nprint(\"Count of rows and column are: \" , test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-01-07T17:16:58.461406Z","iopub.execute_input":"2022-01-07T17:16:58.461987Z","iopub.status.idle":"2022-01-07T17:17:20.223268Z","shell.execute_reply.started":"2022-01-07T17:16:58.461943Z","shell.execute_reply":"2022-01-07T17:17:20.222426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"datetime\"]=pd.to_datetime(test[\"click_time\"])\ntest[\"day_of_week\"]=test[\"datetime\"].dt.dayofweek\ntest[\"day_of_year\"]=test[\"datetime\"].dt.dayofyear\ntest[\"month\"]=test[\"datetime\"].dt.month\ntest[\"hour\"]=test[\"datetime\"].dt.hour\ntest['ip']=test['ip'].astype(int)\ntest['app']=test['app'].astype(int)\ntest['device']=test['device'].astype(int)\ntest['os']=test['os'].astype(int)\ntest['channel']=test['channel'].astype(int)\n\ntest_df=test.drop([\"click_time\",\"datetime\",\"click_id\"], axis=1)\n\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T17:17:20.22483Z","iopub.execute_input":"2022-01-07T17:17:20.225152Z","iopub.status.idle":"2022-01-07T17:17:33.568429Z","shell.execute_reply.started":"2022-01-07T17:17:20.225111Z","shell.execute_reply":"2022-01-07T17:17:33.567536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### XGBoost Model has given us best score we will predict using this model !","metadata":{}},{"cell_type":"code","source":"final_y_ada= XGBC_model.predict_proba(test_df)\nsub1 = pd.DataFrame()\nsub1['click_id'] = test['click_id']\nsub1['is_attributed'] = final_y_ada[:, 1]\nsub1.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-07T17:22:11.480718Z","iopub.execute_input":"2022-01-07T17:22:11.481017Z","iopub.status.idle":"2022-01-07T17:22:20.592077Z","shell.execute_reply.started":"2022-01-07T17:22:11.480988Z","shell.execute_reply":"2022-01-07T17:22:20.591289Z"},"trusted":true},"execution_count":null,"outputs":[]}]}