{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-21T17:39:55.371501Z","iopub.execute_input":"2023-06-21T17:39:55.372045Z","iopub.status.idle":"2023-06-21T17:39:55.380954Z","shell.execute_reply.started":"2023-06-21T17:39:55.372005Z","shell.execute_reply":"2023-06-21T17:39:55.379856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing all the necessary libraries\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport datetime as dt\nfrom collections import Counter\nfrom imblearn.combine import SMOTETomek\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import GridSearchCV\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\n\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:39:55.382874Z","iopub.execute_input":"2023-06-21T17:39:55.383643Z","iopub.status.idle":"2023-06-21T17:39:55.393889Z","shell.execute_reply.started":"2023-06-21T17:39:55.383609Z","shell.execute_reply":"2023-06-21T17:39:55.392833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('../input/talkingdata-adtracking-fraud-detection/train_sample.csv')\ntest_data = pd.read_csv('../input/talkingdata-adtracking-fraud-detection/test_supplement.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:39:55.395399Z","iopub.execute_input":"2023-06-21T17:39:55.396393Z","iopub.status.idle":"2023-06-21T17:40:27.884075Z","shell.execute_reply.started":"2023-06-21T17:39:55.396360Z","shell.execute_reply":"2023-06-21T17:40:27.882499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:27.885877Z","iopub.execute_input":"2023-06-21T17:40:27.886210Z","iopub.status.idle":"2023-06-21T17:40:27.900446Z","shell.execute_reply.started":"2023-06-21T17:40:27.886181Z","shell.execute_reply":"2023-06-21T17:40:27.899185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:27.902114Z","iopub.execute_input":"2023-06-21T17:40:27.902902Z","iopub.status.idle":"2023-06-21T17:40:27.911706Z","shell.execute_reply.started":"2023-06-21T17:40:27.902869Z","shell.execute_reply":"2023-06-21T17:40:27.910810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isna().sum()/len(train_data) * 100","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:27.912736Z","iopub.execute_input":"2023-06-21T17:40:27.912976Z","iopub.status.idle":"2023-06-21T17:40:27.960408Z","shell.execute_reply.started":"2023-06-21T17:40:27.912957Z","shell.execute_reply":"2023-06-21T17:40:27.959410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can get rid of the dataset with more than 80% missing value and hence removing attributed_time\ntrain_data.drop(columns = ['attributed_time'], inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:27.964606Z","iopub.execute_input":"2023-06-21T17:40:27.964909Z","iopub.status.idle":"2023-06-21T17:40:27.979756Z","shell.execute_reply.started":"2023-06-21T17:40:27.964887Z","shell.execute_reply":"2023-06-21T17:40:27.979055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:27.980980Z","iopub.execute_input":"2023-06-21T17:40:27.981433Z","iopub.status.idle":"2023-06-21T17:40:28.001079Z","shell.execute_reply.started":"2023-06-21T17:40:27.981410Z","shell.execute_reply":"2023-06-21T17:40:27.998836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Performing EDA","metadata":{}},{"cell_type":"code","source":"sns.countplot(data = train_data, x='is_attributed')","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.002900Z","iopub.execute_input":"2023-06-21T17:40:28.003403Z","iopub.status.idle":"2023-06-21T17:40:28.215550Z","shell.execute_reply.started":"2023-06-21T17:40:28.003367Z","shell.execute_reply":"2023-06-21T17:40:28.214468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['is_attributed'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.216916Z","iopub.execute_input":"2023-06-21T17:40:28.217207Z","iopub.status.idle":"2023-06-21T17:40:28.229990Z","shell.execute_reply.started":"2023-06-21T17:40:28.217181Z","shell.execute_reply":"2023-06-21T17:40:28.227530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Clearly the data is highly imbalanced and therefore, we might use the balancing technique for minority class","metadata":{}},{"cell_type":"code","source":"train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.232254Z","iopub.execute_input":"2023-06-21T17:40:28.232932Z","iopub.status.idle":"2023-06-21T17:40:28.282829Z","shell.execute_reply.started":"2023-06-21T17:40:28.232889Z","shell.execute_reply":"2023-06-21T17:40:28.281373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The quartiles are increasing and thus, there might be no outliers in the data.","metadata":{}},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.284382Z","iopub.execute_input":"2023-06-21T17:40:28.284740Z","iopub.status.idle":"2023-06-21T17:40:28.316586Z","shell.execute_reply.started":"2023-06-21T17:40:28.284712Z","shell.execute_reply":"2023-06-21T17:40:28.315950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.os.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.317289Z","iopub.execute_input":"2023-06-21T17:40:28.317524Z","iopub.status.idle":"2023-06-21T17:40:28.326180Z","shell.execute_reply.started":"2023-06-21T17:40:28.317504Z","shell.execute_reply":"2023-06-21T17:40:28.325249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fix_dataframe(df):\n    # Dropping the column attributed_time \n#     df.drop(columns=df['attributed_time'], inplace = True)\n    # Converting the click time object to date time column\n    df['click_time'] = pd.to_datetime(df['click_time'])\n    df['month'] = df['click_time'].dt.month\n    df['day'] = df['click_time'].dt.day\n    df['hour'] = df['click_time'].dt.hour\n    df['dayOfWeek'] = df['click_time'].dt.dayofweek\n    df['dayOfYear'] = df['click_time'].dt.dayofyear\n    df['seconds'] = df['click_time'].dt.second\n    df.drop(columns=df['click_time'], inplace = True)\n    ip_count = df.groupby('ip').size().reset_index(name='ip_count').astype('int64')\n    df = pd.merge(df, ip_count, on='ip', how='left', sort=False)\n    df.drop(columns=['ip'], inplace = True)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.326964Z","iopub.execute_input":"2023-06-21T17:40:28.327194Z","iopub.status.idle":"2023-06-21T17:40:28.336990Z","shell.execute_reply.started":"2023-06-21T17:40:28.327175Z","shell.execute_reply":"2023-06-21T17:40:28.336052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['click_time'] = pd.to_datetime(train_data['click_time'])","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.338060Z","iopub.execute_input":"2023-06-21T17:40:28.338392Z","iopub.status.idle":"2023-06-21T17:40:28.368451Z","shell.execute_reply.started":"2023-06-21T17:40:28.338355Z","shell.execute_reply":"2023-06-21T17:40:28.367718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.370040Z","iopub.execute_input":"2023-06-21T17:40:28.370759Z","iopub.status.idle":"2023-06-21T17:40:28.394001Z","shell.execute_reply.started":"2023-06-21T17:40:28.370723Z","shell.execute_reply":"2023-06-21T17:40:28.392194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# extracting all the dates months from the date column\ntrain_data['month'] = train_data['click_time'].dt.month\n\ntrain_data['day'] = train_data['click_time'].dt.day\ntrain_data['hour'] = train_data['click_time'].dt.hour\ntrain_data['dayOfWeek'] = train_data['click_time'].dt.dayofweek\ntrain_data['dayOfYear'] = train_data['click_time'].dt.dayofyear\ntrain_data['seconds'] = train_data['click_time'].dt.second\nip_count = train_data.groupby('ip').size().reset_index(name='ip_count').astype('int64')\ntrain_data = pd.merge(train_data, ip_count, on='ip', how='left', sort=False)\ntrain_data.drop(columns=['ip'], inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.396520Z","iopub.execute_input":"2023-06-21T17:40:28.396944Z","iopub.status.idle":"2023-06-21T17:40:28.486187Z","shell.execute_reply.started":"2023-06-21T17:40:28.396910Z","shell.execute_reply":"2023-06-21T17:40:28.485194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.487249Z","iopub.execute_input":"2023-06-21T17:40:28.487526Z","iopub.status.idle":"2023-06-21T17:40:28.499974Z","shell.execute_reply.started":"2023-06-21T17:40:28.487504Z","shell.execute_reply":"2023-06-21T17:40:28.499152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data = train_data, x='hour')","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.500961Z","iopub.execute_input":"2023-06-21T17:40:28.501344Z","iopub.status.idle":"2023-06-21T17:40:28.819468Z","shell.execute_reply.started":"2023-06-21T17:40:28.501323Z","shell.execute_reply":"2023-06-21T17:40:28.818373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### It can be seen that most clicks are happening at the 4th hour of the day","metadata":{}},{"cell_type":"code","source":"sns.countplot(data = train_data, x='dayOfWeek')","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:28.820706Z","iopub.execute_input":"2023-06-21T17:40:28.820997Z","iopub.status.idle":"2023-06-21T17:40:29.006243Z","shell.execute_reply.started":"2023-06-21T17:40:28.820974Z","shell.execute_reply":"2023-06-21T17:40:29.005179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.device.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:29.009234Z","iopub.execute_input":"2023-06-21T17:40:29.009520Z","iopub.status.idle":"2023-06-21T17:40:29.019956Z","shell.execute_reply.started":"2023-06-21T17:40:29.009498Z","shell.execute_reply":"2023-06-21T17:40:29.019003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_click_from_device = train_data.device.value_counts()[:5]","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:29.021088Z","iopub.execute_input":"2023-06-21T17:40:29.022135Z","iopub.status.idle":"2023-06-21T17:40:29.032677Z","shell.execute_reply.started":"2023-06-21T17:40:29.022081Z","shell.execute_reply":"2023-06-21T17:40:29.031733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_click_from_device","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:29.034032Z","iopub.execute_input":"2023-06-21T17:40:29.034557Z","iopub.status.idle":"2023-06-21T17:40:29.046046Z","shell.execute_reply.started":"2023-06-21T17:40:29.034530Z","shell.execute_reply":"2023-06-21T17:40:29.044514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Device 1 has the most clicks in the distribution","metadata":{}},{"cell_type":"code","source":"train_data.drop(columns=['click_time'], inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:29.048456Z","iopub.execute_input":"2023-06-21T17:40:29.048836Z","iopub.status.idle":"2023-06-21T17:40:29.063037Z","shell.execute_reply.started":"2023-06-21T17:40:29.048803Z","shell.execute_reply":"2023-06-21T17:40:29.061923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Data Preparation for Modelling","metadata":{}},{"cell_type":"code","source":"X = train_data.drop(['is_attributed'], axis = 1 )\ny = train_data['is_attributed']","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:29.070555Z","iopub.execute_input":"2023-06-21T17:40:29.070890Z","iopub.status.idle":"2023-06-21T17:40:29.083267Z","shell.execute_reply.started":"2023-06-21T17:40:29.070863Z","shell.execute_reply":"2023-06-21T17:40:29.081895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:29.084805Z","iopub.execute_input":"2023-06-21T17:40:29.085131Z","iopub.status.idle":"2023-06-21T17:40:29.102689Z","shell.execute_reply.started":"2023-06-21T17:40:29.085106Z","shell.execute_reply":"2023-06-21T17:40:29.102053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, stratify = y, test_size = 0.2, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:29.103636Z","iopub.execute_input":"2023-06-21T17:40:29.104481Z","iopub.status.idle":"2023-06-21T17:40:29.169964Z","shell.execute_reply.started":"2023-06-21T17:40:29.104456Z","shell.execute_reply":"2023-06-21T17:40:29.168741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:29.171502Z","iopub.execute_input":"2023-06-21T17:40:29.171797Z","iopub.status.idle":"2023-06-21T17:40:29.178278Z","shell.execute_reply.started":"2023-06-21T17:40:29.171773Z","shell.execute_reply":"2023-06-21T17:40:29.176611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"smotetk = SMOTETomek()\ncounter = Counter(y_train)\nX_train, y_train = smotetk.fit_resample(X_train, y_train)\nnew_counter = Counter(y_train)\nprint('Before count:', counter)\nprint('After count:', new_counter)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:29.180367Z","iopub.execute_input":"2023-06-21T17:40:29.180984Z","iopub.status.idle":"2023-06-21T17:40:40.342464Z","shell.execute_reply.started":"2023-06-21T17:40:29.180936Z","shell.execute_reply":"2023-06-21T17:40:40.340305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now that the data is balanced, we can proceed for the modelling\n# Since there is high likely the chances where the misclassification may occur and thus model must be updated\n# for every misclassification. Thus, we will use the boosting techniques to encounter this.","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:40.344879Z","iopub.execute_input":"2023-06-21T17:40:40.345263Z","iopub.status.idle":"2023-06-21T17:40:40.349960Z","shell.execute_reply.started":"2023-06-21T17:40:40.345236Z","shell.execute_reply":"2023-06-21T17:40:40.348712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Final check of the X_train\nX_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:40.351971Z","iopub.execute_input":"2023-06-21T17:40:40.352635Z","iopub.status.idle":"2023-06-21T17:40:40.376287Z","shell.execute_reply.started":"2023-06-21T17:40:40.352598Z","shell.execute_reply":"2023-06-21T17:40:40.374635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Model Creation","metadata":{}},{"cell_type":"code","source":"# Model 1 Creating a baseline model to compare other models results with.\nX_train.columns","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:40.378335Z","iopub.execute_input":"2023-06-21T17:40:40.378819Z","iopub.status.idle":"2023-06-21T17:40:40.388292Z","shell.execute_reply.started":"2023-06-21T17:40:40.378778Z","shell.execute_reply":"2023-06-21T17:40:40.386719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.columns","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:40.390392Z","iopub.execute_input":"2023-06-21T17:40:40.390879Z","iopub.status.idle":"2023-06-21T17:40:40.400335Z","shell.execute_reply.started":"2023-06-21T17:40:40.390842Z","shell.execute_reply":"2023-06-21T17:40:40.398997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = LogisticRegression()\nlr.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:40.401833Z","iopub.execute_input":"2023-06-21T17:40:40.402235Z","iopub.status.idle":"2023-06-21T17:40:41.120694Z","shell.execute_reply.started":"2023-06-21T17:40:40.402203Z","shell.execute_reply":"2023-06-21T17:40:41.120008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = lr.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:41.121769Z","iopub.execute_input":"2023-06-21T17:40:41.122164Z","iopub.status.idle":"2023-06-21T17:40:41.129519Z","shell.execute_reply.started":"2023-06-21T17:40:41.122141Z","shell.execute_reply":"2023-06-21T17:40:41.128733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(metrics.classification_report(y_test, y_pred))\n\nprint(metrics.roc_auc_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:41.130757Z","iopub.execute_input":"2023-06-21T17:40:41.131210Z","iopub.status.idle":"2023-06-21T17:40:41.187042Z","shell.execute_reply.started":"2023-06-21T17:40:41.131136Z","shell.execute_reply":"2023-06-21T17:40:41.186182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_probs = lr.predict_proba(X_test)\npreds = lr_probs[:,1]\nfpr_lr, tpr_lr, threshold = metrics.roc_curve(y_test, preds)\nroc_auc = metrics.auc(fpr_lr, tpr_lr)\n\nplt.title('Receiver Operating Characteristic')\nplt.plot(fpr_lr, tpr_lr, label = 'AUC = %0.2f' % roc_auc)\nplt.legend(loc = 'lower right')\nplt.plot([0, 1], [0, 1],'r--')\nplt.xlim([0, 1])\nplt.ylim([0, 1])\nplt.ylabel('True Positive Rate')\nplt.xlabel('False Positive Rate')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:41.191144Z","iopub.execute_input":"2023-06-21T17:40:41.193299Z","iopub.status.idle":"2023-06-21T17:40:41.412476Z","shell.execute_reply.started":"2023-06-21T17:40:41.193259Z","shell.execute_reply":"2023-06-21T17:40:41.411587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model 2 XGBoost Classifier\n\nfolds = 3\n\nparam_grid = {\"learning_rate\":[0.5, 0.6],\n            \"subsample\":[0.6, 0.8],\n            \"n_estimators\":[200, 300],\n            \"max_depth\":[2]}          \n\n\nxgb_clf = XGBClassifier()\n\nxgb_cv = GridSearchCV(estimator = xgb_clf, \n                        param_grid = param_grid, \n                        scoring= 'roc_auc', \n                        cv = folds, \n                        verbose = 1,\n                        return_train_score=True) \n\nxgb_cv.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:40:41.413873Z","iopub.execute_input":"2023-06-21T17:40:41.414354Z","iopub.status.idle":"2023-06-21T17:42:40.635144Z","shell.execute_reply.started":"2023-06-21T17:40:41.414328Z","shell.execute_reply":"2023-06-21T17:42:40.633930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_cv.best_estimator_, xgb_cv.best_params_, xgb_cv.best_score_","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:42:40.638246Z","iopub.execute_input":"2023-06-21T17:42:40.638520Z","iopub.status.idle":"2023-06-21T17:42:40.646712Z","shell.execute_reply.started":"2023-06-21T17:42:40.638498Z","shell.execute_reply":"2023-06-21T17:42:40.645563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(xgb_cv.cv_results_)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:42:40.648503Z","iopub.execute_input":"2023-06-21T17:42:40.648911Z","iopub.status.idle":"2023-06-21T17:42:40.677148Z","shell.execute_reply.started":"2023-06-21T17:42:40.648879Z","shell.execute_reply":"2023-06-21T17:42:40.676221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_final = xgb_cv.best_estimator_","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:42:40.678374Z","iopub.execute_input":"2023-06-21T17:42:40.678620Z","iopub.status.idle":"2023-06-21T17:42:40.683257Z","shell.execute_reply.started":"2023-06-21T17:42:40.678599Z","shell.execute_reply":"2023-06-21T17:42:40.682316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_final.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:42:40.685771Z","iopub.execute_input":"2023-06-21T17:42:40.686043Z","iopub.status.idle":"2023-06-21T17:42:48.781317Z","shell.execute_reply.started":"2023-06-21T17:42:40.686023Z","shell.execute_reply":"2023-06-21T17:42:48.780413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_xgb = xgb_final.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:42:48.782377Z","iopub.execute_input":"2023-06-21T17:42:48.783233Z","iopub.status.idle":"2023-06-21T17:42:48.804163Z","shell.execute_reply.started":"2023-06-21T17:42:48.783209Z","shell.execute_reply":"2023-06-21T17:42:48.803361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(metrics.classification_report(y_test, y_pred_xgb))\n\nprint(metrics.roc_auc_score(y_test, y_pred_xgb))","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:42:48.805359Z","iopub.execute_input":"2023-06-21T17:42:48.806007Z","iopub.status.idle":"2023-06-21T17:42:48.834642Z","shell.execute_reply.started":"2023-06-21T17:42:48.805981Z","shell.execute_reply":"2023-06-21T17:42:48.833181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xg_probs = xgb_final.predict_proba(X_test)\nxg_preds = xg_probs[:,1]\nfpr_xg, tpr_xg, threshold = metrics.roc_curve(y_test, xg_preds)\nroc_auc_xg = metrics.auc(fpr_xg, tpr_xg)\n\nplt.title('Receiver Operating Characteristic')\nplt.plot(fpr_lr, tpr_lr, label = 'AUC_LR = %0.2f' % roc_auc)\nplt.plot(fpr_xg, tpr_xg, 'b', label = 'AUC_XG = %0.2f' % roc_auc_xg, color='orange')\nplt.legend(loc = 'lower right')\nplt.plot([0, 1], [0, 1],'r--')\nplt.xlim([0, 1])\nplt.ylim([0, 1])\nplt.ylabel('True Positive Rate')\nplt.xlabel('False Positive Rate')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:42:48.835992Z","iopub.execute_input":"2023-06-21T17:42:48.836301Z","iopub.status.idle":"2023-06-21T17:42:49.051575Z","shell.execute_reply.started":"2023-06-21T17:42:48.836274Z","shell.execute_reply":"2023-06-21T17:42:49.050098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model 3 : AdaBoost Classifier\n\nfolds = 3\n\nparam_grid = {\"base_estimator__max_depth\" : [2, 5],\n              \"n_estimators\": [200, 400, 600]\n             }\n\n\ntree = DecisionTreeClassifier()\n\nada_clf = AdaBoostClassifier(base_estimator=tree, learning_rate=0.6, algorithm=\"SAMME\")\n\nada_cv = GridSearchCV(estimator = ada_clf, \n                        param_grid = param_grid, \n                        scoring= 'roc_auc', \n                        cv = folds, \n                        verbose = 1,\n                        return_train_score=True) \n\nada_cv.fit(X_train, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T17:42:49.053088Z","iopub.execute_input":"2023-06-21T17:42:49.053388Z","iopub.status.idle":"2023-06-21T18:11:38.127751Z","shell.execute_reply.started":"2023-06-21T17:42:49.053364Z","shell.execute_reply":"2023-06-21T18:11:38.126550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ada_cv.best_estimator_, ada_cv.best_params_, ada_cv.best_score_","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:11:38.129060Z","iopub.execute_input":"2023-06-21T18:11:38.129770Z","iopub.status.idle":"2023-06-21T18:11:38.138108Z","shell.execute_reply.started":"2023-06-21T18:11:38.129741Z","shell.execute_reply":"2023-06-21T18:11:38.136718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ada_final = ada_cv.best_estimator_\n\nada_final.fit(X_train, y_train)\n\ny_pred_ada = ada_final.predict(X_test)\nprint(metrics.classification_report(y_test, y_pred_ada))\n\nprint(metrics.roc_auc_score(y_test, y_pred_ada))","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:11:38.140973Z","iopub.execute_input":"2023-06-21T18:11:38.141250Z","iopub.status.idle":"2023-06-21T18:15:57.589137Z","shell.execute_reply.started":"2023-06-21T18:11:38.141228Z","shell.execute_reply":"2023-06-21T18:15:57.588138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ada_probs = ada_final.predict_proba(X_test)\nada_preds = ada_probs[:,1]\nfpr_ada, tpr_ada, threshold = metrics.roc_curve(y_test, ada_preds)\nroc_auc_ada = metrics.auc(fpr_ada, tpr_ada)\n\nplt.title('Receiver Operating Characteristic')\nplt.plot(fpr_lr, tpr_lr, label = 'AUC_LR = %0.2f' % roc_auc)\nplt.plot(fpr_xg, tpr_xg, 'b', label = 'AUC_XG = %0.2f' % roc_auc_xg, color='orange')\nplt.plot(fpr_ada, tpr_ada, 'b', label = 'AUC_ADA = %0.2f' % roc_auc_ada, color='green')\nplt.legend(loc = 'lower right')\nplt.plot([0, 1], [0, 1],'r--')\nplt.xlim([0, 1])\nplt.ylim([0, 1])\nplt.ylabel('True Positive Rate')\nplt.xlabel('False Positive Rate')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:15:57.590367Z","iopub.execute_input":"2023-06-21T18:15:57.590668Z","iopub.status.idle":"2023-06-21T18:15:58.423709Z","shell.execute_reply.started":"2023-06-21T18:15:57.590632Z","shell.execute_reply":"2023-06-21T18:15:58.422268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## AdaBoost Classifier gave the best result on the given problem, however XGBoost is relatively faster.","metadata":{}},{"cell_type":"code","source":"test_data = pd.read_csv('../input/talkingdata-adtracking-fraud-detection/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:15:58.425267Z","iopub.execute_input":"2023-06-21T18:15:58.425562Z","iopub.status.idle":"2023-06-21T18:16:18.306801Z","shell.execute_reply.started":"2023-06-21T18:15:58.425536Z","shell.execute_reply":"2023-06-21T18:16:18.305728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()\ntest_data = fix_dataframe(test_data)\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:16:18.308226Z","iopub.execute_input":"2023-06-21T18:16:18.308798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:27:27.656731Z","iopub.execute_input":"2023-06-21T18:27:27.657104Z","iopub.status.idle":"2023-06-21T18:27:27.669783Z","shell.execute_reply.started":"2023-06-21T18:27:27.657079Z","shell.execute_reply":"2023-06-21T18:27:27.668491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ip_count = test_data.groupby('ip').size().reset_index(name='ip_count').astype('int64')\ndf = pd.merge(test_data, ip_count, on='ip', how='left', sort=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:27:30.335113Z","iopub.execute_input":"2023-06-21T18:27:30.335434Z","iopub.status.idle":"2023-06-21T18:27:34.031216Z","shell.execute_reply.started":"2023-06-21T18:27:30.335412Z","shell.execute_reply":"2023-06-21T18:27:34.029860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:27:35.751719Z","iopub.execute_input":"2023-06-21T18:27:35.752068Z","iopub.status.idle":"2023-06-21T18:27:35.765456Z","shell.execute_reply.started":"2023-06-21T18:27:35.752044Z","shell.execute_reply":"2023-06-21T18:27:35.764236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(columns=['click_id','ip','click_time'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:27:38.931171Z","iopub.execute_input":"2023-06-21T18:27:38.932335Z","iopub.status.idle":"2023-06-21T18:27:39.339229Z","shell.execute_reply.started":"2023-06-21T18:27:38.932301Z","shell.execute_reply":"2023-06-21T18:27:39.338259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:39:23.574122Z","iopub.execute_input":"2023-06-21T18:39:23.574351Z","iopub.status.idle":"2023-06-21T18:39:23.590072Z","shell.execute_reply.started":"2023-06-21T18:39:23.574331Z","shell.execute_reply":"2023-06-21T18:39:23.589158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_ada = ada_final.predict_proba(df)\nsubmission = pd.DataFrame()\nsubmission['click_id'] = test_data['click_id']\nsubmission['is_attributed'] = preds_ada[:, 1]\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-21T18:27:44.652636Z","iopub.execute_input":"2023-06-21T18:27:44.653965Z","iopub.status.idle":"2023-06-21T18:38:44.518224Z","shell.execute_reply.started":"2023-06-21T18:27:44.653865Z","shell.execute_reply":"2023-06-21T18:38:44.517281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-21T19:01:05.345977Z","iopub.execute_input":"2023-06-21T19:01:05.346396Z","iopub.status.idle":"2023-06-21T19:01:05.353934Z","shell.execute_reply.started":"2023-06-21T19:01:05.346366Z","shell.execute_reply":"2023-06-21T19:01:05.352470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2023-06-21T19:01:24.275495Z","iopub.execute_input":"2023-06-21T19:01:24.275887Z","iopub.status.idle":"2023-06-21T19:01:57.361866Z","shell.execute_reply.started":"2023-06-21T19:01:24.275859Z","shell.execute_reply":"2023-06-21T19:01:57.360796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 18790469 ","metadata":{},"execution_count":null,"outputs":[]}]}