{"metadata":{"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":8540,"databundleVersionId":862041,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.13"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2024-05-30T03:16:37.02816Z","iopub.status.busy":"2024-05-30T03:16:37.027794Z","iopub.status.idle":"2024-05-30T03:16:45.182582Z","shell.execute_reply":"2024-05-30T03:16:45.181614Z","shell.execute_reply.started":"2024-05-30T03:16:37.028132Z"}}},{"cell_type":"code","source":"import os\nimport optuna\nimport gc\nimport math\nimport time\nimport warnings\nimport numpy as np  \nimport pandas as pd  \nimport dask\nimport dask.dataframe as dd\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom tqdm import tqdm\nfrom scipy.stats import mode\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import (\n    GradientBoostingClassifier,\n    RandomForestClassifier,\n    AdaBoostClassifier,\n    StackingClassifier,\n)\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.calibration import CalibratedClassifierCV\nfrom sklearn.preprocessing import MinMaxScaler, LabelEncoder, StandardScaler\nfrom sklearn.utils import resample\nfrom sklearn.model_selection import (\n    cross_val_score,\n    train_test_split,\n    RandomizedSearchCV,\n)\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.metrics import (\n    log_loss,\n    roc_curve,\n    accuracy_score,\n    roc_auc_score,\n    f1_score,\n)\nimport xgboost as xgb\nimport lightgbm as lgb\nfrom catboost import CatBoostClassifier, Pool\nfrom skopt import BayesSearchCV\nimport torch\nfrom imblearn.over_sampling import SMOTE\n\n# Adjust plot settings\nplt.rcParams[\"figure.figsize\"] = [16, 9]\n\n# Ignore warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Iterate through files in the specified directory\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:11:23.892854Z","iopub.status.busy":"2024-05-30T17:11:23.892559Z","iopub.status.idle":"2024-05-30T17:11:33.257392Z","shell.execute_reply":"2024-05-30T17:11:33.256436Z","shell.execute_reply.started":"2024-05-30T17:11:23.892827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the data types for each column in the dataset\ndtypes = {\n    'ip': 'uint32',\n    'app': 'object',\n    'device': 'object',\n    'os': 'object',\n    'channel': 'object',\n    'click_time': 'object',\n    'is_attributed': 'uint8',\n}","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:11:33.259996Z","iopub.status.busy":"2024-05-30T17:11:33.259246Z","iopub.status.idle":"2024-05-30T17:11:33.264809Z","shell.execute_reply":"2024-05-30T17:11:33.263889Z","shell.execute_reply.started":"2024-05-30T17:11:33.259962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the columns to be used from the dataset\nusecols = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:11:33.266365Z","iopub.status.busy":"2024-05-30T17:11:33.266016Z","iopub.status.idle":"2024-05-30T17:11:33.275168Z","shell.execute_reply":"2024-05-30T17:11:33.274331Z","shell.execute_reply.started":"2024-05-30T17:11:33.266335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Data","metadata":{}},{"cell_type":"code","source":"# Define the path to the competition dataset\nfile_path = '../input/talkingdata-adtracking-fraud-detection/train.csv'\n\n# Read the dataset using Dask, specifying data types and columns to use, and parsing date columns\ncompetition_data = dd.read_csv(\n    file_path,\n    dtype=dtypes,\n    # nrows=limit,  # Note: `nrows` is not supported by `dd.read_csv`\n    usecols=usecols,\n    parse_dates=['click_time']\n)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:11:33.278095Z","iopub.status.busy":"2024-05-30T17:11:33.277459Z","iopub.status.idle":"2024-05-30T17:11:33.346423Z","shell.execute_reply":"2024-05-30T17:11:33.345729Z","shell.execute_reply.started":"2024-05-30T17:11:33.278047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display information about the competition_data DataFrame\ncompetition_data.info()","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:11:33.347631Z","iopub.status.busy":"2024-05-30T17:11:33.347359Z","iopub.status.idle":"2024-05-30T17:11:33.35741Z","shell.execute_reply":"2024-05-30T17:11:33.356504Z","shell.execute_reply.started":"2024-05-30T17:11:33.347607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filter the dataset to include only rows where the 'is_attributed' column is equal to 1\ncompetition_data_positive = competition_data[competition_data.is_attributed == 1]","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:11:33.35888Z","iopub.status.busy":"2024-05-30T17:11:33.358536Z","iopub.status.idle":"2024-05-30T17:11:33.366233Z","shell.execute_reply":"2024-05-30T17:11:33.365451Z","shell.execute_reply.started":"2024-05-30T17:11:33.358848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Record the start time\nstart_time = time.time()\n\n# Check the type of the filtered dataset\ndata_type = type(competition_data_positive)\n\n# Print the type of the filtered dataset\nprint(data_type)\n\n# Calculate and print the elapsed time\nprint(\"--- %s seconds ---\" % (time.time() - start_time))","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:11:33.367571Z","iopub.status.busy":"2024-05-30T17:11:33.367287Z","iopub.status.idle":"2024-05-30T17:11:33.375248Z","shell.execute_reply":"2024-05-30T17:11:33.374212Z","shell.execute_reply.started":"2024-05-30T17:11:33.367542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Record the start time\nstart_time = time.time()\n\n# Compute the Dask DataFrame to convert it into a Pandas DataFrame\ncompetition_data_positive = competition_data_positive.compute()\n\n# Calculate and print the elapsed time\nprint(\"--- %s seconds ---\" % (time.time() - start_time))","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:11:33.376967Z","iopub.status.busy":"2024-05-30T17:11:33.376598Z","iopub.status.idle":"2024-05-30T17:20:34.115944Z","shell.execute_reply":"2024-05-30T17:20:34.114919Z","shell.execute_reply.started":"2024-05-30T17:11:33.376937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_data_positive.sample(10)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:23:50.177993Z","iopub.status.busy":"2024-05-30T17:23:50.177568Z","iopub.status.idle":"2024-05-30T17:23:50.212829Z","shell.execute_reply":"2024-05-30T17:23:50.211935Z","shell.execute_reply.started":"2024-05-30T17:23:50.17796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_data_negative = competition_data[competition_data.is_attributed == 0] ","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:23:54.704742Z","iopub.status.busy":"2024-05-30T17:23:54.704375Z","iopub.status.idle":"2024-05-30T17:23:54.711864Z","shell.execute_reply":"2024-05-30T17:23:54.710861Z","shell.execute_reply.started":"2024-05-30T17:23:54.704712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_data_negative = competition_data_negative.sample(frac=0.0025) #number","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:23:56.594692Z","iopub.status.busy":"2024-05-30T17:23:56.594315Z","iopub.status.idle":"2024-05-30T17:23:56.605464Z","shell.execute_reply":"2024-05-30T17:23:56.604552Z","shell.execute_reply.started":"2024-05-30T17:23:56.594663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_time = time.time()\ncompetition_data_negative = competition_data_negative.compute()\nprint(\"--- %s seconds ---\" % (time.time() - start_time))","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:23:58.014692Z","iopub.status.busy":"2024-05-30T17:23:58.01384Z","iopub.status.idle":"2024-05-30T17:33:37.134401Z","shell.execute_reply":"2024-05-30T17:33:37.133194Z","shell.execute_reply.started":"2024-05-30T17:23:58.014655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_data_negative.sample(10)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:33:37.137358Z","iopub.status.busy":"2024-05-30T17:33:37.136528Z","iopub.status.idle":"2024-05-30T17:33:37.16094Z","shell.execute_reply":"2024-05-30T17:33:37.160012Z","shell.execute_reply.started":"2024-05-30T17:33:37.137321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_data = pd.concat([competition_data_positive, competition_data_negative])","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:21.15085Z","iopub.status.busy":"2024-05-30T17:39:21.149891Z","iopub.status.idle":"2024-05-30T17:39:21.253399Z","shell.execute_reply":"2024-05-30T17:39:21.252401Z","shell.execute_reply.started":"2024-05-30T17:39:21.150806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train, test, train_labels, test_labels = train_test_split(competition_data, competition_data.is_attributed, test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:25.465182Z","iopub.status.busy":"2024-05-30T17:39:25.464792Z","iopub.status.idle":"2024-05-30T17:39:25.85377Z","shell.execute_reply":"2024-05-30T17:39:25.852691Z","shell.execute_reply.started":"2024-05-30T17:39:25.465151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering ","metadata":{}},{"cell_type":"markdown","source":"## Click_time column","metadata":{}},{"cell_type":"code","source":"# Add new columns for timestamp features day, hour, minute, and second\ntrain['click_time'] = pd.to_datetime(train['click_time'])\ntrain['day'] = train['click_time'].dt.day.astype('uint8')\n# Fill in the rest\ntrain['hour'] = train['click_time'].dt.hour.astype('uint8')\ntrain['minute'] = train['click_time'].dt.minute.astype('uint8')\ntrain['second'] = train['click_time'].dt.second.astype('uint8')","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:28.234457Z","iopub.status.busy":"2024-05-30T17:39:28.234122Z","iopub.status.idle":"2024-05-30T17:39:28.536891Z","shell.execute_reply":"2024-05-30T17:39:28.536109Z","shell.execute_reply.started":"2024-05-30T17:39:28.234432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add new columns for timestamp features day, hour, minute, and second\ntest['click_time'] = pd.to_datetime(test['click_time'])\ntest['day'] = test['click_time'].dt.day.astype('uint8')\n# Fill in the rest\ntest['hour'] = test['click_time'].dt.hour.astype('uint8')\ntest['minute'] = test['click_time'].dt.minute.astype('uint8')\ntest['second'] = test['click_time'].dt.second.astype('uint8')","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:29.916722Z","iopub.status.busy":"2024-05-30T17:39:29.916341Z","iopub.status.idle":"2024-05-30T17:39:30.062933Z","shell.execute_reply":"2024-05-30T17:39:30.062038Z","shell.execute_reply.started":"2024-05-30T17:39:29.916692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# store the label\ntrain_labels = train.is_attributed.values\ntest_labels = test.is_attributed.values\n\n# drop labels and attributed_time since it represnets the same info as the is_attributed\ntrain.drop(labels = ['click_time','is_attributed'], axis = 1, inplace = True)\ntest.drop(labels = ['click_time', 'is_attributed'], axis = 1, inplace = True)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:31.366667Z","iopub.status.busy":"2024-05-30T17:39:31.36599Z","iopub.status.idle":"2024-05-30T17:39:31.528843Z","shell.execute_reply":"2024-05-30T17:39:31.527814Z","shell.execute_reply.started":"2024-05-30T17:39:31.366637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.fillna(train.max())\ntest = test.fillna(test.max())","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:33.474448Z","iopub.status.busy":"2024-05-30T17:39:33.473584Z","iopub.status.idle":"2024-05-30T17:39:34.720663Z","shell.execute_reply":"2024-05-30T17:39:34.71985Z","shell.execute_reply.started":"2024-05-30T17:39:33.474412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate():\n  \n  fpr, tpr, _ = roc_curve(test_labels,  y_pred_proba)\n\n  # Plot the ROC curve\n  plt.plot(fpr, tpr, color='blue', label='ROC Curve')\n\n  # Fill below the ROC curve with color\n  plt.fill_between(fpr, 0, tpr, color='skyblue', alpha=0.2)\n\n  # Plot the diagonal line (y=x) for reference\n  plt.plot([0, 1], [0, 1], color='red', linestyle='--', label='Random Chance')\n\n  # Add labels and title\n  plt.ylabel('True Positive Rate')\n  plt.xlabel('False Positive Rate')\n  plt.title('ROC Curve')\n    \n\n  # Show the plot\n  plt.show()","metadata":{"execution":{"iopub.execute_input":"2024-05-30T18:01:38.39273Z","iopub.status.busy":"2024-05-30T18:01:38.392339Z","iopub.status.idle":"2024-05-30T18:01:38.399322Z","shell.execute_reply":"2024-05-30T18:01:38.3984Z","shell.execute_reply.started":"2024-05-30T18:01:38.392699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = []","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:37.264504Z","iopub.status.busy":"2024-05-30T17:39:37.263582Z","iopub.status.idle":"2024-05-30T17:39:37.269603Z","shell.execute_reply":"2024-05-30T17:39:37.268794Z","shell.execute_reply.started":"2024-05-30T17:39:37.264469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train, test, train_labels, test_labels = train_test_split(competition_data, competition_data.is_attributed, test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:39.297495Z","iopub.status.busy":"2024-05-30T17:39:39.297127Z","iopub.status.idle":"2024-05-30T17:39:39.692487Z","shell.execute_reply":"2024-05-30T17:39:39.691678Z","shell.execute_reply.started":"2024-05-30T17:39:39.297466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:41.595711Z","iopub.status.busy":"2024-05-30T17:39:41.595019Z","iopub.status.idle":"2024-05-30T17:39:42.008537Z","shell.execute_reply":"2024-05-30T17:39:42.007584Z","shell.execute_reply.started":"2024-05-30T17:39:41.595677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add new columns for timestamp features day, hour, minute, and second\ntrain['click_time'] = pd.to_datetime(train['click_time'])\ntrain['day'] = train['click_time'].dt.day.astype('uint8')\n# Fill in the rest\ntrain['hour'] = train['click_time'].dt.hour.astype('uint8')\ntrain['minute'] = train['click_time'].dt.minute.astype('uint8')\ntrain['second'] = train['click_time'].dt.second.astype('uint8')\n\n# Add new columns for timestamp features day, hour, minute, and second\ntest['click_time'] = pd.to_datetime(test['click_time'])\ntest['day'] = test['click_time'].dt.day.astype('uint8')\n# Fill in the rest\ntest['hour'] = test['click_time'].dt.hour.astype('uint8')\ntest['minute'] = test['click_time'].dt.minute.astype('uint8')\ntest['second'] = test['click_time'].dt.second.astype('uint8')","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:44.170112Z","iopub.status.busy":"2024-05-30T17:39:44.169711Z","iopub.status.idle":"2024-05-30T17:39:44.617616Z","shell.execute_reply":"2024-05-30T17:39:44.616526Z","shell.execute_reply.started":"2024-05-30T17:39:44.170078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Previous click","metadata":{}},{"cell_type":"code","source":"def do_prev_Click( df,agg_suffix='prevClick', agg_type='float32'):\n\n    print(f\">> \\nExtracting {agg_suffix} time calculation features...\\n\")\n    \n    GROUP_BY_NEXT_CLICKS = [\n \n    {'groupby': ['ip', 'channel']},\n    {'groupby': ['ip', 'os']},\n  \n    ]\n\n    # Calculate the time to next click for each group\n    for spec in GROUP_BY_NEXT_CLICKS:\n    \n       # Name of new feature\n        new_feature = '{}_{}'.format('_'.join(spec['groupby']),agg_suffix)    \n    \n        # Unique list of features to select\n        all_features = spec['groupby'] + ['click_time']\n\n        # Run calculation\n        print(f\">> Grouping by {spec['groupby']}, and saving time to {agg_suffix} in: {new_feature}\")\n        df[new_feature] = (df.click_time - df[all_features].groupby(spec[\n                'groupby']).click_time.shift(+1) ).dt.seconds.astype(agg_type)\n        \n        gc.collect()\n    return (df)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:46.416077Z","iopub.status.busy":"2024-05-30T17:39:46.415354Z","iopub.status.idle":"2024-05-30T17:39:46.423543Z","shell.execute_reply":"2024-05-30T17:39:46.422421Z","shell.execute_reply.started":"2024-05-30T17:39:46.416026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = do_prev_Click(train,agg_suffix='prevClick', agg_type='float32'  )\ngc.collect()\ntest = do_prev_Click(test,agg_suffix='prevClick', agg_type='float32'  )\ngc.collect()","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:48.322832Z","iopub.status.busy":"2024-05-30T17:39:48.322134Z","iopub.status.idle":"2024-05-30T17:39:54.88936Z","shell.execute_reply":"2024-05-30T17:39:54.888442Z","shell.execute_reply.started":"2024-05-30T17:39:48.322801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train\ntrain['day'] = train['click_time'].dt.day.astype('uint8')\ntrain['hour'] = train['click_time'].dt.hour.astype('uint8')\n\n# test\ntest['day'] = test['click_time'].dt.day.astype('uint8')\ntest['hour'] = test['click_time'].dt.hour.astype('uint8')\n\ndf_concat = pd.concat([train, test], ignore_index=True)\ndf_concat['day'] = df_concat['click_time'].dt.day.astype('uint8')\ndf_concat['hour'] = df_concat['click_time'].dt.hour.astype('uint8')","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:54.8912Z","iopub.status.busy":"2024-05-30T17:39:54.890914Z","iopub.status.idle":"2024-05-30T17:39:55.081853Z","shell.execute_reply":"2024-05-30T17:39:55.080847Z","shell.execute_reply.started":"2024-05-30T17:39:54.891176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Most frequent hours","metadata":{}},{"cell_type":"code","source":"most_freq_hours_in_test_data = [4, 5, 9, 10, 13, 14]\nleast_freq_hours_in_test_data = [6, 11, 15]\n\ntrain['in_test_hh'] = (\n    3 - 2 * train.hour.isin(most_freq_hours_in_test_data)\n      - 1 * train.hour.isin(least_freq_hours_in_test_data)\n).astype('uint8')\n\ntest['in_test_hh'] = (\n    3 - 2 * test.hour.isin(most_freq_hours_in_test_data)\n      - 1 * test.hour.isin(least_freq_hours_in_test_data)\n).astype('uint8')\n\ndf_concat['in_test_hh'] = (\n    3 - 2 * df_concat.hour.isin(most_freq_hours_in_test_data)\n      - 1 * df_concat.hour.isin(least_freq_hours_in_test_data)\n).astype('uint8')","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:55.083321Z","iopub.status.busy":"2024-05-30T17:39:55.083002Z","iopub.status.idle":"2024-05-30T17:39:55.409965Z","shell.execute_reply":"2024-05-30T17:39:55.409019Z","shell.execute_reply.started":"2024-05-30T17:39:55.083296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"group = ['ip', 'day', 'in_test_hh']\ndf = df_concat.groupby(group).size().astype('uint16')\ndf = pd.DataFrame(df, columns=['ip_day_test_hh_clicks']).reset_index()\ntrain = train.merge(df, how='left', on=group)\ntest = test.merge(df, how='left', on=group)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:55.411947Z","iopub.status.busy":"2024-05-30T17:39:55.411657Z","iopub.status.idle":"2024-05-30T17:39:56.001933Z","shell.execute_reply":"2024-05-30T17:39:56.000948Z","shell.execute_reply.started":"2024-05-30T17:39:55.411923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ip_app_device_clicks\ngroup = ['ip', 'app', 'device']\ndf = df_concat.groupby(group).size().astype('uint16')\ndf = pd.DataFrame(df, columns=['ip_app_device_clicks']).reset_index()\ntrain = train.merge(df, how='left', on=group)\ntest = test.merge(df, how='left', on=group)\n\n# ip_app_device_day_clicks\ngroup = ['ip', 'app', 'device', 'day']\ndf = df_concat.groupby(group).size().astype('uint16')\ndf = pd.DataFrame(df, columns=['ip_app_device_day_clicks']).reset_index()\ntrain = train.merge(df, how='left', on=group)\ntest = test.merge(df, how='left', on=group)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:39:56.333029Z","iopub.status.busy":"2024-05-30T17:39:56.332172Z","iopub.status.idle":"2024-05-30T17:39:59.35188Z","shell.execute_reply":"2024-05-30T17:39:59.350818Z","shell.execute_reply.started":"2024-05-30T17:39:56.332987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# single 'ip' grouper for ['app', 'device', 'channel']\ncount_cols = ['app', 'device', 'channel', 'hour']\ngroup = 'ip'\nfor col in count_cols:\n    df = df_concat.groupby(group)[col].nunique().astype('uint16')\n    df.name = 'ip_nunique_{}'.format(col)\n    df = pd.DataFrame(df).reset_index()\n    train = train.merge(df, how='left', on=group)\n    test = test.merge(df, how='left', on=group)\n    #####################\n    # self-add\n    group_day = ['ip', 'day']\n    df = df_concat.groupby(group_day)[col].nunique().astype('uint16')\n    df.name = 'ip_day_nunique_{}'.format(col)\n    df = pd.DataFrame(df).reset_index()\n    train = train.merge(df, how='left', on=group_day)\n    test = test.merge(df, how='left', on=group_day)\n    #####################\n    \n# single 'app' grouper for 'channel'\ngroup = 'app'\ncol = 'channel'\ndf = df_concat.groupby(group)[col].nunique().astype('uint16')\ndf.name = 'app_nunique_{}'.format(col)\ndf = pd.DataFrame(df).reset_index()\ntrain = train.merge(df, how='left', on=group)\ntest = test.merge(df, how='left', on=group)\n#####################\n\n# self-add\ngroup_day = ['app', 'day']\ncol = 'channel'\ndf = df_concat.groupby(group_day)[col].nunique().astype('uint16')\ndf.name = 'app_day_nunique_{}'.format(col)\ndf = pd.DataFrame(df).reset_index()\ntrain = train.merge(df, how='left', on=group_day)\ntest = test.merge(df, how='left', on=group_day)\n#####################\n\n# duble ['ip', 'app'] grouper for 'os'\ngroup = ['ip', 'app']\ncol = 'os'\ndf = df_concat.groupby(group)[col].nunique().astype('uint16')\ndf.name = 'ip_app_nunique_{}'.format(col)\ndf = pd.DataFrame(df).reset_index()\ntrain = train.merge(df, how='left', on=group)\ntest = test.merge(df, how='left', on=group)\n#####################\n\n# self-add\ngroup_day = ['ip', 'app', 'day']\ncol = 'os'\ndf = df_concat.groupby(group_day)[col].nunique().astype('uint16')\ndf.name = 'ip_app_day_nunique_{}'.format(col)\ndf = pd.DataFrame(df).reset_index()\ntrain = train.merge(df, how='left', on=group_day)\ntest = test.merge(df, how='left', on=group_day)\n#####################\n\n# triple ['ip', 'device', 'os'] grouper for 'app'\ngroup = ['ip', 'device', 'os']\ncol = 'app'\ndf = df_concat.groupby(group)[col].nunique().astype('uint16')\ndf.name = 'ip_device_os_nunique_{}'.format(col)\ndf = pd.DataFrame(df).reset_index()\ntrain = train.merge(df, how='left', on=group)\ntest = test.merge(df, how='left', on=group)\n#####################\n\n# self-add\ngroup_day = ['ip', 'device', 'os'] + ['day']\ncol = 'app'\ndf = df_concat.groupby(group_day)[col].nunique().astype('uint16')\ndf.name = 'ip_device_os_day_nunique_{}'.format(col)\ndf = pd.DataFrame(df).reset_index()\ntrain = train.merge(df, how='left', on=group_day)\ntest = test.merge(df, how='left', on=group_day)\n#####################","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:40:00.701703Z","iopub.status.busy":"2024-05-30T17:40:00.70089Z","iopub.status.idle":"2024-05-30T17:40:10.725989Z","shell.execute_reply":"2024-05-30T17:40:10.724855Z","shell.execute_reply.started":"2024-05-30T17:40:00.701672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define all the groupby transformations\nGROUPBY_AGGREGATIONS = [\n    # Variance in day, for ip-app-channel\n    {'groupby': ['ip','app','channel'], 'select': 'day', 'agg': 'var', 'type': 'float32'},\n    # Variance in day, for ip-app-device\n    {'groupby': ['ip','app','device'], 'select': 'day', 'agg': 'var', 'type': 'float32'},\n    # Variance in day, for ip-app-os\n    {'groupby': ['ip','app','os'], 'select': 'day', 'agg': 'var', 'type': 'float32'},\n   # Count, for ip-day-hour\n#     {'groupby': ['ip','day','hour'], 'select': 'channel', 'agg': 'count', 'type': 'float32'},\n    # Count, for ip-day-hour\n     {'groupby': ['channel'], 'select': 'day', 'agg': 'var', 'type': 'float32'},\n    # Count, for ip-day-hour\n   {'groupby': ['os'], 'select': 'hour', 'agg': 'var', 'type': 'float32'},\n    {'groupby': ['ip','app','os','device'], 'select': 'hour', 'agg': 'var', 'type': 'float32'},\n    {'groupby': ['ip','app','os'], 'select': 'hour', 'agg': 'var', 'type': 'float32'},\n\n    # Mean hour, for ip-app-channel\n    {'groupby': ['ip','app','channel'], 'select': 'hour', 'agg': 'mean', 'type': 'float32', 'type': 'float32'},\n    {'groupby': ['channel'], 'select': 'hour', 'agg': 'mean', 'type': 'float32', 'type': 'float32'}\n    \n    \n    \n]\n# Apply all the groupby transformations\nfor spec in GROUPBY_AGGREGATIONS:\n    print(f\"Grouping by {spec['groupby']}, and aggregating {spec['select']} with {spec['agg']}\")\n    \n    # Unique list of features to select\n    all_features = list(set(spec['groupby'] + [spec['select']]))\n    # Name of new feature\n    new_feature = '{}_{}'.format('_'.join(spec['groupby']), spec['agg'])\n     # Perform the groupby\n    gp = train[all_features]. \\\n        groupby(spec['groupby'])[spec['select']]. \\\n        agg(spec['agg']). \\\n        reset_index(). \\\n        rename(index=str, columns={spec['select']: new_feature}).astype(spec['type'])\n    gp = test[all_features]. \\\n        groupby(spec['groupby'])[spec['select']]. \\\n        agg(spec['agg']). \\\n        reset_index(). \\\n        rename(index=str, columns={spec['select']: new_feature}).astype(spec['type'])","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:40:10.72861Z","iopub.status.busy":"2024-05-30T17:40:10.728308Z","iopub.status.idle":"2024-05-30T17:40:17.344858Z","shell.execute_reply":"2024-05-30T17:40:17.343892Z","shell.execute_reply.started":"2024-05-30T17:40:10.728585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gp['channel'] = gp['channel'].astype('object')\ngp['channel_mean'] = gp['channel_mean'].astype('object')\n# Merge back to X_train\ntrain = train.merge(gp, on=spec['groupby'], how='left')\ntest = test.merge(gp, on=spec['groupby'], how='left')\ndel gp\nprint(\"End\")","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:40:17.346429Z","iopub.status.busy":"2024-05-30T17:40:17.346131Z","iopub.status.idle":"2024-05-30T17:40:17.753262Z","shell.execute_reply":"2024-05-30T17:40:17.752294Z","shell.execute_reply.started":"2024-05-30T17:40:17.346404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LDA","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.decomposition import LatentDirichletAllocation\n\ndef modelTopicsWithDevice(df, ip_column, app_column, device_column):\n    \"\"\"Model LDA topics for each IP based on apps and devices.\"\"\"\n    # Assuming vocab selection is based on app usage frequency\n    vocab = df[app_column].value_counts().nlargest(10).index.tolist()\n\n    # Aggregate app and device values for each IP\n    ip_activities = {}\n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        ip = row[ip_column]\n        app = row[app_column]\n        device = str(row[device_column])  # Convert device to string if it's not already\n        activity = f\"{app}_{device}\"  # Concatenate app and device information\n\n        if app in vocab:\n            ip_activities.setdefault(ip, []).append(activity)\n\n    ips = list(ip_activities.keys())\n    sentences = [' '.join(ip_activities[ip]) for ip in ips]\n    \n    \n    # Vectorize the app and device sentences\n    cv = CountVectorizer()\n    dt_matrix = cv.fit_transform(sentences)\n    \n    # Apply LDA\n    lda = LatentDirichletAllocation(n_components=5, random_state=42)\n    ip_topics = lda.fit_transform(dt_matrix)\n    \n    # Create a DataFrame for the topics\n    topic_cols = [f\"{ip_column}_{app_column}_{device_column}_topic{i+1}\" for i in range(5)]\n    topic_df = pd.DataFrame(ip_topics, columns=topic_cols)\n    topic_df[ip_column] = ips\n    \n    # Merge the topics back to the original DataFrame\n    df = df.merge(topic_df, on=ip_column, how='left')\n    \n    return df","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:40:17.756556Z","iopub.status.busy":"2024-05-30T17:40:17.755907Z","iopub.status.idle":"2024-05-30T17:40:17.778042Z","shell.execute_reply":"2024-05-30T17:40:17.777119Z","shell.execute_reply.started":"2024-05-30T17:40:17.756512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = modelTopicsWithDevice(train, 'ip', 'app', 'device')\ntest = modelTopicsWithDevice(test, 'ip', 'app', 'device')","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:40:17.779789Z","iopub.status.busy":"2024-05-30T17:40:17.779447Z","iopub.status.idle":"2024-05-30T17:45:36.135194Z","shell.execute_reply":"2024-05-30T17:45:36.134254Z","shell.execute_reply.started":"2024-05-30T17:40:17.779759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(['channel_mean'], axis=1, inplace=True)\ntest.drop(['channel_mean'], axis=1, inplace=True)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:45:36.136664Z","iopub.status.busy":"2024-05-30T17:45:36.136393Z","iopub.status.idle":"2024-05-30T17:45:36.294112Z","shell.execute_reply":"2024-05-30T17:45:36.293009Z","shell.execute_reply.started":"2024-05-30T17:45:36.13664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.fillna(train.max())\ntest = test.fillna(test.max())","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:45:36.295863Z","iopub.status.busy":"2024-05-30T17:45:36.295522Z","iopub.status.idle":"2024-05-30T17:45:37.829914Z","shell.execute_reply":"2024-05-30T17:45:37.829116Z","shell.execute_reply.started":"2024-05-30T17:45:36.295832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop('click_time', axis=1, inplace=True)\ntest.drop('click_time', axis=1, inplace=True)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:45:37.831213Z","iopub.status.busy":"2024-05-30T17:45:37.830932Z","iopub.status.idle":"2024-05-30T17:45:38.047898Z","shell.execute_reply":"2024-05-30T17:45:38.047081Z","shell.execute_reply.started":"2024-05-30T17:45:37.83119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Iterative variables","metadata":{}},{"cell_type":"code","source":"# Not the best solution to ValueError: y contains previously unseen labels: [0, 1, 2,...\nunknown_value = -1 #Make sure this is int (as other labels) or you will not be able to predict in the end ⚠️\n\nfrom sklearn import preprocessing\n\ncat_features = ['ip', 'app', 'device', 'os', 'channel']\n#cat_features = ['ip']\n\n#encoder = preprocessing.LabelEncoder() - Incorrect, we need a label encoder for each feature\n# Create new columns in clicks using preprocessing.LabelEncoder()\n\nfor feature in cat_features:\n    start_time = time.time()\n    print(feature)\n    \n    #New encoder for each feature\n    encoder = preprocessing.LabelEncoder()\n    \n    #Fit on all possible values of this feature\n    encoder.fit(train[feature])\n    \n    #Create LabelEncoder of input to output\n    le_dict = dict(zip(encoder.classes_, encoder.transform(encoder.classes_)))\n    \n    #Encode unseen values to the unknown_value label\n    encoded = train[feature].apply(lambda x: le_dict.get(x, unknown_value))\n    train[feature+'_labels'] = encoded\n    \n    #Competition submission\n    competition_encoded = test[feature].apply(lambda x: le_dict.get(x, unknown_value))\n    #ValueError: y contains previously unseen labels: [0, 2, 3, 4, 5,\n    test[feature+'_labels'] = competition_encoded\n    \n    print(\"--- %s seconds ---\" % (time.time() - start_time))","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:45:38.049335Z","iopub.status.busy":"2024-05-30T17:45:38.049035Z","iopub.status.idle":"2024-05-30T17:45:56.595013Z","shell.execute_reply":"2024-05-30T17:45:56.594048Z","shell.execute_reply.started":"2024-05-30T17:45:38.049312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ip_labels_unknowns = sum(train['ip_labels'] == unknown_value)\ntrain_ip_labels_unknowns","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:45:56.598382Z","iopub.status.busy":"2024-05-30T17:45:56.598015Z","iopub.status.idle":"2024-05-30T17:45:56.676847Z","shell.execute_reply":"2024-05-30T17:45:56.675954Z","shell.execute_reply.started":"2024-05-30T17:45:56.598357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compet_test_ip_labels_unknowns = sum(test['ip_labels'] == unknown_value)\ncompet_test_ip_labels_unknowns","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:45:56.678341Z","iopub.status.busy":"2024-05-30T17:45:56.677994Z","iopub.status.idle":"2024-05-30T17:45:56.719581Z","shell.execute_reply":"2024-05-30T17:45:56.718821Z","shell.execute_reply.started":"2024-05-30T17:45:56.678308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_own_metrics={'clicks': train.shape[0], #min(limit, clicks.shape[0]),\n                'competition_test_data':test.shape[0],\n                'train ip_labels unknowns': train_ip_labels_unknowns,\n                'compet_test ip_labels unknowns':compet_test_ip_labels_unknowns,\n                'compet_test ip_labels unknowns %': round(100*compet_test_ip_labels_unknowns/test.shape[0],2)}\nmy_own_metrics","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:45:56.720806Z","iopub.status.busy":"2024-05-30T17:45:56.720542Z","iopub.status.idle":"2024-05-30T17:45:56.727979Z","shell.execute_reply":"2024-05-30T17:45:56.727038Z","shell.execute_reply.started":"2024-05-30T17:45:56.720784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import itertools\nfrom sklearn import preprocessing\n\ncat_features = ['ip', 'app', 'device', 'os', 'channel']\ninteractions = pd.DataFrame(index=train.index)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:45:56.729361Z","iopub.status.busy":"2024-05-30T17:45:56.729087Z","iopub.status.idle":"2024-05-30T17:45:56.739012Z","shell.execute_reply":"2024-05-30T17:45:56.738288Z","shell.execute_reply.started":"2024-05-30T17:45:56.729339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iterate through each pair of features, combine them into interaction features\nfor interaction_feature_tuple in itertools.combinations(cat_features,2):\n    #New feature name as concatination of 2 categorical features\n    interaction_feature  = '_'.join(list(interaction_feature_tuple))\n    print(interaction_feature_tuple, interaction_feature)\n    \n    #New interaction as concatination of the values of each combination of cateforical features\n    interactions_values = train[interaction_feature_tuple[0]].astype(str) + '_' + train[interaction_feature_tuple[1]].astype(str)\n    \n    #New label encoder for each interaction_feature \n    label_enc = preprocessing.LabelEncoder()\n    #interactions = interactions.assign(interaction_feature=label_enc.fit_transform(interactions_values)) ??? uses the string interaction_feature as the column name ???\n    #interactions[interaction_feature] = label_enc.fit_transform(interactions_values)                     #??? index values and how do they relate to the full dataset clicks ???\n\n    #Fit on all possible values of this feature\n    label_enc.fit(interactions_values)\n    #Create LabelEncoder of input to output\n    le_dict = dict(zip(label_enc.classes_, label_enc.transform(label_enc.classes_)))\n    #Encode unseen values to the unknown_value label\n    encoded = interactions_values.apply(lambda x: le_dict.get(x, unknown_value))\n    train[interaction_feature] = encoded\n    \n    print('train.columns')\n    print(train.columns)\n    \n    \n    #Competition submission\n    # Apply encoding to the competition test dataset\n    comp_interactions_values = test[interaction_feature_tuple[0]].astype(str) + '_' + test[interaction_feature_tuple[1]].astype(str)\n    #competition_test_data[interaction_feature] = label_enc.transform(comp_interactions_values)  #??? ValueError: y contains previously unseen labels: '119901_9' ???\n    \n    competition_encoded = comp_interactions_values.apply(lambda x: le_dict.get(x, unknown_value))\n    test[interaction_feature] = competition_encoded\n    print('test.columns')\n    print(test.columns)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:45:56.740415Z","iopub.status.busy":"2024-05-30T17:45:56.740141Z","iopub.status.idle":"2024-05-30T17:46:39.671027Z","shell.execute_reply":"2024-05-30T17:46:39.670106Z","shell.execute_reply.started":"2024-05-30T17:45:56.740392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:46:39.672556Z","iopub.status.busy":"2024-05-30T17:46:39.672256Z","iopub.status.idle":"2024-05-30T17:46:39.679357Z","shell.execute_reply":"2024-05-30T17:46:39.678491Z","shell.execute_reply.started":"2024-05-30T17:46:39.67253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.columns","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:46:39.680755Z","iopub.status.busy":"2024-05-30T17:46:39.680447Z","iopub.status.idle":"2024-05-30T17:46:39.692208Z","shell.execute_reply":"2024-05-30T17:46:39.691298Z","shell.execute_reply.started":"2024-05-30T17:46:39.680731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Drop variables","metadata":{}},{"cell_type":"code","source":"train.drop('channel', axis=1, inplace=True)\ntest.drop('channel', axis=1, inplace=True)\ntrain.drop('device', axis=1, inplace=True)\ntest.drop('device', axis=1, inplace=True)\ntrain.drop('ip', axis=1, inplace=True)\ntest.drop('ip', axis=1, inplace=True)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:46:39.693626Z","iopub.status.busy":"2024-05-30T17:46:39.693361Z","iopub.status.idle":"2024-05-30T17:46:40.327093Z","shell.execute_reply":"2024-05-30T17:46:40.326082Z","shell.execute_reply.started":"2024-05-30T17:46:39.693604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### SMOTE","metadata":{}},{"cell_type":"code","source":"# initialize SMOTE method\nsm = SMOTE(random_state=42)\n# ytrain = train.drop('is_attributed',axis=1)\ntrain_balanced,train_labels_balanced = sm.fit_resample(train.drop('is_attributed',axis=1),train['is_attributed'])\nprint(\"Dimension of X_train_sm Shape:\", train_balanced.shape)\nprint(\"Dimension of y_train_sm Shape:\", train_labels_balanced.shape)","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:46:40.328749Z","iopub.status.busy":"2024-05-30T17:46:40.328425Z","iopub.status.idle":"2024-05-30T17:50:35.094013Z","shell.execute_reply":"2024-05-30T17:50:35.092969Z","shell.execute_reply.started":"2024-05-30T17:46:40.328722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count the number of occurrences of each label\nlabel_counts = train_labels_balanced.value_counts()\n\n# Display the counts\nprint(\"Number of '0' labels:\", label_counts[0])\nprint(\"Number of '1' labels:\", label_counts[1])","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:50:35.095766Z","iopub.status.busy":"2024-05-30T17:50:35.095465Z","iopub.status.idle":"2024-05-30T17:50:35.106375Z","shell.execute_reply":"2024-05-30T17:50:35.105409Z","shell.execute_reply.started":"2024-05-30T17:50:35.095739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Standard Scale","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\ntrain_balanced = scaler.fit_transform(train_balanced)\ntest = scaler.transform(test.drop('is_attributed',axis=1))","metadata":{"execution":{"iopub.execute_input":"2024-05-30T17:50:35.107702Z","iopub.status.busy":"2024-05-30T17:50:35.107448Z","iopub.status.idle":"2024-05-30T17:50:46.326351Z","shell.execute_reply":"2024-05-30T17:50:46.325308Z","shell.execute_reply.started":"2024-05-30T17:50:35.10768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Machine Learning Models","metadata":{}},{"cell_type":"markdown","source":"### 1. Before Tuning - AdaBoost","metadata":{}},{"cell_type":"code","source":"# base estimator\ntree = DecisionTreeClassifier()\n\n# adaboost with the tree as base estimator\n# learning rate is arbitrarily set, we'll discuss learning_rate below\nABC_model = AdaBoostClassifier()\n#     base_estimator=tree,\n#     learning_rate=1.0,\n#     n_estimators=300,\n#     algorithm=\"SAMME\")\n\n# Perform cross-validation\ncv_scores = cross_val_score(ABC_model, train_balanced, train_labels_balanced, cv=5, scoring='roc_auc')\n\n# Train the model on the entire training set\nABC_model.fit(train_balanced, train_labels_balanced)\n\n# Predictions on the test set\nABC_y_pred = ABC_model.predict(test)\ny_pred_proba = ABC_model.predict_proba(test)[:, 1]\n\n# Calculate accuracy and AUC Score on the test set\nf1 = f1_score(test_labels, ABC_y_pred)\nABC_auc_score = roc_auc_score(test_labels, y_pred_proba)\n\nprint(f\"Cross-Validation ROC-AUC Scores: {cv_scores}\")\nprint(f\"Average Cross-Validation ROC-AUC Score: {np.mean(cv_scores)}\")\n\nevaluate()\nresults.append({\"Model\": 'AdaBoost', \"F1 Score\": f1, \"AUC\": ABC_auc_score})","metadata":{"execution":{"iopub.execute_input":"2024-05-31T02:26:49.795237Z","iopub.status.busy":"2024-05-31T02:26:49.794304Z","iopub.status.idle":"2024-05-31T02:26:49.828709Z","shell.execute_reply":"2024-05-31T02:26:49.827215Z","shell.execute_reply.started":"2024-05-31T02:26:49.7952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. After Tuning - AdaBoost","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n    hyperparameters = {\n        'n_estimators': trial.suggest_int('n_estimators', 50, 500),\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 2.0),\n        'algorithm': trial.suggest_categorical('algorithm', ['SAMME', 'SAMME.R'])\n    }\n    \n    model = AdaBoostClassifier(**hyperparameters)\n    scores = cross_val_score(model, train_balanced, train_labels_balanced, cv=5, scoring='roc_auc')\n    return np.mean(scores)\n\n# Create and run the study\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=100)\n\n# Print best trial details\nbest_params = study.best_params\nprint(\"Best params found:\", best_params)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params = study.best_params\n\n# Initialize AdaBoost Classifier with best parameters\nada_model = AdaBoostClassifier(**best_params)\n\n# Train the model on the entire training set\nada_model.fit(train_balanced, train_labels_balanced)\n\n# Predictions on the test set\nada_y_pred = ada_model.predict(test)\ny_pred_proba = ada_model.predict_proba(test)[:, 1]\n\n# Calculate F1-score\nf1 = f1_score(test_labels, ada_y_pred)\nauc_score = roc_auc_score(test_labels, y_pred_proba)\n\n# Print the F1-score and AUC score\nprint(f\"F1 Score: {f1}\")\nprint(f\"AUC Score: {auc_score}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Initialize Random Forest Classifier\n# rf_model = RandomForestClassifier()\n# #     n_estimators=100,\n# #     max_depth=20,\n# #     min_samples_split=2,\n# #     min_samples_leaf=1,\n# #     max_features='sqrt',\n# #     bootstrap=False,\n# #     criterion='gini'\n# # )\n\n# # Perform cross-validation\n# cv_scores = cross_val_score(rf_model, train_balanced, train_labels_balanced, cv=5, scoring='roc_auc')\n\n# # Train the model on the entire training set\n# rf_model.fit(train_balanced, train_labels_balanced)\n\n# # Predictions on the test set\n# rf_y_pred = rf_model.predict(test)\n# y_pred_proba = rf_model.predict_proba(test)[:, 1]\n\n# # Calculate accuracy and AUC Score on the test set\n# f1 = f1_score(test_labels, cat_y_pred)\n# rf_auc_score = roc_auc_score(test_labels, y_pred_proba)\n\n# print(f\"Cross-Validation ROC-AUC Scores: {cv_scores}\")\n# print(f\"Average Cross-Validation ROC-AUC Score: {np.mean(cv_scores)}\")\n\n# evaluate()\n# results.append({\"Model\": 'Random Forest', \"F1 Score\": f1, \"AUC\": rf_auc_score})","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 6. CatBoost","metadata":{}},{"cell_type":"code","source":"# from catboost import CatBoostClassifier\n\n# # Manually select parameters from the parameter grid\n# # params = {\n# #     'depth': 10,\n# #     'learning_rate': 0.2,\n# #     'iterations': 500,\n# #     'l2_leaf_reg': 3,\n# #     'border_count': 50,\n# #     'bagging_temperature': 0.04019881483063329,\n# #     'loss_function': 'Logloss',\n# #     'od_type': 'Iter',\n# #     'od_wait': 50,\n# #     'verbose': False\n# # }\n\n# # Initialize CatBoost Classifier with selected parameters\n# cat_model = CatBoostClassifier()\n# # (**params)\n\n# # Perform cross-validation\n# cv_scores = cross_val_score(cat_model, train_balanced, train_labels_balanced, cv=5, scoring='roc_auc')\n\n# # Train the model on the entire training set\n# cat_model.fit(train_balanced, train_labels_balanced)\n\n# # Predictions on the test set\n# cat_y_pred = cat_model.predict(test)\n# y_pred_proba = cat_model.predict_proba(test)[:, 1]\n\n# # Calculate F1-score\n# f1 = f1_score(test_labels, cat_y_pred)\n# cat_auc_score = roc_auc_score(test_labels, y_pred_proba)\n\n\n# # Print the F1-score (Note: No need for a probability array for F1-score)\n# print(f\"F1 Score: {f1}\")\n# print(f\"AUC Score: {cat_auc_score}\")\n# print(f\"Cross-Validation ROC-AUC Scores: {cv_scores}\")\n# print(f\"Average Cross-Validation ROC-AUC Score: {np.mean(cv_scores)}\")\n\n# evaluate()\n# results.append({\"Model\": 'CatBoost', \"F1 Score\": f1, \"AUC\": cat_auc_score})","metadata":{"execution":{"iopub.status.busy":"2024-05-30T17:20:34.470471Z","iopub.status.idle":"2024-05-30T17:20:34.470803Z","shell.execute_reply":"2024-05-30T17:20:34.470655Z","shell.execute_reply.started":"2024-05-30T17:20:34.470641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7. AdaBoosting Classifier","metadata":{}},{"cell_type":"code","source":"# # base estimator\n# tree = DecisionTreeClassifier()\n\n# # adaboost with the tree as base estimator\n# # learning rate is arbitrarily set, we'll discuss learning_rate below\n# ABC_model = AdaBoostClassifier()\n# #     base_estimator=tree,\n# #     learning_rate=1.0,\n# #     n_estimators=300,\n# #     algorithm=\"SAMME\")\n\n# # Perform cross-validation\n# cv_scores = cross_val_score(ABC_model, train_balanced, train_labels_balanced, cv=5, scoring='roc_auc')\n\n# # Train the model on the entire training set\n# ABC_model.fit(train_balanced, train_labels_balanced)\n\n# # Predictions on the test set\n# ABC_y_pred = ABC_model.predict(test)\n# y_pred_proba = ABC_model.predict_proba(test)[:, 1]\n\n# # Calculate accuracy and AUC Score on the test set\n# f1 = f1_score(test_labels, ABC_y_pred)\n# ABC_auc_score = roc_auc_score(test_labels, y_pred_proba)\n\n# # print(f\"Cross-Validation ROC-AUC Scores: {cv_scores}\")\n# # print(f\"Average Cross-Validation ROC-AUC Score: {np.mean(cv_scores)}\")\n\n# evaluate()\n# results.append({\"Model\": 'AdaBoost', \"F1 Score\": f1, \"AUC\": ABC_auc_score})","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.ensemble import StackingClassifier\n\n# # Define base models\n# base_models = [\n#     ('Logistic Regression', lr_model),\n#     ('XGBoost', xgb_model),\n#     ('Gradient Boosting', gb_model),\n#     ('LightGBM', lgb_model),\n#     ('CatBoost', cat_model),\n#     ('RF', rf_model),\n#     ('AdaBoost', ABC_model)\n# ]\n\n# # Define meta-model\n# meta_model = LogisticRegression()\n\n# # Initialize Stacking Classifier\n# stacked_model = StackingClassifier(estimators=base_models, final_estimator=meta_model, cv=5)\n\n# # Train the stacked model (Note: The pre-trained model won't be re-trained)\n# stacked_model.fit(train_balanced, train_labels_balanced)\n\n# # Predictions and evaluation\n# y_pred = stacked_model.predict(test)\n# y_pred_proba = stacked_model.predict_proba(test)[:, 1]  # Probabilities for the positive class\n\n# # Calculate accuracy and AUC Score\n# f1 = f1_score(test_labels, y_pred)\n# auc_score = roc_auc_score(test_labels, y_pred_proba)\n\n# print(f\"AUC Score: {auc_score}\")\n# evaluate()\n# results.append({\"Model\": 'Stacked Model', \"F1 Score\": f1, \"AUC\": auc_score})","metadata":{"execution":{"iopub.status.busy":"2024-05-30T17:20:34.475113Z","iopub.status.idle":"2024-05-30T17:20:34.475439Z","shell.execute_reply":"2024-05-30T17:20:34.475292Z","shell.execute_reply.started":"2024-05-30T17:20:34.475279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 8. Majority Vote","metadata":{}},{"cell_type":"code","source":"# from scipy.stats import mode\n\n# # Stack predictions for ease of calculation\n# stacked_predictions = np.column_stack((gb_y_pred, xgb_y_pred, lgb_y_pred, cat_y_pred, ABC_y_pred))\n\n# # Perform majority voting\n# majority_votes = mode(stacked_predictions, axis=1)[0]\n\n# # Flatten to get a 1D array of final predictions\n# final_predictions = np.ravel(majority_votes)\n\n# # Calculate accuracy and AUC Score\n# f1 = f1_score(test_labels, final_predictions)\n# auc_score = roc_auc_score(test_labels, final_predictions)\n# print(f\"AUC Score: {auc_score}\")\n# evaluate()\n# results.append({\"Model\": 'Majority Vote', \"F1 Score\": f1, \"AUC\": auc_score})","metadata":{"execution":{"iopub.status.busy":"2024-05-30T17:20:34.476662Z","iopub.status.idle":"2024-05-30T17:20:34.476965Z","shell.execute_reply":"2024-05-30T17:20:34.476826Z","shell.execute_reply.started":"2024-05-30T17:20:34.476814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 9. XGBoost + Random Forest\n\n","metadata":{}},{"cell_type":"code","source":"# # Define base models\n# from sklearn.ensemble import RandomForestClassifier\n# import xgboost as xgb\n\n# # Define and train XGBoost model\n# xgb_model = xgb.XGBClassifier(n_estimators=100, random_state=42)\n# xgb_model.fit(train_balanced, train_labels_balanced)  \n\n# # Define and train Random Forest model\n# rf_model = RandomForestClassifier(n_estimators=100, random_state=42)\n# rf_model.fit(train_balanced, train_labels_balanced) \n\n# base_models = [\n#     ('XGBoost', xgb_model),\n#     ('Random Forest', rf_model),\n# ]\n\n# # Define meta-model\n# meta_model = LogisticRegression()\n\n# # Initialize Stacking Classifier\n# stacked_model = StackingClassifier(estimators=base_models, final_estimator=meta_model, cv=5)\n\n# # Train the stacked model (Note: The pre-trained model won't be re-trained)\n# stacked_model.fit(train_balanced, train_labels_balanced)\n\n# # Predictions and evaluation\n# y_pred = stacked_model.predict(test)\n# y_pred_proba = stacked_model.predict_proba(test)[:, 1]  # Probabilities for the positive class\n\n# # Calculate accuracy and AUC Score\n# f1 = f1_score(test_labels, y_pred)\n# auc_score = roc_auc_score(test_labels, y_pred_proba)\n\n# print(f\"AUC Score: {auc_score}\")\n# results.append({\"Model\": 'XGBoost + Random Forest', \"F1 Score\": f1, \"AUC\": auc_score})\n# evaluate()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T17:20:34.478021Z","iopub.status.idle":"2024-05-30T17:20:34.478362Z","shell.execute_reply":"2024-05-30T17:20:34.478218Z","shell.execute_reply.started":"2024-05-30T17:20:34.478204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# results = pd.DataFrame(results)\n# results","metadata":{"execution":{"iopub.status.busy":"2024-05-30T17:20:34.479352Z","iopub.status.idle":"2024-05-30T17:20:34.479666Z","shell.execute_reply":"2024-05-30T17:20:34.479523Z","shell.execute_reply.started":"2024-05-30T17:20:34.479501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import optuna\n# from sklearn.ensemble import GradientBoostingClassifier\n\n# def objective(trial):\n#     hyperparameters = {\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.5),\n#         'n_estimators': trial.suggest_int('n_estimators', 50, 500),\n#         'max_depth': trial.suggest_int('max_depth', 3, 10),\n#         'min_samples_split': trial.suggest_int('min_samples_split', 2, 20),\n#         'min_samples_leaf': trial.suggest_int('min_samples_leaf', 1, 20),\n#         'subsample': trial.suggest_uniform('subsample', 0.5, 1.0)\n#     }\n    \n#     model = GradientBoostingClassifier(**hyperparameters)\n#     scores = cross_val_score(model, train_balanced, train_labels_balanced, cv=5, scoring='roc_auc')\n#     return np.mean(scores)\n\n# study = optuna.create_study(direction='maximize')\n# study.optimize(objective, n_trials=10) \n\n\n# best_params = study.best_params\n# print(\"Best params found:\", best_params)","metadata":{},"execution_count":null,"outputs":[]}]}