{"cells":[{"metadata":{},"cell_type":"markdown","source":"## Please select an option before submitting results to the competition"},{"metadata":{"trusted":true},"cell_type":"code","source":"submit_flag = True #False #True\nprint(submit_flag)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# TalkingData AdTracking Fraud Detection Challenge\n# Can you detect fraudulent click traffic for mobile app ads?\n# https://www.kaggle.com/c/talkingdata-adtracking-fraud-detection"},{"metadata":{},"cell_type":"markdown","source":"**This notebook is inspired by an exercise in the [Feature Engineering](https://www.kaggle.com/learn/feature-engineering) course**  \n**You can reference the tutorial at [this link](https://www.kaggle.com/matleonard/feature-generation)**  \n**You can reference my notebook at [this link](https://www.kaggle.com/georgezoto/feature-engineering-feature-generation)**  \n\n---\n"},{"metadata":{},"cell_type":"markdown","source":"<center><a href=\"https://www.kaggle.com/c/talkingdata-adtracking-fraud-detection\"><img src=\"https://i.imgur.com/srKxEkD.png\" width=600px></a></center>"},{"metadata":{},"cell_type":"markdown","source":"# Introduction\n\nIn this set of exercises, you'll create new features from the existing data. Again you'll compare the score lift for each new feature compared to a baseline model. First off, run the cells below to set up a baseline dataset and model."},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn import preprocessing, metrics\nimport lightgbm as lgb\n\nimport matplotlib.pyplot as plt\nplt.rcParams[\"figure.figsize\"] = (16,9)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create features from   timestamps\nclick_data = pd.read_csv('../input/feature-engineering-data/train_sample.csv', \n                         parse_dates=['click_time'])\nclick_times = click_data['click_time']\nclicks = click_data.assign(day=click_times.dt.day.astype('uint8'),\n                           hour=click_times.dt.hour.astype('uint8'),\n                           minute=click_times.dt.minute.astype('uint8'),\n                           second=click_times.dt.second.astype('uint8'))\n\n# Label encoding for categorical features\ncat_features = ['ip', 'app', 'device', 'os', 'channel']\nfor feature in cat_features:\n    label_encoder = preprocessing.LabelEncoder()\n    clicks[feature] = label_encoder.fit_transform(clicks[feature])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clicks.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clicks.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clicks['is_attributed'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clicks['is_attributed'].value_counts(normalize=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Competition data"},{"metadata":{"trusted":true},"cell_type":"code","source":"#Read only first limit rows\nlimit = 20_000_000\n\n#Read only these columns - skip attributed_time \nusecols = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_data = pd.read_csv('../input/talkingdata-adtracking-fraud-detection/train.csv', \n                               nrows=limit, \n                               usecols=usecols, \n                               parse_dates=['click_time'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_data['is_attributed'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_data['is_attributed'].value_counts(normalize=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_test_data = pd.read_csv('../input/talkingdata-adtracking-fraud-detection/test.csv', \n                                    parse_dates=['click_time'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Add new columns for timestamp features day, hour, minute, and second\ncompetition_test_data = competition_test_data.copy()\ncompetition_test_data['day'] = competition_test_data['click_time'].dt.day.astype('uint8')\n# Fill in the rest\ncompetition_test_data['hour'] = competition_test_data['click_time'].dt.hour.astype('uint8')\ncompetition_test_data['minute'] = competition_test_data['click_time'].dt.minute.astype('uint8')\ncompetition_test_data['second'] = competition_test_data['click_time'].dt.second.astype('uint8')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_test_data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_test_data.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Helpful content packed methods used throughout the notebook 😀"},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_data_splits(dataframe, valid_fraction=0.1):\n\n    dataframe = dataframe.sort_values('click_time')\n    valid_rows = int(len(dataframe) * valid_fraction)\n    train = dataframe[:-valid_rows * 2]\n    # valid size == test size, last two sections of the data\n    valid = dataframe[-valid_rows * 2:-valid_rows]\n    test = dataframe[-valid_rows:]\n    \n    return train, valid, test\n\ndef train_model(train, valid, test=None, feature_cols=None, valid_name_model='Baseline Model'):\n    if feature_cols is None:\n        feature_cols = train.columns.drop(['click_time', 'attributed_time',\n                                           'is_attributed'])\n    dtrain = lgb.Dataset(train[feature_cols], label=train['is_attributed'])\n    dvalid = lgb.Dataset(valid[feature_cols], label=valid['is_attributed'])\n    \n    param = {'num_leaves': 64, 'objective': 'binary', \n             'metric': 'auc', 'seed': 7}\n    num_round = 1000\n    \n    #Record eval results for plotting\n    validation_metrics = {} \n    \n    print(\"Training model. Hold on a minute to see the validation score\")\n    bst = lgb.train(param, dtrain, num_round, valid_sets=[dvalid], valid_names=valid_name_model,\n                    early_stopping_rounds=20, evals_result=validation_metrics, verbose_eval=False)\n    \n    valid_pred = bst.predict(valid[feature_cols])\n    valid_score = metrics.roc_auc_score(valid['is_attributed'], valid_pred)\n    print(f\"Validation AUC score: {valid_score}\")\n    \n    if test is not None: \n        test_pred = bst.predict(test[feature_cols])\n        test_score = metrics.roc_auc_score(test['is_attributed'], test_pred)\n        return bst, valid_score, test_score, validation_metrics\n    else:\n        return bst, valid_score, validation_metrics","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def my_own_train_plot_model(clicks, valid_name_model, my_own_metrics):\n    #valid_name_model='V11 FI Numerical ip_past_6hr_counts Model'\n    print(valid_name_model+' score')\n\n    train, valid, test = get_data_splits(clicks)\n    bst, valid_score, validation_metrics = train_model(train, valid, valid_name_model=valid_name_model)\n\n    my_own_metrics[valid_name_model] = valid_score\n    print(my_own_metrics)\n    plot_model_information(bst, validation_metrics, my_own_metrics)\n    \n    return bst","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Model information"},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_model_information(bst, validation_metrics, my_own_metrics):\n    print('Number of trees:', bst.num_trees())\n    \n    print('Plot model performance')\n    ax = lgb.plot_metric(validation_metrics, metric='auc');\n    plt.show()\n    \n    print('Plot feature importances...')\n    ax = lgb.plot_importance(bst, max_num_features=15)\n    plt.show()\n    \n    def plot_my_own_metrics(my_own_metrics):\n        x=list(my_own_metrics.keys())\n        y=list(my_own_metrics.values())\n        plt.barh(x, y);\n\n        for index, value in enumerate(y):\n            plt.text(value, index, str(value))\n\n    print('plot_my_own_metrics')    \n    plot_my_own_metrics(my_own_metrics)\n    plt.show()\n    \n    tree_index = 0\n    print('Plot '+str(tree_index)+'th tree...')  # one tree use categorical feature to split\n    ax = lgb.plot_tree(bst, tree_index=tree_index, figsize=(64, 36), show_info=['split_gain'])\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_own_metrics = {}\nvalid_name_model='Baseline LightGBM Model'\nbst = my_own_train_plot_model(clicks, valid_name_model, my_own_metrics)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 1) Add interaction features\n\nHere you'll add interaction features for each pair of categorical features (ip, app, device, os, channel). The easiest way to iterate through the pairs of features is with `itertools.combinations`. For each new column, join the values as strings with an underscore, so 13 and 47 would become `\"13_47\"`. As you add the new columns to the dataset, be sure to label encode the values."},{"metadata":{"trusted":true},"cell_type":"code","source":"clicks.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_test_data.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Data leakage using clicks/entire dataset to LabelEncode ???"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Not the best solution to ValueError: y contains previously unseen labels: [0, 1, 2,...\nunknown_value = -1 #Make sure this is int (as other labels) or you will not be able to predict in the end ⚠️","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import itertools\n\ncat_features = ['ip', 'app', 'device', 'os', 'channel']\ninteractions = pd.DataFrame(index=clicks.index)\n\n# Iterate through each pair of features, combine them into interaction features\nfor interaction_feature_tuple in itertools.combinations(cat_features,2):\n    #New feature name as concatination of 2 categorical features\n    interaction_feature  = '_'.join(list(interaction_feature_tuple))\n    print(interaction_feature_tuple, interaction_feature)\n    \n    #New interaction as concatination of the values of each combination of cateforical features\n    interactions_values = clicks[interaction_feature_tuple[0]].astype(str) + '_' + clicks[interaction_feature_tuple[1]].astype(str)\n    \n    #New label encoder for each interaction_feature \n    label_enc = preprocessing.LabelEncoder()\n    #interactions = interactions.assign(interaction_feature=label_enc.fit_transform(interactions_values)) ??? uses the string interaction_feature as the column name ???\n    #interactions[interaction_feature] = label_enc.fit_transform(interactions_values)                     #??? index values and how do they relate to the full dataset clicks ???\n\n    #Fit on all possible values of this feature\n    label_enc.fit(interactions_values)\n    #Create LabelEncoder of input to output\n    le_dict = dict(zip(label_enc.classes_, label_enc.transform(label_enc.classes_)))\n    #Encode unseen values to the unknown_value label\n    encoded = interactions_values.apply(lambda x: le_dict.get(x, unknown_value))\n    clicks[interaction_feature] = encoded\n    \n    print('clicks.head()')\n    print(clicks.head())\n\n    #Competition submission\n    # Apply encoding to the competition test dataset\n    comp_interactions_values = competition_test_data[interaction_feature_tuple[0]].astype(str) + '_' + competition_test_data[interaction_feature_tuple[1]].astype(str)\n    #competition_test_data[interaction_feature] = label_enc.transform(comp_interactions_values)  #??? ValueError: y contains previously unseen labels: '119901_9' ???\n    \n    competition_encoded = comp_interactions_values.apply(lambda x: le_dict.get(x, unknown_value))\n    competition_test_data[interaction_feature] = competition_encoded\n    print('competition_test_data.head()')\n    print(competition_test_data.head())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## How many unknown_value did we get in the test dataset?"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_ip_labels_unknowns = sum(clicks['ip_app'] == unknown_value)\ntrain_ip_labels_unknowns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"compet_test_ip_labels_unknowns = sum(competition_test_data['ip_app'] == unknown_value)\ncompet_test_ip_labels_unknowns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_name_model='V10 FI Categorical Model'\nmy_own_train_plot_model(clicks, valid_name_model, my_own_metrics)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Generating numerical features\n\nAdding interactions is a quick way to create more categorical features from the data. It's also effective to create new numerical features, you'll typically get a lot of improvement in the model. This takes a bit of brainstorming and experimentation to find features that work well.\n\nFor these exercises I'm going to have you implement functions that operate on Pandas Series. It can take multiple minutes to run these functions on the entire data set so instead I'll provide feedback by running your function on a smaller dataset."},{"metadata":{},"cell_type":"markdown","source":"### 2) Number of events in the past six hours\n\nThe first feature you'll be creating is the number of events from the same IP in the last six hours. It's likely that someone who is visiting often will download the app.\n\nImplement a function `count_past_events` that takes a Series of click times (timestamps) and returns another Series with the number of events in the last six hours. **Tip:** The `rolling` method is useful for this."},{"metadata":{"trusted":true},"cell_type":"code","source":"def count_past_events(series):\n    new_series = pd.Series(index=series, data=series.index, name=\"count_6_hours\").sort_index()\n    print(new_series.head())\n    count_6_hours = new_series.rolling('6h').count() - 1\n    return count_6_hours","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Because this can take a while to calculate on the full data, we'll load pre-calculated versions in the cell below to test model performance."},{"metadata":{"trusted":true},"cell_type":"code","source":"# Loading in from saved Parquet file\npast_events = pd.read_parquet('../input/feature-engineering-data/past_6hr_events.pqt')\nclicks['ip_past_6hr_counts'] = past_events\n\n#train, valid, test = get_data_splits(clicks)\n#_ = train_model(train, valid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_name_model='V11 FIN ip_past_6hr_counts'\nmy_own_train_plot_model(clicks, valid_name_model, my_own_metrics)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 3) Features from future information\n\nIn the last exercise you created a feature that looked at past events. You could also make features that use information from events in the future. Should you use future events or not? "},{"metadata":{},"cell_type":"markdown","source":"### 4) Time since last event\n\nImplement a function `time_diff` that calculates the time since the last event in seconds from a Series of timestamps. This will be ran like so:\n\n```python\ntimedeltas = clicks.groupby('ip')['click_time'].transform(time_diff)\n```"},{"metadata":{"trusted":true},"cell_type":"code","source":"def time_diff(series):\n    \"\"\"Returns a series with the time since the last timestamp in seconds.\"\"\"\n    time_since_last_event = series.diff().dt.total_seconds()\n    return time_since_last_event","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We'll again load pre-computed versions of the data, which match what your function would return"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Loading in from saved Parquet file\npast_events = pd.read_parquet('../input/feature-engineering-data/time_deltas.pqt')\nclicks['past_events_6hr'] = past_events\n\n#train, valid, test = get_data_splits(clicks.join(past_events))\n#_ = train_model(train, valid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_name_model='V12 FIN time_since_last_event'\nmy_own_train_plot_model(clicks, valid_name_model, my_own_metrics)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 5) Number of previous app downloads\n\nIt's likely that if a visitor downloaded an app previously, it'll affect the likelihood they'll download one again. Implement a function `previous_attributions` that returns a Series with the number of times an app has been downloaded (`'is_attributed' == 1`) before the current event."},{"metadata":{"trusted":true},"cell_type":"code","source":"def previous_attributions(series):\n    \"\"\"Returns a series with the number of times an app has been downloaded.\"\"\"\n    print(series)\n    print(series.expanding(min_periods=2).sum())\n    sums = series.expanding(min_periods=2).sum() - series\n    return sums","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Again loading pre-computed data."},{"metadata":{"trusted":true},"cell_type":"code","source":"# Loading in from saved Parquet file\npast_events = pd.read_parquet('../input/feature-engineering-data/downloads.pqt')\n#clicks['ip_past_6hr_counts'] = past_events ??? Typo to overwrite ???\nclicks['prev_app_downloads'] = past_events \n       \n#train, valid, test = get_data_splits(clicks)\n#_ = train_model(train, valid)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_name_model='V13 FIN prev_app_downloads'\nmy_own_train_plot_model(clicks, valid_name_model, my_own_metrics)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### 6) Tree-based vs Neural Network Models\n\nSo far we've been using LightGBM, a tree-based model. Would these features we've generated work well for neural networks as well as tree-based models?"},{"metadata":{},"cell_type":"markdown","source":"Now that you've generated a bunch of different features, you'll typically want to remove some of them to reduce the size of the model and potentially improve the performance. Next, I'll show you how to do feature selection using a few different methods such as L1 regression and Boruta."},{"metadata":{},"cell_type":"markdown","source":"# Keep Going\n\nYou know how to generate a lot of features. In practice, you'll frequently want to pare them down for modeling. Learn to do that in the **[Feature Selection lesson](https://www.kaggle.com/matleonard/feature-selection)**."},{"metadata":{},"cell_type":"markdown","source":"---\n\n\n\n\n*Have questions or comments? Visit the [Learn Discussion forum](https://www.kaggle.com/learn-forum/161443) to chat with other Learners.*"},{"metadata":{},"cell_type":"markdown","source":"## Constrain additional training features to just categorical feature engineering features"},{"metadata":{"trusted":true},"cell_type":"code","source":"clicks.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clicks.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_test_data.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_test_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_name_model='V10 Feature Eng Categorical Model'\nbst = my_own_train_plot_model(clicks, valid_name_model, my_own_metrics)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_cols = clicks.columns.drop(['click_time', 'attributed_time','is_attributed'])\nfeature_cols","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Submit test predictions to TalkingData AdTracking Fraud Detection Challenge competition using the limited train.csv records from this notebook"},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_test_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bst","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_predictions = bst.predict(competition_test_data[feature_cols])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_predictions_df = pd.DataFrame(competition_predictions, columns=['is_attributed'])\ncompetition_predictions_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"competition_predictions_df['click_id'] = competition_test_data['click_id']\ncompetition_predictions_df = competition_predictions_df[['click_id', 'is_attributed']]\ncompetition_predictions_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.cut(competition_predictions_df['is_attributed'], bins=10).value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.cut(competition_predictions_df['is_attributed'], bins=10).value_counts().plot(kind='bar', rot=45);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"if submit_flag == True:\n    competition_predictions_df.to_csv('submission.csv', index=False)\n    print('submission.csv generated successfully :)')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}