{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","collapsed":true,"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np","execution_count":3,"outputs":[]},{"metadata":{"_cell_guid":"aa593346-74c3-4e2e-9348-0b1314c06144","_uuid":"09d954a6d37fbf8800826a7655a6a13e1dd9cd21"},"cell_type":"markdown","source":"## Pre-processing and feature engineering"},{"metadata":{"_cell_guid":"5dff3a99-5778-452e-81f4-d95cb38a5af8","_uuid":"caf70953925059a073546452b73f3cf8e57aeefc"},"cell_type":"markdown","source":"We engineer the following features:\n* extract the `click_hour`, `click_day` and `click_month` from the `click_time` datetime.\n* encode the time based features cyclically (for an explanation, see [here](https://www.kaggle.com/avanwyk/encoding-cyclical-features-for-deep-learning)).\n* aggregate clicks by `ip` (`total_clicks`), `ip` and `month` (`clicks_in_month`) and `ip` and `day` (`clicks_in_day`).\n* aggregate unique counts of `os`, `app`, `device` and `channel` by `ip`.\n\nSee https://www.kaggle.com/avanwyk/talkingdata-data-exploration-and-class-weights for the visualization of the features.\n\nWe are only training on a sample of the training data, due to resource constraints."},{"metadata":{"_cell_guid":"565ab8a7-674c-4016-8a7b-6236df90168a","_uuid":"977e07cc5eff9759ccfec206c8913be4f50c55fd","trusted":true,"collapsed":true},"cell_type":"code","source":"train_df = pd.read_csv('../input/train.csv', parse_dates=['click_time', 'attributed_time'], nrows=1000000)","execution_count":4,"outputs":[]},{"metadata":{"_cell_guid":"70f20329-a0bd-433a-92e1-0e55c6d13734","_uuid":"84ce8c51b4c3fe7909df88ce381fc7bb72f29f2e","collapsed":true,"trusted":true},"cell_type":"code","source":"def encode_cyclical(frame, col, max_val):\n    frame[col + '_sin'] = np.sin(2 * np.pi * frame[col]/max_val)\n    frame[col + '_cos'] = np.cos(2 * np.pi * frame[col]/max_val)\n    return frame\n\ndef create_click_aggregate(frame, name, idxs):\n    aggregate = frame.groupby(by=idxs, as_index=False).click_time.count()\n    aggregate = aggregate.rename(columns={'click_time': name})\n    return frame.merge(aggregate, on=idxs)\n\ndef unique_values_by_ip(frame, value):\n    n_values_by_ip = frame.groupby(by='ip')[value].nunique()\n    frame.set_index('ip', inplace=True)\n    frame['n_' + value] = n_values_by_ip\n    frame.reset_index(inplace=True)\n    return frame\n\ndef impute_features(df):\n    df['click_hour'] = df['click_time'].dt.hour + df['click_time'].dt.minute / 60\n    df['click_day'] = df['click_time'].dt.day\n    df['click_month'] = df['click_time'].dt.month\n    cyclical_features = [('click_hour', 24), ('click_day', 31), ('click_month', 12)]\n    for f in cyclical_features:\n        df = encode_cyclical(df, *f)\n        \n    df = create_click_aggregate(df, 'total_clicks', ['ip'])\n    df = create_click_aggregate(df, 'clicks_in_day', ['ip', 'click_month', 'click_day'])\n    df = create_click_aggregate(df, 'clicks_in_hour', ['ip', 'click_month', 'click_day', 'click_hour'])\n\n    df = unique_values_by_ip(df, 'os')\n    df = unique_values_by_ip(df, 'app')\n    df = unique_values_by_ip(df, 'device')\n    df = unique_values_by_ip(df, 'channel')\n    return df","execution_count":5,"outputs":[]},{"metadata":{"_cell_guid":"783843f3-255f-46cc-85c9-4c8784cc73e5","_uuid":"f32ce31d28f915c7f3ffccaadbbcb7d7bf04e574"},"cell_type":"markdown","source":"### Create datasets"},{"metadata":{"_cell_guid":"7b479836-f766-4f01-a623-bff6b1b9c62f","_uuid":"392e4220bb84f878aa9771602a0f0837d645eec0"},"cell_type":"markdown","source":"We can no create the training, validation and test datasets from the features imputed above. We use a training to validation set split of 2:1."},{"metadata":{"_cell_guid":"d13224fb-cf7b-4dce-bd9c-3ae533dbadba","_uuid":"26a145786b4a24f94a97b504558a7df5a9b56336","collapsed":true,"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nfeatures = ['ip', 'app', 'device', 'os', 'channel', 'click_hour_sin', 'click_hour_cos', 'click_day_sin',\n                'click_day_cos', 'click_month_sin', 'click_month_cos', 'total_clicks', 'clicks_in_day', 'clicks_in_hour', 'n_os', 'n_app', 'n_device', 'n_channel']\n\ndef create_dataset(df, test=False):\n    X = df[features].values\n    \n    if test:\n        ids = df['click_id']\n        return X, ids\n    \n    y = df['is_attributed'].values\n    \n    return X, y\n\ndef create_train_val_data(df):\n    imputed_df = impute_features(df)\n    X, y = create_dataset(imputed_df)\n    return train_test_split(X, y, test_size=0.33)\n\ndef create_test_data(df):\n    imputed_df = impute_features(df)\n    X, ids = create_dataset(imputed_df, test=True)\n    return X, ids","execution_count":6,"outputs":[]},{"metadata":{"_cell_guid":"f1ecc33b-5d11-4e31-b22d-cf39d872b65a","_uuid":"33cf36efaf2789c5be6a62500092c6cb5b51c59a","collapsed":true},"cell_type":"markdown","source":"## Baseline LightGBM model"},{"metadata":{"_cell_guid":"a9af8088-f81a-4c72-8da8-eaba7c7b245a","_uuid":"f68bf2975f51e1ff9c11337cd44500d76ab25a67"},"cell_type":"markdown","source":"As a baseline a LightGBM model is created. To compensate for the class imbalance we calculated the class weights for each instance."},{"metadata":{"_cell_guid":"844fb773-ea5f-4bf0-8e13-6a7bda5c06ce","_uuid":"1d58137018e1402efcadaaca6ae5fd1192d46550","collapsed":true,"trusted":true},"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.utils import class_weight","execution_count":7,"outputs":[]},{"metadata":{"_cell_guid":"672b264b-6e5c-491d-8a61-b3710b3118ae","_uuid":"fde2cb55dceb496495202b67bd40256ca1a9c2cb","collapsed":true,"trusted":true},"cell_type":"code","source":"X_train, X_val, y_train, y_val = create_train_val_data(train_df)","execution_count":8,"outputs":[]},{"metadata":{"_cell_guid":"c5ca350c-e691-4090-a18b-0bf22264b7d4","_uuid":"ab3922aa238de5e24bfc0ca787db6bd5bee79241","collapsed":true,"trusted":true},"cell_type":"code","source":"def weigh_instances(y):\n    class_weights = class_weight.compute_class_weight('balanced', np.unique(y), y)\n    y_weighted = y.copy().astype(float)\n    y_weighted[y==0] = class_weights[0]\n    y_weighted[y==1] = class_weights[1]\n    return y_weighted","execution_count":9,"outputs":[]},{"metadata":{"_cell_guid":"bb988637-1938-4e24-a03a-ea2aa437e88b","_uuid":"b4b61cb959825ebee2be4ec4bfabaae0d65f2ba6","collapsed":true,"scrolled":true,"trusted":true},"cell_type":"code","source":"y_train_weights = weigh_instances(y_train)","execution_count":10,"outputs":[]},{"metadata":{"_cell_guid":"6096fd6a-0b37-4ac8-bf97-79a46cef6b4e","_uuid":"cf3094cf41214456431a4e12bf25c6c837a60ff5","collapsed":true,"trusted":true},"cell_type":"code","source":"y_val_weights = weigh_instances(y_val)","execution_count":11,"outputs":[]},{"metadata":{"_cell_guid":"b1dae4b1-5c44-4648-8310-21d6668b76d4","_uuid":"096925b44b6e0b62d783b89e6b2f00184d7f4556","collapsed":true,"trusted":true},"cell_type":"code","source":"categorical_features = [idx for idx in range(0, 5)]","execution_count":12,"outputs":[]},{"metadata":{"_cell_guid":"617d3158-b0ee-4802-8991-e989532bb82f","_uuid":"a32e85e20284c7446eef3b938220a7e6efb597d1","collapsed":true,"trusted":true},"cell_type":"code","source":"lgb_train = lgb.Dataset(X_train, y_train, weight=y_train_weights,\n                        categorical_feature=categorical_features, free_raw_data=False)\nlgb_val = lgb.Dataset(X_val, y_val, weight=y_val_weights, reference=lgb_train,\n                       categorical_feature=categorical_features, free_raw_data=False)","execution_count":13,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"334336d8a63a82ace18f12f23c7e4b874179e5a1"},"cell_type":"code","source":"gbm = None","execution_count":22,"outputs":[]},{"metadata":{"_cell_guid":"b8839997-b27a-4cfd-846e-e561fb14c638","_uuid":"8976cd6658f1a4140f7bebd417bc888bd677399e","collapsed":true,"trusted":true},"cell_type":"code","source":"params = {\n    'boosting_type': 'gbdt',\n    'objective': 'binary',\n    'metric': 'binary_logloss',\n    'learning_rate': 0.01,\n    'num_leaves': 31,\n    'max_depth': -1,\n    'min_child_samples': 20,\n    'max_bin': 255,\n    'subsample': 0.6,\n    'subsample_freq': 0,\n    'colsample_bytree': 0.3,\n    'min_child_weight': 5,\n    'subsample_for_bin': 200000,\n    'min_split_gain': 0,\n    'reg_alpha': 0.99,\n    'reg_lambda': 0.9,\n    'nthread': 8,\n    'verbose': 0\n}","execution_count":27,"outputs":[]},{"metadata":{"_cell_guid":"4c5d891f-c124-469b-8bbd-c7227b23a0d4","_uuid":"9933d55e804615f1525130b9c424db4b60abe998","scrolled":true,"trusted":true},"cell_type":"code","source":"gbm = lgb.train(params,\n                lgb_train,\n                init_model=gbm,\n                num_boost_round=40,\n                valid_sets=lgb_val,\n                feature_name=features)","execution_count":28,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"ba1f8c40f178286b13b3e284947168a8d59fa2e4"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}