{"nbformat_minor": 1, "metadata": {"language_info": {"file_extension": ".py", "version": "3.6.3", "codemirror_mode": {"version": 3, "name": "ipython"}, "mimetype": "text/x-python", "name": "python", "pygments_lexer": "ipython3", "nbconvert_exporter": "python"}, "kernelspec": {"display_name": "Python 3", "name": "python3", "language": "python"}}, "cells": [{"metadata": {"_uuid": "7d76e8d6f55dd51420857a11cba092c840b911a4", "_cell_guid": "5216d274-8031-4b2f-9dc8-f2351992eaf5"}, "source": ["Note that the autoencoder code are borrowed from the following notebook: https://github.com/curiousily/Credit-Card-Fraud-Detection-using-Autoencoders-in-Keras/blob/master/fraud_detection.ipynb\n", "\n", "The code used for summary statistics / dtype fixing belongs to ZihaoXu."], "cell_type": "markdown"}, {"source": ["# important packages to import\n", "import pandas as pd\n", "import numpy as np\n", "import tensorflow as tf\n", "import matplotlib.pyplot as plt\n", "import seaborn as sns\n", "import sys, os\n", "import pickle\n", "import gc; gc.enable()\n", "\n", "from scipy import stats\n", "from pylab import rcParams\n", "from keras.models import Model, load_model\n", "from keras.layers import Input, Dense\n", "from keras.callbacks import ModelCheckpoint, TensorBoard\n", "from keras import regularizers\n", "from matplotlib import offsetbox\n", "from matplotlib.ticker import NullFormatter\n", "from sklearn import preprocessing, cross_validation, svm, manifold\n", "from sklearn.cross_validation import cross_val_score, KFold\n", "from sklearn.metrics import roc_curve, roc_auc_score, auc\n", "from sklearn.ensemble import RandomForestClassifier # Load scikit's random forest classifier library\n", "from sklearn.grid_search import GridSearchCV\n", "from time import time\n", "from datetime import datetime, timedelta\n", "from collections import defaultdict\n", "from multiprocessing import Pool, cpu_count\n", "\n", "from subprocess import check_output\n", "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))"], "metadata": {"collapsed": true, "_uuid": "429a16b108ea544970eed55093467fa23b0254ec", "_cell_guid": "596c0675-6567-459c-9fde-f08060ae37bd"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# Helper Functions\n", "\n", "# holistic summary of the given data set. \n", "# \"remove_bad_rowCol\" can be turned on to remove non-informative col / row\n", "def holistic_summary(df, remove_bad_rowCol = False, verbose = True):\n", "    # remove non-informative columns\n", "    if(remove_bad_rowCol):\n", "        df = df.drop(df.columns[df.isnull().sum() >= .9 * len(df)], axis = 1)\n", "        df = df.drop(df.index[df.isnull().sum(axis = 1) >= .5* len(df.columns)], axis = 0)\n", "        \n", "    # fix column names:\n", "    df.columns = [c.replace(\" \", \"_\").lower() for c in df.columns]\n", "    \n", "    print('***************************************************************')\n", "    print('Begin holistic summary: ')\n", "    print('***************************************************************\\n')\n", "    \n", "    print('Dimension of df: ' + str(df.shape))\n", "    print('Percentage of good observations: ' + str(1 - df.isnull().any(axis = 1).sum()/len(df)))\n", "    print('---------------------------------------------------------------\\n')\n", "    \n", "    print(\"Rows with nan values: \" + str(df.isnull().any(axis = 1).sum()))\n", "    print(\"Cols with nan values: \" + str(df.isnull().any(axis = 0).sum()))\n", "    print('Breakdown:')\n", "    print(df.isnull().sum()[df.isnull().sum()!=0])\n", "    print('---------------------------------------------------------------\\n')\n", "    \n", "    print('Columns details: ')\n", "    print('Columns with known dtypes: ')\n", "    good_cols = pd.DataFrame(df.dtypes[df.dtypes!='object'], columns = ['type'])\n", "    good_cols['nan_num'] = [df[col].isnull().sum() for col in good_cols.index]\n", "    good_cols['unique_val'] = [df[col].nunique() for col in good_cols.index]\n", "    good_cols['example'] = [df[col][1] for col in good_cols.index]\n", "    good_cols = good_cols.reindex(good_cols['type'].astype(str).str.len().sort_values().index)\n", "    print(good_cols)\n", "    print('\\n')\n", "    \n", "    try:\n", "        print('Columns with unknown dtypes:')\n", "        bad_cols = pd.DataFrame(df.dtypes[df.dtypes=='object'], columns = ['type'])\n", "        bad_cols['nan_num'] = [df[col].isnull().sum() for col in bad_cols.index]\n", "        bad_cols['unique_val'] = [df[col].nunique() for col in bad_cols.index]\n", "        bad_cols['example(sliced)'] = [str(df[col][1])[:10] for col in bad_cols.index]\n", "        bad_cols = bad_cols.reindex(bad_cols['example(sliced)'].str.len().sort_values().index)\n", "        print(bad_cols)\n", "    except Exception as e:\n", "        print('No columns with unknown dtypes!')\n", "    print('_______________________________________________________________\\n\\n\\n')\n", "    #if not verbose: enablePrint()\n", "    return df\n", "\n", "# fixing dtypes: time and numeric variables\n", "def fix_dtypes(df, time_cols, num_cols):\n", "    \n", "    print('***************************************************************')\n", "    print('Begin fixing data types: ')\n", "    print('***************************************************************\\n')\n", "    \n", "    def fix_time_col(df, time_cols):\n", "        for time_col in time_cols:\n", "            df[time_col] = pd.to_datetime(df[time_col], errors = 'coerce', format = '%Y%m%d')\n", "        print('---------------------------------------------------------------')\n", "        print('The following time columns has been fixed: ')\n", "        print(time_cols)\n", "        print('---------------------------------------------------------------\\n')\n", "\n", "    def fix_num_col(df, num_cols):\n", "        for col in num_cols:\n", "            df[col] = pd.to_numeric(df[col], errors = 'coerce')\n", "        print('---------------------------------------------------------------')\n", "        print('The following number columns has been fixed: ')\n", "        print(num_cols)\n", "        print('---------------------------------------------------------------\\n')\n", "        \n", "    if(len(num_cols) > 0):\n", "        fix_num_col(df, num_cols)\n", "    fix_time_col(df, time_cols)\n", "\n", "    print('---------------------------------------------------------------')\n", "    print('Final data types:')\n", "    result = pd.DataFrame(df.dtypes, columns = ['type'])\n", "    result = result.reindex(result['type'].astype(str).str.len().sort_values().index)\n", "    print(result)\n", "    print('_______________________________________________________________\\n\\n\\n')\n", "    return df\n", "\n", "# Load in user_logs\n", "def transform_df(df):\n", "    df = pd.DataFrame(df)\n", "    df = df.sort_values(by=['date'], ascending=[False])\n", "    df = df.reset_index(drop=True)\n", "    df = df.drop_duplicates(subset=['msno'], keep='first')\n", "    return df\n", "\n", "def transform_df2(df):\n", "    df = df.sort_values(by=['date'], ascending=[False])\n", "    df = df.reset_index(drop=True)\n", "    df = df.drop_duplicates(subset=['msno'], keep='first')\n", "    return df\n", "\n", "# Memory Reduction\n", "def change_datatype(df):\n", "    int_cols = list(df.select_dtypes(include=['int']).columns)\n", "    for col in int_cols:\n", "        if ((np.max(df[col]) <= 127) and(np.min(df[col] >= -128))):\n", "            df[col] = df[col].astype(np.int8)\n", "        elif ((np.max(df[col]) <= 32767) and(np.min(df[col] >= -32768))):\n", "            df[col] = df[col].astype(np.int16)\n", "        elif ((np.max(df[col]) <= 2147483647) and(np.min(df[col] >= -2147483648))):\n", "            df[col] = df[col].astype(np.int32)\n", "        else:\n", "            df[col] = df[col].astype(np.int64)\n", "            \n", "def plot_roc_curve(svm_clf, X_test, y_test, preds, isRF = False):\n", "    from sklearn.metrics import roc_curve, roc_auc_score\n", "    \n", "    if isRF:\n", "        y_score = svm_clf.predict_proba(X_test)[:,1]\n", "    else:\n", "        y_score = svm_clf.decision_function(X_test)\n", "    (false_positive_rate, true_positive_rate, threshold) = roc_curve(y_test, y_score)\n", "    roc_auc = auc(false_positive_rate, true_positive_rate)\n", "\n", "    # Plot ROC curve\n", "    plt.title('Receiver Operating Characteristic')\n", "    plt.plot(false_positive_rate, true_positive_rate, label='ROC curve (area = %0.2f)' % roc_auc)\n", "    plt.plot([0, 1], ls=\"--\")\n", "    # plt.plot([0, 0], [1, 0] , c=\".7\"), plt.plot([1, 1] , c=\".7\")\n", "    plt.ylabel('True Positive Rate')\n", "    plt.xlabel('False Positive Rate')\n", "    plt.legend(loc=\"lower right\")\n", "    plt.show()\n", "\n", "\n", "# Print out the memory usage\n", "def memo(df):\n", "    mem = df.memory_usage(index=True).sum()\n", "    print(mem/ 1024**2,\" MB\")\n", "\n", "# Print all the available files\n", "def print_file():\n", "    print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))"], "metadata": {"collapsed": true, "_uuid": "61b81a75737450a0e07045a10e5fe74a8f67fd55", "_cell_guid": "71148697-53e3-4095-b1c9-96ac40566689"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["print_file()"], "metadata": {"collapsed": true, "_uuid": "926f0b347220cb5ca5a9236b1226c57fe2f2bf2e", "_cell_guid": "074519aa-3cc5-4ffb-b67d-46d2b941b8c2"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# Load in train and test\n", "train = pd.read_csv('../input/train.csv')\n", "train = train.append(pd.read_csv('../input/train_v2.csv'))\n", "train.index = range(len(train))\n", "test = pd.read_csv('../input/sample_submission_v2.csv')\n", "# test = test.append(pd.read_csv('../input/sample_submission_zero.csv'))\n", "# test.index = range(len(test))\n", "\n", "# Load in other files\n", "members = pd.read_csv('../input/members_v3.csv')\n", "change_datatype(members)\n", "print(\"Memo of members: \")\n", "memo(members)\n", "\n", "trans = pd.read_csv('../input/transactions.csv')\n", "trans = trans.append(pd.read_csv('../input/transactions_v2.csv'))\n", "trans.index = range(len(trans))\n", "change_datatype(trans)\n", "print(\"Memo of trans: \")\n", "memo(trans)\n", "\n", "# Loading in user_logs_v2.csv\n", "df_iter = pd.read_csv('../input/user_logs.csv', low_memory=False, iterator=True, chunksize=10000000)\n", "last_user_logs = []\n", "i = 0 #~400 Million Records - starting at the end but remove locally if needed\n", "for df in df_iter:\n", "    if i>35: # used to be 35, just testing\n", "        if len(df)>0:\n", "            print(df.shape)\n", "            p = Pool(cpu_count())\n", "            df = p.map(transform_df, np.array_split(df, cpu_count()))   \n", "            df = pd.concat(df, axis=0, ignore_index=True).reset_index(drop=True)\n", "            df = transform_df2(df)\n", "            p.close(); p.join()\n", "            last_user_logs.append(df)\n", "            print('...', df.shape)\n", "            df = []\n", "    i+=1\n", "\n", "last_user_logs = pd.concat(last_user_logs, axis=0, ignore_index=True).reset_index(drop=True)\n", "last_user_logs = transform_df2(last_user_logs)\n", "# last_user_logs =  last_user_logs[['msno','num_100', 'num_25', 'num_unq', 'total_secs', 'date']]\n", "print(\"Memo of last_user_logs: \")\n", "memo(last_user_logs)"], "metadata": {"collapsed": true, "_uuid": "6d6197ebb2146593fa350860829cb72d15b07403", "_cell_guid": "403e1451-ff1d-4648-af48-2ed88c3b37b9"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["last_user_logs = last_user_logs.rename(columns = {'date':'last_user_log_date'})\n", "last_user_logs.head()"], "metadata": {"collapsed": true, "_uuid": "546e46b99af5492be5c72b48e67528b8101e58cd", "_cell_guid": "43105ed7-a2f2-47d2-8a41-697fe3aa9168"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# Only select 1/5% of train, merge with bigger csvs\n", "# np.random.seed(47)\n", "# samp = train  #.sample(frac = 1, replace = False)\n", "train = train.merge(members, on = 'msno', how = 'left')\n", "test = test.merge(members, on = 'msno', how = 'left')\n", "\n", "temp_trans = trans.sort_values(by=['transaction_date'], ascending=[False]).reset_index(drop=True)\n", "temp_trans = temp_trans.drop_duplicates(subset=['msno'], keep='first')\n", "temp_trans['discount'] = temp_trans['plan_list_price'] - temp_trans['actual_amount_paid']\n", "temp_trans['amt_per_day'] = temp_trans['actual_amount_paid'] / temp_trans['payment_plan_days']\n", "temp_trans['is_discount'] = temp_trans.discount.apply(lambda x: 1 if x > 0 else 0)\n", "temp_trans['membership_days'] = pd.to_datetime(temp_trans['membership_expire_date']).subtract(pd.to_datetime(temp_trans['transaction_date'])).dt.days.astype(int)\n", "train = train.merge(temp_trans, on = 'msno', how = 'left')\n", "test = test.merge(temp_trans, on = 'msno', how = 'left')\n", "\n", "temp_trans = []\n", "\n", "train = train.merge(last_user_logs, on = 'msno', how = 'left')\n", "test = test.merge(last_user_logs, on = 'msno', how = 'left')\n", "\n", "last_user_logs = []"], "metadata": {"collapsed": true, "_uuid": "a86c171d0bcbc634fba4956a3f4650e08b67e875", "_cell_guid": "4b6ad7b3-1156-47b4-91b1-ba6edab273e1"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["print(\"Train shape: \", train.shape)\n", "# print(\"Samp shape: \", samp.shape)\n", "print(\"Test shape: \", test.shape)\n", "\n", "pd.set_option('max_columns', 100)\n", "train.head()"], "metadata": {"collapsed": true, "_uuid": "e495f195e41f4e3f207a236f7ca2fc9b8edf3634", "_cell_guid": "14f102c3-509f-47df-87bd-ff3e773c4b5c"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["test.head()"], "metadata": {"collapsed": true, "_uuid": "345697526334aaf9fac2657ac91f85eb04ccf76f", "_cell_guid": "addd1c38-9cb9-4880-bc64-ec529cebcde6"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# samp = holistic_summary(samp)\n", "train['last_user_log_date'] = train['last_user_log_date'].fillna(20170105.0)\n", "train = fix_dtypes(train, time_cols = ['transaction_date', 'membership_expire_date', 'registration_init_time', 'last_user_log_date'], num_cols = [])"], "metadata": {"collapsed": true, "_uuid": "e22f709a3d95657bca33f528489368a5d4ceadbf", "_cell_guid": "38f2d0b4-4c1a-478e-a237-ed3623332b6a", "scrolled": false}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# test = holistic_summary(test)\n", "test['last_user_log_date'] = test['last_user_log_date'].fillna(20170105.0)\n", "test = fix_dtypes(test, time_cols = ['transaction_date', 'membership_expire_date', 'registration_init_time', 'last_user_log_date'], num_cols = [])"], "metadata": {"collapsed": true, "_uuid": "c68bc840b7e340508621116c4a4bd1c068a6e938", "_cell_guid": "13cecbcb-47df-4939-af2c-1320ad5937f6", "scrolled": false}, "outputs": [], "cell_type": "code", "execution_count": null}, {"metadata": {"_uuid": "e95e6e028778621945b756308907287d018567ef", "_cell_guid": "0af69bf9-21c8-4edd-bc86-f7959a105cd1"}, "source": ["# Feature engineering:"], "cell_type": "markdown"}, {"source": ["# 0. Date columns\n", "print(\"Creating date columns... \")\n", "# samp['last_user_log_date'] = samp['last_user_log_date'].fillna(np.mean(samp['last_user_log_date']))\n", "date_dict = {'t_':'transaction_date', 'm_':'membership_expire_date', \\\n", "             'r_':'registration_init_time', 'l_':'last_user_log_date'}\n", "for key in date_dict:  \n", "    if key == 'r_':\n", "        train[key+'month'] = [d.month for d in train[date_dict[key]]]\n", "        train[key+'day'] = [d.day for d in train[date_dict[key]]]\n", "#         samp[key+'wday'] = [d.weekday() for d in samp[date_dict[key]]]\n", "    else:\n", "        train[key+'day'] = [d.day for d in train[date_dict[key]]]\n", "#         samp[key+'wday'] = [d.weekday() for d in samp[date_dict[key]]]\n", "train['transaction_date'] = [d.year + (d.month-1) / 12 + d.day / 365 for d in train['transaction_date']]\n", "train['membership_expire_date'] = [d.year + (d.month-1) / 12 + d.day / 365 for d in train['membership_expire_date']]\n", "train['registration_init_time'] = [d.year + (d.month-1) / 12 + d.day / 365 for d in train['registration_init_time']]\n", "train['last_user_log_date'] = [d.year + (d.month-1) / 12 + d.day / 365 for d in train['last_user_log_date']]\n", "print(\"Done!\")\n", "\n", "print(\"Creating date columns for test... \")\n", "for key in date_dict:  \n", "    if key == 'r_':\n", "        test[key+'month'] = [d.month for d in test[date_dict[key]]]\n", "        test[key+'day'] = [d.day for d in test[date_dict[key]]]\n", "#         test[key+'wday'] = [d.weekday() for d in test[date_dict[key]]]\n", "    else:\n", "        test[key+'day'] = [d.day for d in test[date_dict[key]]]\n", "#         test[key+'wday'] = [d.weekday() for d in test[date_dict[key]]]\n", "test['transaction_date'] = [d.year + (d.month-1) / 12 + d.day / 365 for d in test['transaction_date']]\n", "test['membership_expire_date'] = [d.year + (d.month-1) / 12 + d.day / 365 for d in test['membership_expire_date']]\n", "test['registration_init_time'] = [d.year + (d.month-1) / 12 + d.day / 365 for d in test['registration_init_time']]\n", "test['last_user_log_date'] = [d.year + (d.month-1) / 12 + d.day / 365 for d in test['last_user_log_date']]\n", "print(\"Done!\")"], "metadata": {"collapsed": true, "_uuid": "46fc9b3604aaa31a988b42646fb4b5ad5e8ccda9", "_cell_guid": "4ad0a2a9-3866-46fa-ae44-29e08a2a3e18"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# 1. number of transactions\n", "print(\"Creating number of transactions... \")\n", "ttemp = trans[['msno']]\n", "temp = pd.DataFrame(ttemp['msno'].value_counts().reset_index())\n", "temp.columns = ['msno','trans_count']\n", "# train = pd.merge(train, transactions, how='left', on='msno')\n", "train = pd.merge(train, temp, how='left', on='msno')\n", "test = pd.merge(test, temp, how='left', on='msno')\n", "temp = []; ttemp = []\n", "print(\"Done!\")"], "metadata": {"collapsed": true, "_uuid": "e33fb05c3526e729ad51a449bd5bc39fad38da81", "_cell_guid": "c71a6425-39e4-452b-ae55-13e752debe47"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# 2. number of user logs in user_logs_v2 only\n", "print(\"Creating number of user logs... \")\n", "user_logs = pd.read_csv('../input/user_logs_v2.csv', usecols=['msno'])\n", "user_logs = pd.DataFrame(user_logs['msno'].value_counts().reset_index())\n", "user_logs.columns = ['msno','logs_count']\n", "# train = pd.merge(train, user_logs, how='left', on='msno')\n", "train = pd.merge(train, user_logs, how='left', on='msno')\n", "test = pd.merge(test, user_logs, how='left', on='msno')\n", "user_logs = []\n", "print(\"Done!\")"], "metadata": {"collapsed": true, "_uuid": "157ef24ada188f18e37ff56fb89aaeee0f56996a", "_cell_guid": "65064e95-9565-47d5-9f04-ae22e614e43c"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# 3. dummy encodings\n", "print(\"Creating dummy encodings... \")\n", "train['unique_index'] = range(len(train))\n", "# Creat dummy variables for for following columns\n", "prefix_dict = {'pm_id':'payment_method_id', 'pp_days':'payment_plan_days', 'city':'city',\\\n", "             'gender':'gender', 'reg_via':'registered_via', 'pl_price':'plan_list_price'}\n", "dummy_dict = {'pm_id':[41,38], 'pp_days':[30], 'city':[22],\n", "             'pl_price':[99, 149], 'reg_via':[7,4]} # ,22,38,39,35,29,36\n", "for key in prefix_dict:\n", "    if key in ['gender']:\n", "        dummmm_df = pd.get_dummies(train[prefix_dict[key]])\n", "        dummmm_df.columns = [key+'_'+str(s) for s in dummmm_df.columns]\n", "        dummmm_df['unique_index'] = train['unique_index']\n", "        train = train.merge(dummmm_df, on = 'unique_index', how = 'inner')\n", "    else:\n", "        for unique_val in dummy_dict[key]:\n", "            train[key+'_'+str(unique_val)] = np.where(train[prefix_dict[key]] == unique_val, 1, 0)\n", "#         samp[key+'_other'] = np.where(samp[prefix_dict[key]].isin(dummy_dict[key]), 0, 1)\n", "    train = train.drop(prefix_dict[key], 1)\n", "train = train.drop('unique_index', 1)\n", "print(\"Done!\")\n", "\n", "print(\"Creating dummy encodings for test... \")\n", "test['unique_index'] = range(len(test))\n", "for key in prefix_dict:\n", "    if key in ['gender']:\n", "        dummmm_df = pd.get_dummies(test[prefix_dict[key]])\n", "        dummmm_df.columns = [key+'_'+str(s) for s in dummmm_df.columns]\n", "        dummmm_df['unique_index'] = test['unique_index']\n", "        test = test.merge(dummmm_df, on = 'unique_index', how = 'inner')\n", "    else:\n", "        for unique_val in dummy_dict[key]:\n", "            test[key+'_'+str(unique_val)] = np.where(test[prefix_dict[key]] == unique_val, 1, 0)\n", "#         samp[key+'_other'] = np.where(samp[prefix_dict[key]].isin(dummy_dict[key]), 0, 1)\n", "    test = test.drop(prefix_dict[key], 1)\n", "test = test.drop('unique_index', 1)\n", "print(\"Done!\")\n"], "metadata": {"collapsed": true, "_uuid": "b00fbde1dc0cf6386b7372623f49978a07d20241", "_cell_guid": "d41078f7-a303-497e-b060-e9e325f12640"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# 4. interaction terms\n", "train['autorenew_&_not_cancel'] = ((train.is_auto_renew == 1) == (train.is_cancel == 0)).astype(np.int8)\n", "test['autorenew_&_not_cancel'] = ((test.is_auto_renew == 1) == (test.is_cancel == 0)).astype(np.int8)\n", "\n", "train['notAutorenew_&_cancel'] = ((train.is_auto_renew == 0) == (train.is_cancel == 1)).astype(np.int8)\n", "test['notAutorenew_&_cancel'] = ((test.is_auto_renew == 0) == (test.is_cancel == 1)).astype(np.int8)\n", "\n", "memo(train)\n", "memo(test)"], "metadata": {"collapsed": true, "_uuid": "57f95e37559bb742cd454906fc33a51911c5a71c", "_cell_guid": "d09fc599-85aa-497b-a6cd-6610ada04eb6"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"metadata": {"_uuid": "126b7b0c42471e0f2d06c921252cc50ef9f6ddf6", "_cell_guid": "d9016a3a-feaa-4957-950f-b2d649ec0f08"}, "source": ["## Replace infs and imputing missing values by mean"], "cell_type": "markdown"}, {"source": ["feature_cols = [col for col in train.columns if col not in ['is_churn', 'msno']]\n", "\n", "train[feature_cols] = train[feature_cols].applymap(lambda x: np.nan if np.isinf(x) else x)\n", "test[feature_cols] = test[feature_cols].applymap(lambda x: np.nan if np.isinf(x) else x)"], "metadata": {"collapsed": true, "_uuid": "27adf310e994cffcbc5a9aef347addd2673ef764", "_cell_guid": "351824dd-82b5-44dd-9c62-c9d5b355b755"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["print(train.columns[train.isnull().any()].tolist())\n", "fill_dict = {}\n", "for col in train.columns[train.isnull().any()].tolist():\n", "    fill_dict[col] = np.mean(train[col])\n", "train = train.fillna(value = fill_dict)\n", "print(train.columns[train.isnull().any()].tolist())\n", "\n", "print(test.columns[test.isnull().any()].tolist())\n", "# fill_dict = {}\n", "# for col in test.columns[test.isnull().any()].tolist():\n", "#     fill_dict[col] = np.mean(test[col])\n", "test = test.fillna(value = fill_dict)\n", "print(test.columns[test.isnull().any()].tolist())"], "metadata": {"collapsed": true, "_uuid": "a2a412d61f9988138b29b04063dcc3c96b0454da", "_cell_guid": "12b3c383-e389-4b12-ac07-1c8807f1afba"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["pd.set_option('max_columns', 100)\n", "print(train.shape)\n", "train.head()"], "metadata": {"collapsed": true, "_uuid": "7d6f62b3cad3336953c90475c9170a69659921e3", "_cell_guid": "9ef5661f-767d-415f-9316-3792e0e73f25"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["print(test.shape)\n", "test.head()"], "metadata": {"collapsed": true, "_uuid": "ef01544c1fdc93f50f444bb5488a54a26f15cfa2", "_cell_guid": "db4f0f5c-386e-436d-95c1-22c086dade34"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["trans = []\n", "members = []"], "metadata": {"collapsed": true, "_uuid": "cd2a523e0619615784e8d2978e0e07d42223e978", "_cell_guid": "d3564b41-3568-4b83-8370-dd6d41351ead"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"metadata": {"_uuid": "b2908e16012335fe1ff64ba5b3a67c65cebeeae0", "_cell_guid": "c3430b8c-5f2d-4ed7-a1c2-7ceaf155c354"}, "source": ["The dataset is quite imbalanced..."], "cell_type": "markdown"}, {"source": ["print(\"Churn ratio\", len(train[train['is_churn'] == 1])/len(train))"], "metadata": {"collapsed": true, "_uuid": "00f79f4c2c2f887d4700dacb52b981c5bb8ef20e", "_cell_guid": "7534c489-63a3-40ba-b714-6bb05ba37fd2"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"metadata": {"_uuid": "6b603325590bf142f208fbd6390f13351910a1af", "_cell_guid": "20b60f94-baf5-47e5-9711-826b19e85351"}, "source": ["# RF for feature selection"], "cell_type": "markdown"}, {"source": ["from sklearn.metrics import (confusion_matrix, precision_recall_curve, auc,\n", "                             roc_curve, recall_score, classification_report, f1_score,\n", "                             precision_recall_fscore_support)"], "metadata": {"collapsed": true, "_uuid": "1426392ecf286c3db79168fc20fb46748cf6e81a", "_cell_guid": "ae3b15a9-e47f-4c0f-9ca8-5badfe00dd07"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["from math import log\n", "def log_loss(preds, trues):\n", "    preds = [max(min(s, 1-10**(-15)), 10**(-15)) for s in preds]\n", "    return -np.mean([y * log(p) + (1-y) * log(1-p) for y,p in zip(trues, preds)])\n", "trues  = [1,1,0,0]\n", "preds  = [1,1,0,0]\n", "\n", "print(log_loss(trues, preds))"], "metadata": {"collapsed": true, "_uuid": "56657afae3a1e175f06d0b650b570b74fb196478", "_cell_guid": "a267b59b-ea56-41ca-8ba7-32624b3a8a52"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["feature_cols = [col for col in train.columns if col not in ['is_churn', 'msno']]\n", "features = np.array(train[feature_cols])\n", "response = np.array(train['is_churn'])\n", "features_test = np.array(test[feature_cols])\n", "\n", "print(features.shape)\n", "print(response.shape)\n", "print(features_test.shape)"], "metadata": {"collapsed": true, "_uuid": "27e80a89bb5860d4fac48e9853d144426b02a6fe", "_cell_guid": "75857f0a-eee6-427b-b6c0-fcf37b01141e"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["from sklearn.preprocessing import StandardScaler\n", "from sklearn import decomposition\n", "\n", "# Create standardized features\n", "features_std = StandardScaler().fit_transform(features)\n", "features_std_test = StandardScaler().fit_transform(features_test)"], "metadata": {"collapsed": true, "_uuid": "4c0cf9f22c241824598d6dff0362d7a6b2d06b7d", "_cell_guid": "16466eb8-639d-44dc-abce-32bb494e2d34"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["from sklearn.metrics import roc_auc_score, roc_curve\n", "\n", "X_train, X_test, y_train, y_test = cross_validation.train_test_split(features_std, response, test_size = 0.2)"], "metadata": {"collapsed": true, "_uuid": "bca50d4b5c2d34a1fc25d58fd55dd27626aa32c2", "_cell_guid": "b874ac48-4cba-4197-a74a-bfb3130d9920"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# res = pd.DataFrame(res).sort_values('oob_score_', ascending = False)\n", "leaf_size = 1\n", "n_feautres = 10\n", "# res.head(1)"], "metadata": {"collapsed": true, "_uuid": "258ee77460017ca43d02f703dc79382ae8c4395f", "_cell_guid": "73f34ba6-e7a9-4d39-8515-e8ef1a126842"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["def RF(X_train, X_test, y_train, y_test):\n", "    clf = RandomForestClassifier(n_estimators = 250, n_jobs=-1, class_weight=\"balanced\", \\\n", "                                 max_features = n_feautres, min_samples_leaf = leaf_size,\\\n", "                                 random_state = 47)\n", "    print(\"Starting training...\")\n", "    clf.fit(X_train, y_train)\n", "    print(\"Done\")\n", "    print(\"Prediction...\")\n", "    preds = clf.predict(X_test)\n", "    # Print confusion matrix and plot ROC curve\n", "    print(pd.crosstab(y_test, preds, rownames=['Actual'], colnames=['Predicted']))\n", "    plot_roc_curve(clf, X_test, y_test, preds, isRF = True)\n", "    return clf\n", "\n", "rf_clf = RF(X_train, X_test, y_train, y_test)"], "metadata": {"collapsed": true, "_uuid": "9da83e1e695a226c7070f0eac653ed3aec476c70", "_cell_guid": "c466b1df-28c0-469d-9a51-0d6dffd048b0", "scrolled": false}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["LABELS = ['not churn', 'churn']\n", "# X_train, X_test, y_train, y_test = cross_validation.train_test_split(features_std, response, test_size = 0.2)\n", "\n", "y_pred = rf_clf.predict(X_test)\n", "print(log_loss(y_test, y_pred))\n", "conf_matrix = confusion_matrix(y_test, y_pred)\n", "\n", "plt.figure(figsize=(10, 8))\n", "sns.heatmap(conf_matrix, xticklabels=LABELS, yticklabels=LABELS, annot=True, fmt=\"d\");\n", "plt.title(\"Confusion matrix\")\n", "plt.ylabel('True class')\n", "plt.xlabel('Predicted class')\n", "plt.show()"], "metadata": {"collapsed": true, "_uuid": "be3501b6b368addc5ad9350ac40973d8439e9274", "_cell_guid": "69e0ab1c-701e-41db-b065-b6a0059f6c42"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# final_preds = rf_clf.predict(test[feature_cols])\n", "# test['is_churn'] = final_preds.clip(0.+1e-15, 1-1e-15)\n", "# test[['msno','is_churn']].to_csv('pred-rf.csv.gz', index=False, compression='gzip')"], "metadata": {"collapsed": true, "_uuid": "f3a13067355da232042883b259aa2cdd1fb95e4e", "_cell_guid": "37037d02-8dd3-4c8f-aca3-72434e4aa299"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# Plot the feature importances of RF\n", "importance = rf_clf.feature_importances_\n", "importance = pd.DataFrame(importance, index=feature_cols, columns=[\"importance\"])\n", "importance[\"std\"] = np.std([tree.feature_importances_ for tree in rf_clf.estimators_], axis=0)\n", "importance = importance.sort_values('importance', ascending = False)\n", "importance['col_names'] = importance.index\n", "\n", "RF_important_cols = list(importance.index)[:20]\n", "print(\"Most importance 15 features: \", RF_important_cols)\n", "plt.figure(figsize=(10, 12))\n", "sns.barplot(data = importance, y = 'col_names', x = 'importance')\n", "# plt.xticks(rotation=90)"], "metadata": {"collapsed": true, "_uuid": "ac4719110fde4f33d38d0afcc40d58337c0100b0", "_cell_guid": "2d0e0489-7416-4442-93c6-07db8cda4fec"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# from sklearn.svm import LinearSVC\n", "\n", "# def SVC(features, response):\n", "#     X_train, X_test, y_train, y_test = cross_validation.train_test_split(features, response, test_size = 0.2)\n", "\n", "#     clf = LinearSVC(class_weight=\"balanced\")\n", "#     clf.fit(X_train, y_train)\n", "#     preds = clf.predict(X_test)\n", "#     # Print confusion matrix and plot ROC curve\n", "#     print(pd.crosstab(y_test, preds, rownames=['Actual'], colnames=['Predicted']))\n", "#     plot_roc_curve(clf, X_test, y_test, preds)\n", "\n", "# SVC(features, response)\n", "# SVC(features_std, response)\n", "# # SVC(features_pca, response)"], "metadata": {"collapsed": true, "_uuid": "deb3799d4ba6122f10cd262bbfcc1de961951126", "_cell_guid": "83751644-6941-4a7f-94ec-241d4f32268a"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"metadata": {"_uuid": "57a0fdbebfc4d1c8c39f13b3ddc16e6d6f7f7ce8", "_cell_guid": "c4ad60a3-46bb-4b25-b6f4-b888f7e0dc6b"}, "source": ["# Keras Autoencoder"], "cell_type": "markdown"}, {"source": ["np.random.seed(47)\n", "train = train.sample(frac = 0.2, replace = False)"], "metadata": {"collapsed": true, "_uuid": "d3c73254674790db6729eaca2b53ecd9bfcdd058", "_cell_guid": "ebb8c66d-b562-4978-b527-cec212afd52f"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["from sklearn import decomposition\n", "from sklearn.preprocessing import StandardScaler\n", "\n", "# Redefine the featuresets for Autoencoder\n", "features = np.array(train[RF_important_cols])\n", "features_test = np.array(test[RF_important_cols])\n", "response = np.array(train['is_churn'])\n", "\n", "# Create standardized features\n", "features_std = StandardScaler().fit_transform(features)\n", "features_std_test = StandardScaler().fit_transform(features_test)\n", "\n", "# Create a pca object with the 10 components\n", "pca = decomposition.PCA(n_components=10)\n", "\n", "# Fit the PCA and transform the data\n", "features_pca = pca.fit_transform(features_std)\n", "features_pca_test = pca.fit_transform(features_std)"], "metadata": {"collapsed": true, "_uuid": "7d53c55985b4118ca52ec4cfaf565d8260618634", "_cell_guid": "924b5e66-6d98-42ae-ad75-084fcadaa9cb"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["# Convert features_pca back to a df and add the is_churn column\n", "features_std = pd.DataFrame(features_std)\n", "features_std['is_churn'] = np.array(train['is_churn'])\n", "\n", "features_std_test = pd.DataFrame(features_std).values\n", "\n", "\n", "# features_std = pd.DataFrame(features_pca)\n", "# features_std['is_churn'] = np.array(train['is_churn'])\n", "\n", "# features_std_test = pd.DataFrame(features_pca_test).values"], "metadata": {"collapsed": true, "_uuid": "96ce67d1e0741805293f984bcdc4796bfbbbeb48", "_cell_guid": "86b84c9c-cf00-4657-abff-a96f24a3c6da"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["from sklearn.cross_validation import train_test_split\n", "X_train, X_test = train_test_split(features_std, test_size=0.2, random_state=47)\n", "X_train = X_train[X_train['is_churn'] == 0]\n", "X_train = X_train.drop(['is_churn'], axis=1)\n", "\n", "y_test = X_test['is_churn']\n", "X_test = X_test.drop(['is_churn'], axis=1)\n", "\n", "X_train = X_train.values\n", "X_test = X_test.values"], "metadata": {"collapsed": true, "_uuid": "568c56e375415342fa5a2d69387eb4b17f7c3ed5", "_cell_guid": "bb4d21f4-4021-4c74-8105-8d707afbe29e"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["print(X_train.shape)\n", "print(X_test.shape)"], "metadata": {"collapsed": true, "_uuid": "e58689c62a36e39d2da353c9afaea80d37694222", "_cell_guid": "b9300c51-536c-4c10-8e8d-a467afc27df2"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"metadata": {"_uuid": "2035b7dbe6f6529630d411b1cf3511a29b97181a", "_cell_guid": "1095a30e-2bdb-4661-82db-79a17c6c5efc"}, "source": ["## Building the model"], "cell_type": "markdown"}, {"source": ["input_dim = X_train.shape[1]\n", "encoding_dim = 8"], "metadata": {"collapsed": true, "_uuid": "b8a9c37b99b1775a8ebc36c67a57510c3a9252fa", "_cell_guid": "7f85167c-50ba-4e16-b901-eee9821063d3"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["input_layer = Input(shape=(input_dim, ))\n", "\n", "encoder = Dense(encoding_dim, activation=\"tanh\", \n", "                activity_regularizer=regularizers.l1(10e-5))(input_layer)\n", "encoder = Dense(int(encoding_dim / 2), activation=\"relu\")(encoder)\n", "\n", "decoder = Dense(int(encoding_dim / 2), activation='tanh')(encoder)\n", "decoder = Dense(input_dim, activation='relu')(decoder)\n", "\n", "autoencoder = Model(inputs=input_layer, outputs=decoder)"], "metadata": {"collapsed": true, "_uuid": "760a5ac69247ae5ebe9676c91a7e2d8ccfcdc1d1", "_cell_guid": "d5fd75bc-75ab-4cab-9285-862a65578c10"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["nb_epoch = 100\n", "batch_size = 32\n", "\n", "autoencoder.compile(optimizer='adam', \n", "                    loss='mean_squared_error', \n", "                    metrics=['accuracy'])\n", "\n", "checkpointer = ModelCheckpoint(filepath=\"model.h5\",\n", "                               verbose=0,\n", "                               save_best_only=True)\n", "tensorboard = TensorBoard(log_dir='./logs',\n", "                          histogram_freq=0,\n", "                          write_graph=True,\n", "                          write_images=True)"], "metadata": {"collapsed": true, "_uuid": "025b0218858f700e1d5f6aa986bc7184838be59c", "_cell_guid": "f94b0b96-12f1-4fc6-ac50-e4e04b7567b2", "scrolled": false}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["history = autoencoder.fit(X_train, X_train,\n", "                    epochs=nb_epoch,\n", "                    batch_size=batch_size,\n", "                    shuffle=True,\n", "                    validation_data=(X_test, X_test),\n", "                    verbose=1,\n", "                    callbacks=[checkpointer, tensorboard]).history"], "metadata": {"collapsed": true, "_uuid": "27a18879cf96819f582231bff9af38c0faadc272", "_cell_guid": "30a11121-ce22-4aea-8874-5936cc046150"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["print(checkpointer)\n", "print(tensorboard)\n", "print(nb_epoch)\n", "print(batch_size)\n", "print(X_train)\n"], "metadata": {"collapsed": true, "_uuid": "d4c453841a50bfd6bfd3d92258128b737ae8bbe8", "_cell_guid": "20e2bc49-100a-4920-88a7-ffd7dcfdf36e"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["autoencoder = load_model('model.h5')"], "metadata": {"collapsed": true, "_uuid": "74c8bbcbacdb134ad61c4fe54bdbbf06e1459d86", "_cell_guid": "30384add-4700-4084-983f-fe02e1effbc9"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"metadata": {"_uuid": "135d67f70737840367462cb09110f9455a11749e", "_cell_guid": "9add179c-c60c-425e-98fa-accc87096fce"}, "source": ["# Evaluate the Model"], "cell_type": "markdown"}, {"source": ["plt.plot(history['loss'])\n", "plt.plot(history['val_loss'])\n", "plt.title('model loss')\n", "plt.ylabel('loss')\n", "plt.xlabel('epoch')\n", "plt.legend(['train', 'test'], loc='upper right');"], "metadata": {"collapsed": true, "_uuid": "38a6c8ea3e8989c08bb6d697f2e09905dffe86f9", "_cell_guid": "8580fa90-01b7-4894-a126-37ccc149e458"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["predictions = autoencoder.predict(X_test)\n", "mse = np.mean(np.power(X_test - predictions, 2), axis=1)\n", "error_df = pd.DataFrame({'reconstruction_error': mse,\n", "                        'true_class': y_test})"], "metadata": {"collapsed": true, "_uuid": "dd2aeb568944d234651e930102892a17249eb72e", "_cell_guid": "acfadaa5-91a3-4e88-a12d-7805a3523813"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["error_df.describe()"], "metadata": {"collapsed": true, "_uuid": "3578fb3568fcba97406a4ad1fb0839befbf6048d", "_cell_guid": "e0e4cf8e-808d-44b1-a902-201edfe58cb0"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["fig = plt.figure()\n", "ax = fig.add_subplot(111)\n", "normal_error_df = error_df[(error_df['true_class']== 0) & (error_df['reconstruction_error'] < 10)]\n", "_ = ax.hist(normal_error_df.reconstruction_error.values, bins=10, normed = True)\n", "plt.title('Reconstruction error without churn group')\n", "sns.despine()"], "metadata": {"collapsed": true, "_uuid": "e74ed5a635e7318b7fef01293c593680ea1c9672", "_cell_guid": "14a9ed40-0687-4458-a569-ea0611e79066"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["fig = plt.figure()\n", "ax = fig.add_subplot(111)\n", "fraud_error_df = error_df[error_df['true_class'] == 1]\n", "_ = ax.hist(fraud_error_df.reconstruction_error.values, bins=10, normed = True)\n", "plt.title('Reconstruction error with churn group')\n", "sns.despine()"], "metadata": {"collapsed": true, "_uuid": "059bf0d1b3311470c3cae44c237052e83c6033ee", "_cell_guid": "3cc7e9c5-f97a-4055-8112-c4bcf34a6a20"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["fpr, tpr, thresholds = roc_curve(error_df.true_class, error_df.reconstruction_error)\n", "roc_auc = auc(fpr, tpr)\n", "\n", "plt.title('Receiver Operating Characteristic')\n", "plt.plot(fpr, tpr, label='AUC = %0.4f'% roc_auc)\n", "plt.legend(loc='lower right')\n", "plt.plot([0,1],[0,1],'g--')\n", "plt.xlim([-0.001, 1])\n", "plt.ylim([0, 1.001])\n", "plt.ylabel('True Positive Rate')\n", "plt.xlabel('False Positive Rate')\n", "plt.show();"], "metadata": {"collapsed": true, "_uuid": "65b32585322b32c837cf69a9f46308c5266dc068", "_cell_guid": "1359fb43-c33c-492e-af55-e8a151867cd9"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["precision, recall, th = precision_recall_curve(error_df.true_class, error_df.reconstruction_error)\n", "plt.plot(recall, precision, 'b', label='Precision-Recall curve')\n", "plt.title('Recall vs Precision')\n", "plt.xlabel('Recall')\n", "plt.ylabel('Precision')\n", "plt.show()"], "metadata": {"collapsed": true, "_uuid": "0c9ceededb3ce81e9708e349c5f95c15540db4b2", "_cell_guid": "82e80b6b-5629-4aa2-9b3b-5f19c159bd14"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["plt.plot(th, precision[1:], 'b', label='Threshold-Precision curve')\n", "plt.title('Precision for different threshold values')\n", "plt.xlabel('Threshold')\n", "plt.ylabel('Precision')\n", "plt.show()"], "metadata": {"collapsed": true, "_uuid": "68ec29e59ffd13d6a616ed8fd223b708ac012f93", "_cell_guid": "e22b457d-1d3a-405d-b599-5b0a18a234ab"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["plt.plot(th, recall[1:], 'b', label='Threshold-Recall curve')\n", "plt.title('Recall for different threshold values')\n", "plt.xlabel('Reconstruction error')\n", "plt.ylabel('Recall')\n", "plt.show()"], "metadata": {"collapsed": true, "_uuid": "63c35818d8658ba4773a72e64d7e2941fbf4bb60", "_cell_guid": "7b4b6d81-30b3-4977-b130-c6b434d58759"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"metadata": {"_uuid": "fde86829e9efe58cce3c0d5d1181dcdb8a703136", "_cell_guid": "b8246f8e-10fb-4093-b90a-f85edaba723e"}, "source": ["# Prediction"], "cell_type": "markdown"}, {"source": ["threshold = 1"], "metadata": {"collapsed": true, "_uuid": "0b88beb36b33ccebcdfefc80ee71a4e1f16da9eb", "_cell_guid": "ad6c4ba9-157d-42bc-94bb-6e021faf7438"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["groups = error_df.groupby('true_class')\n", "fig, ax = plt.subplots()\n", "\n", "for name, group in groups:\n", "    ax.plot(group.index, group.reconstruction_error, marker='o', ms=3.5, linestyle='',\n", "            label= \"churn\" if name == 1 else \"not churn\", alpha = 0.5)\n", "ax.hlines(threshold, ax.get_xlim()[0], ax.get_xlim()[1], colors=\"r\", zorder=100, label='Threshold')\n", "ax.legend()\n", "plt.title(\"Reconstruction error for different classes\")\n", "plt.ylabel(\"Reconstruction error\")\n", "plt.xlabel(\"Data point index\")\n", "plt.ylim(0,8)\n", "sns.despine()\n", "plt.show();"], "metadata": {"collapsed": true, "_uuid": "8c9fc84b5767744009a9a4de653a01a343b289df", "_cell_guid": "63ef41f3-4bf2-42fa-bc0d-7abc8f79e87a"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"source": ["LABELS = ['not churn', 'churn']\n", "\n", "y_pred = [1 if e > threshold else 0 for e in error_df.reconstruction_error.values]\n", "print(log_loss(error_df.true_class, y_pred))\n", "\n", "conf_matrix = confusion_matrix(error_df.true_class, y_pred)\n", "plt.figure(figsize=(10,8))\n", "sns.heatmap(conf_matrix, xticklabels=LABELS, yticklabels=LABELS, annot=True, fmt=\"d\");\n", "plt.title(\"Confusion matrix\")\n", "plt.ylabel('True class')\n", "plt.xlabel('Predicted class')\n", "plt.show()"], "metadata": {"collapsed": true, "_uuid": "7836721427595436909ecfef9d26b15886056d00", "_cell_guid": "9b963411-257e-4bbd-b472-b26abbf0ac73"}, "outputs": [], "cell_type": "code", "execution_count": null}, {"metadata": {"_uuid": "03df556861265544b0882229cbf7bd430385a6d8", "_cell_guid": "69fa7ee3-941f-4824-b54d-6e3fcc269e55"}, "source": ["# Prediction"], "cell_type": "markdown"}, {"source": ["predictions = autoencoder.predict(features_std_test)\n", "mse = np.mean(np.power(features_std_test - predictions, 2), axis=1)\n", "error_df = pd.DataFrame({'reconstruction_error': mse})\n", "\n", "y_pred = [1 if e > threshold else 0 for e in error_df.reconstruction_error.values]\n", "\n", "test['is_churn'] = y_pred.clip(0.+1e-15, 1-1e-15)\n", "test[['msno','is_churn']].to_csv('pred-auto.csv.gz', index=False, compression='gzip')"], "metadata": {"collapsed": true, "_uuid": "f4d9eebe1fd131a7e5805468d07139589c85d2c2", "_cell_guid": "e2eb55ff-8b42-4a11-b734-2a728bd89127"}, "outputs": [], "cell_type": "code", "execution_count": null}], "nbformat": 4}