{"metadata": {"language_info": {"mimetype": "text/x-python", "version": "3.6.3", "pygments_lexer": "ipython3", "file_extension": ".py", "codemirror_mode": {"version": 3, "name": "ipython"}, "nbconvert_exporter": "python", "name": "python"}, "kernelspec": {"display_name": "Python 3", "language": "python", "name": "python3"}}, "cells": [{"source": ["# Import library"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {"collapsed": true}, "cell_type": "code", "source": ["import pandas as pd\n", "import numpy as np\n", "import seaborn as sns\n", "import matplotlib.pyplot as plt\n", "\n", "from sklearn.linear_model import LogisticRegression\n", "from sklearn.model_selection import StratifiedKFold,RandomizedSearchCV\n", "from sklearn.metrics import roc_auc_score,confusion_matrix,roc_curve\n", "from sklearn.preprocessing import LabelEncoder\n", "from sklearn.feature_extraction.text import TfidfVectorizer\n", "\n", "import datetime as dt\n", "\n", "% matplotlib inline\n", "seed = 129"], "execution_count": 1}, {"source": ["# Import Dataset"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {"collapsed": true}, "cell_type": "code", "source": ["path = '../input/'\n", "#path = ''\n", "train = pd.read_csv(path+'train_v2.csv',dtype={'is_churn':np.int8})\n", "test = pd.read_csv(path+'sample_submission_v2.csv',dtype={'is_churn':np.int8})\n", "members = pd.read_csv(path+'members_v3.csv',parse_dates=['registration_init_time'],dtype={'city':np.int8,'bd':np.int8,\n", "                                                                                         'registered_via':np.int8})\n", "transactions = pd.read_csv(path+'transactions_v2.csv',parse_dates=['transaction_date','membership_expire_date'],\n", "                          dtype={'payment_method_id':np.int8,'payment_plan_days':np.int8,'plan_list_price':np.int8,\n", "                                'actual_amount_paid':np.int8,'is_auto_renew':np.int8,'is_cancel':np.int8})\n", "\n", "user_log = pd.read_csv(path+'user_logs_v2.csv',parse_dates=['date'],dtype={'num_25':np.int16,'num_50':np.int16,\n", "                                    'num_75':np.int16,'num_985':np.int16,'num_100':np.int16,'num_unq':np.int16})"], "execution_count": 2}, {"source": ["# Explore data set"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["print('Number of rows  & columns',train.shape)\n", "train.head()"], "execution_count": 3}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["print('Number of rows  & columns',test.shape)\n", "test.head()"], "execution_count": 4}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["print('Number of rows  & columns',members.shape)\n", "members.head()"], "execution_count": 5}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["print('Number of rows & columns',transactions.shape)\n", "transactions.head()"], "execution_count": 6}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["print('Number of rows & columns',user_log.shape)\n", "user_log.head()"], "execution_count": 7}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["print('\\nTrain:',train.describe().T)\n", "print('\\nTest:',test.describe().T)\n", "print('\\nMembers:',members.describe().T)\n", "print('\\nTransactions:',transactions.describe().T)\n", "print('\\nUser log:',user_log.describe().T)\n"], "execution_count": 8}, {"source": ["# Merge data set"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["train = pd.merge(train,members,on='msno',how='left')\n", "test = pd.merge(test,members,on='msno',how='left')\n", "train = pd.merge(train,transactions,how='left',on='msno',left_index=True, right_index=True)\n", "test = pd.merge(test,transactions,how='left',on='msno',left_index=True, right_index=True,)\n", "train = pd.merge(train,user_log,how='left',on='msno',left_index=True, right_index=True)\n", "test = pd.merge(test,user_log,how='left',on='msno',left_index=True, right_index=True)\n", "\n", "del members,transactions,user_log\n", "print('Number of rows & columns',train.shape)\n", "print('Number of rows & columns',test.shape)\n"], "execution_count": 9}, {"source": ["# Date feature"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["train[['registration_init_time' ,'transaction_date','membership_expire_date','date']].describe()"], "execution_count": 10}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["train[['registration_init_time' ,'transaction_date','membership_expire_date','date']].isnull().sum()"], "execution_count": 11}, {"outputs": [], "metadata": {"collapsed": true}, "cell_type": "code", "source": ["train['registration_init_time'] = train['registration_init_time'].fillna(value=pd.to_datetime('09/10/2015'))\n", "test['registration_init_time'] = test['registration_init_time'].fillna(value=pd.to_datetime('09/10/2015'))"], "execution_count": 12}, {"outputs": [], "metadata": {"collapsed": true}, "cell_type": "code", "source": ["def date_feature(df):\n", "    \n", "    col = ['registration_init_time' ,'transaction_date','membership_expire_date','date']\n", "    var = ['reg','trans','mem_exp','user_']\n", "    #df['duration'] = (df[col[1]] - df[col[0]]).dt.days \n", "    \n", "    for i ,j in zip(col,var):\n", "        df[j+'_day'] = df[i].dt.day.astype('uint8')\n", "        df[j+'_weekday'] = df[i].dt.weekday.astype('uint8')        \n", "        df[j+'_month'] = df[i].dt.month.astype('uint8') \n", "        df[j+'_year'] =df[i].dt.year.astype('uint16') \n", "\n", "date_feature(train)\n", "date_feature(test)"], "execution_count": 13}, {"source": ["# Data analysis "], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["train.columns"], "execution_count": 14}, {"source": ["# Missing value"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["train.isnull().sum()"], "execution_count": 15}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["train.info()"], "execution_count": 16}, {"outputs": [], "metadata": {"collapsed": true}, "cell_type": "code", "source": ["col = [ 'city', 'bd', 'gender', 'registered_via']\n", "def missing(df,columns):\n", "    col = columns\n", "    for i in col:\n", "        df[i].fillna(df[i].mode()[0],inplace=True)\n", "\n", "missing(train,col)\n", "missing(test,col)"], "execution_count": 17}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["def unique_value(df):\n", "    col = df.columns\n", "    for i in col:\n", "        print('Number of unique value in {} is {}'.format(i,df[i].nunique()))\n", "\n", "unique_value(train)"], "execution_count": 18}, {"source": ["# is_churn"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["plt.figure(figsize=(8,6))\n", "sns.set_style('ticks')\n", "sns.countplot(train['is_churn'],palette='summer')\n", "plt.xlabel('The subscription within 30 days of expiration is True/False')"], "execution_count": 19}, {"source": ["Imbalanced data set\n", "\n", "msno: user id\n", "\n", "is_churn: This is the target variable. Churn is defined as whether the user did not continue the subscription within 30 days of expiration. \n", "is_churn = 1 means churn,is_churn = 0 means renewal."], "metadata": {}, "cell_type": "markdown"}, {"source": ["## Univariate analysis"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["print(train['city'].unique())\n", "fig,ax = plt.subplots(2,2,figsize=(16,8))\n", "ax1,ax2,ax3,ax4 = ax.flatten()\n", "\n", "sns.set(style=\"ticks\")\n", "sns.countplot(train['city'],palette='summer',ax=ax1)\n", "#ax1.set_yscale('log')\n", "\n", "ax1.set_xlabel('City')\n", "#ax1.set_xticks(rotation=45)\n", "\n", "sns.countplot(x='gender',data = train,palette='winter',ax=ax2)\n", "#ax2.set_yscale('log')\n", "ax2.set_xlabel('Gender')\n", "\n", "sns.countplot(x='registered_via',data=train,palette='winter',ax=ax3)\n", "#ax3.set_yscale('')\n", "ax3.set_xlabel('Register via')\n", "\n", "sns.countplot(x='payment_method_id',data= train,palette='winter',ax=ax4)\n", "ax4.set_xlabel('Payment_method_id')\n", "\n"], "execution_count": 20}, {"source": ["# bd  (birth day)"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["print(train['bd'].describe())"], "execution_count": 21}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["fig,ax = plt.subplots(1,2,figsize=(16,8))\n", "ax1,ax2 = ax.flatten()\n", "sns.set_style('ticks')\n", "sns.distplot(train['bd'].fillna(train['bd'].mode()[0]),bins=100,color='r',ax=ax1)\n", "plt.title('Distribution of birth day')"], "execution_count": 22}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["plt.figure(figsize=(14,6))\n", "sns.distplot(train.loc[train['bd'].value_counts()]['bd'].fillna(0),bins=50,color='b')"], "execution_count": 23}, {"source": ["# Gender"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["print(pd.crosstab(train['is_churn'],train['gender']))"], "execution_count": 24}, {"source": ["# registration_init_time"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["regi = train.groupby('registration_init_time').count()['is_churn']\n", "plt.subplot(211)\n", "plt.plot(regi,color='b',label='count')\n", "plt.legend(loc='center')\n", "regi = train.groupby('registration_init_time').mean()['is_churn']\n", "plt.subplot(212)\n", "plt.plot(regi,color='r',label='mean')\n", "plt.legend(loc='center')\n", "plt.tight_layout()"], "execution_count": 25}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["regi = train.groupby('registration_init_time').mean()['is_churn']\n", "plt.figure(figsize=(14,6))\n", "sns.distplot(regi,bins=100,color='r')"], "execution_count": 26}, {"source": ["# registration"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["fig,ax = plt.subplots(2,2,figsize=(16,8))\n", "ax1,ax2,ax3,ax4 = ax.flatten()\n", "sns.countplot(train['reg_day'],palette='Set2',ax=ax1)\n", "sns.countplot(data=train,x='reg_month',palette='Set1',ax=ax2)\n", "sns.countplot(data=train,x='reg_year',palette='magma',ax=ax3)\n"], "execution_count": 27}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["cor = train.corr()\n", "plt.figure(figsize=(16,12))\n", "sns.heatmap(cor,cmap='Set1',annot=False)\n", "plt.xticks(rotation=45);"], "execution_count": 28}, {"source": ["# Encoder"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {"collapsed": true}, "cell_type": "code", "source": ["le = LabelEncoder()\n", "train['gender'] = le.fit_transform(train['gender'])\n", "test['gender'] = le.fit_transform(test['gender'])"], "execution_count": 29}, {"source": ["# One Hot Encoding"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["def OHE(df):\n", "    #col = df.select_dtypes(include=['category']).columns\n", "    col = ['city','gender','registered_via']\n", "    print('Categorical columns in dataset',col)\n", "    \n", "    c2,c3 = [],{}\n", "    for c in col:\n", "        if df[c].nunique()>2 :\n", "            c2.append(c)\n", "            c3[c] = 'ohe_'+c\n", "    \n", "    df = pd.get_dummies(df,columns=c2,drop_first=True,prefix=c3)\n", "    print(df.shape)\n", "    return df\n", "train1 = OHE(train)\n", "test1 = OHE(test)"], "execution_count": 30}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["train1.columns"], "execution_count": 31}, {"source": ["# Split data set"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {"collapsed": true}, "cell_type": "code", "source": ["unwanted = ['msno','is_churn','registration_init_time','transaction_date','membership_expire_date','date']\n", "\n", "X = train1.drop(unwanted,axis=1)\n", "y = train1['is_churn'].astype('category')\n", "x_test = test1.drop(unwanted,axis=1)\n"], "execution_count": 32}, {"source": ["## Hyper parameter tuning"], "metadata": {}, "cell_type": "markdown"}, {"source": ["log_reg = LogisticRegression(class_weight='balanced')\n", "param = {'C':[0.001,0.005,0.01,0.05,0.1,0.5,1,1.5,2,3]}\n", "rs_cv = RandomizedSearchCV(estimator=log_reg,param_distributions=param,random_state=seed)\n", "rs_cv.fit(X,y)\n", "print('Best parameter :{} Best score :{}'.format(rs_cv.best_params_,rs_cv.best_score_))"], "metadata": {}, "cell_type": "markdown"}, {"source": ["# Logistic regression model with Stratified KFold split"], "metadata": {}, "cell_type": "markdown"}, {"source": ["#\n", "\n", "kf = StratifiedKFold(n_splits=5,shuffle=True,random_state=seed)\n", "pred_test_full =0\n", "cv_score =[]\n", "i=1\n", "for train_index,test_index in kf.split(X,y):\n", "    print('{} of KFold {}'.format(i,kf.n_splits))\n", "    xtr,xvl = X.loc[train_index],X.loc[test_index]\n", "    ytr,yvl = y.loc[train_index],y.loc[test_index]    \n", "    #model\n", "    lr = LogisticRegression(class_weight='balanced',C=1)\n", "    lr.fit(xtr,ytr)\n", "    score = lr.score(xvl,yvl)\n", "    print('ROC AUC score:',score)\n", "    cv_score.append(score)    \n", "    pred_test = lr.predict_proba(x_test)[:,1]\n", "    pred_test_full +=pred_test\n", "    i+=1"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {}, "cell_type": "code", "source": ["lr = LogisticRegression(class_weight='balanced',C=1)\n", "lr.fit(X,y)\n", "y_pred = lr.predict_proba(x_test)[:,1]\n", "lr.score(X,y)"], "execution_count": null}, {"source": ["# Model validation"], "metadata": {}, "cell_type": "markdown"}, {"source": ["print(cv_score)\n", "print('\\nMean accuracy',np.mean(cv_score))\n", "confusion_matrix(yvl,lr.predict(xvl))"], "metadata": {}, "cell_type": "markdown"}, {"source": ["## Reciever Operating Charactaristics"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {"collapsed": true}, "cell_type": "code", "source": ["y_proba = lr.predict_proba(X)[:,1]\n", "fpr,tpr,th = roc_curve(y,y_proba)\n", "\n", "plt.figure(figsize=(14,6))\n", "plt.plot(fpr,tpr,color='r')\n", "plt.plot([0,1],[0,1],color='b')\n", "plt.title('Reciever operating Charactaristics')\n", "plt.xlabel('False positive rate')\n", "plt.ylabel('True positive rate')"], "execution_count": null}, {"source": ["# Predict for unseen data set"], "metadata": {}, "cell_type": "markdown"}, {"outputs": [], "metadata": {"collapsed": true}, "cell_type": "code", "source": ["#y_pred = pred_test_full/5\n", "submit = pd.DataFrame({'msno':test['msno'],'is_churn':y_pred})\n", "submit.to_csv('kk_pred.csv',index=False)\n", "#submit.to_csv('kk_pred.csv.gz',index=False,compression='gzip')"], "execution_count": null}, {"source": ["# Thank you for visiting"], "metadata": {}, "cell_type": "markdown"}], "nbformat": 4, "nbformat_minor": 1}