{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c8887209-fece-ace3-c60d-183a2e4e12e9"
      },
      "outputs": [],
      "source": [
        "# Ignore warnings\n",
        "import warnings\n",
        "warnings.filterwarnings('ignore')\n",
        "\n",
        "# Handle table-like data and matrices\n",
        "import numpy as np\n",
        "import pandas as pd\n",
        "\n",
        "# Modelling Algorithms\n",
        "from sklearn.tree import DecisionTreeClassifier\n",
        "from sklearn.linear_model import LogisticRegression\n",
        "from sklearn.neighbors import KNeighborsClassifier\n",
        "from sklearn.naive_bayes import GaussianNB\n",
        "from sklearn.svm import SVC, LinearSVC\n",
        "from sklearn.ensemble import RandomForestClassifier , GradientBoostingClassifier\n",
        "\n",
        "# Modelling Helpers\n",
        "from sklearn.preprocessing import Imputer , Normalizer , scale\n",
        "from sklearn.cross_validation import train_test_split , StratifiedKFold\n",
        "from sklearn.feature_selection import RFECV\n",
        "\n",
        "# Visualisation\n",
        "import matplotlib as mpl\n",
        "import matplotlib.pyplot as plt\n",
        "import matplotlib.pylab as pylab\n",
        "import seaborn as sns\n",
        "\n",
        "# Configure visualisations\n",
        "%matplotlib inline\n",
        "mpl.style.use( 'ggplot' )\n",
        "sns.set_style( 'white' )\n",
        "pylab.rcParams[ 'figure.figsize' ] = 8 , 6"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "185b12dd-f6ea-80db-8d18-0a1b610d5157"
      },
      "outputs": [],
      "source": [
        "def plot_histograms( df , variables , n_rows , n_cols ):\n",
        "    fig = plt.figure( figsize = ( 16 , 12 ) )\n",
        "    for i, var_name in enumerate( variables ):\n",
        "        ax=fig.add_subplot( n_rows , n_cols , i+1 )\n",
        "        df[ var_name ].hist( bins=10 , ax=ax )\n",
        "        ax.set_title( 'Skew: ' + str( round( float( df[ var_name ].skew() ) , ) ) ) # + ' ' + var_name ) #var_name+\" Distribution\")\n",
        "        ax.set_xticklabels( [] , visible=False )\n",
        "        ax.set_yticklabels( [] , visible=False )\n",
        "    fig.tight_layout()  # Improves appearance a bit.\n",
        "    plt.show()\n",
        "\n",
        "def plot_distribution( df , var , target , **kwargs ):\n",
        "    row = kwargs.get( 'row' , None )\n",
        "    col = kwargs.get( 'col' , None )\n",
        "    facet = sns.FacetGrid( df , hue=target , aspect=4 , row = row , col = col )\n",
        "    facet.map( sns.kdeplot , var , shade= True )\n",
        "    facet.set( xlim=( 0 , df[ var ].max() ) )\n",
        "    facet.add_legend()\n",
        "\n",
        "def plot_categories( df , cat , target , **kwargs ):\n",
        "    row = kwargs.get( 'row' , None )\n",
        "    col = kwargs.get( 'col' , None )\n",
        "    facet = sns.FacetGrid( df , row = row , col = col )\n",
        "    facet.map( sns.barplot , cat , target )\n",
        "    facet.add_legend()\n",
        "\n",
        "def plot_correlation_map( df ):\n",
        "    corr = titanic.corr()\n",
        "    _ , ax = plt.subplots( figsize =( 12 , 10 ) )\n",
        "    cmap = sns.diverging_palette( 220 , 10 , as_cmap = True )\n",
        "    _ = sns.heatmap(\n",
        "        corr, \n",
        "        cmap = cmap,\n",
        "        square=True, \n",
        "        cbar_kws={ 'shrink' : .9 }, \n",
        "        ax=ax, \n",
        "        annot = True, \n",
        "        annot_kws = { 'fontsize' : 12 }\n",
        "    )\n",
        "\n",
        "def describe_more( df ):\n",
        "    var = [] ; l = [] ; t = []\n",
        "    for x in df:\n",
        "        var.append( x )\n",
        "        l.append( len( pd.value_counts( df[ x ] ) ) )\n",
        "        t.append( df[ x ].dtypes )\n",
        "    levels = pd.DataFrame( { 'Variable' : var , 'Levels' : l , 'Datatype' : t } )\n",
        "    levels.sort_values( by = 'Levels' , inplace = True )\n",
        "    return levels\n",
        "\n",
        "def plot_variable_importance( X , y ):\n",
        "    tree = DecisionTreeClassifier( random_state = 99 )\n",
        "    tree.fit( X , y )\n",
        "    plot_model_var_imp( tree , X , y )\n",
        "    \n",
        "def plot_model_var_imp( model , X , y ):\n",
        "    imp = pd.DataFrame( \n",
        "        model.feature_importances_  , \n",
        "        columns = [ 'Importance' ] , \n",
        "        index = X.columns \n",
        "    )\n",
        "    imp = imp.sort_values( [ 'Importance' ] , ascending = True )\n",
        "    imp[ : 10 ].plot( kind = 'barh' )\n",
        "    print (model.score( X , y ))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f6015875-892e-f8c5-f0b8-1c34dcaf3f11"
      },
      "outputs": [],
      "source": [
        "# get titanic & test csv files as a DataFrame\n",
        "train = pd.read_csv(\"../input/train.csv\")\n",
        "test    = pd.read_csv(\"../input/test.csv\")\n",
        "\n",
        "full = train.append( test , ignore_index = True )\n",
        "titanic = full[ :891 ]\n",
        "\n",
        "del train , test\n",
        "\n",
        "print ('Datasets:' , 'full:' , full.shape , 'titanic:' , titanic.shape)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f026aa69-1b06-d614-f14a-025a1464e5c6"
      },
      "outputs": [],
      "source": [
        "# Run the code to see the variables, then read the variable description below to understand them.\n",
        "titanic.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "38f773ed-f16b-4faf-b15c-59ff7d7f4c51"
      },
      "outputs": [],
      "source": [
        ""
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a6a1da2f-ca1e-c40a-7eca-cb37a66920e4"
      },
      "outputs": [],
      "source": [
        "titanic.describe()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "bb8be1a1-a92d-b318-ff61-741604821499"
      },
      "outputs": [],
      "source": [
        "plot_correlation_map(titanic)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "28f37db5-d686-c103-c291-532280670b13"
      },
      "outputs": [],
      "source": [
        "plot_distribution(titanic, var='Age', target='Survived', row='Sex')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "72818b12-6a85-f115-6503-a089266723b8"
      },
      "outputs": [],
      "source": [
        "plot_distribution(titanic, var='Fare', target='Survived', row='Sex')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "dc10055e-3293-f38b-8f5f-0685a9764955"
      },
      "outputs": [],
      "source": [
        "plot_categories(titanic, cat='Embarked', target='Survived')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "603feee9-2515-f016-17a3-e532eca17e12"
      },
      "outputs": [],
      "source": [
        "plot_categories(titanic, cat='Sex', target='Survived')\n",
        "plot_categories(titanic, cat='Pclass', target='Survived')\n",
        "plot_categories(titanic, cat='SibSp', target='Survived')\n",
        "plot_categories(titanic, cat='Parch', target='Survived')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a6379969-67ea-bd00-5e1c-143b5e2553c8"
      },
      "outputs": [],
      "source": [
        "sex = pd.Series(np.where(full.Sex == 'male', 1, 0), name = 'Sex')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4648e2fb-ab4c-ffd9-10fa-bb654fe86dfc"
      },
      "outputs": [],
      "source": [
        "embarked = pd.get_dummies(full.Embarked, prefix='Embarked')\n",
        "embarked.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "65e692bf-476e-47df-a004-9347f12684d0"
      },
      "outputs": [],
      "source": [
        "pclass = pd.get_dummies(full.Pclass, prefix='Pclass')\n",
        "pclass.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a723e968-a874-dcdd-6c81-8468751d9c16"
      },
      "outputs": [],
      "source": [
        "imputed = pd.DataFrame()\n",
        "imputed['Age'] = full.Age.fillna(full.Age.mean())\n",
        "imputed['Fare'] = full.Fare.fillna(full.Fare.mean())\n",
        "imputed.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "dca09ebe-e5de-8367-6d89-f5470acac8d1"
      },
      "outputs": [],
      "source": [
        "title = pd.DataFrame()\n",
        "title['Title'] = full['Name'].map(lambda name: name.split(',')[1].split('.')[0].strip())\n",
        "\n",
        "Title_Dictionary = {\n",
        "    \"Capt\":       \"Officer\",\n",
        "    \"Col\":        \"Officer\",\n",
        "    \"Major\":      \"Officer\",\n",
        "    \"Jonkheer\":   \"Royalty\",\n",
        "    \"Don\":        \"Royalty\",\n",
        "    \"Sir\" :       \"Royalty\",\n",
        "    \"Dr\":         \"Officer\",\n",
        "    \"Rev\":        \"Officer\",\n",
        "    \"the Countess\":\"Royalty\",\n",
        "    \"Dona\":       \"Royalty\",\n",
        "    \"Mme\":        \"Mrs\",\n",
        "    \"Mlle\":       \"Miss\",\n",
        "    \"Ms\":         \"Mrs\",\n",
        "    \"Mr\" :        \"Mr\",\n",
        "    \"Mrs\" :       \"Mrs\",\n",
        "    \"Miss\" :      \"Miss\",\n",
        "    \"Master\" :    \"Master\",\n",
        "    \"Lady\" :      \"Royalty\"\n",
        "}\n",
        "\n",
        "title['Title'] = title.Title.map(Title_Dictionary)\n",
        "title = pd.get_dummies(title.Title)\n",
        "title.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "52b30713-fd8e-5047-be32-c98625571339"
      },
      "outputs": [],
      "source": [
        "cabin = pd.DataFrame()\n",
        "\n",
        "cabin['Cabin'] = full.Cabin.fillna('U')\n",
        "cabin['Cabin'] = cabin['Cabin'].map(lambda c: c[0])\n",
        "cabin = pd.get_dummies(cabin['Cabin'], prefix='Cabin')\n",
        "cabin.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "816d8c3b-221f-7ca9-88c3-190886675aa3"
      },
      "outputs": [],
      "source": [
        "def cleanTicket(ticket):\n",
        "    ticket = ticket.replace('.','')\n",
        "    ticket = ticket.replace('/','')\n",
        "    ticket = ticket.split()\n",
        "    ticket = map(lambda t:t.strip(), ticket)\n",
        "    ticket = list(filter(lambda t : not t.isdigit(), ticket))\n",
        "    if len(ticket) > 0:\n",
        "        return ticket[0]\n",
        "    else:\n",
        "        return 'XXX'\n",
        "    \n",
        "ticket = pd.DataFrame()\n",
        "\n",
        "ticket['Ticket'] = full['Ticket'].map(cleanTicket)\n",
        "ticket = pd.get_dummies(ticket['Ticket'], prefix = 'Ticket')\n",
        "\n",
        "ticket.shape\n",
        "ticket.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "31d4adc3-8004-b552-2fa9-ab569b5689e6"
      },
      "outputs": [],
      "source": [
        "family = pd.DataFrame()\n",
        "\n",
        "family['FamilySize'] = full['Parch'] + full['SibSp'] + 1\n",
        "\n",
        "family['Family_Single'] = family['FamilySize'].map(lambda s : 1 if s == 1 else 0)\n",
        "family['Family_Small'] = family['FamilySize'].map(lambda s : 1 if 2 <= s <= 4 else 0)\n",
        "family['Fmaily_Large'] = family['FamilySize'].map(lambda s : 1 if 5 <= s else 0)\n",
        "family.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "21b5c12f-2f68-0e8d-5315-523c5864b9aa"
      },
      "outputs": [],
      "source": [
        "full_X = pd.concat([imputed, embarked, cabin, sex], axis=1)\n",
        "full_X.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8e10eefc-853e-ec31-f6c3-5862057655e5"
      },
      "outputs": [],
      "source": [
        "train_valid_x = full_X[0:891]\n",
        "train_valid_y = titanic.Survived\n",
        "test_X = full_X[891:]\n",
        "train_X, valid_X, train_y, valid_y = train_test_split(train_valid_x, train_valid_y, train_size = 0.7)\n",
        "\n",
        "print(full_X.shape, train_X.shape, valid_X.shape, train_y.shape, valid_y.shape, test_X.shape)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1dcaab17-3d55-c19c-25b8-5bbe88299411"
      },
      "outputs": [],
      "source": [
        "plot_variable_importance(train_X, train_y)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "dfbfae16-a2ad-d6f6-ae09-7bbc7ce1df2e"
      },
      "outputs": [],
      "source": [
        "model = RandomForestClassifier(n_estimators=100)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0670a40b-4159-92c9-16c5-41c97d3c364e"
      },
      "outputs": [],
      "source": [
        "model = SVC()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c81b3a7c-c285-0000-0e2d-92eb1edf4b14"
      },
      "outputs": [],
      "source": [
        "model = GradientBoostingClassifier()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2acbfada-e518-435a-78cd-7cfc1cf8dc7d"
      },
      "outputs": [],
      "source": [
        "model = KNeighborsClassifier(n_neighbors = 3)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0a6fb3db-ed5b-ca22-d57b-a9c59805d6b9"
      },
      "outputs": [],
      "source": [
        "model = GaussianNB()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "35550142-0613-7027-00c4-5b045edb8954"
      },
      "outputs": [],
      "source": [
        "model = LogisticRegression()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4d2d1579-2e80-da87-35ec-d4bef870f9f9"
      },
      "outputs": [],
      "source": [
        "model.fit(train_X, train_y)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "296b9c6d-969f-49ec-a07c-4a05befed28a"
      },
      "outputs": [],
      "source": [
        "print(model.score(train_X, train_y), model.score(valid_X, valid_y))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "cb7b9a06-785e-7a84-37d2-7763d4e7c773"
      },
      "outputs": [],
      "source": [
        "rfecv = RFECV(estimator = model, step = 1, cv = StratifiedKFold(train_y, 2), scoring = 'accuracy')\n",
        "rfecv.fit(train_X, train_y)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ac327f82-9f52-0149-e0e2-c6e1514f0ddf"
      },
      "outputs": [],
      "source": [
        "test_Y = model.predict(test_X)\n",
        "passenger_id = full[891:].PassengerId\n",
        "test = pd.DataFrame({'PassengerId': passenger_id, 'Survived': test_Y})\n",
        "test.shape\n",
        "test.head()\n",
        "test.to_csv('titanic_pred.csv', index=False)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d7f1f06c-a70d-331d-4add-a089a4a45b86"
      },
      "outputs": [],
      "source": [
        ""
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}