{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "754603d1-6faa-8c5f-4fed-2e8881c92171"
      },
      "source": [
        "**<h1>First Bit of Code</h1>**"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "49b13e2c-3c7d-2050-05c4-6848de1d8064"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "bc914698-5891-800b-028c-7d5da66b1aa1"
      },
      "outputs": [],
      "source": [
        "# Ignore warnings\n",
        "import warnings\n",
        "warnings.filterwarnings('ignore')\n",
        "\n",
        "# Handle table-like data and matrices\n",
        "import numpy as np\n",
        "import pandas as pd\n",
        "\n",
        "# Modelling Algorithms\n",
        "from sklearn.tree import DecisionTreeClassifier\n",
        "from sklearn.linear_model import LogisticRegression\n",
        "from sklearn.neighbors import KNeighborsClassifier\n",
        "from sklearn.naive_bayes import GaussianNB\n",
        "from sklearn.svm import SVC, LinearSVC\n",
        "from sklearn.ensemble import RandomForestClassifier , GradientBoostingClassifier\n",
        "\n",
        "# Modelling Helpers\n",
        "from sklearn.preprocessing import Imputer , Normalizer , scale\n",
        "from sklearn.cross_validation import train_test_split , StratifiedKFold\n",
        "from sklearn.feature_selection import RFECV\n",
        "\n",
        "# Visualisation\n",
        "import matplotlib as mpl\n",
        "import matplotlib.pyplot as plt\n",
        "import matplotlib.pylab as pylab\n",
        "import seaborn as sns\n",
        "\n",
        "# Configure visualisations\n",
        "%matplotlib inline\n",
        "mpl.style.use( 'ggplot' )\n",
        "sns.set_style( 'white' )\n",
        "pylab.rcParams[ 'figure.figsize' ] = 8 , 6"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fef8e781-76b3-4b58-0bf8-c01238327695"
      },
      "outputs": [],
      "source": [
        "def plot_histograms( df , variables , n_rows , n_cols ):\n",
        "    fig = plt.figure( figsize = ( 16 , 12 ) )\n",
        "    for i, var_name in enumerate( variables ):\n",
        "        ax=fig.add_subplot( n_rows , n_cols , i+1 )\n",
        "        df[ var_name ].hist( bins=10 , ax=ax )\n",
        "        ax.set_title( 'Skew: ' + str( round( float( df[ var_name ].skew() ) , ) ) ) # + ' ' + var_name ) #var_name+\" Distribution\")\n",
        "        ax.set_xticklabels( [] , visible=False )\n",
        "        ax.set_yticklabels( [] , visible=False )\n",
        "    fig.tight_layout()  # Improves appearance a bit.\n",
        "    plt.show()\n",
        "\n",
        "def plot_distribution( df , var , target , **kwargs ):\n",
        "    row = kwargs.get( 'row' , None )\n",
        "    col = kwargs.get( 'col' , None )\n",
        "    facet = sns.FacetGrid( df , hue=target , aspect=4 , row = row , col = col )\n",
        "    facet.map( sns.kdeplot , var , shade= True )\n",
        "    facet.set( xlim=( 0 , df[ var ].max() ) )\n",
        "    facet.add_legend()\n",
        "\n",
        "def plot_categories( df , cat , target , **kwargs ):\n",
        "    row = kwargs.get( 'row' , None )\n",
        "    col = kwargs.get( 'col' , None )\n",
        "    facet = sns.FacetGrid( df , row = row , col = col )\n",
        "    facet.map( sns.barplot , cat , target )\n",
        "    facet.add_legend()\n",
        "\n",
        "def plot_correlation_map( df ):\n",
        "    corr = titanic.corr()\n",
        "    _ , ax = plt.subplots( figsize =( 12 , 10 ) )\n",
        "    cmap = sns.diverging_palette( 220 , 10 , as_cmap = True )\n",
        "    _ = sns.heatmap(\n",
        "        corr, \n",
        "        cmap = cmap,\n",
        "        square=True, \n",
        "        cbar_kws={ 'shrink' : .9 }, \n",
        "        ax=ax, \n",
        "        annot = True, \n",
        "        annot_kws = { 'fontsize' : 12 }\n",
        "    )\n",
        "\n",
        "def describe_more( df ):\n",
        "    var = [] ; l = [] ; t = []\n",
        "    for x in df:\n",
        "        var.append( x )\n",
        "        l.append( len( pd.value_counts( df[ x ] ) ) )\n",
        "        t.append( df[ x ].dtypes )\n",
        "    levels = pd.DataFrame( { 'Variable' : var , 'Levels' : l , 'Datatype' : t } )\n",
        "    levels.sort_values( by = 'Levels' , inplace = True )\n",
        "    return levels\n",
        "\n",
        "def plot_variable_importance( X , y ):\n",
        "    tree = DecisionTreeClassifier( random_state = 99 )\n",
        "    tree.fit( X , y )\n",
        "    plot_model_var_imp( tree , X , y )\n",
        "    \n",
        "def plot_model_var_imp( model , X , y ):\n",
        "    imp = pd.DataFrame( \n",
        "        model.feature_importances_  , \n",
        "        columns = [ 'Importance' ] , \n",
        "        index = X.columns \n",
        "    )\n",
        "    imp = imp.sort_values( [ 'Importance' ] , ascending = True )\n",
        "    imp[ : 10 ].plot( kind = 'barh' )\n",
        "    print (model.score( X , y ))\n",
        "    "
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ddb17a98-cbb8-1e80-94fa-02b9651b732f"
      },
      "outputs": [],
      "source": [
        "# get titanic & test csv files as a DataFrame\n",
        "train = pd.read_csv(\"../input/train.csv\")\n",
        "test    = pd.read_csv(\"../input/test.csv\")\n",
        "\n",
        "full = train.append( test , ignore_index = True )\n",
        "titanic = full[ :891 ]\n",
        "\n",
        "del train , test\n",
        "\n",
        "print ('Datasets:' , 'full:' , full.shape , 'titanic:' , titanic.shape)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "00a6c5bb-b090-4600-cf7e-1bd0bd66b0d2"
      },
      "outputs": [],
      "source": [
        "titanic.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8038f04c-80a4-5305-15a9-fb2e5b36b652"
      },
      "outputs": [],
      "source": [
        "titanic.describe()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a23b63eb-527c-4681-8cf4-83cf4644d58e"
      },
      "outputs": [],
      "source": [
        "plot_correlation_map( titanic )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3dc51c1e-d808-6258-7744-7a3093b09265"
      },
      "outputs": [],
      "source": [
        "# Plot distributions of Age of passangers who survived or did not survive\n",
        "plot_distribution( titanic , var = 'Age' , target = 'Survived' , row = 'Sex' )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a5017228-b0bf-731b-a07b-fed74a290d82"
      },
      "outputs": [],
      "source": [
        "# Plot distributions of Age of passangers who survived or did not survive\n",
        "plot_distribution( titanic , var = 'Fare' , target = 'Survived' )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "bd70be88-e0ad-e5a6-3aa2-ba6d34b61b32"
      },
      "outputs": [],
      "source": [
        "# Plot survival rate by Embarked\n",
        "plot_categories( titanic , cat = 'Embarked' , target = 'Survived' )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1929b6e4-c79b-2a94-7270-cabf182018c1"
      },
      "outputs": [],
      "source": [
        "# Excersise 2\n",
        "# Plot survival rate by Sex\n",
        "plot_categories( titanic , cat = 'Sex' , target = 'Survived' )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a3c731e7-5c7e-6a72-ab34-08cd00f86b1a"
      },
      "outputs": [],
      "source": [
        "# Excersise 3\n",
        "# Plot survival rate by Pclass\n",
        "plot_categories( titanic , cat = 'Pclass' , target = 'Survived' )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d95cbb52-8ae0-af11-a784-b4ef0dbb68b4"
      },
      "outputs": [],
      "source": [
        "# Excersise 4\n",
        "# Plot survival rate by SibSp\n",
        "plot_categories( titanic , cat = 'SibSp' , target = 'Survived' )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "41b5caed-cca8-398a-3fa2-30789b6e2c8a"
      },
      "outputs": [],
      "source": [
        "# Excersise 5\n",
        "# Plot survival rate by Parch\n",
        "plot_categories( titanic , cat = 'Parch' , target = 'Survived' )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9144f6f1-9db1-c16d-ad4e-30199756d3d7"
      },
      "outputs": [],
      "source": [
        "# Transform Sex into binary values 0 and 1\n",
        "sex = pd.Series( np.where( full.Sex == 'male' , 1 , 0 ) , name = 'Sex' )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a9dd9909-cbc2-cefd-443d-f6e59d627e32"
      },
      "outputs": [],
      "source": [
        "# Create a new variable for every unique value of Embarked\n",
        "embarked = pd.get_dummies( full.Embarked , prefix='Embarked' )\n",
        "embarked.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1f0224df-65c2-7a6e-16c6-ab9cd3c41f2e"
      },
      "outputs": [],
      "source": [
        "# Create a new variable for every unique value of Embarked\n",
        "pclass = pd.get_dummies( full.Pclass , prefix='Pclass' )\n",
        "pclass.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "553e3d8d-3fbc-7618-024e-c0b29ae0570d"
      },
      "outputs": [],
      "source": [
        "# Create dataset\n",
        "imputed = pd.DataFrame()\n",
        "\n",
        "# Fill missing values of Age with the average of Age (mean)\n",
        "imputed[ 'Age' ] = full.Age.fillna( full.Age.mean() )\n",
        "\n",
        "# Fill missing values of Fare with the average of Fare (mean)\n",
        "imputed[ 'Fare' ] = full.Fare.fillna( full.Fare.mean() )\n",
        "\n",
        "imputed.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "66878c1d-ab87-951c-b246-ac3c37557096"
      },
      "outputs": [],
      "source": [
        "title = pd.DataFrame()\n",
        "# we extract the title from each name\n",
        "title[ 'Title' ] = full[ 'Name' ].map( lambda name: name.split( ',' )[1].split( '.' )[0].strip() )\n",
        "\n",
        "# a map of more aggregated titles\n",
        "Title_Dictionary = {\n",
        "                    \"Capt\":       \"Officer\",\n",
        "                    \"Col\":        \"Officer\",\n",
        "                    \"Major\":      \"Officer\",\n",
        "                    \"Jonkheer\":   \"Royalty\",\n",
        "                    \"Don\":        \"Royalty\",\n",
        "                    \"Sir\" :       \"Royalty\",\n",
        "                    \"Dr\":         \"Officer\",\n",
        "                    \"Rev\":        \"Officer\",\n",
        "                    \"the Countess\":\"Royalty\",\n",
        "                    \"Dona\":       \"Royalty\",\n",
        "                    \"Mme\":        \"Mrs\",\n",
        "                    \"Mlle\":       \"Miss\",\n",
        "                    \"Ms\":         \"Mrs\",\n",
        "                    \"Mr\" :        \"Mr\",\n",
        "                    \"Mrs\" :       \"Mrs\",\n",
        "                    \"Miss\" :      \"Miss\",\n",
        "                    \"Master\" :    \"Master\",\n",
        "                    \"Lady\" :      \"Royalty\"\n",
        "\n",
        "                    }\n",
        "\n",
        "# we map each title\n",
        "title[ 'Title' ] = title.Title.map( Title_Dictionary )\n",
        "title = pd.get_dummies( title.Title )\n",
        "#title = pd.concat( [ title , titles_dummies ] , axis = 1 )\n",
        "\n",
        "title.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "302aa6ad-5fa0-8579-8748-178ee85e26d8"
      },
      "outputs": [],
      "source": [
        "cabin = pd.DataFrame()\n",
        "\n",
        "# replacing missing cabins with U (for Uknown)\n",
        "cabin[ 'Cabin' ] = full.Cabin.fillna( 'U' )\n",
        "\n",
        "# mapping each Cabin value with the cabin letter\n",
        "cabin[ 'Cabin' ] = cabin[ 'Cabin' ].map( lambda c : c[0] )\n",
        "\n",
        "# dummy encoding ...\n",
        "cabin = pd.get_dummies( cabin['Cabin'] , prefix = 'Cabin' )\n",
        "\n",
        "cabin.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ee9d5ae4-5136-a2af-f496-0a9fbbacb9f8"
      },
      "outputs": [],
      "source": [
        "# a function that extracts each prefix of the ticket, returns 'XXX' if no prefix (i.e the ticket is a digit)\n",
        "def cleanTicket( ticket ):\n",
        "    ticket = ticket.replace( '.' , '' )\n",
        "    ticket = ticket.replace( '/' , '' )\n",
        "    ticket = ticket.split()\n",
        "    ticket = map( lambda t : t.strip() , ticket )\n",
        "    ticket = list(filter( lambda t : not t.isdigit() , ticket ))\n",
        "    if len( ticket ) > 0:\n",
        "        return ticket[0]\n",
        "    else: \n",
        "        return 'XXX'\n",
        "\n",
        "ticket = pd.DataFrame()\n",
        "\n",
        "# Extracting dummy variables from tickets:\n",
        "ticket[ 'Ticket' ] = full[ 'Ticket' ].map( cleanTicket )\n",
        "ticket = pd.get_dummies( ticket[ 'Ticket' ] , prefix = 'Ticket' )\n",
        "\n",
        "ticket.shape\n",
        "ticket.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "65d8835d-331c-b12a-85a3-39efb84d4bc5"
      },
      "outputs": [],
      "source": [
        "family = pd.DataFrame()\n",
        "\n",
        "# introducing a new feature : the size of families (including the passenger)\n",
        "family[ 'FamilySize' ] = full[ 'Parch' ] + full[ 'SibSp' ] + 1\n",
        "\n",
        "# introducing other features based on the family size\n",
        "family[ 'Family_Single' ] = family[ 'FamilySize' ].map( lambda s : 1 if s == 1 else 0 )\n",
        "family[ 'Family_Small' ]  = family[ 'FamilySize' ].map( lambda s : 1 if 2 <= s <= 4 else 0 )\n",
        "family[ 'Family_Large' ]  = family[ 'FamilySize' ].map( lambda s : 1 if 5 <= s else 0 )\n",
        "\n",
        "family.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fad87cc9-cf99-6389-5eaa-d5cd3ef5aea8"
      },
      "outputs": [],
      "source": [
        "full_X = pd.concat( [ imputed , embarked , cabin , sex ] , axis=1 )\n",
        "full_X.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "737b72c1-b060-3800-c297-0e1c624fc69e"
      },
      "outputs": [],
      "source": [
        "# Create all datasets that are necessary to train, validate and test models\n",
        "train_valid_X = full_X[ 0:891 ]\n",
        "train_valid_y = titanic.Survived\n",
        "test_X = full_X[ 891: ]\n",
        "train_X , valid_X , train_y , valid_y = train_test_split( train_valid_X , train_valid_y , train_size = .7 )\n",
        "\n",
        "print (full_X.shape , train_X.shape , valid_X.shape , train_y.shape , valid_y.shape , test_X.shape)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3195f4ee-e928-dbf5-7598-1853acb86008"
      },
      "outputs": [],
      "source": [
        "plot_variable_importance(train_X, train_y)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8d73595d-79dd-c8a7-cc89-df46177f2b8c"
      },
      "outputs": [],
      "source": [
        "model = RandomForestClassifier(n_estimators=100)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "74f928da-4c21-b426-2dc2-c35d76f001a5"
      },
      "outputs": [],
      "source": [
        "model = SVC()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "69e6366c-ad1f-048c-8753-64c069b1e5f3"
      },
      "outputs": [],
      "source": [
        "model = GradientBoostingClassifier()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "6757437e-b1c5-b480-4924-5351d03c016f"
      },
      "outputs": [],
      "source": [
        "model = KNeighborsClassifier(n_neighbors = 3)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "cd9ced88-1075-8ffc-734f-8f13947b468d"
      },
      "outputs": [],
      "source": [
        "model = LogisticRegression()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d0b46cf6-e849-4f10-38b0-a1d61689c17e"
      },
      "outputs": [],
      "source": [
        "model.fit( train_X , train_y )"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9853e1c6-70bc-e95c-f94e-b012b6210495"
      },
      "outputs": [],
      "source": [
        "# Score the model\n",
        "print (model.score( train_X , train_y ) , model.score( valid_X , valid_y ))"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9e5e2a08-5f89-3178-1358-1ea255a29c7d"
      },
      "outputs": [],
      "source": [
        "#plot_model_var_imp(model, train_X, train_y)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3966cfc4-0bd1-b2c2-6112-2bb9bbb5c66c"
      },
      "outputs": [],
      "source": [
        "rfecv = RFECV( estimator = model , step = 1 , cv = StratifiedKFold( train_y , 2 ) , scoring = 'accuracy' )\n",
        "rfecv.fit( train_X , train_y )\n",
        "\n",
        "#print (rfecv.score( train_X , train_y ) , rfecv.score( valid_X , valid_y ))\n",
        "#print( \"Optimal number of features : %d\" % rfecv.n_features_ )\n",
        "\n",
        "# Plot number of features VS. cross-validation scores\n",
        "#plt.figure()\n",
        "#plt.xlabel( \"Number of features selected\" )\n",
        "#plt.ylabel( \"Cross validation score (nb of correct classifications)\" )\n",
        "#plt.plot( range( 1 , len( rfecv.grid_scores_ ) + 1 ) , rfecv.grid_scores_ )\n",
        "#plt.show()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4b29b66b-0183-969e-75b0-34fd6c14a03f"
      },
      "outputs": [],
      "source": [
        "test_Y = model.predict( test_X )\n",
        "passenger_id = full[891:].PassengerId\n",
        "test = pd.DataFrame( { 'PassengerId': passenger_id , 'Survived': test_Y } )\n",
        "test.shape\n",
        "test.head()\n",
        "test.to_csv( 'titanic_pred.csv' , index = False )"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}