{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "bba7eab1-29b2-ac70-acf2-8888c9272133"
      },
      "source": [
        "### Next meeting goals [2017-03-21]\n",
        "\n",
        "* Read some more of the tutorial material.\n",
        "* Explore the data in order to:\n",
        "  * classify existing data (e.g. clear up and classify the cabins)\n",
        "  * see if there are implied features (e.g. Mrs., Mr., Miss, from the name, or location on the ship based on ticket or cabin)\n",
        "  * do we need each column? Should it be left continuous? Quantized?\n",
        "* Do some research on the Titanic layout and accident."
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "b04d3b9e-8517-c0c8-0f5d-4a238449f577"
      },
      "source": [
        ""
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "870f187d-65ed-358c-26b8-993f4ec3a42e"
      },
      "outputs": [],
      "source": [
        "import numpy as np\n",
        "import pandas as pd\n",
        "import matplotlib.pyplot as plt\n",
        "import seaborn as sns\n",
        "%matplotlib inline\n",
        "\n",
        "raw_data_train = pd.read_csv('../input/train.csv')\n",
        "raw_data_test = pd.read_csv('../input/test.csv')\n",
        "data_combined = pd.concat([raw_data_train, raw_data_test])\n",
        "\n",
        "data_train.sample(3)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d0608567-7796-508c-1149-d59600601ff0"
      },
      "outputs": [],
      "source": [
        "def simplify_ages(df):\n",
        "    df.Age = df.Age.fillna(-0.5)\n",
        "    bins = (-1, 0, 5, 12, 18, 25, 35, 60, 120)\n",
        "    group_names = ['Unknown', 'Baby', 'Child', 'Teenager', 'Student', 'Young Adult', 'Adult', 'Senior']\n",
        "    categories = pd.cut(df.Age, bins, labels=group_names)\n",
        "    df.Age = categories\n",
        "    return df\n",
        "\n",
        "def simplify_cabins(df):\n",
        "    df.Cabin = df.Cabin.fillna('N')\n",
        "    df.Cabin = df.Cabin.apply(lambda x: x[0])\n",
        "    return df\n",
        "\n",
        "def simplify_fares(df):\n",
        "    df.Fare = df.Fare.fillna(-0.5)\n",
        "    bins = (-1, 0, 8, 15, 31, 1000)\n",
        "    #bins = (-1, 0, 20, 40, 60, 80, 100, 200, 400, 800)\n",
        "    group_names = ['Unknown', '1_quartile', '2_quartile', '3_quartile', '4_quartile']\n",
        "    #group_names = ['Unknown', '1_quartile', '2_quartile', '3_quartile', '4_quartile', 'A', 'B', 'C', 'D']\n",
        "    categories = pd.cut(df.Fare, bins, labels=group_names)\n",
        "    df['FareCat'] = categories\n",
        "    return df\n",
        "\n",
        "def format_name(df):\n",
        "    df['Lname'] = df.Name.apply(lambda x: x.split(',')[0])\n",
        "    df['NamePrefix'] = df.Name.apply(lambda x: (x.split(',')[1]).split('.')[0])\n",
        "    return df    \n",
        "    \n",
        "def drop_features(df):\n",
        "    return df.drop(['Ticket', 'Name', 'Embarked'], axis=1)\n",
        "\n",
        "def transform_features(df):\n",
        "    df = simplify_ages(df)\n",
        "    df = simplify_cabins(df)\n",
        "    df = simplify_fares(df)\n",
        "    df = format_name(df)\n",
        "    # df = drop_features(df)\n",
        "    return df\n",
        "\n",
        "data_combined = transform_features(data_combined)\n",
        "data_combined.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b0abed86-4a5f-faa2-a370-74e34dbbc70d"
      },
      "outputs": [],
      "source": [
        "plt.figure(figsize=(10,10))\n",
        "plot = sns.countplot(y=\"NamePrefix\", data=data_combined)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5cf4f5bb-3a9e-510a-e89d-750e8a295fd2"
      },
      "outputs": [],
      "source": [
        "plt.figure(figsize=(10,10))\n",
        "sns.barplot(y=\"NamePrefix\", x=\"Survived\", data=data_combined)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "84a25274-4301-1af8-b655-68cbd112328c"
      },
      "outputs": [],
      "source": [
        "plt.figure(figsize=(10,10))\n",
        "sns.countplot(x=\"FareCat\", hue=\"Sex\", data=data_combined)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4d9e74cc-af7d-a8cb-faaf-efed8d595d45"
      },
      "outputs": [],
      "source": [
        "plt.figure(figsize=(10,10))\n",
        "sns.countplot(y=\"NamePrefix\", hue=\"Sex\", data=data_combined)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "499fead1-4d37-4e41-9593-694ba7345382"
      },
      "outputs": [],
      "source": [
        "data_combined.loc[data_combined[\"NamePrefix\"] == \" Dona\"]"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}