{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "cf40f2a7-201e-bfca-779d-430de5f89053"
      },
      "source": [
        "Notes and Links\n",
        "---------\n",
        "\n",
        "+ [scikit tutorial](https://www.kaggle.com/jeffd23/titanic/scikit-learn-ml-from-start-to-finish)\n",
        "+ [step by step tutorial](https://www.kaggle.com/startupsci/titanic/titanic-data-science-solutions)\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "897380c6-2aea-541d-b0ff-df6c288da86a"
      },
      "outputs": [],
      "source": [
        "# Libs to work with the data\n",
        "import pandas as pd\n",
        "import numpy as np\n",
        "import random as rnd\n",
        "import seaborn as sns\n",
        "sns.set(style=\"darkgrid\")\n",
        "import matplotlib.pyplot as plt\n",
        "\n",
        "# Print out what files are available to work with\n",
        "from subprocess import check_output\n",
        "print(\"Files:\\n\", check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Load the data\n",
        "train_raw = pd.read_csv(\"../input/train.csv\")\n",
        "test_raw = pd.read_csv(\"../input/test.csv\")\n",
        "combined_raw = pd.concat([train_raw, train_raw])\n",
        "\n",
        "# Let's look at the possible data to even play with\n",
        "print(\"Columns:\\n\", combined_raw.columns.values)\n",
        "\n",
        "# Let's look at the data\n",
        "# Categorical Features\n",
        "print(\"\\nCategorical:\\n\", combined_raw.describe(include=['O']))\n",
        "# Numerical Features\n",
        "combined_raw.describe()"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "4b80eb90-11a0-5840-1106-fc1b18823a24"
      },
      "source": [
        "###Meeting March 21 2017 - Cleaning Data, maybe looking at Cabins"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f2a7653c-fae4-430c-e110-2f546ed996e7"
      },
      "outputs": [],
      "source": [
        "def simplify_ages(df):\n",
        "    df.Age = df.Age.fillna(-0.5)\n",
        "    bins = (-1, 0, 5, 12, 18, 25, 35, 60, 120)\n",
        "    group_names = ['Unknown', 'Baby', 'Child', 'Teenager', 'Student', 'Young Adult', 'Adult', 'Senior']\n",
        "    categories = pd.cut(df.Age, bins, labels=group_names)\n",
        "    df.Age = categories\n",
        "    return df\n",
        "\n",
        "def simplify_cabins(df):\n",
        "    df.Cabin = df.Cabin.fillna('N')\n",
        "    df.Cabin = df.Cabin.apply(lambda x: x[0])\n",
        "    return df\n",
        "\n",
        "def simplify_fares(df):\n",
        "    df.Fare = df.Fare.fillna(-0.5)\n",
        "    bins = (-1, 0, 8, 15, 31, 1000)\n",
        "    group_names = ['Unknown', '1_quartile', '2_quartile', '3_quartile', '4_quartile']\n",
        "    categories = pd.cut(df.Fare, bins, labels=group_names)\n",
        "    df.Fare = categories\n",
        "    return df\n",
        "\n",
        "def format_name(df):\n",
        "    df['Lname'] = df.Name.apply(lambda x: x.split(' ')[0])\n",
        "    df['NamePrefix'] = df.Name.apply(lambda x: (x.split(',')[1]).split('.')[0])\n",
        "    return df    \n",
        "    \n",
        "def drop_features(df):\n",
        "    return df.drop(['Ticket', 'Embarked'], axis=1)\n",
        "\n",
        "def transform_features(df):\n",
        "    df = simplify_ages(df)\n",
        "    df = simplify_cabins(df)\n",
        "    df = simplify_fares(df)\n",
        "    df = format_name(df)\n",
        "    df = drop_features(df)\n",
        "    return df\n",
        "\n",
        "train_formatted = transform_features(train_raw)\n",
        "test_formatted = transform_features(test_raw)\n",
        "combined_formatted = pd.concat([train_formatted, test_formatted])\n",
        "combined_formatted.head()\n",
        "\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a45decff-7282-aa1c-01a2-414ddd74ac27"
      },
      "outputs": [],
      "source": [
        "plt.figure(figsize=(8,8))\n",
        "sns.countplot(y=\"NamePrefix\", data=combined_formatted)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e0b995fd-8ee2-7b07-5e19-183623cd4490"
      },
      "outputs": [],
      "source": [
        "combined_formatted.loc[combined_formatted['NamePrefix'] == \"y\"]"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}