{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "94cf8498-4d4b-a089-0878-49a45b0d97f2"
      },
      "source": [
        "# 1 Data import"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "8541bdb4-01b2-6692-733c-45fc596f2558"
      },
      "outputs": [],
      "source": [
        "import numpy as np\n",
        "import pandas as pd\n",
        "import seaborn as sns\n",
        "import matplotlib.pyplot as plt\n",
        "%matplotlib inline\n",
        "\n",
        "titanic_df = pd.read_csv('../input/train.csv')\n",
        "test_df = pd.read_csv('../input/test.csv')\n",
        "\n",
        "titanic_df = titanic_df.drop(['PassengerId'], axis = 1)"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "0962f54b-76e2-00ee-eabc-47e3ae3eb2a9"
      },
      "source": [
        "# 2 Exploration"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "c4322e5d-beb6-13ab-4f58-bae72d834280"
      },
      "source": [
        "## 2.1 Sex vs. Survival"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d488d76a-ba95-7700-9c68-84ee644447b9"
      },
      "outputs": [],
      "source": [
        "sns.factorplot(\"Sex\", data = titanic_df, kind = 'count',hue = \"Survived\")"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "017834c6-cb6e-dc1b-941b-e42acab80153"
      },
      "source": [
        "## 2.2 Family size vs. Survival\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "685e066e-511e-d4d1-2707-0363fa03fc7d"
      },
      "outputs": [],
      "source": [
        "sns.factorplot(\"SibSp\", data = titanic_df, kind = 'count',hue = \"Survived\")"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "71b770ca-fb89-c992-c304-ccf315ac313f"
      },
      "outputs": [],
      "source": [
        "sns.factorplot(\"Parch\", data = titanic_df, kind = 'count',hue = \"Survived\")"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "8f896da2-dc8d-0b60-0a01-d845f43f1c45"
      },
      "source": [
        "## 3 Cleanup data"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "10838d2f-0d71-daa5-5d6e-b81e2dc98ad2"
      },
      "outputs": [],
      "source": [
        "def df_cleanup(df):\n",
        "    df = df.drop(['Cabin', 'Name', 'Ticket'], axis = 1)\n",
        "    \n",
        "    for passenger in df[(df['Age'].isnull())].index:\n",
        "        df.loc[passenger, 'Age'] = np.average(df[(df['Age'].notnull())]['Age'])\n",
        "\n",
        "    for passenger in df[(df['Fare'].isnull())].index:\n",
        "        df.loc[passenger, 'Fare'] = np.average(df[(df['Fare'].notnull())]['Fare'])\n",
        "\n",
        "    df = sex_category(df)\n",
        "    df = embark_category(df)\n",
        "    df = family_category(df)\n",
        "    df = child_category(df)\n",
        "    \n",
        "    df[['Sex','Embarked']] = df[['Sex','Embarked']].apply(pd.to_numeric)\n",
        "    return df"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a609dc7d-1c78-c2dd-2d68-33f221c9c221"
      },
      "outputs": [],
      "source": [
        "def sex_category(df):\n",
        "    df.loc[(df['Sex'] == 'male'), 'Sex'] = 0\n",
        "    df.loc[(df['Sex'] == 'female'), 'Sex'] = 1\n",
        "    df.loc[(df['Sex'].isnull()), 'Sex'] = 2\n",
        "    return df"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e62d6a8b-4c9f-289f-c24d-865998ee12ab"
      },
      "outputs": [],
      "source": [
        "def embark_category(df):\n",
        "    df.loc[(df['Embarked'] == 'S'), 'Embarked'] = 0\n",
        "    df.loc[(df['Embarked'] == 'C'), 'Embarked'] = 1\n",
        "    df.loc[(df['Embarked'] == 'Q'), 'Embarked'] = 2\n",
        "    df.loc[(df['Embarked'].isnull()), 'Embarked'] = 3\n",
        "    return df"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e3e06adb-63ea-0906-49a6-c17f18ca264c"
      },
      "outputs": [],
      "source": [
        "def family_category(df):\n",
        "    df[\"FamilySize\"] = df[\"SibSp\"] + df[\"Parch\"]\n",
        "    return df"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f35a3bda-c3ec-673e-6520-a25ff34dc654"
      },
      "outputs": [],
      "source": [
        "def child_category(df):\n",
        "    df['Children'] = df['Age'].map(lambda x: 1 if x < 6.0 else 0)\n",
        "    return df"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "cdcf0469-caa7-05c5-0d30-3b25e2183c17"
      },
      "outputs": [],
      "source": [
        "titanic_df = df_cleanup(titanic_df)\n",
        "test_df = df_cleanup(test_df)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9be636d7-a292-701f-1046-336009ead859"
      },
      "outputs": [],
      "source": [
        "features_list = list(titanic_df.columns.values)\n",
        "\n",
        "X_train = titanic_df.drop(\"Survived\",axis=1)\n",
        "Y_train = titanic_df[\"Survived\"]\n",
        "X_test  = test_df.drop(\"PassengerId\",axis=1).copy()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c160c5a8-8399-c924-c8a8-898fb33af8f0"
      },
      "outputs": [],
      "source": [
        "from sklearn.pipeline import Pipeline\n",
        "from sklearn.feature_selection import SelectPercentile, SelectKBest, f_classif, chi2\n",
        "from sklearn.model_selection import StratifiedShuffleSplit\n",
        "from sklearn.model_selection import GridSearchCV\n",
        "from sklearn.metrics import classification_report\n",
        "from sklearn.preprocessing import MinMaxScaler\n",
        "from sklearn.ensemble import AdaBoostClassifier, RandomForestClassifier\n",
        "from sklearn import tree\n",
        "\n",
        "classifier = \"Rforest\"\n",
        "\n",
        "if classifier == \"Ada\":\n",
        "    pipe = Pipeline([\n",
        "           ('k_best', SelectKBest()),\n",
        "           ('classify', AdaBoostClassifier())\n",
        "        ])\n",
        "        \n",
        "    param_grid = ([\n",
        "            {\n",
        "                'k_best__k': [2,3,4,5,6,7,8],\n",
        "                'classify__n_estimators':[10, 15],\n",
        "                'classify__algorithm': ['SAMME', 'SAMME.R'],\n",
        "                'classify__learning_rate': [0.2, 0.5, 1.0, 1.5, 2.0]\n",
        "            }\n",
        "        ])\n",
        "        \n",
        "elif classifier == \"DTree\":\n",
        "    pipe = Pipeline([\n",
        "           ('k_best', SelectKBest()),\n",
        "           ('classify', tree.DecisionTreeClassifier())\n",
        "        ])\n",
        "        \n",
        "    param_grid = ([\n",
        "            {\n",
        "                'k_best__k': ['all'],\n",
        "                'classify__max_features': [0.5, 1.0, 'sqrt', 'auto'],\n",
        "                'classify__max_depth': [4, 6, 8, 10, 12, None]\n",
        "            }\n",
        "        ])\n",
        "        \n",
        "elif classifier == \"Rforest\":\n",
        "    pipe = Pipeline([\n",
        "           ('k_best', SelectKBest()),\n",
        "           ('classify', RandomForestClassifier())\n",
        "        ])\n",
        "        \n",
        "    param_grid = ([\n",
        "            {\n",
        "                'k_best__k': [2,3,4,5,6,7,8],\n",
        "                'classify__criterion':['gini', 'entropy'],\n",
        "                'classify__max_features': [0.5, 1.0, 'sqrt', 'auto'],\n",
        "                'classify__max_depth': [4, 6, 8, 10, 12, None]\n",
        "            }\n",
        "        ])\n",
        "\n",
        "\n",
        "sss = StratifiedShuffleSplit()\n",
        "clf = GridSearchCV(pipe, param_grid = param_grid, cv = sss, scoring='roc_auc')\n",
        "clf.fit(X_train, Y_train)\n",
        "\n",
        "print(\"(clf.best_estimator_.steps): \", (clf.best_estimator_.steps))\n",
        "print( \"(clf.best_score_): \", (clf.best_score_))\n",
        "print( \"(clf.best_params_): \", (clf.best_params_))\n",
        "print( \"(clf.scorer_): \", (clf.scorer_))\n",
        "\n",
        "chosen_features = clf.best_estimator_.named_steps['k_best'].get_support(indices=True)\n",
        "finalFeatureList = [features_list[i+1] for i in chosen_features]\n",
        "print(finalFeatureList)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "422680ea-b434-de19-5ad4-dc007139e725"
      },
      "outputs": [],
      "source": [
        "Y_pred = clf.predict(X_test)\n",
        "\n",
        "print(clf.score(X_train, Y_train))\n",
        "\n",
        "submission = pd.DataFrame({\n",
        "        \"PassengerId\": test_df[\"PassengerId\"],\n",
        "        \"Survived\": Y_pred\n",
        "    })\n",
        "submission.to_csv('titanic.csv', index=False)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b81decfc-8b7b-4e23-86a8-1337b1f5615a",
        "collapsed": true
      },
      "outputs": [],
      "source": [
        ""
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}