{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "3e534246-7c5b-1676-2193-745cfeb4ce54"
      },
      "source": [
        "read input files"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1f00d01f-4139-9bb5-08c5-a1bc210f7cfb"
      },
      "outputs": [],
      "source": [
        "import pandas as pd\n",
        "train_df = pd.read_csv('../input/train.csv')\n",
        "test_df = pd.read_csv('../input/test.csv')"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "74113fe3-3376-177a-5df6-0182744728a1"
      },
      "source": [
        "mapping functions"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "77ecee62-56cf-e44b-c496-e4fb8fea9b4e"
      },
      "outputs": [],
      "source": [
        "def sex_map(sex):\n",
        "    if sex == 'female':\n",
        "        return 0\n",
        "    else:\n",
        "        return 1"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "fcd4a41c-06ed-2ae0-c587-b6f4432bf486"
      },
      "source": [
        "transform data field"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b8817d7f-0de6-f9d4-ed8a-a4622ca0c7a2"
      },
      "outputs": [],
      "source": [
        "combine = [train_df, test_df]\n",
        "for d in combine:\n",
        "    d['Sex'] = d['Sex'].map(sex_map)"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "94fc4615-b278-e10a-94e9-1764fb7f7121"
      },
      "source": [
        "select feature vectors"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f9932c18-12f5-acaf-8511-014ef86f502a"
      },
      "outputs": [],
      "source": [
        "fields = ['PassengerId','Sex']\n",
        "X_train = train_df.select(lambda f : f in fields, axis=1)\n",
        "Y_train = train_df['Survived']\n",
        "X_test = test_df.select(lambda f : f in fields, axis=1)\n",
        "X_train.shape,X_test.shape,Y_train.shape"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "db5fd425-b4e5-ed90-3f3b-3238de6fa813"
      },
      "source": [
        "prediction using naive bayes"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f1e7253a-b150-3841-da53-56f787f2d668"
      },
      "outputs": [],
      "source": [
        "from sklearn.naive_bayes import GaussianNB\n",
        "gaussian = GaussianNB()\n",
        "gaussian.fit(X_train, Y_train)\n",
        "Y_pred = gaussian.predict(X_test)\n",
        "gaussian.score(X_train, Y_train)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3ece765b-3b70-4781-10d0-677ea02e0ffd"
      },
      "outputs": [],
      "source": [
        "submission = pd.DataFrame({\n",
        "        \"PassengerId\": test_df[\"PassengerId\"],\n",
        "        \"Survived\": Y_pred\n",
        "    })\n",
        "submission.to_csv('titanic-output.csv', index=False)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}