{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "739fd23d-abe0-b7f7-fd57-6e2b4c40f1f0"
      },
      "source": [
        "Using scikit learn for mlp classification"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e1ef14c3-b4e9-4c16-8675-09770f94b4f0"
      },
      "outputs": [],
      "source": ""
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "37389a2b-a083-d186-d332-9bc67298ea6f"
      },
      "outputs": [],
      "source": [
        "import pandas as pd\n",
        "import numpy as np\n",
        "from sklearn import cross_validation\n",
        "from sklearn import preprocessing\n",
        "from sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier\n",
        "\n",
        "df = pd.read_csv(\"../input/train.csv\", error_bad_lines=True)\n",
        "df2 = pd.read_csv('../input/test.csv')\n",
        "df = df[np.isfinite(df['Age'])]\n",
        "df = df[np.isfinite(df['Pclass'])]\n",
        "df = df[np.isfinite(df['Fare'])]\n",
        "df = df[np.isfinite(df['Survived'])]\n",
        "\n",
        "\n",
        "\n",
        "\n",
        "for index, row in df.iterrows():\n",
        "      df.loc[index, 'Sex'] = 0 if row['Sex'] == 'female' else 1\n",
        " \n",
        "for index, row in df2.iterrows():\n",
        "      df2.loc[index, 'Sex'] = 0 if row['Sex'] == 'female' else 1\n",
        "\n",
        "\n",
        "x = []\n",
        "y = []\n",
        "\n",
        "x2 = []\n",
        "\n",
        "for index, row in df.iterrows():\n",
        "    x.append([float(row['Sex']), row['Parch'], float(row['Age']), float(row['Fare']), float(row['Pclass'])])\n",
        "    y.append(row['Survived'])\n",
        "    \n",
        "for index, row in df.iterrows():\n",
        "    x2.append([float(row['Sex']), row['Parch'], float(row['Age']), float(row['Fare']), float(row['Pclass'])])\n",
        "    \n",
        "x = np.array(x)\n",
        "y = np.array(y)\n",
        "\n",
        "x2 = np.array(x2)\n",
        "\n",
        "\n",
        "\n",
        "clf = RandomForestClassifier(n_estimators=25, min_samples_split=2)\n",
        "clf.fit(x, y)\n",
        "\n",
        "result = clf.predict(x2)\n",
        "\n",
        "res = pd.DataFrame()\n",
        "for index, row in df2.iterrows():\n",
        "    res = res.append(\n",
        "        {\n",
        "            'PassengerId': str(row['PassengerId']),\n",
        "            'Survived': str(result[index]) \n",
        "        }, \n",
        "        ignore_index=True\n",
        "    )\n",
        "res.head()\n",
        "res.to_csv('submission.csv', index=False)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}