{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "4aa46d40-988c-04ba-36d5-f483d74513f4"
      },
      "source": [
        "solving titanic "
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0bc83fc9-32e4-1f1d-850c-24dd22ddde20"
      },
      "outputs": [],
      "source": [
        "import numpy as np\n",
        "import matplotlib.pyplot as plt\n",
        "import pandas as pd"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7c5be0a9-17f2-666a-db74-4d3bf5e4fe43"
      },
      "outputs": [],
      "source": [
        "dataset = pd.read_csv('../input/train.csv')\n",
        "\n",
        "dataset.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "10048947-205c-1405-0e53-b1c8b4528970"
      },
      "outputs": [],
      "source": [
        "X = dataset.iloc[:, [4 ,2,6,5 ]].values\n",
        "y = dataset.iloc[:, 1].values"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "749f749a-3042-220c-9e02-617189485050"
      },
      "outputs": [],
      "source": [
        "from sklearn.preprocessing import Imputer\n",
        "imputer = Imputer(missing_values =  \"NaN\" , strategy = \"mean\" , axis = 0)\n",
        "imputer = imputer.fit(X[:, 1:])\n",
        "X[:, 1:] = imputer.transform(X[:,1:])"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5f4e84f4-d0b6-d7a0-c6ed-115699e63659"
      },
      "outputs": [],
      "source": [
        "from sklearn.preprocessing import LabelEncoder\n",
        "labelencoder_X = LabelEncoder()\n",
        "X[:,0] = labelencoder_X.fit_transform(X[:,0])"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "59c872b0-3cee-9d31-cb9a-469285813745"
      },
      "outputs": [],
      "source": [
        "from mpl_toolkits.mplot3d import Axes3D\n",
        "import matplotlib.pyplot as plt\n",
        "import numpy as np\n",
        "\n",
        "fig = plt.figure()\n",
        "fig.set_size_inches(98.5, 90.5)\n",
        "ax = fig.add_subplot(999, projection='3d')\n",
        "\n",
        "\n",
        "i=0\n",
        "for g , xs ,ys , zs  in X:\n",
        "    c = y[i]\n",
        "    i +=1\n",
        "    if c == 1:\n",
        "        cc='blue'  #servive\n",
        "    else:\n",
        "        cc='yellow'  #died\n",
        "    if g == 1:\n",
        "        m = \"^\"   # male\n",
        "    else:\n",
        "        m = \"v\"   #famle\n",
        "    \n",
        "    ax.scatter(xs, ys, zs , c=cc , marker=m)\n",
        "\n",
        "ax.set_xlabel('class')\n",
        "ax.set_ylabel('sibsp')\n",
        "ax.set_zlabel('age')\n",
        "plt.show()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4e8a6167-5a1e-2929-4d31-c49a4725485f"
      },
      "outputs": [],
      "source": [
        "from sklearn.model_selection import train_test_split\n",
        "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.20, random_state = 0)\n",
        "from sklearn.preprocessing import StandardScaler\n",
        "sc = StandardScaler()\n",
        "X_train = sc.fit_transform(X_train)\n",
        "X_test = sc.transform(X_test)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "665ed442-e26f-2990-e57d-408b044a3348"
      },
      "outputs": [],
      "source": [
        "from sklearn.neighbors import KNeighborsClassifier\n",
        "classifier = KNeighborsClassifier(n_neighbors = 4 , metric = 'minkowski' , p = 2)\n",
        "classifier.fit(X_train , y_train)\n",
        "y_pred = classifier.predict(X_test)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "23845bd4-e3a9-3576-ef58-a74cf37d5922"
      },
      "outputs": [],
      "source": [
        "from sklearn.metrics import confusion_matrix\n",
        "cm = confusion_matrix(y_test, y_pred)\n",
        "cm"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "06a7fdf9-6a04-6fbd-4929-a3c6280dda55"
      },
      "outputs": [],
      "source": [
        "def getscore(orgin , predicted):\n",
        "    score = len(orgin)\n",
        "    print(score)\n",
        "    j = 0\n",
        "    for i in predicted:\n",
        "        if i == orgin[j]:\n",
        "            j +=1\n",
        "        else:\n",
        "            score -=1\n",
        "            j +=1\n",
        "    result = (score / float(len(orgin))  ) * 100\n",
        "    print(score)\n",
        "    result = str(result)\n",
        "    return result+'%'"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "20017f99-b342-1f00-0bc9-361c9ceacf2e"
      },
      "outputs": [],
      "source": [
        "getscore(y_test,y_pred)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "26a13d4e-d8de-5f88-c25d-4f4c0a61f904"
      },
      "outputs": [],
      "source": [
        "fig = plt.figure()\n",
        "fig.set_size_inches(98.5, 90.5)\n",
        "ax = fig.add_subplot(999, projection='3d')\n",
        "\n",
        "\n",
        "i=0\n",
        "for g , xs ,ys , zs  in X_test:\n",
        "    c = y_test[i]\n",
        "    if c == 1:\n",
        "        cc='blue'  #servive\n",
        "    else:\n",
        "        cc='red'  #died\n",
        "    if g == 1:\n",
        "        m = \"^\"   # male\n",
        "    else:\n",
        "        m = \"v\"   #famle\n",
        "    if y_pred[i] == y_test[i]:\n",
        "        if y_pred[i] == 1:\n",
        "            cc = 'green'\n",
        "        else:\n",
        "            cc = 'yellow'\n",
        "        m = \"+\"\n",
        "    i +=1\n",
        "    \n",
        "    ax.scatter(xs, ys, zs , c=cc , marker=m )\n",
        "\n",
        "ax.set_xlabel('class')\n",
        "ax.set_ylabel('sibsp')\n",
        "ax.set_zlabel('age')\n",
        "plt.show()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fcef1926-7c0f-2fc7-376f-9fc0d6b36c57"
      },
      "outputs": [],
      "source": [
        "realdata = pd.read_csv('../input/test.csv')\n",
        "realdata.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fe90d2ee-7695-24d4-fd8a-d504f9779077"
      },
      "outputs": [],
      "source": [
        "#y_real = pd.read_csv('../input/gender_submission.csv')\n",
        "#y_real.head()\n",
        "# i try to include these file but its not work so  these step  skipped"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d725477a-01a3-39f1-53f2-feea6d9bcdff"
      },
      "outputs": [],
      "source": [
        "\n",
        "X_real = realdata.iloc[:, [3 ,1,5,4 ]].values\n",
        "#y_real = y_real.iloc[:, 1].values\n",
        "X_real"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "82922a27-47ba-d529-d3c1-999dce0eb4b3"
      },
      "outputs": [],
      "source": [
        "from sklearn.preprocessing import Imputer\n",
        "imputer = Imputer(missing_values =  \"NaN\" , strategy = \"mean\" , axis = 0)\n",
        "imputer = imputer.fit(X_real[:, 1:])\n",
        "X_real[:, 1:] = imputer.transform(X_real[:,1:])\n",
        "X_real"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "22ea60ed-ea2b-5400-0c38-5f02c0d4a84f"
      },
      "outputs": [],
      "source": [
        "from sklearn.preprocessing import LabelEncoder\n",
        "labelencoder_X_real = LabelEncoder()\n",
        "X_real[:,0] = labelencoder_X_real.fit_transform(X_real[:,0])"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5e2c32bc-3c3e-7381-caa6-215327c332c4"
      },
      "outputs": [],
      "source": [
        "from sklearn.preprocessing import StandardScaler\n",
        "sc = StandardScaler()\n",
        "X_real = sc.fit_transform(X_real)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "728af89c-8b82-2b65-ea33-47a33f477691"
      },
      "outputs": [],
      "source": [
        "y_real_pred = classifier.predict(X_real)\n",
        "fig = plt.figure()\n",
        "fig.set_size_inches(98.5, 90.5)\n",
        "ax = fig.add_subplot(999, projection='3d')\n",
        "i=0\n",
        "for g , xs ,ys , zs  in X_real:\n",
        "    c = y_real_pred[i]\n",
        "    if c == 1:\n",
        "        cc='blue'  #servive\n",
        "    else:\n",
        "        cc='red'  #died\n",
        "    if g == 1:\n",
        "        m = \"^\"   # male\n",
        "    else:\n",
        "        m = \"v\"   #famle\n",
        "\n",
        "    i +=1\n",
        "    \n",
        "    ax.scatter(xs, ys, zs , c=cc , marker=m )\n",
        "\n",
        "ax.set_xlabel('class')\n",
        "ax.set_ylabel('sibsp')\n",
        "ax.set_zlabel('age')\n",
        "plt.show()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "92c8f9e8-cb54-8ea2-0b7b-6800b62cfb02"
      },
      "outputs": [],
      "source": [
        "y_real_pred"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "943b7184-6a27-4d98-4756-a3b9b3cdb071"
      },
      "outputs": [],
      "source": [
        "ids = realdata.iloc[:, 0].values\n",
        "mamon_predcation = np.vstack((ids, y_real_pred)).T\n",
        "np.savetxt(\"mamon.csv\", mamon_predcation, delimiter=\",\")"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "58ac941c-aa4d-0a34-7ea3-15bac3fbc465"
      },
      "outputs": [],
      "source": [
        "mamon_predcation"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4db74a81-dc7a-a6a1-8df9-7f351411b174"
      },
      "outputs": [],
      "source": [
        ""
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}