{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2e63d8fd-1069-ee2a-dbdf-dd6e0a047ad3"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "import math\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "cc5a6303-f43a-6082-5711-f9b1c48114a1"
      },
      "outputs": [],
      "source": [
        "train_data = pd.read_csv(\"../input/train.csv\")\n",
        "test_data = pd.read_csv(\"../input/test.csv\")\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "def892dd-65fa-2f3e-3f59-b2c81e8f7777"
      },
      "outputs": [],
      "source": [
        "test_data.columns"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "16b64b45-d95c-7f8c-61ea-304dadf14c3d"
      },
      "outputs": [],
      "source": [
        "train_data.columns"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3d60c833-e4a6-6da8-e90a-d746fb1921c6"
      },
      "outputs": [],
      "source": [
        "train_data.head()"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "a60dbced-94ac-7c06-0de9-d8542899406d"
      },
      "source": [
        "My first step would be to find our which of the pessangers attributes have higher weights on the result than others. I will do this by splitting the dataset into survivors and killed people and compare the data entropie for each attribute."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "268d33e2-a838-62ba-4b68-9906e34813da"
      },
      "outputs": [],
      "source": [
        "surv_mask = train_data.apply(lambda x: True if x[\"Survived\"]==1 else False, axis=1)\n",
        "survived = train_data[surv_mask]\n",
        "killed = train_data[~surv_mask]\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0602384f-ab24-f920-6a86-ca0878d32501"
      },
      "outputs": [],
      "source": [
        "def get_series_entropie(series):\n",
        "    series_prob_distr = series.value_counts(normalize=True)\n",
        "    series_calc = series_prob_distr.apply(lambda x: x*math.log(x,2))\n",
        "    return -series_calc.sum()\n",
        "\n",
        "def get_df_entropie(df):\n",
        "    return df.apply(lambda x: get_series_entropie(x))\n",
        "    "
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2990a2d5-aade-d4a5-11c5-a389c2b8e35b"
      },
      "outputs": [],
      "source": [
        "entr_diff = ((get_df_entropie(survived)+ get_df_entropie(killed))/2) - get_df_entropie(test_data) \n",
        "entr_diff.sort_values()\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2062d848-7e07-0880-0b8f-e6390b4e8a3c"
      },
      "outputs": [],
      "source": [
        "get_df_entropie(test_data)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "1674b1d0-6aa2-8053-b770-850392a69f71"
      },
      "outputs": [],
      "source": [
        "get_df_entropie(survived)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a24b6f43-4479-c998-ac3b-a86c26866b39"
      },
      "outputs": [],
      "source": [
        "get_df_entropie(killed)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "503a474c-3363-4baf-2251-2263294db18f"
      },
      "outputs": [],
      "source": [
        ""
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}