{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "27866c51-8d58-7ef0-4ae4-38953c59af44"
      },
      "outputs": [],
      "source": [
        "import os\n",
        "\n",
        "import numpy as np\n",
        "import pandas as pd\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "ef625a10-d9f6-6eea-21a2-e9f159657d83"
      },
      "outputs": [],
      "source": [
        "df_train = pd.read_csv(\"../input/train_users_2.csv\")\n",
        "df_train.sample(n=5)\n",
        "\n",
        "\n",
        "#Quoi qu'est ce que c'est qu'un pipeline?\n",
        "#- C'est un tube qui enchaine les traitement de donn\u00e9es, \u00e9tape par \u00e9tape. Il ex\u00e9cute la routine que nous cr\u00e9ons nous aussi \u00e9tape par \u00e9tape.\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9c2ff406-5793-08fa-e2d8-9bd5e3613e96"
      },
      "outputs": [],
      "source": [
        "df_test = pd.read_csv(\"../input/test_users.csv\")\n",
        "df_test.sample(n=5)\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "44868b42-101b-604f-e1c4-5fb19dd60bbd"
      },
      "outputs": [],
      "source": [
        "df_all = pd.concat((df_train, df_test), axis=0, ignore_index=True)\n",
        "df_all.head(n=5)\n",
        "#concatene 2 tableaux l'un au dessus de l'autre en ignorant l'index"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "67affa8f-38d8-dddb-7002-df63ac5acfb0"
      },
      "outputs": [],
      "source": [
        "#supprimer la cellule 1ere reseration\n",
        "df_all.drop('date_first_booking',axis=1, inplace=True)\n",
        "df_all.sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9f7b0019-f380-f718-7873-8e4257affc3c"
      },
      "outputs": [],
      "source": [
        "df_all['date_account_created'] = pd.to_datetime(df_all['date_account_created'], format = '%Y-%m-%d')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f59a6919-7620-7aa5-9521-2ee0da437805"
      },
      "outputs": [],
      "source": [
        "df_all['timestamp_first_active'] = pd.to_datetime(df_all['timestamp_first_active'], format = '%Y%m%d%H%M%S')"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9ef11900-0d07-144d-f59f-65a87081bee5"
      },
      "outputs": [],
      "source": [
        "df_all.sample(n=5)\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "53e8b33f-58d5-12d2-b6b8-81b4dbb2c8e5"
      },
      "outputs": [],
      "source": [
        "def remove_age_outliers(x, min_value=15, max_value=90):\n",
        "    if np.logical_or(x<=min_value, x>=max_value):\n",
        "        return np.nan\n",
        "    else:\n",
        "        return x"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "b5d95a75-faff-ae4b-8e6c-642d702f6d88"
      },
      "outputs": [],
      "source": [
        "#fonction lambda\n",
        "#lambda x: True if(X==0) else False\n",
        "#              applique a toutes les lignes du tableau\n",
        "#               |\n",
        "#              \\/\n",
        "df_all['age'] = df_all['age'].apply(lambda x: remove_age_outliers(x) if(not np.isnan(x)) else x)\n",
        "df_all['age'].fillna(-1, inplace=True)\n",
        "df_all['age'].sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "602c9764-f6fb-54ec-fdca-1b9697694f15"
      },
      "outputs": [],
      "source": [
        "df_all.age = df_all.age.astype(int)\n",
        "df_all['age'].sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "12e2ef30-5bc6-0bf2-8d73-d5a9ee8125f4"
      },
      "outputs": [],
      "source": [
        "def check_NaN_Values_in_df(df):\n",
        "    for col in df:\n",
        "        nan_count = df[col].isnull().sum()\n",
        "        \n",
        "        if nan_count != 0:\n",
        "            print (col+\" -> \" + str(nan_count) + \"NaN Values\")\n",
        "        #for x in df"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "986d267c-c58a-c139-b244-8727b9200ea6"
      },
      "outputs": [],
      "source": [
        "check_NaN_Values_in_df(df_all)\n"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "aed84b07-bbc9-c834-941a-173a8e767abd"
      },
      "outputs": [],
      "source": [
        "df_all['first_affiliate_tracked'].fillna(-1, inplace=True)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d2a1f870-02c4-ba95-2670-a72983b670a7"
      },
      "outputs": [],
      "source": [
        "check_NaN_Values_in_df(df_all)\n",
        "df_all.sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "e9828c2c-433f-b14f-8d7f-0aea7d194362"
      },
      "outputs": [],
      "source": [
        "df_all.drop('language',axis=1, inplace=True)\n",
        "df_all.sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a4d0c10a-a6d9-0850-cfc1-b7eddec3e172"
      },
      "outputs": [],
      "source": [
        "df_all = df_all[df_all['date_account_created'] > '2013-02-01']\n",
        "df_all.sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "0c0262db-89e2-bdf0-9d3a-ded7bd6b11a5"
      },
      "outputs": [],
      "source": [
        "if  not os.path.exists(\"output\"):\n",
        "    os.makedirs(\"ouput\")\n",
        "    \n",
        "df_all.to_csv(\"output/cleanned.csv\", sep=',', index=False)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}