{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "272f5eed-e0a8-ddf8-d078-255661662506"
      },
      "outputs": [],
      "source": [
        "import os\n",
        "import numpy as np\n",
        "import pandas as pd"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "bda1e999-b63d-60e9-884d-4525d70bc0cc"
      },
      "outputs": [],
      "source": [
        "df_train = pd.read_csv(\"../input/train_users_2.csv\")\n",
        "df_train.sample(n=5)\n",
        "##df_train.head(n=5) # les 5 premiere ligne"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "17746796-023f-15a3-7a04-afeb707cde83"
      },
      "outputs": [],
      "source": [
        "df_test = pd.read_csv(\"../input/test_users.csv\")\n",
        "df_test.sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "82c4ff29-26ed-7cf3-6895-e6aa2161d117"
      },
      "outputs": [],
      "source": [
        "df_all = pd.concat((df_train, df_test), axis=0, ignore_index = True)\n",
        "df_all.head(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "155641c8-71ea-6d63-6219-55f17703e9f9"
      },
      "outputs": [],
      "source": [
        "df_all.drop('date_first_booking',axis = 1, inplace = True)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "f2a19c0d-58aa-42ea-a184-97d30743f6c9"
      },
      "outputs": [],
      "source": [
        "df_all['date_account_created'] = pd.to_datetime(df_all['date_account_created'], format='%Y-%m-%d')\n",
        "df_all.sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "3e5cf038-028e-230a-7d16-390266ca64d8"
      },
      "outputs": [],
      "source": [
        "df_all['timestamp_first_active'] = pd.to_datetime(df_all['timestamp_first_active'], format='%Y%m%d%H%M%S')\n",
        "df_all.sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c1f870ae-e560-4982-f420-515bff95351f"
      },
      "outputs": [],
      "source": [
        "def remove_age_incorrect(x ,min_value= 15, max_value =90):\n",
        "    if np.logical_or(x <=min_value, x>=max_value):\n",
        "        return np.nan\n",
        "    return x\n",
        "\n",
        "df_all['age'] = df_all['age'].apply(lambda x: remove_age_incorrect(x) if(not np.isnan(x)) else x)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9be29a49-1e1e-04f2-1eee-539afd4d2ea5"
      },
      "outputs": [],
      "source": [
        "def check_NaN_Values_in_df(df):\n",
        "    for col in df:\n",
        "        nan_count = df[col].isnull().sum()\n",
        "        if nan_count != 0:\n",
        "            print(col + \" \" + str(nan_count) + \" Nan Values\")\n",
        "\n",
        "df_all['age'].fillna(-1, inplace=True)\n",
        "df_all['first_affiliate_tracked'].fillna(-1, inplace=True)\n",
        "check_NaN_Values_in_df(df_all)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2edcb28b-7fde-caf0-f6c9-7acdd88394f8"
      },
      "outputs": [],
      "source": [
        "df_all.drop('timestamp_first_active',axis=1,inplace =True)\n",
        "df_all.drop('language',axis=1,inplace =True)\n",
        "df_all.sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7e9c04bf-8af9-d579-444d-212841401529"
      },
      "outputs": [],
      "source": [
        "df_all = df_all[df_all['date_account_created'] > '2013-02-01']\n",
        "df_all.sample(n=5)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "9fcf875c-dcc4-4b3a-6246-3273b8178975"
      },
      "outputs": [],
      "source": [
        "if not os.path.exists(\"outpout\"):\n",
        "    os.makedirs(\"output\")\n",
        "    \n",
        "df_all.to_csv(\"output/cleaned.csv\",sep=',', index=False)"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}