{"metadata": {"kernelspec": {"language": "python", "display_name": "Python 3", "name": "python3"}, "language_info": {"mimetype": "text/x-python", "name": "python", "nbconvert_exporter": "python", "codemirror_mode": {"name": "ipython", "version": 3}, "file_extension": ".py", "version": "3.6.1", "pygments_lexer": "ipython3"}}, "nbformat": 4, "nbformat_minor": 1, "cells": [{"cell_type": "code", "metadata": {"_uuid": "19a75eefe9ea447d42b2a96ad4f7dd424110c2d1", "collapsed": true, "_cell_guid": "b74d4556-2fee-4246-96c7-1a069ac05da0"}, "execution_count": null, "outputs": [], "source": ["%matplotlib inline\n", "import numpy as np\n", "import pandas as pd\n", "import matplotlib.pyplot as plt\n", "import xgboost as xgb"]}, {"cell_type": "code", "metadata": {"_uuid": "6a8c347faa9978c88d4c0686b4cfcb88191392b9", "collapsed": true, "_cell_guid": "79385736-aa96-44cf-886c-2b8a62a2e04a"}, "execution_count": null, "outputs": [], "source": ["train = pd.read_csv(\"../input/train.csv\").merge(\n", "    pd.read_csv(\"../input/members.csv\"),\n", "    on=\"msno\"\n", ")\n", "test = pd.read_csv(\"../input/sample_submission_zero.csv\").merge(\n", "    pd.read_csv(\"../input/members.csv\"),\n", "    on=\"msno\",\n", "    how=\"left\"\n", ")"]}, {"cell_type": "markdown", "metadata": {"_uuid": "dc9bc1542a29e5db052771dd35b01cbe4f410a7c", "_cell_guid": "cd650f7d-6c40-4a65-9d59-997c233145c6"}, "source": ["# Data\n", "\n", "Let's explore our data (include description and correlations)"]}, {"cell_type": "code", "metadata": {"_uuid": "ca7f34dbbf4d13cf615a7d84572b53fa232b8c8c", "_cell_guid": "548ba31f-3ba5-4cb4-a9aa-d66c1b0ae3c2"}, "execution_count": null, "outputs": [], "source": ["train.head()"]}, {"cell_type": "code", "metadata": {"_uuid": "a2b6b6710285a44883ad6098efa954b77b2f395a", "_cell_guid": "6f40c780-a123-48d3-8e57-c40c5b8a974f"}, "execution_count": null, "outputs": [], "source": ["train.describe(include=\"all\")"]}, {"cell_type": "code", "metadata": {"_uuid": "fcfacc2208023d8cf390746cc035ce97e977d414", "_cell_guid": "da9b0563-db26-452c-93bb-ec79f340d2b1"}, "execution_count": null, "outputs": [], "source": ["train.corr()"]}, {"cell_type": "markdown", "metadata": {"_uuid": "e9afe5d107f2af6ffff93d52bfd822b6a069bd03", "_cell_guid": "5b27cc5c-b81a-4552-880f-9e802e858105"}, "source": ["So - where no features with big correlation with is_churn. Anyway, continue.\n", "\n", "# Crossvalidation\n", "\n", "Let's define out cross-validation function (I'll use it because we haven't \"predict_proba\" on DummyRegressor and we need some changes in log_loss)"]}, {"cell_type": "code", "metadata": {"_uuid": "ef80fbad2023450aaccf1625f67b7bf628fcb765", "collapsed": true, "_cell_guid": "57e561a8-e53f-4cb9-ab1f-b36a7f0c7ea6"}, "execution_count": null, "outputs": [], "source": ["from sklearn.model_selection import StratifiedKFold\n", "from sklearn.metrics import log_loss\n", "\n", "def cv(model, X, y, predictor, random_state=None):\n", "    kfold = StratifiedKFold(shuffle=True,\n", "                            random_state=random_state)\n", "    initial_params = model.get_params()\n", "    losses = []\n", "    for i, indices in enumerate(kfold.split(X, y)):\n", "        print(\"Fold {0}\".format(i + 1))\n", "        train_index, test_index = indices\n", "        model.set_params(**initial_params)\n", "        X_train, X_test = X[train_index], X[test_index]\n", "        y_train, y_test = y[train_index], y[test_index]\n", "        model.fit(X_train, y_train)\n", "        p_predicted = predictor(model, X_test)\n", "        p_predicted[p_predicted > 1-10**(-15)] = 1-10**(-15)\n", "        p_predicted[p_predicted < 10**(-15)] = 10**(-15)\n", "        losses.append(log_loss(y_test, p_predicted))\n", "    return np.array(losses)"]}, {"cell_type": "markdown", "metadata": {"_uuid": "65b7638440600fb3fc82a943913b5e18054e8d11", "_cell_guid": "4defb782-fb9d-4bad-890c-9c52995104af"}, "source": ["# Dummy model"]}, {"cell_type": "code", "metadata": {"_uuid": "83bea02e92ae00e33c859f32d51015cc8767cd77", "_cell_guid": "35f5edb2-6e29-4ce3-ba07-142852bdddf0"}, "execution_count": null, "outputs": [], "source": ["from sklearn.dummy import DummyRegressor\n", "\n", "SEED = 42\n", "\n", "cv(DummyRegressor(\"constant\", constant=train[\"is_churn\"].mean()),\n", "   np.zeros([len(train),1]),\n", "   train[\"is_churn\"],\n", "   random_state=SEED,\n", "   predictor=lambda estimator, X: estimator.predict(X))"]}, {"cell_type": "markdown", "metadata": {"_uuid": "23a63fb703394891b6b1d453e2fd503610882305", "_cell_guid": "11576191-6108-4d73-be13-f6865928f6a1"}, "source": ["# Features\n", "\n", "Let's try to build XGBClassifier base on our features.\n", "\n", "## registered_via\n", "\n", "I'll start from feature with bigger absolute corellation value"]}, {"cell_type": "code", "metadata": {"_uuid": "e242a3f2263bcb948e773a33a80050b2b3c954fd", "_cell_guid": "a8365d65-f4d4-4731-9f01-42a6da007fc7"}, "execution_count": null, "outputs": [], "source": ["cv(xgb.XGBClassifier(),\n", "   np.array(train[[\"registered_via\"]]),\n", "   train[\"is_churn\"],\n", "   random_state=SEED,\n", "   predictor=lambda estimator, X: estimator.predict_proba(X)[:, 1])"]}, {"cell_type": "markdown", "metadata": {"_uuid": "e46543b6d962144d3822d9699ea7c8576ca0af72", "_cell_guid": "6dc5995b-fa05-427e-9bba-e0388b2fd4b2"}, "source": ["# city\n", "Let's add city feature."]}, {"cell_type": "code", "metadata": {"_uuid": "e9d3021c2c4415fcb4e9cf2b0082c59dc89c56ec", "_cell_guid": "46843eb3-815c-4354-8726-c725cdfcb13f"}, "execution_count": null, "outputs": [], "source": ["cv(xgb.XGBClassifier(),\n", "   np.array(train[[\"registered_via\", \"city\"]]),\n", "   train[\"is_churn\"],\n", "   random_state=SEED,\n", "   predictor=lambda estimator, X: estimator.predict_proba(X)[:, 1])"]}, {"cell_type": "markdown", "metadata": {"_uuid": "714477b52c61bf23052ecfbae10507249d2b3b15", "_cell_guid": "8a6c43be-6714-48aa-9dfa-357d128aaccd"}, "source": ["Seems like it must be good idea to make set of categorical features city0 - cityN. Let's check it:"]}, {"cell_type": "code", "metadata": {"_uuid": "0d47f028cdd7ae8610f73ec092559a6aff57943b", "_cell_guid": "fa4cf928-43a4-4423-a3cc-4215c5922011"}, "execution_count": null, "outputs": [], "source": ["city_values = list(set(train[\"city\"]))\n", "city_values.sort()\n", "\n", "city_features = [\"city{0}\".format(city) for city in city_values]\n", "for city in city_values:\n", "    train[\"city{0}\".format(city)] = train[\"city\"] == city\n", "    test[\"city{0}\".format(city)] = test[\"city\"] == city\n", "\n", "cv(xgb.XGBClassifier(),\n", "   np.array(train[[\"registered_via\"] + city_features]),\n", "   train[\"is_churn\"],\n", "   random_state=SEED,\n", "   predictor=lambda estimator, X: estimator.predict_proba(X)[:, 1])"]}, {"cell_type": "markdown", "metadata": {"_uuid": "f6a6c81cd5f9a805111cef26c1b0266b42f45b26", "_cell_guid": "4675c573-ce17-498f-80af-449a7aa819e5"}, "source": ["## bd\n", "\n", "bd - age feature. As written in description - it have some outliners, so let's see histogram"]}, {"cell_type": "code", "metadata": {"_uuid": "8edcdd1f78ddcc91c3b455fb5065180a06365747", "_cell_guid": "8ff774f5-7c62-44ba-9fdf-b851b6e8f0a8"}, "execution_count": null, "outputs": [], "source": ["plt.hist(train[\"bd\"]);"]}, {"cell_type": "code", "metadata": {"_uuid": "9b0b8724782e7e85791266f86fd482978b378aa8", "_cell_guid": "91b27041-c399-45d0-8771-88d8f102d7ff"}, "execution_count": null, "outputs": [], "source": ["plt.hist(np.log(np.abs(train[\"bd\"]) + 0.001));"]}, {"cell_type": "code", "metadata": {"_uuid": "f4c07153c653ea67cca5ce9d5808cfd19681ec4c", "_cell_guid": "6f8e4ffb-06f0-4b21-9f24-fe1afbbbc97b"}, "execution_count": null, "outputs": [], "source": ["train[\"bd\"].min(), train[\"bd\"].max(), train[\"bd\"].mean(), train[\"bd\"].std()"]}, {"cell_type": "code", "metadata": {"_uuid": "1d44703c019f7759f5954663657b93ae8c0ad378", "_cell_guid": "8d7f04db-9886-4f7e-b621-66f2ef7c85b3"}, "execution_count": null, "outputs": [], "source": ["cv(xgb.XGBClassifier(),\n", "   np.array(train[[\"registered_via\", \"bd\"] + city_features]),\n", "   train[\"is_churn\"],\n", "   random_state=SEED,\n", "   predictor=lambda estimator, X: estimator.predict_proba(X)[:, 1])"]}, {"cell_type": "markdown", "metadata": {"_uuid": "b0775bd5a021c0a10313cac5b156b22a68c37a26", "_cell_guid": "bb8c4e3a-49ec-49e9-8187-3547f56f2614"}, "source": ["I didn't find a way to replace outliners with better cv score, so continue"]}, {"cell_type": "markdown", "metadata": {"_uuid": "94ec10a9363d4f520e48f0ce9f8c44ce0d967511", "_cell_guid": "229d1030-727d-4df1-ada8-e0212883e078"}, "source": ["## gender\n", "\n", "There we have categorical feature (male/female). So let's convert it to binary feature - but note that we have missing values"]}, {"cell_type": "code", "metadata": {"_uuid": "969c47373270d7cfc6f2f18d5923b7a6e7d48832", "collapsed": true, "_cell_guid": "08f5de32-1d64-40f1-968f-09826c8e7e5b"}, "execution_count": null, "outputs": [], "source": ["def gender(val):\n", "    if val == \"male\":\n", "        return 1\n", "    elif val == \"female\":\n", "        return -1\n", "    else:\n", "        return float(\"NaN\")\n", "    \n", "train[\"gender_converted\"] = train[\"gender\"].apply(gender)\n", "test[\"gender_converted\"] = test[\"gender\"].apply(gender)"]}, {"cell_type": "code", "metadata": {"_uuid": "6357be4704f4b7bc8e6fecf0b7a35e9eca5c3c30", "_cell_guid": "422738ae-bcfc-4706-bbfe-fc713a6eee4e"}, "execution_count": null, "outputs": [], "source": ["cv(xgb.XGBClassifier(),\n", "   np.array(train[[\"registered_via\", \"bd\", \"gender_converted\"] + city_features]),\n", "   train[\"is_churn\"],\n", "   random_state=SEED,\n", "   predictor=lambda estimator, X: estimator.predict_proba(X)[:, 1])"]}, {"cell_type": "markdown", "metadata": {"_uuid": "eaab4c4d4111959288dc4651a341e2780d6c9aa9", "_cell_guid": "65e2bec9-8ccf-41d8-8366-2b175c87a430"}, "source": ["## registration_init_time\n", "\n", "Let's build set of features:\n", "- parse date\n", "- convert date to unix timestamp (to make numerical feature)\n", "- add year/month/day features (e.g. to find season changes)\n", "\n", "### unix time"]}, {"cell_type": "code", "metadata": {"_uuid": "9529f7b2c460f32f6aab4cd876743d7ce3a4cbb7", "collapsed": true, "_cell_guid": "9423a57d-4723-4c0e-bcc6-a53fa17beac6"}, "execution_count": null, "outputs": [], "source": ["def datetime_to_unix(dt):\n", "    epoch = pd.to_datetime('1970-01-01')\n", "    return (dt - epoch).total_seconds()\n", "\n", "\n", "train[\"registration_init_time_date\"] = pd.to_datetime(train[\"registration_init_time\"], format=\"%Y%m%d\")\n", "test[\"registration_init_time_date\"] = pd.to_datetime(test[\"registration_init_time\"], format=\"%Y%m%d\")\n", "train[\"registration_init_time_unix\"] = train[\"registration_init_time_date\"].apply(datetime_to_unix)\n", "test[\"registration_init_time_unix\"] = test[\"registration_init_time_date\"].apply(datetime_to_unix)"]}, {"cell_type": "markdown", "metadata": {"_uuid": "2626a424be76457d8280b2150330ffe2d62add78", "_cell_guid": "20409590-f887-48b7-aeb6-a34c95dcbeea"}, "source": ["### year/month/day"]}, {"cell_type": "code", "metadata": {"_uuid": "408ed51f2fce9f29f48bfc4f564f85e5b9242aed", "_cell_guid": "f6a6eaca-1152-49e8-90a4-3cb3f58a428e"}, "execution_count": null, "outputs": [], "source": ["cv(xgb.XGBClassifier(),\n", "   np.array(train[[\"registered_via\", \"bd\", \"gender_converted\", \"registration_init_time_unix\"] + \n", "                  city_features]),\n", "   train[\"is_churn\"],\n", "   random_state=SEED,\n", "   predictor=lambda estimator, X: estimator.predict_proba(X)[:, 1])"]}, {"cell_type": "code", "metadata": {"_uuid": "1851c949df8b888284ed85b09baa17a64fc8066f", "collapsed": true, "_cell_guid": "6956bba9-152c-4fdd-830e-a3cb87302753"}, "execution_count": null, "outputs": [], "source": ["train[\"registration_init_time_year\"] = train[\"registration_init_time_date\"].apply(lambda date: date.year)\n", "test[\"registration_init_time_year\"] = test[\"registration_init_time_date\"].apply(lambda date: date.year)\n", "train[\"registration_init_time_month\"] = train[\"registration_init_time_date\"].apply(lambda date: date.month)\n", "test[\"registration_init_time_month\"] = test[\"registration_init_time_date\"].apply(lambda date: date.month)\n", "train[\"registration_init_time_day\"] = train[\"registration_init_time_date\"].apply(lambda date: date.day)\n", "test[\"registration_init_time_day\"] = test[\"registration_init_time_date\"].apply(lambda date: date.day)"]}, {"cell_type": "markdown", "metadata": {"_uuid": "a5bbbf6771809c1482880b901e44d3d67a1f5d92", "_cell_guid": "5e8c8f08-69e6-45bf-985c-2f3da257fcea"}, "source": ["Let's train model on all data and make prediction for test records"]}, {"cell_type": "code", "metadata": {"_uuid": "7a61323ed5e859fed5c53bbb55359f1c22100be3", "_cell_guid": "9e019689-21bd-4547-8626-f85ecebe6c83"}, "execution_count": null, "outputs": [], "source": ["cv(xgb.XGBClassifier(),\n", "   np.array(train[[\"registered_via\", \"bd\", \"gender_converted\", \"registration_init_time_unix\",\n", "                   \"registration_init_time_year\", \"registration_init_time_month\", \"registration_init_time_day\"] + \n", "                  city_features]),\n", "   train[\"is_churn\"],\n", "   random_state=SEED,\n", "   predictor=lambda estimator, X: estimator.predict_proba(X)[:, 1])"]}, {"cell_type": "code", "metadata": {"_uuid": "6c3731205b6438ffe3a48a316872de0b078b4074", "_cell_guid": "c6ac734e-616c-4ecb-af66-031ae4306136"}, "execution_count": null, "outputs": [], "source": ["from collections import OrderedDict\n", "\n", "clf = xgb.XGBClassifier()\n", "clf.fit(\n", "    np.array(train[[\"registered_via\", \"bd\", \"gender_converted\", \"registration_init_time_unix\",\n", "                   \"registration_init_time_year\", \"registration_init_time_month\", \"registration_init_time_day\"] + \n", "                  city_features]),\n", "    np.array(train[\"is_churn\"])\n", ")\n", "prediction = clf.predict_proba(np.array(test[[\"registered_via\", \"bd\",\n", "                                              \"gender_converted\", \n", "                                              \"registration_init_time_unix\",\n", "                                              \"registration_init_time_year\", \n", "                                              \"registration_init_time_month\",\n", "                                              \"registration_init_time_day\"] + \n", "                                             city_features]))[:, 1]\n", "prediction_df = pd.DataFrame(OrderedDict([ (\"msno\", test[\"msno\"]), (\"is_churn\", prediction) ]))\n", "prediction_df.head()"]}, {"cell_type": "code", "metadata": {"_uuid": "d98d6e6165138831713a2ca5514e04db4e486420", "collapsed": true, "_cell_guid": "aa75b86d-2dc2-4403-ad04-b11a64c763de"}, "execution_count": null, "outputs": [], "source": ["prediction_df.to_csv(\"prediction.csv\", index=False)"]}, {"cell_type": "markdown", "metadata": {"_uuid": "ee29a3611db10b3bf51b05037cea5191336dddb9", "_cell_guid": "813fdd5f-01c9-49c0-bcc6-1bdc087fce94"}, "source": ["## expiration_date\n", "\n", "Let's try to build unix time feature from expiration_date"]}, {"cell_type": "code", "metadata": {"_uuid": "d510f0f65a2e2aafa49f37ef5393d42ee0d84a11", "collapsed": true, "_cell_guid": "29ffb32d-8378-4e82-9f88-ed1c6ffc5553"}, "execution_count": null, "outputs": [], "source": ["train[\"expiration_date\"] = pd.to_datetime(train[\"expiration_date\"], format=\"%Y%m%d\")\n", "test[\"expiration_date\"] = pd.to_datetime(test[\"expiration_date\"], format=\"%Y%m%d\")"]}, {"cell_type": "code", "metadata": {"_uuid": "db02f247a0f45d3509258d0c4845222f81aa7f3a", "collapsed": true, "_cell_guid": "99375446-5ae4-4d5e-81b8-c8b485914d71"}, "execution_count": null, "outputs": [], "source": ["train[\"expiration_date_unix\"] = train[\"expiration_date\"].apply(datetime_to_unix)\n", "test[\"expiration_date_unix\"] = test[\"expiration_date\"].apply(datetime_to_unix)"]}, {"cell_type": "code", "metadata": {"_uuid": "348a6349d46fc445ca910a029580f77ea5017e34", "_cell_guid": "65e379cd-04fd-4f9c-b8c0-d982ae1554e1"}, "execution_count": null, "outputs": [], "source": ["cv(xgb.XGBClassifier(),\n", "   np.array(train[[\"registered_via\", \"bd\", \"gender_converted\", \"registration_init_time_unix\",\n", "                   \"registration_init_time_year\", \"registration_init_time_month\", \"registration_init_time_day\",\n", "                   \"expiration_date_unix\"] + \n", "                  city_features]),\n", "   train[\"is_churn\"],\n", "   random_state=SEED,\n", "   predictor=lambda estimator, X: estimator.predict_proba(X)[:, 1])"]}, {"cell_type": "markdown", "metadata": {"_uuid": "17abf0f717fffe70b84796d8fcf2d48af31b7204", "_cell_guid": "73d6d75d-71ba-486b-9ef2-3b02adfd7024"}, "source": ["Good score? maybe not, in public leatherboard it gives ~= 0.8 score :-)"]}]}