{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "02f999f1-9b67-a535-a483-abcf95c81c5d"
      },
      "source": [
        "What combination of location, time, and advertisers are likely to receive clicks?"
      ]
    },
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "75617bb3-ab5a-7fe8-c65b-f7ea014024ec"
      },
      "source": [
        "filedir = '../input/'\n",
        "import pandas as pd\n",
        "import matplotlib.pyplot as plt\n",
        "import numpy as np\n",
        "import datetime\n",
        "import time\n",
        "from sklearn import tree\n",
        "from sklearn import linear_model\n",
        "from sklearn.model_selection import train_test_split\n",
        "from sklearn.model_selection import cross_val_score\n",
        "from sklearn import ensemble\n",
        "from sklearn import svm\n",
        "from sklearn import metrics\n",
        "from sklearn.model_selection import learning_curve\n",
        "from sklearn.model_selection import ShuffleSplit\n",
        "\n",
        "N_ROWS = 1000000\n",
        "\n",
        "'''\n",
        "#load documents csv's\n",
        "documents_categories_df = pd.read_csv(filedir+'documents_categories.csv')\n",
        "documents_entities_df = pd.read_csv(filedir+'documents_entities.csv')\n",
        "documents_meta_df = pd.read_csv(filedir+'documents_meta.csv')\n",
        "document_topics_df = pd.read_csv(filedir+'documents_topics.csv')\n",
        "'''\n",
        "\n",
        "#load events, clicks_train, promoted_content and page_views_sample csv's\n",
        "events_df = pd.read_csv(filedir+'events.csv', nrows=N_ROWS)\n",
        "promoted_content_df = pd.read_csv(filedir+'promoted_content.csv', nrows=N_ROWS)\n",
        "promoted_content_df = promoted_content_df.rename(columns={'document_id': 'ad_document_id'})\n",
        "clicks_train_df = pd.read_csv(filedir+'clicks_train.csv', nrows=N_ROWS)\n",
        "page_views_sample_df = pd.read_csv(filedir+'page_views_sample.csv', usecols=['uuid','document_id', 'timestamp', 'traffic_source'], nrows=N_ROWS)\n",
        "\n",
        "\n",
        "#merge events and clicks_train on display_id value\n",
        "#in order to find what ad was clicked for each event/display_id\n",
        "#this links platform and geo_location data to each ad that was clicked\n",
        "events_x_clicks_train_df = pd.merge(events_df, clicks_train_df, on='display_id', how='outer').dropna()\n",
        "#print(events_x_clicks_train_df)\n",
        "\n",
        "\n",
        "#convert geo_location to int data\n",
        "events_x_clicks_train_df['geo_location'] = events_x_clicks_train_df['geo_location'].replace('[^0-9]', '', regex=True).replace('', np.nan, regex=True)\n",
        "pd.to_numeric(events_x_clicks_train_df['geo_location'])\n",
        "\n",
        "\n",
        "#merge previous with page_views on uuid and document_id\n",
        "#in order to find traffic source for each event/display_id\n",
        "all_available_data_for_each_event = pd.merge(events_x_clicks_train_df, page_views_sample_df, on=['uuid', 'document_id','timestamp'])\n",
        "all_available_data_for_each_event = pd.merge(all_available_data_for_each_event, promoted_content_df, on='ad_id', how='outer').dropna()\n",
        "all_available_data_for_each_event = all_available_data_for_each_event.sort_values(by='display_id')\n",
        "#print(all_available_data_for_each_event)\n",
        "\n",
        "X = all_available_data_for_each_event[['display_id', 'document_id', 'timestamp', 'platform', 'geo_location', 'ad_id', 'traffic_source', 'ad_document_id', 'campaign_id', 'advertiser_id']]\n",
        "y = all_available_data_for_each_event[['clicked']]\n",
        "\n",
        "\n",
        "#rectify timestamp column and convert to hour of day clicked\n",
        "pd.to_numeric(X['timestamp'])\n",
        "X['true_time'] = X['timestamp']+1465876799998\n",
        "X['date'] = pd.to_datetime(X['true_time'], unit='ms')\n",
        "X['hour'] = X['date'].apply(lambda x: x.strftime('%H'))\n",
        "X = X.drop('date', axis=1)\n",
        "X = X.drop('timestamp', axis=1)\n",
        "X = X.drop('true_time', axis=1)\n",
        "\n",
        "\n",
        "#get platform & traffic source columns\n",
        "X['platform'] = X['platform'].astype(str).convert_objects(convert_numeric=True)\n",
        "X['traffic_source'] = X['traffic_source'].astype(str).convert_objects(convert_numeric=True)\n",
        "X['traffic_source'] = X['traffic_source'].astype(int)\n",
        "\n",
        "#print(X)\n",
        "#print(y)\n",
        "'''\n",
        "all_document_ids = X.document_id.unique()\n",
        "all_geo_locations = X.geo_location.unique()\n",
        "all_ad_ids = X.ad_id.unique()\n",
        "all_ad_document_ids = X.ad_document_id.unique()\n",
        "all_campaign_ids = X.campaign_id.unique()\n",
        "all_advertiser_ids = X.advertiser_id.unique()\n",
        "print(len(all_geo_locations)+len(all_advertiser_ids))\n",
        "'''\n",
        "# Get one hot encoding of columns geo_location, platform, hour, advertiser_id, and traffic_source\n",
        "# Drop column as it is now encoded, then drop irrelevant columns\n",
        "X = X.join(pd.get_dummies(X['platform'], prefix='pl_'))\n",
        "X = X.drop('platform', axis=1)\n",
        "X = X.join(pd.get_dummies(X['geo_location'], prefix='ge_'))\n",
        "X = X.drop('geo_location', axis=1)\n",
        "X = X.join(pd.get_dummies(X['advertiser_id'], prefix='ad_'))\n",
        "X = X.drop('advertiser_id', axis=1)\n",
        "X = X.join(pd.get_dummies(X['hour'], prefix='hr_'))\n",
        "X = X.drop('hour', axis=1)\n",
        "X = X.join(pd.get_dummies(X['traffic_source'], prefix='ts_'))\n",
        "X = X.drop('traffic_source', axis=1)\n",
        "\n",
        "#X = X.join(pd.get_dummies(X['campaign_id']))\n",
        "X = X.drop('campaign_id', axis=1)\n",
        "X = X.drop('ad_id', axis=1)\n",
        "X = X.drop('document_id', axis=1)\n",
        "X = X.drop('ad_document_id', axis=1)\n",
        "X = X.drop('display_id', axis=1)\n",
        "\n",
        "print(X)\n",
        "print(y)\n",
        "\n",
        "\n",
        "#sklearn stuff\n",
        "\n",
        "#clf = tree.DecisionTreeClassifier() \n",
        "#0.78\n",
        "\n",
        "#clf = linear_model.LogisticRegression(solver='sag', multi_class='ovr') \n",
        "#0.82\n",
        "\n",
        "#clf = linear_model.LogisticRegressionCV(Cs=100, fit_intercept=True, cv=None, \n",
        "#    dual=False, penalty='l2', \n",
        "#    solver='liblinear', tol=0.0001, max_iter=1000, refit=True, multi_class='ovr')\n",
        "#0.82\n",
        "\n",
        "#clf = linear_model.SGDClassifier() \n",
        "#0.77\n",
        "\n",
        "clf = ensemble.RandomForestClassifier(n_estimators=50, criterion='gini', max_depth=None, \n",
        "    min_samples_split=20, min_samples_leaf=1, min_weight_fraction_leaf=0.0, \n",
        "    max_features='auto', max_leaf_nodes=None, min_impurity_split=1e-02, bootstrap=True, \n",
        "    class_weight=None)\n",
        "#0.82\n",
        "\n",
        "#tree.export_graphviz(clf, out_file='tree.dot')  \n",
        "\n",
        "#clf = svm.SVC(C=1.0, cache_size=200, class_weight=None, coef0=0.0,\n",
        "#    decision_function_shape=None, degree=3, gamma='auto', kernel='rbf',\n",
        "#    max_iter=-1, probability=False, random_state=None, shrinking=True,\n",
        "#    tol=0.001, verbose=False)\n",
        "#0.82\n",
        "\n",
        "#clf = svm.LinearSVC(penalty='l1', loss='squared_hinge', dual=False, \n",
        "#    tol=0.00001, C=1.0, multi_class='crammer_singer', fit_intercept=False, \n",
        "#    intercept_scaling=1, max_iter=2000)\n",
        "#0.81\n",
        "\n",
        "#clf = svm.NuSVC(nu=0.5, kernel='rbf', degree=3, gamma='auto', coef0=0.1, shrinking=True, \n",
        "#    probability=False, tol=0.0001, cache_size=20,\n",
        "#    max_iter=1000, decision_function_shape='ovr', random_state=None)\n",
        "#0.60\n",
        "\n",
        "#clf = linear_model.PassiveAggressiveClassifier(C=1.0, fit_intercept=True, n_iter=5, \n",
        "#    shuffle=True, verbose=0, loss='hinge', n_jobs=1, random_state=None, \n",
        "#    warm_start=False, class_weight=None)\n",
        "#0.81\n",
        "\n",
        "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=2)\n",
        "clf.fit(X_train.as_matrix(), y_train.as_matrix())\n",
        "print(clf.score(X_test, y_test))\n",
        "np.set_printoptions(threshold=np.nan)\n",
        "#print(clf.predict(X_test))\n",
        "#print(y_test)\n",
        "print(metrics.f1_score(y_test, clf.predict(X_test)))\n",
        "\n",
        "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=3)\n",
        "clf.fit(X_train.as_matrix(), y_train.as_matrix())\n",
        "print(clf.score(X_test, y_test))\n",
        "print(metrics.f1_score(y_test, clf.predict(X_test)))\n",
        "\n",
        "X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.4, random_state=4)\n",
        "clf.fit(X_train.as_matrix(), y_train.as_matrix())\n",
        "print(clf.score(X_test, y_test))\n",
        "print(metrics.f1_score(y_test, clf.predict(X_test)))\n",
        "\n",
        "\n",
        "def plot_learning_curve(estimator, title, X, y, ylim=None, cv=None, n_jobs=1, train_sizes=np.linspace(.1, 1.0, 5)):\n",
        "    \"\"\"\n",
        "    Generate a simple plot of the test and training learning curve.\n",
        "\n",
        "    Parameters\n",
        "    ----------\n",
        "    estimator : object type that implements the \"fit\" and \"predict\" methods\n",
        "        An object of that type which is cloned for each validation.\n",
        "\n",
        "    title : string\n",
        "        Title for the chart.\n",
        "\n",
        "    X : array-like, shape (n_samples, n_features)\n",
        "        Training vector, where n_samples is the number of samples and\n",
        "        n_features is the number of features.\n",
        "\n",
        "    y : array-like, shape (n_samples) or (n_samples, n_features), optional\n",
        "        Target relative to X for classification or regression;\n",
        "        None for unsupervised learning.\n",
        "\n",
        "    ylim : tuple, shape (ymin, ymax), optional\n",
        "        Defines minimum and maximum yvalues plotted.\n",
        "\n",
        "    cv : int, cross-validation generator or an iterable, optional\n",
        "        Determines the cross-validation splitting strategy.\n",
        "        Possible inputs for cv are:\n",
        "          - None, to use the default 3-fold cross-validation,\n",
        "          - integer, to specify the number of folds.\n",
        "          - An object to be used as a cross-validation generator.\n",
        "          - An iterable yielding train/test splits.\n",
        "\n",
        "        For integer/None inputs, if ``y`` is binary or multiclass,\n",
        "        :class:`StratifiedKFold` used. If the estimator is not a classifier\n",
        "        or if ``y`` is neither binary nor multiclass, :class:`KFold` is used.\n",
        "\n",
        "        Refer :ref:`User Guide <cross_validation>` for the various\n",
        "        cross-validators that can be used here.\n",
        "\n",
        "    n_jobs : integer, optional\n",
        "        Number of jobs to run in parallel (default 1).\n",
        "    \"\"\"\n",
        "    plt.figure()\n",
        "    plt.title(title)\n",
        "    if ylim is not None:\n",
        "        plt.ylim(*ylim)\n",
        "    plt.xlabel(\"Training examples\")\n",
        "    plt.ylabel(\"Score\")\n",
        "    train_sizes, train_scores, test_scores = learning_curve(estimator, X, y, cv=cv, n_jobs=n_jobs, train_sizes=train_sizes)\n",
        "    train_scores_mean = np.mean(train_scores, axis=1)\n",
        "    train_scores_std = np.std(train_scores, axis=1)\n",
        "    test_scores_mean = np.mean(test_scores, axis=1)\n",
        "    test_scores_std = np.std(test_scores, axis=1)\n",
        "    plt.grid()\n",
        "\n",
        "    plt.fill_between(train_sizes, train_scores_mean - train_scores_std, train_scores_mean + train_scores_std, alpha=0.1, color=\"r\")\n",
        "    plt.fill_between(train_sizes, test_scores_mean - test_scores_std, test_scores_mean + test_scores_std, alpha=0.1, color=\"g\")\n",
        "    plt.plot(train_sizes, train_scores_mean, 'o-', color=\"r\", label=\"Training score\")\n",
        "    plt.plot(train_sizes, test_scores_mean, 'o-', color=\"g\", label=\"Cross-validation score\")\n",
        "    plt.legend(loc=\"best\")\n",
        "    plt.show()\n",
        "    return plt\n",
        "\n",
        "title = \"Learning Curves (Random Forest)\"\n",
        "# Cross validation with 100 iterations to get smoother mean test and train\n",
        "# score curves, each time with 20% data randomly selected as a validation set.\n",
        "cvx = ShuffleSplit(n_splits=100, test_size=0.2, random_state=0)\n",
        "y = y.as_matrix().ravel()\n",
        "plot_learning_curve(clf, title, X.as_matrix(), y, ylim=(0.7, 1.01), cv=cvx, n_jobs=4)\n",
        "\n",
        "#title = \"Learning Curves (SVM, RBF kernel, $\\gamma=0.001$)\"\n",
        "# SVC is more expensive so we do a lower number of CV iterations:\n",
        "#cv = ShuffleSplit(n_splits=10, test_size=0.2, random_state=0)\n",
        "#estimator = SVC(gamma=0.001)\n",
        "#plot_learning_curve(estimator, title, X, y, (0.7, 1.01), cv=cv, n_jobs=4)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "7a1c7365-2822-0b32-2417-bea2c6b96850"
      },
      "outputs": [],
      "source": [
        "I seem to get an accuracy of about 0.82 no matter which algorithm I use."
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.5.2"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}