{
  "cells": [
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "db8273d5-66bb-9147-ca4f-fc8c42d76745"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "d4501df2-2d9f-1c9f-44b7-6dbd74360d25"
      },
      "outputs": [],
      "source": [
        "from sklearn.feature_extraction import DictVectorizer"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "94487bd6-d5ee-6eb1-7d77-89e4903bedc9"
      },
      "outputs": [],
      "source": [
        "nRows = 100000"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a6df54b6-90fa-1a58-275f-291f32ac3efe"
      },
      "outputs": [],
      "source": [
        "events = pd.read_csv('../input/events.csv', nrows=nRows)\n",
        "events.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "87f75f19-0df1-ae75-34ff-2d83e38c11d8"
      },
      "outputs": [],
      "source": [
        "len(events.display_id.unique())"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5ffa1e6b-0dff-4ee1-e610-bdc76fa3b9ff"
      },
      "outputs": [],
      "source": [
        "geoUpdate = events.geo_location.ravel()\n",
        "print(geoUpdate)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a4f051fe-5c48-8759-8691-c58469414d08"
      },
      "outputs": [],
      "source": [
        "geoNumeric = DictVectorizer()\n",
        "\n",
        "X_train_categ = geoNumeric.fit_transform(geoUpdate) #\u041e\u0431\u0443\u0447\u0430\u044e\u0449\u0430\u044f\n",
        "#X_test_categ = geoNumeric.transform(test) #\u043f\u0440\u043e\u0432\u0435\u0440\u043e\u0447\u043d\u0430\u044f\n",
        "print(X_train_categ)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4cf39227-0cfe-21d9-5d2d-fcb194d78109"
      },
      "outputs": [],
      "source": ""
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c2beb383-4222-7218-65d6-690026941214"
      },
      "outputs": [],
      "source": [
        "clicks_train = pd.read_csv('../input/clicks_train.csv', nrows=nRows)\n",
        "clicks_train.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "35694437-820e-38d2-0413-b3f94d8c9b0f"
      },
      "outputs": [],
      "source": [
        "len(clicks_train.ad_id.unique())"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "fdd837f4-acb4-d573-b6d0-aac53fad6776"
      },
      "outputs": [],
      "source": [
        "page_views_sample = pd.read_csv('../input/page_views_sample.csv', nrows=nRows)\n",
        "page_views_sample.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "44cb6ece-9a48-9829-a46a-ee2643da2783"
      },
      "outputs": [],
      "source": [
        "promoted_content = pd.read_csv('../input/promoted_content.csv', nrows=nRows)\n",
        "promoted_content.head()"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "06dc5986-4e19-3e39-6cf8-7e230a52f527"
      },
      "outputs": [],
      "source": [
        "len(promoted_content.ad_id.unique())"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "35949778-2e96-f115-2c27-85744159dbce"
      },
      "outputs": [],
      "source": [
        "len(promoted_content.document_id.unique())"
      ]
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.5.2"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}