{
  "cells": [
    {
      "cell_type": "markdown",
      "metadata": {
        "_cell_guid": "1166bdf4-b6af-7423-a56a-0de9269e020e"
      },
      "source": [
        "### Introduction\n",
        "\n",
        "This notebook is going to detail my initial EDA attempts on this problem. \n",
        "\n",
        "For a great introduction on the problem see here (I have borrowed the starting code to base this off..): https://www.kaggle.com/wendykan/youtube8m/starter-explore-youtube8m-sample-data \n",
        "\n",
        "## Things to do:\n",
        "\n",
        "- Average RGB/audio for the most popular labels\n",
        "- Try and measure the similarity of some of the labels\n",
        "- What next? Video recognition seems to be bloody tough"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2c91a446-7e24-4d64-a58c-642c524f396a"
      },
      "outputs": [],
      "source": [
        "# This Python 3 environment comes with many helpful analytics libraries installed\n",
        "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n",
        "# For example, here's several helpful packages to load in \n",
        "\n",
        "import numpy as np # linear algebra\n",
        "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n",
        "\n",
        "# Input data files are available in the \"../input/\" directory.\n",
        "# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n",
        "\n",
        "from subprocess import check_output\n",
        "print(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n",
        "\n",
        "# Any results you write to the current directory are saved as output."
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "2c86285d-e36a-0160-2988-7d2b920fd3ba"
      },
      "outputs": [],
      "source": [
        "import tensorflow as tf\n",
        "import numpy as np\n",
        "\n",
        "video_lvl_record = \"../input/video_level/train-1.tfrecord\"\n",
        "frame_lvl_record = \"../input/frame_level/train-1.tfrecord\"\n",
        "\n",
        "vid_ids = []\n",
        "labels = []\n",
        "mean_rgb = []\n",
        "mean_audio = []\n",
        "\n",
        "for example in tf.python_io.tf_record_iterator(video_lvl_record):\n",
        "    tf_example = tf.train.Example.FromString(example)\n",
        "\n",
        "    vid_ids.append(tf_example.features.feature['video_id'].bytes_list.value[0].decode(encoding='UTF-8'))\n",
        "    labels.append(tf_example.features.feature['labels'].int64_list.value)\n",
        "    mean_rgb.append(tf_example.features.feature['mean_rgb'].float_list.value)\n",
        "    mean_audio.append(tf_example.features.feature['mean_audio'].float_list.value)"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "5da817a7-e893-87bd-f169-92ad2bae6bcf"
      },
      "outputs": [],
      "source": [
        "n=20\n",
        "from collections import Counter\n",
        "label_mapping = pd.Series.from_csv('../input/label_names.csv',header=0)\n",
        "label_dict = label_mapping.to_dict()\n",
        "\n",
        "top_n = Counter([item for sublist in labels for item in sublist]).most_common(n)\n",
        "top_n_labels = [int(i[0]) for i in top_n]\n",
        "top_n_label_count = [int(i[1]) for i in top_n]\n",
        "top_n_label_names = [label_dict[x] for x in top_n_labels]\n",
        "\n",
        "top_labels = pd.DataFrame(data = top_n_labels, columns = ['label_num'])\n",
        "top_labels['count'] = top_n_label_count\n",
        "top_labels['label_name'] = top_n_label_names\n",
        "\n",
        "top_labels = top_labels.drop('label_num', axis = 1)\n",
        "\n",
        "import seaborn as sns\n",
        "import matplotlib.pyplot as plt\n",
        "\n",
        "ax = sns.barplot(x='label_name', y='count', data=top_labels)\n",
        "ax.set(xlabel='Label Name', ylabel='Label Count')\n",
        "ax.set_xticklabels(ax.xaxis.get_majorticklabels(), rotation=45)\n",
        "_ = plt.show();"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "4fc2a7f6-3a5f-f05b-3fce-3108eb7a6604"
      },
      "outputs": [],
      "source": [
        "#### Next step:\n",
        "#### Find a way of taking the lists generated above to be data frames\n",
        "#### Do more EDA on that"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "c6d9f4bd-2d7a-3f11-9ede-9175949c93c6"
      },
      "outputs": [],
      "source": [
        "### Trying to figure out how to get a list of videos in the top 20..\n",
        "\n",
        "test = labels[:5]\n",
        "label_test = [item for sublist in test for item in sublist]\n",
        "label_test"
      ]
    },
    {
      "cell_type": "code",
      "execution_count": null,
      "metadata": {
        "_cell_guid": "a2b4e988-a2be-f0c6-21c6-b133c0516e90"
      },
      "outputs": [],
      "source": ""
    }
  ],
  "metadata": {
    "_change_revision": 0,
    "_is_fork": false,
    "kernelspec": {
      "display_name": "Python 3",
      "language": "python",
      "name": "python3"
    },
    "language_info": {
      "codemirror_mode": {
        "name": "ipython",
        "version": 3
      },
      "file_extension": ".py",
      "mimetype": "text/x-python",
      "name": "python",
      "nbconvert_exporter": "python",
      "pygments_lexer": "ipython3",
      "version": "3.6.0"
    }
  },
  "nbformat": 4,
  "nbformat_minor": 0
}