{"nbformat_minor": 1, "nbformat": 4, "cells": [{"execution_count": null, "cell_type": "code", "outputs": [], "source": ["# This Python 3 environment comes with many helpful analytics libraries installed\n", "# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n", "# For example, here's several helpful packages to load in \n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "import io\n", "import bson     \n", "prod_to_category = dict()"], "metadata": {"collapsed": true, "_uuid": "c6f20e9f35e99236e5095f734dfa4cc2d9c73ba5", "_cell_guid": "01706550-69c1-4abc-b47a-c1cde3f64cf8"}}, {"execution_count": null, "cell_type": "code", "outputs": [], "source": ["def calculate_ratio(panda_frame,n,total):\n", "    panda_10k = panda_frame.sort_values(\n", "    \"frequency\", ascending=False).head(n).reset_index()\n", "    top_total = 0\n", "    for data in panda_10k['frequency']:\n", "        top_total += data\n", "    print(\"Total Image {}, Ratio : {}\".format(top_total,1.0*top_total/total))\n", "    return top_total"], "metadata": {"collapsed": true, "_uuid": "d6173fe81d060fb30bf0de3bf08fc6a9c480d37a", "_cell_guid": "047e8651-75cd-4772-a19e-39a8843ec801"}}, {"execution_count": null, "cell_type": "code", "outputs": [], "source": ["data = bson.decode_file_iter(open('../input/train.bson', 'rb'))\n", "category_frequency = dict()\n", "total_product = 0\n", "for c, d in enumerate(data):\n", "    total_product += 1\n", "    product_id = d['_id']\n", "    category_id = d['category_id'] # This won't be in Test data\n", "    if category_id not in category_frequency:\n", "        category_frequency[category_id] = 1\n", "    else:\n", "        category_frequency[category_id] += 1"], "metadata": {"collapsed": true, "_uuid": "043b4fa50d5d9e98d04cb1e627ead180a9389be1", "_cell_guid": "5f9fcf19-e553-45c3-b845-65e4a3e8b0d7"}}, {"execution_count": null, "cell_type": "code", "outputs": [], "source": ["table_frame = []\n", "for key,item in category_frequency.items():\n", "    table_frame.append([key,item])\n", "panda_frame = pd.DataFrame(table_frame, columns=['category_id','frequency'])"], "metadata": {"collapsed": true, "_uuid": "e8dd85bdf3bfa3ef27484025e1efa8efb3046b4c", "_cell_guid": "d87e17ab-d25c-4f8a-8981-7455408c52a4"}}, {"execution_count": null, "cell_type": "code", "outputs": [], "source": ["n = 1000\n", "calculate_ratio(panda_frame,n,total_product)"], "metadata": {"_uuid": "fdf0d11e45c2e521cfeeebc4783ba16f62e28f36", "_cell_guid": "7e8150d0-3c8f-433f-98fb-ebe6617d80b3"}}, {"execution_count": null, "cell_type": "code", "outputs": [], "source": ["n = 2000\n", "calculate_ratio(panda_frame,n,total_product)"], "metadata": {"_uuid": "431dba23b12175bdebc9232ea21592da46487d2d", "_cell_guid": "a709f429-f77f-4c03-ac5e-6c4e5b10d9be"}}, {"execution_count": null, "cell_type": "code", "outputs": [], "source": ["panda_10k = panda_frame.sort_values(\n", "    \"frequency\", ascending=False).head(n).reset_index()\n"], "metadata": {"collapsed": true, "_uuid": "e39ce0f1b6639fbf6a7da499dd4db7e71aca789b", "_cell_guid": "2512f096-5fc3-4443-b11e-56fca5735a43"}}, {"execution_count": null, "cell_type": "code", "outputs": [], "source": ["data = bson.decode_file_iter(open('../input/train.bson', 'rb'))\n", "image_frequency = dict()\n", "total_images = 0\n", "for c, d in enumerate(data):\n", "    product_id = d['_id']\n", "    category_id = d['category_id'] # This won't be in Test data\n", "    if category_id not in image_frequency:\n", "        image_frequency[category_id] = 1\n", "    for e, pic in enumerate(d['imgs']):\n", "        total_images += 1\n", "        image_frequency[category_id] += 1"], "metadata": {"collapsed": true, "_uuid": "b6d55c7091d922da433cf34a75c0359cb8e19320", "_cell_guid": "9f03d509-b67e-41c7-ab62-fa29a417baa9"}}, {"execution_count": null, "cell_type": "code", "outputs": [], "source": ["table_frame_2 = []\n", "for key,item in image_frequency.items():\n", "    table_frame_2.append([key,item])\n", "panda_frame_2 = pd.DataFrame(table_frame_2, columns=['category_id','frequency'])"], "metadata": {"collapsed": true, "_uuid": "b7909524e38d9885d3f5711d88c23fbc8b9dc0e5", "_cell_guid": "0d6b3d29-9ac9-4090-9b0f-fe56b0a349b7"}}, {"execution_count": null, "cell_type": "code", "outputs": [], "source": ["n = 1000\n", "total_images = calculate_ratio(panda_frame_2,n,total_images)"], "metadata": {"collapsed": true, "_uuid": "b0cf707d0d1ae4e3c702c4c24e9ce2d5a211bb6f", "_cell_guid": "337aba0f-b69d-47e4-a1d0-e5176b8f4492"}}, {"execution_count": null, "cell_type": "code", "outputs": [], "source": ["n = 2000\n", "calculate_ratio(panda_frame_2,n,total_images)"], "metadata": {"collapsed": true, "_uuid": "3246ad0faab8e901e02f3aeaa2fc51c1da343376", "_cell_guid": "ab06f3b0-07ed-469b-9291-8e13f21e27e0"}}, {"cell_type": "markdown", "source": ["## Summary\n", "* Top 1000 categories takes roughly 85% of overall training dataset\n", "* Top 2000 categories takes roughly 95% of overall training dataset\n"], "metadata": {}}], "metadata": {"kernelspec": {"language": "python", "name": "python3", "display_name": "Python 3"}, "language_info": {"pygments_lexer": "ipython3", "mimetype": "text/x-python", "file_extension": ".py", "version": "3.6.1", "codemirror_mode": {"version": 3, "name": "ipython"}, "name": "python", "nbconvert_exporter": "python"}}}