{"nbformat": 4, "cells": [{"cell_type": "code", "outputs": [], "metadata": {"_uuid": "77404a7b41ae6bd04813caa3d2bb7a4d05e393fd", "collapsed": true, "_cell_guid": "06e9ac82-8827-4576-973d-cf25c9dc3e42"}, "execution_count": null, "source": ["import numpy as np\n", "import pandas as pd\n", "import io\n", "import bson\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread\n", "from tqdm import tqdm_notebook"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "51a77ccff36bb14bd6b5487a853236b76b2f8355", "collapsed": true, "_cell_guid": "73f7b276-37cc-407c-b3f5-5bdc680ee8d8"}, "execution_count": null, "source": ["categories = pd.read_csv('../input/category_names.csv', index_col='category_id')"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "58d3493a5d1e5f51efb20143538fe1ae1194e92d", "_cell_guid": "1c35da51-0623-45df-b1fe-da626c0fe030"}, "execution_count": null, "source": ["prod_id = []\n", "prod_category = []\n", "prod_num_imgs = []\n", "\n", "num_dicts = 7069896 # according to data page\n", "\n", "# This will take about 02m15s to complete\n", "with open('../input/train.bson', 'rb') as f, tqdm_notebook(total=num_dicts) as bar:\n", "        \n", "    data = bson.decode_file_iter(f)\n", "\n", "    for c, d in enumerate(data):\n", "        bar.update()\n", "        prod_id.append(d['_id'])\n", "        prod_category.append(d['category_id'])\n", "        prod_num_imgs.append(len(d['imgs']))"]}, {"cell_type": "markdown", "metadata": {"_uuid": "d9d2f538009aeb675fff970be709c33676110eb1", "_cell_guid": "f9e8bd93-f59f-48c7-b1c7-fec360ec1043"}, "source": ["Create the dataframe"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "b0d9b31aca3e6edad630f00b4bfda0a6e7922100", "collapsed": true, "_cell_guid": "4d3f6ca1-22e1-44ff-9eaa-37086c9dac67"}, "execution_count": null, "source": ["df_dict = {\n", "    'category': prod_category,\n", "    'num_imgs': prod_num_imgs\n", "}\n", "df = pd.DataFrame(df_dict, index=prod_id)\n", "del df_dict # Free memory"]}, {"cell_type": "markdown", "metadata": {"_uuid": "dec5561fccdfa4e80d44351cbf8de9a6f70c4acd", "_cell_guid": "f3bab2d7-9678-47fa-97b0-c26930821b3b"}, "source": ["### Number or images"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "bf724dd28730de64a2bde1b30e76248877779a38", "_cell_guid": "0c3bc9b9-77fd-4c3a-8a4c-aca19e918ef1"}, "execution_count": null, "source": ["df.num_imgs.value_counts().plot(kind='bar');"]}, {"cell_type": "markdown", "metadata": {"_uuid": "0b40b68e9d99891a0cdffecffb1effce4bf347d5", "_cell_guid": "f979bf1e-3c83-4503-ad72-34acc06db410"}, "source": ["### Most common categories\n", "\n", "Mos common on all lavels:"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "26bd499ba3b3a8b4c1b48ea6869386a00a2cd884", "_cell_guid": "8eb74819-a9e3-468b-8d3a-5cb9bdafa1ff"}, "execution_count": null, "source": ["df.category.value_counts().to_frame().head(25).join(categories)"]}, {"cell_type": "markdown", "metadata": {}, "source": ["Most common level 1"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["df.category.value_counts().to_frame().join(categories) \\\n", "    .groupby('category_level1')['category'].sum() \\\n", "    .sort_values(ascending=False).head(10).to_frame().reset_index()"]}, {"cell_type": "markdown", "metadata": {}, "source": ["Most common level 2"]}, {"cell_type": "code", "outputs": [], "metadata": {}, "execution_count": null, "source": ["df.category.value_counts().to_frame().join(categories) \\\n", "    .groupby(['category_level1', 'category_level2'])['category'].sum() \\\n", "    .sort_values(ascending=False).head(15).to_frame().reset_index()"]}, {"cell_type": "markdown", "metadata": {"_uuid": "6fb118654e619328aa29c4875f705356ae02ed99", "_cell_guid": "52f9bdd9-abfd-4628-aefd-e797e23828a7"}, "source": ["### Least common categories"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "b5ec7da4f50464df7f2eb0dd683ba77ef96eac1e", "_cell_guid": "23773540-1ae4-49d0-b7b0-0749cdb559b5"}, "execution_count": null, "source": ["df.category.value_counts().to_frame().tail(15).join(categories)"]}, {"cell_type": "markdown", "metadata": {"_uuid": "6e5e23a9745f2453fca2276dc60abdf5d669f1c0", "_cell_guid": "42287c76-89fd-46fd-b4a2-af72378f648c"}, "source": ["### Is there any relation between id and category?"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "af78079cc5fccbc4033ff9472a5fdd6dfeb5c5fa", "_cell_guid": "077585a2-75a3-4398-9b4d-da1ddf4816c1"}, "execution_count": null, "source": ["df_cat = df.category.sample(100000, random_state=42).reset_index()\n", "\n", "fig, ax = plt.subplots(1,1, figsize=(10, 10))\n", "ax.set_xlabel('index')\n", "ax.set_ylabel('category')\n", "ax.scatter(df_cat.index.values, df_cat.category.values, alpha=0.02);"]}, {"cell_type": "markdown", "metadata": {"_uuid": "47656eea490ad831daee7393d57bbd682ded85c4", "_cell_guid": "e7598c61-a84b-4320-a7ea-a159f83dd66e"}, "source": ["The category seens to be very unrelated to the identifier, which are good news (no leak here)."]}, {"cell_type": "markdown", "metadata": {"_uuid": "d02f777a36bf70fe4a726aab1cb48c43b1734f1b", "_cell_guid": "de49db2d-4a68-4258-9e44-8cf3b3030614"}, "source": ["### Number of categories and accuracy\n", "\n", "We may try to reduce the 5270 categories in order to ease the train. But what's the cost of accuracy for this approach? The chart bellow show this relation:"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "be4c6c4fe3e044d4177d502bd53e2296e55e9ceb", "_cell_guid": "7d896458-3ebb-4c44-9b25-27681eebaa4d"}, "execution_count": null, "source": ["cum_sum = df.category.value_counts().cumsum().reset_index(drop=True)\n", "cum_sum /= cum_sum.max()\n", "ax = cum_sum.plot(figsize=(12, 8))\n", "ax.grid()\n", "ax.set_xlabel('Num. of Categories')\n", "ax.set_ylabel('Max. Accuracy');"]}, {"cell_type": "markdown", "metadata": {"_uuid": "6c9b7569aed212c28055856d238d565e318a0058", "_cell_guid": "5a38d19e-35b9-46e0-bb99-715dbe121f2b"}, "source": ["And here is a table of some values."]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "cd377f2bf9803b69d3a2df54c3d884674e2ada26", "_cell_guid": "958ce2c1-42b7-4463-94ab-06a4041e8fa1"}, "execution_count": null, "source": ["max_accuracies = [0.2, 0.4, 0.6, 0.8, 0.9, 0.95, 0.98, 0.99, 1.0]\n", "num_cat = map(lambda a: (cum_sum.values <= a).sum(), max_accuracies)\n", "\n", "pd.DataFrame({'Num. Categories': list(num_cat), 'Max. Accuracy': max_accuracies})"]}, {"cell_type": "markdown", "metadata": {"_uuid": "e83b63037c64735686c9558ec8cd7f3a1969d5d3", "_cell_guid": "ee5975e6-6426-460e-bb06-df300297e862"}, "source": ["### Most common category submission\n", "\n", "This will score 0.01121 on LB"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "91a0b293813135fbdf759aa56a2be79473e042bf", "collapsed": true, "_cell_guid": "b0e0c969-1ac7-4a4c-9746-8f6c321a2cc7"}, "execution_count": null, "source": ["submission = pd.read_csv('../input/sample_submission.csv')"]}, {"cell_type": "code", "outputs": [], "metadata": {"_uuid": "3405157198e5ef57e0036378d80abf5863b22fa4", "_cell_guid": "594b9960-efeb-40c2-b6b5-031431d01caa"}, "execution_count": null, "source": ["submission['category_id'] = 1000018296\n", "\n", "submission.to_csv('most_common_benchmark.csv.gz', compression='gzip', index=False)"]}], "metadata": {"kernelspec": {"language": "python", "name": "python3", "display_name": "Python 3"}, "language_info": {"file_extension": ".py", "codemirror_mode": {"name": "ipython", "version": 3}, "pygments_lexer": "ipython3", "version": "3.6.1", "name": "python", "nbconvert_exporter": "python", "mimetype": "text/x-python"}}, "nbformat_minor": 1}