{"metadata": {"kernelspec": {"language": "python", "name": "python3", "display_name": "Python 3"}, "language_info": {"version": "3.6.3", "file_extension": ".py", "name": "python", "pygments_lexer": "ipython3", "nbconvert_exporter": "python", "mimetype": "text/x-python", "codemirror_mode": {"version": 3, "name": "ipython"}}}, "nbformat_minor": 1, "cells": [{"execution_count": null, "metadata": {"_cell_guid": "06e9ac82-8827-4576-973d-cf25c9dc3e42", "collapsed": true, "_uuid": "77404a7b41ae6bd04813caa3d2bb7a4d05e393fd"}, "source": ["import numpy as np\n", "import pandas as pd\n", "import io\n", "import bson\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread\n", "from tqdm import tqdm_notebook"], "cell_type": "code", "outputs": []}, {"execution_count": null, "metadata": {"_cell_guid": "73f7b276-37cc-407c-b3f5-5bdc680ee8d8", "collapsed": true, "_uuid": "51a77ccff36bb14bd6b5487a853236b76b2f8355"}, "source": ["categories = pd.read_csv('../input/category_names.csv', index_col='category_id')"], "cell_type": "code", "outputs": []}, {"execution_count": null, "metadata": {"_cell_guid": "1c35da51-0623-45df-b1fe-da626c0fe030", "_uuid": "58d3493a5d1e5f51efb20143538fe1ae1194e92d"}, "source": ["prod_id = []\n", "prod_category = []\n", "prod_num_imgs = []\n", "\n", "num_dicts = 7069896 # according to data page\n", "\n", "# This will take about 02m15s to complete\n", "with open('../input/train.bson', 'rb') as f, tqdm_notebook(total=num_dicts) as bar:\n", "        \n", "    data = bson.decode_file_iter(f)\n", "\n", "    for c, d in enumerate(data):\n", "        bar.update()\n", "        prod_id.append(d['_id'])\n", "        prod_category.append(d['category_id'])\n", "        prod_num_imgs.append(len(d['imgs']))"], "cell_type": "code", "outputs": []}, {"source": ["Create the dataframe"], "metadata": {"_cell_guid": "f9e8bd93-f59f-48c7-b1c7-fec360ec1043", "_uuid": "d9d2f538009aeb675fff970be709c33676110eb1"}, "cell_type": "markdown"}, {"execution_count": null, "metadata": {"_cell_guid": "4d3f6ca1-22e1-44ff-9eaa-37086c9dac67", "collapsed": true, "_uuid": "b0d9b31aca3e6edad630f00b4bfda0a6e7922100"}, "source": ["df_dict = {\n", "    'category': prod_category,\n", "    'num_imgs': prod_num_imgs\n", "}\n", "df = pd.DataFrame(df_dict, index=prod_id)\n", "del df_dict # Free memory"], "cell_type": "code", "outputs": []}, {"source": ["### Number or images"], "metadata": {"_cell_guid": "f3bab2d7-9678-47fa-97b0-c26930821b3b", "_uuid": "dec5561fccdfa4e80d44351cbf8de9a6f70c4acd"}, "cell_type": "markdown"}, {"execution_count": null, "metadata": {"scrolled": true, "_cell_guid": "0c3bc9b9-77fd-4c3a-8a4c-aca19e918ef1", "_uuid": "bf724dd28730de64a2bde1b30e76248877779a38"}, "source": ["df.num_imgs.value_counts().plot(kind='bar');\n", "print(\"## Total number of images: {:d}\".format(df.num_imgs.sum()))\n", "print(\"## Total number of categories: {:d}\".format(len(pd.unique(df.category))))"], "cell_type": "code", "outputs": []}, {"source": ["Calculating the destribution of each category in the train data."], "metadata": {"_cell_guid": "5d6f05c5-10b4-406d-b4c8-147a7d8c5a8f", "_uuid": "390e4eaf3ab9ab1f6163aec276cef926ef3287ac"}, "cell_type": "markdown"}, {"execution_count": null, "metadata": {"_cell_guid": "726b0e4d-8fce-486d-934d-bbb0cc570aef", "_uuid": "ef76dc543de668311c55a79b4b9ddaf10edb2803"}, "source": ["cat_counts = df.category.value_counts().to_frame()\n", "cat_counts = cat_counts / cat_counts[\"category\"].sum()\n", "print(cat_counts.head())\n", "print(pd.unique(df.category))"], "cell_type": "code", "outputs": []}, {"execution_count": null, "metadata": {}, "source": ["cat_counts.sort_values(by=\"category\",inplace=True)\n", "bot_5_categories = cat_counts.head()\n", "top_5_categories = cat_counts.tail()\n", "print(bot_5_categories)\n", "print(top_5_categories)"], "cell_type": "code", "outputs": []}, {"source": ["Now creating a biased random sample based on the distribution of data."], "metadata": {"_cell_guid": "5a38d19e-35b9-46e0-bb99-715dbe121f2b", "_uuid": "6c9b7569aed212c28055856d238d565e318a0058"}, "cell_type": "markdown"}, {"execution_count": null, "metadata": {"_cell_guid": "958ce2c1-42b7-4463-94ab-06a4041e8fa1", "collapsed": true, "_uuid": "cd377f2bf9803b69d3a2df54c3d884674e2ada26"}, "source": ["samp_sub_df = pd.read_csv(\"../input/sample_submission.csv\")\n", "samp_sub_df[\"category_id\"] = np.random.choice(cat_counts.index,size=len(samp_sub_df),p=cat_counts[\"category\"].values)\n", "samp_sub_df.to_csv(\"baised_rand_submission.csv\", index=False)"], "cell_type": "code", "outputs": []}, {"execution_count": null, "metadata": {"collapsed": true}, "source": [], "cell_type": "code", "outputs": []}], "nbformat": 4}