{"nbformat": 4, "cells": [{"cell_type": "markdown", "metadata": {"_uuid": "5399c9ff99efd7f4443f5a24902a5b641ed86841", "_cell_guid": "56077885-c637-41ea-8412-15191fd510a8"}, "source": ["I wanted to test out random selection score. I just created this simple notebook to test that out. It scored really poorly as expected."]}, {"outputs": [], "cell_type": "code", "metadata": {"_uuid": "d5a18adb3bd7dbcb68e74eeed12cc10362222cf3", "collapsed": true, "_cell_guid": "6b824222-6a94-4ed1-be75-0ba131c535f9"}, "execution_count": null, "source": ["# Imports\n", "import numpy as np # linear algebra\n", "import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n", "\n", "import io\n", "import bson                       # this is installed with the pymongo package\n", "import matplotlib.pyplot as plt\n", "from skimage.data import imread   # or, whatever image library you prefer\n", "import multiprocessing as mp      # will come in handy due to the size of the data"]}, {"outputs": [], "cell_type": "code", "metadata": {"_uuid": "cab67e86a08e1c8c3197a545e3eb7d0939a339b9", "_cell_guid": "b4fe9af5-b936-4f97-998d-5dda96fa3c45"}, "execution_count": null, "source": ["cat_names_df = pd.read_csv(\"../input/category_names.csv\")\n", "samp_sub_df = pd.read_csv(\"../input/sample_submission.csv\")"]}, {"outputs": [], "cell_type": "code", "metadata": {"_uuid": "192f15a8015fc134b4c2b164ce9222c741f21bba", "_cell_guid": "14edb74f-de79-4e36-8cd0-1ca291c482cf"}, "execution_count": null, "source": ["print(cat_names_df.head())\n", "print(samp_sub_df.head())"]}, {"outputs": [], "cell_type": "code", "metadata": {"_uuid": "851d4406948a21d5e91aab884b516195b28bb3bf", "_cell_guid": "ca4a30d4-eb40-45e5-ae53-f2cf85bd18fe"}, "execution_count": null, "source": ["print(len(cat_names_df))"]}, {"outputs": [], "cell_type": "code", "metadata": {"_uuid": "de42b09e0a6ec570f6c20706da5283e557cb32c3", "_cell_guid": "91031050-2e4a-4394-ad32-42db23548ac0"}, "execution_count": null, "source": ["print(len(samp_sub_df))"]}, {"outputs": [], "cell_type": "code", "metadata": {"_uuid": "ef01984aeb17059f45d53db776f5e0fce6321bb5", "collapsed": true, "_cell_guid": "344a3f84-2daa-493f-9be7-8c91e690feab"}, "execution_count": null, "source": ["samp_sub_df[\"category_id\"] = np.random.choice(cat_names_df[\"category_id\"].values,len(samp_sub_df))"]}, {"outputs": [], "cell_type": "code", "metadata": {"_uuid": "76b3ca8c83c66055e16d877702503a5201405984", "_cell_guid": "ae2f7d6b-2b16-415f-9d2c-8c8c7dec369e"}, "execution_count": null, "source": ["print(samp_sub_df.head())"]}, {"outputs": [], "cell_type": "code", "metadata": {"_uuid": "a6f1df27950afd75cc2ded18558ca440c7bc188a", "collapsed": true, "_cell_guid": "bc0ba2e2-0e2d-48a7-80d2-295a740a6166"}, "execution_count": null, "source": ["#samp_sub_df.to_csv(\"rand_submission.csv.gz\", compression=\"gzip\", index=False)\n", "samp_sub_df.to_csv(\"rand_submission.csv\", index=False)"]}, {"outputs": [], "cell_type": "code", "metadata": {}, "execution_count": null, "source": ["data = bson.decode_file_iter(open('../input/train_example.bson', 'rb'))\n", "\n", "for c, d in enumerate(data):\n", "    product_id = d['_id']\n", "    category_id = d['category_id'] # This won't be in Test data\n", "    \n", "    for e, pic in enumerate(d['imgs']):\n", "        pic = imread(io.BytesIO(pic['picture']))\n", "        plt.imshow(pic);\n", "        plt.show()\n", "        # do something with the picture, etc"]}], "metadata": {"language_info": {"name": "python", "codemirror_mode": {"name": "ipython", "version": 3}, "nbconvert_exporter": "python", "version": "3.6.3", "mimetype": "text/x-python", "pygments_lexer": "ipython3", "file_extension": ".py"}, "kernelspec": {"name": "python3", "language": "python", "display_name": "Python 3"}}, "nbformat_minor": 1}