{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\n\nfrom collections import Counter\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport skimage\nimport seaborn as sns\n\nfrom IPython.display import Image\n\nINPUT_DIR = '/kaggle/input/kuzushiji-recognition'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Load training data\nAvailable data is described in the [competition description](https://www.kaggle.com/c/kuzushiji-recognition/data)"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_metadata = pd.read_csv(os.path.join(INPUT_DIR, 'train.csv'), index_col='image_id')\nunicode_translations = pd.read_csv(os.path.join(INPUT_DIR, 'unicode_translation.csv'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_metadata.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'We have {train_metadata.shape[0]} images in the train dataset')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# There are some images that do not contain any character, let's have a look at one\nimg_no_chars = train_metadata[train_metadata.labels.isnull()].reset_index().iloc[0].image_id\nImage(os.path.join(INPUT_DIR, 'train_images', '{}.jpg'.format(img_no_chars)), width=500)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Would be usefult to visualize the images with bounding boxes, let's write a function to do so"},{"metadata":{"trusted":true},"cell_type":"code","source":"def show_page(img_id, bounding_boxes=False):\n    \"\"\" Shows a page (image)\n    \n    :param img_id - str: ID of the page to show\n    :param bounding_boxes - boolean: If True, will drow the bounding boxes on the characters contained in the image\n    \"\"\"\n    img = skimage.io.imread(os.path.join(INPUT_DIR, 'train_images', '{}.jpg'.format(img_id)))\n    plt.figure(figsize=(15,15))\n    plt.imshow(img)\n    \n    if bounding_boxes:\n        def _chunks(l, n):\n            for i in range(0, len(l), n):\n                yield l[i:i+n]\n        ax = plt.gca()\n        chars = train_metadata.loc[img_id].labels.split()\n        for char, x, y, height, width in chars[::5]:\n            ax.add_patch(plt.Rectangle(x, y, height, width, color='blue', fill=False, linewidth=2))\n            ax.text(x, y, char, size='x-large', color='white', bbox={'facecolor':'blue', 'alpha':1.0})\n            plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_page('100241706_00004_2', bounding_boxes=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unicode_translations.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's try to understand a bit the available data by answering to the following questions:\n\n1. What's the distribution of characters per image?\n    * Min characters per image\n    * Average characters per image\n    * Max characters per image\n    * Total numbers of different characters (classes)\n    * Overall distribution of classes"},{"metadata":{"trusted":true},"cell_type":"code","source":"def num_chars_in_image(labels):\n    \"\"\" Return the number of characters in an image given the labels string\n    \n    Labels is a string with all the characters in the image. The format is labels = ['unicode char', 'X', 'Y', 'width', 'height'] for each character.\n    So if we jump the list in steps of length 5, we can get all the unicodes\n    \"\"\"\n    return 0 if labels is np.nan else len(labels[::5])\n    \ntrain_metadata['num_chars'] = train_metadata.labels.apply(lambda l: num_chars_in_image(l))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def count_characters(df):\n    counter = Counter()\n    for labels in df[df.num_chars > 0].labels:\n        for c in labels.split()[::5]:\n            counter[c]+=1\n    return counter\n        \nchars_count = count_characters(train_metadata)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f'Min characters per image: {train_metadata.num_chars.min()}')\nprint(f'Max characters per image: {train_metadata.num_chars.max()}')\nprint(f'Avg characters per image: {np.average(train_metadata.num_chars)} (std of {np.std(train_metadata.num_chars)})')\nprint(f'Total number of different characters (classes) is {len(chars_count)}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"_ = sns.distplot(train_metadata.num_chars, kde=False)\n_ = plt.title('Number of characters per page count distribution')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"If we don't take into account the pages without characters at all, this looks like a binomial distribution, where the biggest concentration is around 800 characters. The distribution is quite wide though"},{"metadata":{"trusted":true},"cell_type":"code","source":"_ = sns.distplot(train_metadata[train_metadata.num_chars > 0].num_chars, kde=False)\n_ = plt.title('Number of characters per page count distribution (no blank pages)')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"A random classifier would have an accuracy of `1/num_classes`, which is `0.0237%` Let's see if we can beat that without _much_ effort"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}