{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\nimport PIL\n\n%matplotlib inline\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/kuzushiji-recognition/train.csv\"); len(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"character_dict = pd.read_csv(\"/kaggle/input/kuzushiji-recognition/unicode_translation.csv\"); len(character_dict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ims = os.listdir(\"/kaggle/input/kuzushiji-recognition/train_images\"); len(ims)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Class used to store each individual character on a page"},{"metadata":{"trusted":true},"cell_type":"code","source":"class JapChar():\n    def __init__(self, char_data, im_id):\n        self.char = char_data[0]\n        self.x = int(char_data[1])\n        self.y = int(char_data[2])\n        self.width = int(char_data[3])\n        self.height = int(char_data[4])\n        self.im_id = im_id\n        \n    def get_area(self):\n        return self.width * self.height\n\n    def get_file(self):\n        return \"/kaggle/input/kuzushiji-recognition/train_images/\" + self.im_id + \".jpg\";\n    \n    def get_top_left(self):\n        return [self.x, self.y]\n    \n    def get_bottom_right(self):\n        return [self.x + self.width, self.y + self.height]\n    \n    def show(self):\n        plt.figure(figsize = (6, 6))\n        im = PIL.Image.open(self.get_file())\n        im = im.crop(self.get_top_left()  + self.get_bottom_right())\n        plt.imshow(im)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Class used to store a page of characters"},{"metadata":{"trusted":true},"cell_type":"code","source":"class ScripturePage:\n    def __init__(self, im_data):\n        self.id = im_data[0]\n        if type(im_data[1]) is not float:\n            split_labels = im_data[1].split()\n            self.labels = [JapChar(split_labels[i: i+5], self.id) for i in range(0, len(split_labels), 5)]\n        else:\n            self.labels = []\n        \n    def get_file(self):\n        return \"/kaggle/input/kuzushiji-recognition/train_images/\" + self.id + \".jpg\";\n    \n    def show(self):\n        plt.figure(figsize  = (10, 10))\n        plt.imshow(plt.imread(self.get_file()))\n        \n    def get_im(self):\n        return PIL.Image.open(self.get_file());\n    \n    def show_labeled(self):\n        plt.figure(figsize  = (10, 10))\n        ax = plt.gca()\n        plt.imshow(self.get_im())\n        \n        for label in self.labels:\n            box = Rectangle((label.x, label.y), label.width, label.height, fill = False, edgecolor = 'r')\n            ax.add_patch(box)\n            \n        plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = [ScripturePage(train.loc[i]) for i in range(len(train))]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"page = data[0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Example of what one character might look like"},{"metadata":{"trusted":true},"cell_type":"code","source":"page.labels[25].show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Example of what one page of scripture might look like"},{"metadata":{"trusted":true},"cell_type":"code","source":"page.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Same page as above but labeled"},{"metadata":{"trusted":true},"cell_type":"code","source":"page.show_labeled()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Puts all of the characters from the training set into one list called all_chars"},{"metadata":{"trusted":true},"cell_type":"code","source":"all_chars = []\nfor page in data:\n    all_chars = all_chars + page.labels","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Total number of characters in the training set"},{"metadata":{"trusted":true},"cell_type":"code","source":"len(all_chars)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"A histogram of the areas of the characters"},{"metadata":{"trusted":true},"cell_type":"code","source":"all_char_areas = [char.get_area() for char in all_chars]\nplt.figure(figsize=(10, 10))\n_,_,_ = plt.hist(all_char_areas, 20)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Frequency of each character in the training set"},{"metadata":{"trusted":true},"cell_type":"code","source":"char_freq = character_dict.copy()\ncodes = np.array(char_freq[\"Unicode\"])\nfreqs = np.zeros(len(char_freq))\n\nfor char in all_chars:\n    freqs[np.where(codes == char.char)[0]] += 1\n    \nchar_freq[\"Frequency\"] = freqs\nchar_freq.describe()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}