{"cells":[{"metadata":{},"cell_type":"markdown","source":"### If you find this kernel usefull, Do upvote."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sb\nimport matplotlib.pyplot as plt\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Train File"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/bengaliai-cv19/train.csv')\nprint('Train Data Shape: ', df_train.shape)\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Test File"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/bengaliai-cv19/test.csv')\nprint('Test Data Shape: ', df_test.shape)\ndf_test.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Class-Map File"},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map = pd.read_csv('/kaggle/input/bengaliai-cv19/class_map.csv')\nprint('Test Data Shape: ', class_map.shape)\nclass_map.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Image Data\nImage Data in in parquet files and contrain grayscale images of below mentioned dimentions. If you want to read more about this file format then try: \nhttps://acadgild.com/blog/parquet-file-format-hadoop\nNote that the file it self conatins values of all the 32332 pixels (137*236) in each row coresponding to a image.  \n###### Image Height = 137\n###### Image Width = 236"},{"metadata":{},"cell_type":"markdown","source":"### Image Utils"},{"metadata":{"trusted":true},"cell_type":"code","source":"HEIGHT = 137\nWIDTH = 236\n\ndef load_npa(file):\n    df = pd.read_parquet(file)\n    return df.iloc[:, 1:].values.reshape(-1, HEIGHT, WIDTH)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## loading one of the parquest file for analysis\ndummy_images = load_npa('/kaggle/input/bengaliai-cv19/train_image_data_0.parquet')\nprint(\"Shape of loaded files: \", dummy_images.shape)\nprint(\"Number of images in loaded files: \", dummy_images.shape[0])\nprint(\"Shape of first loaded image: \", dummy_images[0].shape)\nprint(\"\\n\\nFirst image looks like:\\n\\n\", dummy_images[0])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Must say hard to tell anything by looking at this image... Better Way to look at it"},{"metadata":{"trusted":true},"cell_type":"code","source":"## View the pixel values as image\nplt.imshow(dummy_images[0], cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"##### Some More images from loaded data"},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(5, 5, figsize=(16, 8))\nfor i in range(5):\n    for j in range(5):\n        ax[i][j].imshow(dummy_images[i*5+j], cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Train Data Analysis\n### Number of Samples: 200840\n#### File Contains:\n* image_id: the foreign key for the parquet files\n* grapheme_root: the first of the three target classes\n* vowel_diacritic: the second target class\n* consonant_diacritic: the third target class\n* grapheme: the complete character. Provided for informational purposes only, you should not need to use this."},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Unique Grapheme-Root in train data: \", df_train.grapheme_root.nunique())\nprint(\"Unique Vowel-Diacritic in train data: \", df_train.vowel_diacritic.nunique())\nprint(\"Unique Consonant-Diacritic in train data: \", df_train.consonant_diacritic.nunique())\nprint(\"Unique Grapheme (Combination of three) in train data: \", df_train.grapheme.nunique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"### Majority of Images per Grapheme count is below 180. Only 1 grapheme has 283 images in it.\nimages_per_grapheme = df_train.groupby('grapheme')[['image_id']].count().reset_index().reset_index()\nsb.catplot(x='index', y='image_id', data=images_per_grapheme)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_per_grapheme_root = df_train.groupby('grapheme_root')[['image_id']].count().reset_index().reset_index()\nsb.catplot(x='index', y='image_id', data=images_per_grapheme_root)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_per_grapheme_diacritic = df_train.groupby('vowel_diacritic')[['image_id']].count().reset_index()\nsb.catplot(x='vowel_diacritic', y='image_id', data=images_per_grapheme_diacritic)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_per_grapheme_diacritic = df_train.groupby('consonant_diacritic')[['image_id']].count().reset_index()\nsb.catplot(x='consonant_diacritic', y='image_id', data=images_per_grapheme_diacritic)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Above Plots can be used to observe class imbalance at different levels"},{"metadata":{},"cell_type":"markdown","source":"### Thanks for reading."}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}