{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/bengaliai-cv19/train.csv')\ntest = pd.read_csv('/kaggle/input/bengaliai-cv19/test.csv')\nclass_map_df = pd.read_csv('/kaggle/input/bengaliai-cv19/class_map.csv')\nsample_sub_df = pd.read_csv('/kaggle/input/bengaliai-cv19/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## 3 outputs or prediction from 1 image. So total 12 images in test set.\nlen(test)/3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_sub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df.component_type.unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df.component_type.value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Consonant Diacritic"},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df.component[class_map_df.component_type=='consonant_diacritic']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df[class_map_df.component_type=='consonant_diacritic']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"const_diac = class_map_df.component[class_map_df.component_type=='consonant_diacritic'].values\nconst_diac","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for diac in const_diac:\n    for i in diac:\n        print(i,end=' ')\n    print()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"These writings are too tiny, can't see."},{"metadata":{"trusted":true},"cell_type":"code","source":"HEIGHT = 236\nWIDTH = 236\nimport PIL.Image as Image, PIL.ImageDraw as ImageDraw, PIL.ImageFont as ImageFont\nimport matplotlib.pyplot as plt\n\ndef image_from_char(char):\n    image = Image.new('RGB', (WIDTH, HEIGHT))\n    draw = ImageDraw.Draw(image)\n    myfont = ImageFont.truetype('/kaggle/input/kalpurush-fonts/kalpurush-2.ttf', 120)\n    w, h = draw.textsize(char, font=myfont)\n    draw.text(((WIDTH - w) / 2,(HEIGHT - h) / 3), char, font=myfont)\n\n    return image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(1, 7, figsize=(16, 8))\nax = ax.flatten()\n\nfor i,diac in enumerate(const_diac):\n    ax[i].imshow(image_from_char(diac), cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"1. Lets see them in action, maybe then I'll be clearer."},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"const_sample = train.sort_values(['consonant_diacritic']).groupby('consonant_diacritic').head(2).reset_index()\nconst_sample","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for j,i in enumerate(range(0,len(const_sample),2)):\n    print(j,i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(2, 7, figsize=(16, 8))\nax = ax.flatten()\n\nfor i,diac in enumerate(const_diac):\n    ax[i].axis(\"off\")\n    print(diac)\n    ax[i].imshow(image_from_char(diac), cmap='Greys')\nfor j,i in enumerate(range(0,len(const_sample),2)):\n    x = const_sample.iloc[i].grapheme\n    print(x)\n    ax[j+7].axis(\"off\")\n#     ax[j+7].title.set_text(x)\n    ax[j+7].imshow(image_from_char(x), cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Consonant diacritic names:\n    - 0\n    - chondro-bindu\n    - ref\n    - ref + jofola\n    - jofola\n    - rofola\n    - rofola + jofola"},{"metadata":{},"cell_type":"markdown","source":"## Vowel Diacritic"},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df.component[class_map_df.component_type=='vowel_diacritic']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Explore Images"},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train), len(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Use a small subset for visualization"},{"metadata":{"trusted":true},"cell_type":"code","source":"ok = pd.read_parquet(f'/kaggle/input/bengaliai-cv19/train_image_data_0.parquet')\nok.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ok = pd.merge(ok, train, on='image_id')\nok.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(ok), len(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"50210*4 # 4 parquet files","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"only_imgs = ok.drop(columns=['image_id','grapheme_root','vowel_diacritic','consonant_diacritic','grapheme'])\nonly_imgs.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(only_imgs.columns)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"HEIGHT = 137\nWIDTH = 236\nHEIGHT*WIDTH","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(5, 5, figsize=(16, 8))\nax = ax.flatten()\n\nfor i in range(25):\n    ax[i].axis(\"off\")\n    ax[i].imshow(only_imgs.iloc[i].values.reshape(HEIGHT,WIDTH), cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Even I can't understand some. Wonder how machine will! Some images are cut :(\n\nTry to use: https://www.kaggle.com/c/bengaliai-cv19/discussion/122731"},{"metadata":{"trusted":true},"cell_type":"code","source":"# for i in range(4):\n#     ### inner train will remove other rows\n#     train_df = pd.merge(pd.read_parquet(f'/kaggle/input/bengaliai-cv19/train_image_data_{i}.parquet'), train_df, on='image_id')#.drop(['image_id'], axis=1)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}