{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport PIL.Image as Image, PIL.ImageDraw as ImageDraw, PIL.ImageFont as ImageFont","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import plotly.graph_objects as go\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"HEIGHT=137\nWIDTH=236","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_as_npa(file):\n    df=pd.read_parquet(file)\n    return df.iloc[:,0],df.iloc[:,1:].values.reshape(-1,HEIGHT, WIDTH)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def image_from_char(char):\n    image = Image.new('RGB', (WIDTH, HEIGHT))\n    draw = ImageDraw.Draw(image)\n    myfont = ImageFont.truetype('/kaggle/input/bengali-fonts/hind_siliguri_normal_500.ttf', 120)\n    w, h = draw.textsize(char, font=myfont)\n    draw.text(((WIDTH - w) / 2,(HEIGHT - h) / 2), char, font=myfont)\n\n    return image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_ids0, images0 = load_as_npa('/kaggle/input/bengaliai-cv19/train_image_data_0.parquet')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(5,5,figsize=(16,8))\nax = ax.flatten()\n\nfor i in range(25):\n    ax[i].imshow(images0[i], cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/bengaliai-cv19/train.csv')\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_ids0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df = pd.read_csv('/kaggle/input/bengaliai-cv19/class_map.csv')\nclass_map_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of unique grapheme_root: {}\".format(train_df['grapheme_root'].nunique()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = go.Figure(data=[go.Histogram(x=train_df['grapheme_root'])])\nfig.update_layout(title_text='`grapheme_root`values')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It seems that \"grapheme_root\" is highly imbalanced."},{"metadata":{},"cell_type":"markdown","source":"# Most common  'grapheme_root' values"},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train_df['grapheme_root'].value_counts().sort_values()[-20:].index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y = train_df['grapheme_root'].value_counts().sort_values()[-20:].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig=go.Figure(data=[go.Bar(x=x,y=y)])\nfig.update_layout(title_text='Most Common grapheme_root values.Top 20')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"common_gr=class_map_df[(class_map_df['component_type']=='grapheme_root')&(class_map_df['label'].isin(x))]['component']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(4,5,figsize=(16,8))\nax = ax.flatten()\nfor i in range(20):\n    ax[i].imshow(image_from_char(common_gr.values[i]),cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train_df['grapheme_root'].value_counts().sort_values()[:20].index\ny = train_df['grapheme_root'].value_counts().sort_values()[:20].values\nfig = go.Figure(data=[go.Bar(x=x,y=y)])\nfig.update_layout(title_text='Least common grapheme_root values')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"notcommon_gr=class_map_df[(class_map_df['component_type']=='grapheme_root')&(class_map_df['label'].isin(x))]['component']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(4,5,figsize=(16,8))\nax = ax.flatten()\n\nfor i in range(20):\n    ax[i].imshow(image_from_char(notcommon_gr.values[i]),cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"# Vowel diacritic"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['vowel_diacritic'].nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x=train_df['vowel_diacritic'].value_counts().sort_values().index\ny=train_df['vowel_diacritic'].value_counts().sort_values().values\nfig = go.Figure(data=[go.Bar(x=x,y=y)])\nfig.update_layout(title_text='Vowel_diacritic values')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":""},{"metadata":{"trusted":true},"cell_type":"code","source":"#train Data의 component\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#각 component의 label과 생김새\nclass_map_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"vowels = class_map_df[(class_map_df['component_type']=='vowel_diacritic')&(class_map_df['label'].isin(x))]['component']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f,ax = plt.subplots(3,5,figsize=(16,8))\nax = ax.flatten()\n\nfor i in range(15):\n    if i < len(vowels):\n        ax[i].imshow(image_from_char(vowels.values[i]),cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"vowels.values","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Consonant diacritic "},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['consonant_diacritic'].nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train_df['consonant_diacritic'].value_counts().sort_values().index\ny = train_df['consonant_diacritic'].value_counts().sort_values().values\nfig = go.Figure(data=[go.Bar(x=x,y=y)])\nfig.update_layout(title_text='consonant_diacritic values')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"consonants=class_map_df[(class_map_df['component_type']=='consonant_diacritic')&(class_map_df['label'].isin(x))]['component']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(consonants)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f,ax = plt.subplots(1,7,figsize=(16,8))\nax = ax.flatten()\n\nfor i in range(7):\n    ax[i].imshow(image_from_char(consonants.values[i]),cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Similar Graphemes "},{"metadata":{},"cell_type":"markdown","source":"The most common grapheme_root is . Let's check some variants"},{"metadata":{"trusted":true},"cell_type":"code","source":"common_gr.values[9]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(image_from_char(common_gr.values[9]),cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = train_df[0:50000]\n\ngr_root_component=class_map_df[(class_map_df['component_type']=='grapheme_root')&(class_map_df['label']==72)]['component']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type(class_map_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type(class_map_df['component'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gr_root_component","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"## Digital variants of the most common *grapheme_root*"},{"metadata":{"trusted":true},"cell_type":"code","source":"samples=train_df[train_df['grapheme_root']==72].sample(n=25)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"samples.reset_index(drop=True, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f,ax = plt.subplots(5,5,figsize=(16,8))\nax=ax.flatten()\nk=0\nfor i, row in samples.iterrows():\n    ax[k].imshow(image_from_char(row['grapheme']),cmap='Greys')\n    k=k+1","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Handwritten variants of the most common *grapheme_root*"},{"metadata":{"trusted":true},"cell_type":"code","source":"f,ax = plt.subplots(5,5,figsize=(16,8))\nax=ax.flatten()\nk=0\nfor i, row in samples.iterrows():\n    ax[k].imshow(images0[i], cmap='Greys')\n    k=k+1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i, row in samples.iterrows():\n    print(i)\n    print(row)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"samples=train_df[\n    (train_df['grapheme_root']==72)&\n    (train_df['vowel_diacritic']==0)&\n    (train_df['consonant_diacritic']==0)\n].sample(n=25)\n\nf, ax = plt.subplots(5,5,figsize=(16,8))\nax = ax.flatten()\nk=0\nfor i, row in samples.iterrows():\n    ax[k].imshow(images0[i],cmap='Greys')\n    k=k+1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}