{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Bengali.AI\n-----\n### Quick data exploration\nIn the next couple of days, I'll continue to explore the BengaliAI dataset, stay tuned."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport PIL.Image as Image, PIL.ImageDraw as ImageDraw, PIL.ImageFont as ImageFont\n\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"HEIGHT = 137\nWIDTH = 236","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_as_npa(file):\n    df = pd.read_parquet(file)\n    return df.iloc[:, 0], df.iloc[:, 1:].values.reshape(-1, HEIGHT, WIDTH)\n\ndef image_from_char(char):\n    image = Image.new('RGB', (WIDTH, HEIGHT))\n    draw = ImageDraw.Draw(image)\n    myfont = ImageFont.truetype('/kaggle/input/bengaliai/hind_siliguri_normal_500.ttf', 120)\n    w, h = draw.textsize(char, font=myfont)\n    draw.text(((WIDTH - w) / 2,(HEIGHT - h) / 2), char, font=myfont)\n\n    return image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_ids0, images0 = load_as_npa('/kaggle/input/bengaliai-cv19/train_image_data_0.parquet')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(5, 5, figsize=(16, 8))\nax = ax.flatten()\n\nfor i in range(25):\n    ax[i].imshow(images0[i], cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### train.csv\n- **image_id**: the foreign key for the parquet files\n- **grapheme_root**: the first of the three target classes\n- **vowel_diacritic**: the second target class\n- **consonant_diacritic**: the third target class\n- **grapheme**: the complete character. Provided for informational purposes only, you should not need to use this."},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/bengaliai-cv19/train.csv')\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df = pd.read_csv('/kaggle/input/bengaliai-cv19/class_map.csv')\nclass_map_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Grapheme root"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of unique grapheme_root: {}\".format(train_df['grapheme_root'].nunique()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = go.Figure(data=[go.Histogram(x=train_df['grapheme_root'])])\nfig.update_layout(title_text='`grapheme_root` values')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"It seems that `grapheme_root` is highly imbalanced."},{"metadata":{},"cell_type":"markdown","source":"## Most common `grapheme_root` values"},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train_df['grapheme_root'].value_counts().sort_values()[-20:].index\ny = train_df['grapheme_root'].value_counts().sort_values()[-20:].values\nfig = go.Figure(data=[go.Bar(x=x, y=y)])\nfig.update_layout(title_text='Most common `grapheme_root` values')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"common_gr = class_map_df[(class_map_df['component_type'] == 'grapheme_root') & (class_map_df['label'].isin(x))]['component']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(4, 5, figsize=(16, 8))\nax = ax.flatten()\n\nfor i in range(20):\n    ax[i].imshow(image_from_char(common_gr.values[i]), cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Least common `grapheme_root` values"},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train_df['grapheme_root'].value_counts().sort_values()[:20].index\ny = train_df['grapheme_root'].value_counts().sort_values()[:20].values\nfig = go.Figure(data=[go.Bar(x=x, y=y)])\nfig.update_layout(title_text='Least common `grapheme_root` values')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"notcommon_gr = class_map_df[(class_map_df['component_type'] == 'grapheme_root') & (class_map_df['label'].isin(x))]['component']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(4, 5, figsize=(16, 8))\nax = ax.flatten()\n\nfor i in range(20):\n    ax[i].imshow(image_from_char(notcommon_gr.values[i]), cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"# Vowel diacritic"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['vowel_diacritic'].nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train_df['vowel_diacritic'].value_counts().sort_values().index\ny = train_df['vowel_diacritic'].value_counts().sort_values().values\nfig = go.Figure(data=[go.Bar(x=x, y=y)])\nfig.update_layout(title_text='`vowel_diacritic` values')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"vowels = class_map_df[(class_map_df['component_type'] == 'vowel_diacritic') & (class_map_df['label'].isin(x))]['component']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(3, 5, figsize=(16, 8))\nax = ax.flatten()\n\nfor i in range(15):\n    if i < len(vowels):\n        ax[i].imshow(image_from_char(vowels.values[i]), cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Consonant diacritic"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['consonant_diacritic'].nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = train_df['consonant_diacritic'].value_counts().sort_values().index\ny = train_df['consonant_diacritic'].value_counts().sort_values().values\nfig = go.Figure(data=[go.Bar(x=x, y=y)])\nfig.update_layout(title_text='`consonant_diacritic` values')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"consonants = class_map_df[(class_map_df['component_type'] == 'consonant_diacritic') & (class_map_df['label'].isin(x))]['component']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(1, 7, figsize=(16, 8))\nax = ax.flatten()\n\nfor i in range(7):\n    ax[i].imshow(image_from_char(consonants.values[i]), cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Similar Graphemes\nThe most common `grapheme_root` is `দ`. Let's check some variants."},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"train_df = train_df[0:50000]\n\n# Most common grapheme_root\ngr_root_component = class_map_df[(class_map_df['component_type'] == 'grapheme_root') & (class_map_df['label'] == 72)]['component']\nplt.imshow(image_from_char(gr_root_component[72]), cmap='Greys')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Digital variants of the most common `grapheme_root`"},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"samples = train_df[train_df['grapheme_root'] == 72].sample(n=25)\n# samples.reset_index(drop=True, inplace=True)\n\nf, ax = plt.subplots(5, 5, figsize=(16, 8))\nax = ax.flatten()\nk = 0\nfor i, row in samples.iterrows():\n    ax[k].imshow(image_from_char(row['grapheme']), cmap='Greys')\n    k = k + 1","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Handwritten variants of the most common `grapheme_root`\n\nThe samples below are the handwritten pairs of the digital ones above."},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"f, ax = plt.subplots(5, 5, figsize=(16, 8))\nax = ax.flatten()\nk = 0\nfor i, row in samples.iterrows():\n    ax[k].imshow(images0[i], cmap='Greys')\n    k = k + 1","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Examples of grapheme root `দ` without vowel_diacritic and consonant_diacritic components."},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"samples = train_df[\n    (train_df['grapheme_root'] == 72) &\n    (train_df['vowel_diacritic'] == 0) &\n    (train_df['consonant_diacritic'] == 0)\n].sample(n=25)\n\nf, ax = plt.subplots(5, 5, figsize=(16, 8))\nax = ax.flatten()\nk = 0\nfor i, row in samples.iterrows():\n    ax[k].imshow(images0[i], cmap='Greys')\n    k = k + 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"----------------\n**Thanks for reading. Please vote if you find this notebook useful.**"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}