{"cells":[{"metadata":{"_uuid":"0bd288c7-d583-4f6f-ae3e-2de9f39aa8d2","_cell_guid":"520363dd-a298-479f-8105-90dac49b7209","trusted":true},"cell_type":"code","source":"#!pip install -U tensorflow","execution_count":0,"outputs":[]},{"metadata":{"_uuid":"2ed45852-b8fc-408e-a1c9-0ed7561650a4","_cell_guid":"56afca24-02f0-4e5a-8b3f-178e21f18460","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport time, gc\nimport tensorflow as tf\nfrom PIL import Image\nprint(tf.__version__)\n\nfrom sklearn.model_selection import train_test_split\nfrom matplotlib import pyplot as plt\nimport matplotlib\nmatplotlib.use('Agg')\n\n# import the necessary keras and sklearn packages\n\nfrom sklearn.preprocessing import LabelBinarizer\nfrom sklearn.model_selection import train_test_split\n\nimport random\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"68c1661a-4019-4b9d-a95b-70c25382837b","_cell_guid":"8ac35d16-621a-4709-ab1d-7415eea8dad8","trusted":true},"cell_type":"code","source":"train_df_ = pd.read_csv('/kaggle/input/bengaliai-cv19/train.csv')\ntest_df_ = pd.read_csv('/kaggle/input/bengaliai-cv19/test.csv')\nclass_map_df = pd.read_csv('/kaggle/input/bengaliai-cv19/class_map.csv')\nsample_sub_df = pd.read_csv('/kaggle/input/bengaliai-cv19/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"eb8c77ad-d354-4eb1-9863-af3d677147aa","_cell_guid":"add3e14b-9cd0-4435-8913-2022fb488b98","trusted":true},"cell_type":"markdown","source":"## Exploratory Data Analysis"},{"metadata":{"_uuid":"7492cd01-aeb4-42d5-b7f0-f85e543e59c5","_cell_guid":"48a17262-00de-4be5-a356-5abfb76e26d3","trusted":true},"cell_type":"code","source":"train_df_.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_50310 = train_df_[train_df_.image_id=='Train_50310']\nimg_50220 = train_df_[train_df_.image_id=='Train_50220']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_50310","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_50220","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"949e6b64-399d-4e14-96f8-09a16dc103df","_cell_guid":"5141b32a-6e6d-43a6-bf2c-a765d27c9cd3","trusted":true},"cell_type":"code","source":"len(train_df_)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"66690611-7a72-4a38-9299-98143d35e0ed","_cell_guid":"3d0384c6-161c-4ddb-ae63-38dd81a9c9a0","trusted":true},"cell_type":"code","source":"class_map_df.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d4f959cf-cbd1-4338-ad13-340a131b6e09","_cell_guid":"f9f1639b-e737-4ada-b36d-33836b432075","trusted":true},"cell_type":"code","source":"class_map_df.component_type.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c9391856-e02f-4e59-a086-f1a82c082f1d","_cell_guid":"ff3b339e-dd3b-4f4e-bef6-46884024cdb8","trusted":true},"cell_type":"code","source":"class_map_df_root = class_map_df[class_map_df.component_type=='grapheme_root']\nclass_map_df_vowel = class_map_df[class_map_df.component_type=='vowel_diacritic']\nclass_map_df_cons = class_map_df[class_map_df.component_type=='consonant_diacritic']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df_root.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df_vowel.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_map_df_cons.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"10e8c10a-fda5-4ff5-bf69-557636134af4","_cell_guid":"333bf698-f616-4845-b8e3-aca695ecbec1","trusted":true},"cell_type":"markdown","source":" ### Top 10 Grapheme Roots in training set"},{"metadata":{"_uuid":"c6c4890f-2e66-41ed-8437-bd500096a155","_cell_guid":"a5aa7950-4489-4ac1-ae74-4f17a4bbeed6","trusted":true},"cell_type":"code","source":"train_df_groot = train_df_.groupby(['grapheme_root']).size().reset_index()\ntrain_df_groot=train_df_groot.rename(columns={0:'count'})","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2d8e5518-2be0-438b-b0fb-8f18f7923e50","_cell_guid":"93c61507-4924-403b-aa18-4985d503c4b3","trusted":true},"cell_type":"code","source":"class_map_df_groot = class_map_df[class_map_df.component_type=='grapheme_root']\ngroot_merged = pd.merge(train_df_groot,class_map_df_groot[['label','component']],left_on='grapheme_root',right_on='label',how='inner')\ngroot_merged.sort_values(by=\"count\",ascending=False)[:10]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"891cf254-916d-4187-859f-6f46950d3e8a","_cell_guid":"00776742-dc28-4aa8-b73a-b6715fb1c144","trusted":true},"cell_type":"markdown","source":"### Vowel Diacritic in taining data (There are only 11)"},{"metadata":{"_uuid":"edb95192-bebc-40b9-9f7b-e2c50d328133","_cell_guid":"baa732ea-0803-45df-a585-478ecf44bf39","trusted":true},"cell_type":"code","source":"train_df_vd = train_df_.groupby(['vowel_diacritic']).size().reset_index()\ntrain_df_vd=train_df_vd.rename(columns={0:'count'})\nclass_map_df_vd = class_map_df[class_map_df.component_type=='vowel_diacritic']\nvd_merged = pd.merge(train_df_vd,class_map_df_vd[['label','component']],left_on='vowel_diacritic',right_on='label',how='inner')\nvd_merged.sort_values(by=\"count\",ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"457e0747-7c99-4ab1-941a-c9757be63cf0","_cell_guid":"e6861563-bd79-489f-a1de-7a7ee228594d","trusted":true},"cell_type":"markdown","source":"### Consonant Diacritic in training data (There are only 7)"},{"metadata":{"_uuid":"a59bf5e5-6a75-4b55-9a7c-5dd271be9b8e","_cell_guid":"476fe1f2-2f5d-4474-b97e-d00c527e82d3","trusted":true},"cell_type":"code","source":"train_df_cd = train_df_.groupby(['consonant_diacritic']).size().reset_index()\ntrain_df_cd=train_df_cd.rename(columns={0:'count'})\nclass_map_df_cd = class_map_df[class_map_df.component_type=='consonant_diacritic']\ncd_merged = pd.merge(train_df_cd,class_map_df_cd[['label','component']],left_on='consonant_diacritic',right_on='label',how='inner')\ncd_merged.sort_values(by=\"count\",ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"808be361-c28f-4820-aa41-6c132b330bd0","_cell_guid":"ad50ac5c-5ae1-493d-bbaa-956b64b92724","trusted":true},"cell_type":"markdown","source":"### Read the image file data from feather file, and check few images"},{"metadata":{"_uuid":"8068cfea-3160-4b1a-bdb6-8aa541805c0a","_cell_guid":"f04b2406-ae9b-48b8-b3e1-bf222fe9e1d3","trusted":true},"cell_type":"code","source":"def read_data(nf):\n    nf=int(nf)\n    train_df = pd.read_feather(f'/kaggle/input/bengaliaicv19feather/train_image_data_{nf}.feather')\n    return train_df\n\ndef read_test_data(nf):\n    nf=int(nf)\n    test_df = pd.read_feather(f'/kaggle/input/bengaliaicv19feather/test_image_data_{nf}.feather')\n    return test_df","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"34973c7d-289b-45af-81ff-3fd887719066","_cell_guid":"8c55ea48-d8c8-4028-9316-2f7d297fab16","trusted":true},"cell_type":"code","source":"train_df=read_data(1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_df.columns)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bdae18d7-6c32-48c1-862f-f87f6e0119aa","_cell_guid":"b4ef6a68-2f20-450b-a894-1f91f0c27fbd","trusted":true},"cell_type":"markdown","source":"### Check few sample images"},{"metadata":{"_uuid":"d41a4158-aef9-4b7f-8827-0d3328de2a94","_cell_guid":"38dd2dfe-14be-4145-9e86-dbbcb1b6c0b1","trusted":true},"cell_type":"code","source":"%matplotlib inline\nimport sys\nlabel = train_df.iloc[100,0]\nprint(label)\nimg = train_df.iloc[100,1:]\nimg=img.astype('uint8')\nimg = np.array(img).reshape(137,236)\nplt.imshow(img);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ef2d3b8b-19e2-455e-9e9a-a17da7c666aa","_cell_guid":"2fba6a18-6ed7-4508-884d-06a573c1b0d0","trusted":true},"cell_type":"code","source":"img = train_df.iloc[10,1:]\nlabel = train_df.iloc[10,0]\nimg=img.astype('uint8')\nprint(label)\nimg = np.array(img).reshape(137,236)\nimg = Image.fromarray(img)\nplt.imshow(img);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"df287e20-beaf-4edb-bb9e-8039e04ba5ed","_cell_guid":"910ddab8-4937-48d7-bf2d-9620602237a2","trusted":true},"cell_type":"markdown","source":"## Resize an image using PIL Image and check (resize to (96,96) input shape for CNN)"},{"metadata":{"_uuid":"453e1d86-7a71-490d-a9df-6d34eace8eef","_cell_guid":"b071be03-63b9-438a-b941-77d76544a621","trusted":true},"cell_type":"code","source":"fig = plt.figure()\nimg_resized = img.resize((96,96))\nplt.imshow(img_resized);","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}