{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport cv2\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = '../input/bengaliai-cv19/train_image_data_0.parquet'\ndata = pd.read_parquet(path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_path = '../input/bengaliai-cv19/train.csv'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_df = pd.read_csv(label_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"grapheme_1 = label_df.groupby('grapheme').get_group('প্র')\ngrapheme_1.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_lexicon = pd.read_csv('../input/indicword/indicword_bengali_lexicon_grapheme.csv')\ndf_lexicon.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_data(df,ind):\n    img_id = df.iloc[ind].values[0]\n    img = df.iloc[ind].values[1:].reshape(137,236).astype(np.uint8)\n    return img, img_id","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_img_bb(img):\n    kernel = np.ones((5,5),np.uint8)\n    opening = cv2.morphologyEx(img, cv2.MORPH_OPEN, kernel)\n    mask = opening<(255//2)\n\n    maskx = np.any(mask, axis=0)\n    masky = np.any(mask, axis=1)\n    x1 = np.argmax(maskx)\n    y1 = np.argmax(masky)\n    x2 = len(maskx) - np.argmax(maskx[::-1])\n    y2 = len(masky) - np.argmax(masky[::-1])\n    sub_image = img[y1:y2, x1:x2]\n    return sub_image, (y1,y2,x1,x2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_bb = []\nind = 10\nprint(df_lexicon.iloc[ind]['lexicon'])\ngraphemes = df_lexicon.iloc[ind]['graphemes'].split('-')\nindices = [label_df.groupby('grapheme').get_group(g).index[2] for g in graphemes]\nfor ind in indices:\n    img, img_id = get_data(data,ind)\n#     print(img_id)\n    sub_image,_ = get_img_bb(img)\n    img_bb.append(sub_image)\n    \nimg_start = img_bb[0]\nfor img in img_bb[1:]:\n    if img.shape[0]>img_start.shape[0]:\n        pad = img.shape[0]-img_start.shape[0]\n        img_start = np.concatenate(\n            (img_start,np.ones((pad,img_start.shape[1]),dtype=np.uint8)*255),\n            axis = 0\n        )\n        \n    elif img.shape[0]<img_start.shape[0]:\n        pad = img_start.shape[0] - img.shape[0]\n        img = np.concatenate(\n            (img,np.ones((pad,img.shape[1]),dtype=np.uint8)*255),\n            axis = 0\n        )\n    img_start = np.concatenate((img_start,img), axis = 1)\nplt.imshow(img_start)\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}