{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom keras.layers import Dense,Conv2D,Flatten,MaxPool2D,Dropout,BatchNormalization, Input\nfrom keras.optimizers import Adam\nfrom keras.models import Model\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2021-09-05T14:57:42.239475Z","iopub.execute_input":"2021-09-05T14:57:42.239871Z","iopub.status.idle":"2021-09-05T14:57:46.516786Z","shell.execute_reply.started":"2021-09-05T14:57:42.239790Z","shell.execute_reply":"2021-09-05T14:57:46.515916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/bengaliai-cv19/train.csv')\ntest_df = pd.read_csv('/kaggle/input/bengaliai-cv19/test.csv')\nclass_map_df = pd.read_csv('/kaggle/input/bengaliai-cv19/class_map.csv')\nsample_sub_df = pd.read_csv('/kaggle/input/bengaliai-cv19/sample_submission.csv')","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2021-09-05T14:57:52.679691Z","iopub.execute_input":"2021-09-05T14:57:52.680042Z","iopub.status.idle":"2021-09-05T14:57:52.970143Z","shell.execute_reply.started":"2021-09-05T14:57:52.680006Z","shell.execute_reply":"2021-09-05T14:57:52.969219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:58:10.964419Z","iopub.execute_input":"2021-09-05T14:58:10.964803Z","iopub.status.idle":"2021-09-05T14:58:10.989279Z","shell.execute_reply.started":"2021-09-05T14:58:10.964771Z","shell.execute_reply":"2021-09-05T14:58:10.988262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:58:14.940172Z","iopub.execute_input":"2021-09-05T14:58:14.940518Z","iopub.status.idle":"2021-09-05T14:58:14.953422Z","shell.execute_reply.started":"2021-09-05T14:58:14.940471Z","shell.execute_reply":"2021-09-05T14:58:14.952643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:58:57.208000Z","iopub.execute_input":"2021-09-05T14:58:57.208330Z","iopub.status.idle":"2021-09-05T14:58:57.216918Z","shell.execute_reply.started":"2021-09-05T14:58:57.208299Z","shell.execute_reply":"2021-09-05T14:58:57.215977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Size of training data: {train_df.shape}')\nprint(f'Size of test data: {test_df.shape}')","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:58:25.385957Z","iopub.execute_input":"2021-09-05T14:58:25.386269Z","iopub.status.idle":"2021-09-05T14:58:25.394069Z","shell.execute_reply.started":"2021-09-05T14:58:25.386242Z","shell.execute_reply":"2021-09-05T14:58:25.393285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport numpy as np\nimport h5py\nimport matplotlib.pyplot as plt\nimport scipy\nfrom PIL import Image\nfrom scipy import ndimage\nimport tensorflow as tf\nfrom tensorflow.python.framework import ops\nfrom tqdm.auto import tqdm\nfrom glob import glob\nimport time, gc\nimport cv2\n\n\n%matplotlib inline\nnp.random.seed(1)","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:59:02.315281Z","iopub.execute_input":"2021-09-05T14:59:02.315616Z","iopub.status.idle":"2021-09-05T14:59:02.512550Z","shell.execute_reply.started":"2021-09-05T14:59:02.315586Z","shell.execute_reply":"2021-09-05T14:59:02.511752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_grapheme_root=train_df[\"grapheme_root\"]\ny_vowel_diacritic=train_df[\"vowel_diacritic\"]\ny_cons_diacritic=train_df[\"consonant_diacritic\"]","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:59:07.745343Z","iopub.execute_input":"2021-09-05T14:59:07.745684Z","iopub.status.idle":"2021-09-05T14:59:07.751264Z","shell.execute_reply.started":"2021-09-05T14:59:07.745656Z","shell.execute_reply":"2021-09-05T14:59:07.750513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_grapheme_root.max())\nprint(y_vowel_diacritic.max())\nprint(y_cons_diacritic.max())","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:59:11.233984Z","iopub.execute_input":"2021-09-05T14:59:11.234296Z","iopub.status.idle":"2021-09-05T14:59:11.240625Z","shell.execute_reply.started":"2021-09-05T14:59:11.234268Z","shell.execute_reply":"2021-09-05T14:59:11.239598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_to_one_hot(Y, C):\n    Y = np.eye(C)[np.reshape(Y,-1)]\n    return Y","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:59:18.094300Z","iopub.execute_input":"2021-09-05T14:59:18.094991Z","iopub.status.idle":"2021-09-05T14:59:18.103181Z","shell.execute_reply.started":"2021-09-05T14:59:18.094947Z","shell.execute_reply":"2021-09-05T14:59:18.102204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_root = convert_to_one_hot(y_grapheme_root, y_grapheme_root.max()+1).T\nY_cons = convert_to_one_hot(y_cons_diacritic,y_cons_diacritic.max()+1).T\nY_vowel = convert_to_one_hot(y_vowel_diacritic, y_vowel_diacritic.max()+1).T","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:59:21.636830Z","iopub.execute_input":"2021-09-05T14:59:21.637140Z","iopub.status.idle":"2021-09-05T14:59:21.722644Z","shell.execute_reply.started":"2021-09-05T14:59:21.637113Z","shell.execute_reply":"2021-09-05T14:59:21.721798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE=64\nN_CHANNELS=1","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:59:25.593233Z","iopub.execute_input":"2021-09-05T14:59:25.593565Z","iopub.status.idle":"2021-09-05T14:59:25.597819Z","shell.execute_reply.started":"2021-09-05T14:59:25.593537Z","shell.execute_reply":"2021-09-05T14:59:25.596863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs = Input(shape = (IMG_SIZE, IMG_SIZE, 1))\nmodel = Conv2D(filters=8, kernel_size=(4,4), padding='SAME',strides = [1,1] ,activation='relu', input_shape=(IMG_SIZE, IMG_SIZE, 1))(inputs)\nmodel = MaxPool2D(pool_size=(4,4),strides=[4,4],padding='SAME')(model)\nmodel = Flatten()(model)\nhead_root = Dense(168, activation = None)(model)\nhead_vowel = Dense(11, activation = None)(model)\nhead_consonant = Dense(7, activation = None)(model)\n\nmodel = Model(inputs=inputs, outputs=[head_root, head_vowel, head_consonant])\n\n","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:59:41.089750Z","iopub.execute_input":"2021-09-05T14:59:41.090073Z","iopub.status.idle":"2021-09-05T14:59:41.129592Z","shell.execute_reply.started":"2021-09-05T14:59:41.090045Z","shell.execute_reply":"2021-09-05T14:59:41.128835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T14:59:45.824669Z","iopub.execute_input":"2021-09-05T14:59:45.825006Z","iopub.status.idle":"2021-09-05T14:59:45.835028Z","shell.execute_reply.started":"2021-09-05T14:59:45.824978Z","shell.execute_reply":"2021-09-05T14:59:45.833861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:00:02.631005Z","iopub.execute_input":"2021-09-05T15:00:02.631357Z","iopub.status.idle":"2021-09-05T15:00:02.646299Z","shell.execute_reply.started":"2021-09-05T15:00:02.631326Z","shell.execute_reply":"2021-09-05T15:00:02.645307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64\nepochs = 100","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:00:07.814314Z","iopub.execute_input":"2021-09-05T15:00:07.814754Z","iopub.status.idle":"2021-09-05T15:00:07.818717Z","shell.execute_reply.started":"2021-09-05T15:00:07.814712Z","shell.execute_reply":"2021-09-05T15:00:07.817614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_root=Y_root.T\nY_cons =Y_cons.T\nY_vowel=Y_vowel.T","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:00:14.117286Z","iopub.execute_input":"2021-09-05T15:00:14.117628Z","iopub.status.idle":"2021-09-05T15:00:14.121679Z","shell.execute_reply.started":"2021-09-05T15:00:14.117597Z","shell.execute_reply":"2021-09-05T15:00:14.120672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize(df, size=64, need_progress_bar=True):\n    resized = {}\n    resize_size=64\n    if need_progress_bar:\n        for i in tqdm(range(df.shape[0])):\n            image=df.loc[df.index[i]].values.reshape(137,236)\n            _, thresh = cv2.threshold(image, 30, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n            contours, _ = cv2.findContours(thresh,cv2.RETR_LIST,cv2.CHAIN_APPROX_SIMPLE)[-2:]\n\n            idx = 0 \n            ls_xmin = []\n            ls_ymin = []\n            ls_xmax = []\n            ls_ymax = []\n            for cnt in contours:\n                idx += 1\n                x,y,w,h = cv2.boundingRect(cnt)\n                ls_xmin.append(x)\n                ls_ymin.append(y)\n                ls_xmax.append(x + w)\n                ls_ymax.append(y + h)\n            xmin = min(ls_xmin)\n            ymin = min(ls_ymin)\n            xmax = max(ls_xmax)\n            ymax = max(ls_ymax)\n\n            roi = image[ymin:ymax,xmin:xmax]\n            resized_roi = cv2.resize(roi, (resize_size, resize_size),interpolation=cv2.INTER_AREA)\n            resized[df.index[i]] = resized_roi.reshape(-1)\n    else:\n        for i in range(df.shape[0]):\n            #image = cv2.resize(df.loc[df.index[i]].values.reshape(137,236),(size,size),None,fx=0.5,fy=0.5,interpolation=cv2.INTER_AREA)\n            image=df.loc[df.index[i]].values.reshape(137,236)\n            _, thresh = cv2.threshold(image, 30, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n            contours, _ = cv2.findContours(thresh,cv2.RETR_LIST,cv2.CHAIN_APPROX_SIMPLE)[-2:]\n\n            idx = 0 \n            ls_xmin = []\n            ls_ymin = []\n            ls_xmax = []\n            ls_ymax = []\n            for cnt in contours:\n                idx += 1\n                x,y,w,h = cv2.boundingRect(cnt)\n                ls_xmin.append(x)\n                ls_ymin.append(y)\n                ls_xmax.append(x + w)\n                ls_ymax.append(y + h)\n            xmin = min(ls_xmin)\n            ymin = min(ls_ymin)\n            xmax = max(ls_xmax)\n            ymax = max(ls_ymax)\n\n            roi = image[ymin:ymax,xmin:xmax]\n            resized_roi = cv2.resize(roi, (resize_size, resize_size),interpolation=cv2.INTER_AREA)\n            resized[df.index[i]] = resized_roi.reshape(-1)\n    resized = pd.DataFrame(resized).T\n    return resized","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:00:16.909488Z","iopub.execute_input":"2021-09-05T15:00:16.909851Z","iopub.status.idle":"2021-09-05T15:00:16.925274Z","shell.execute_reply.started":"2021-09-05T15:00:16.909820Z","shell.execute_reply":"2021-09-05T15:00:16.924335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_=pd.DataFrame()\nfor i in range(4):\n    train_df_= pd.merge(pd.read_parquet(f'/kaggle/input/bengaliai-cv19/train_image_data_{i}.parquet'), train_df, on='image_id')\n    print(train_df_.shape)\n    X_train = train_df_.drop(['image_id','grapheme_root', 'vowel_diacritic', 'consonant_diacritic','grapheme'], axis=1)\n    X_train=resize(X_train)/255\n    X_train = X_train.values.reshape(-1, IMG_SIZE, IMG_SIZE, N_CHANNELS)\n    print(X_train.shape)\n    model.fit(X_train,{'dense_1': Y_root[i*50210:(i+1)*50210,:], 'dense_2': Y_vowel[i*50210:(i+1)*50210,:], 'dense_3': Y_cons[i*50210:(i+1)*50210,:]},batch_size=batch_size,epochs = epochs)\n    print(i)","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:00:23.359219Z","iopub.execute_input":"2021-09-05T15:00:23.359545Z","iopub.status.idle":"2021-09-05T15:02:20.752690Z","shell.execute_reply.started":"2021-09-05T15:00:23.359512Z","shell.execute_reply":"2021-09-05T15:02:20.750674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df,Y_root,Y_cons ,Y_vowel\nprint(train_df_.shape)\ndel train_df_","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:03:27.983428Z","iopub.execute_input":"2021-09-05T15:03:27.983795Z","iopub.status.idle":"2021-09-05T15:03:27.999286Z","shell.execute_reply.started":"2021-09-05T15:03:27.983763Z","shell.execute_reply":"2021-09-05T15:03:27.998206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:03:35.640899Z","iopub.execute_input":"2021-09-05T15:03:35.641224Z","iopub.status.idle":"2021-09-05T15:03:35.651185Z","shell.execute_reply.started":"2021-09-05T15:03:35.641192Z","shell.execute_reply":"2021-09-05T15:03:35.650324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_dict = {\n    'grapheme_root': [],\n    'vowel_diacritic': [],\n    'consonant_diacritic': []\n}","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:03:47.984736Z","iopub.execute_input":"2021-09-05T15:03:47.985062Z","iopub.status.idle":"2021-09-05T15:03:47.988922Z","shell.execute_reply.started":"2021-09-05T15:03:47.985033Z","shell.execute_reply":"2021-09-05T15:03:47.988063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"components = ['consonant_diacritic', 'grapheme_root', 'vowel_diacritic']\ntarget=[] # model predictions placeholder\nrow_id=[] # row_id place holder","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:04:00.989686Z","iopub.execute_input":"2021-09-05T15:04:00.990015Z","iopub.status.idle":"2021-09-05T15:04:00.994321Z","shell.execute_reply.started":"2021-09-05T15:04:00.989976Z","shell.execute_reply":"2021-09-05T15:04:00.993242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_=pd.DataFrame()\nfor i in range(4):\n    test_df_= pd.read_parquet('/kaggle/input/bengaliai-cv19/test_image_data_{}.parquet'.format(i)) \n    test_df_.set_index('image_id', inplace=True)\n    X_test=resize(test_df_)/255\n    print(X_test.shape)\n    X_test = X_test.values.reshape(-1, IMG_SIZE, IMG_SIZE, N_CHANNELS)\n    print(X_test.shape)\n    preds=model.predict(X_test)\n    print(preds)\n    for i, p in enumerate(preds_dict):\n        preds_dict[p] = np.argmax(preds[i], axis=1)\n        \n    for k,id in enumerate(test_df_.index.values):  \n        for i,comp in enumerate(components):\n            id_sample=id+'_'+comp\n            row_id.append(id_sample)\n            target.append(preds_dict[comp][k])","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:04:09.299407Z","iopub.execute_input":"2021-09-05T15:04:09.299765Z","iopub.status.idle":"2021-09-05T15:05:23.944139Z","shell.execute_reply.started":"2021-09-05T15:04:09.299733Z","shell.execute_reply":"2021-09-05T15:05:23.943229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample = pd.DataFrame(\n    {\n        'row_id': row_id,\n        'target':target\n    },\n    columns = ['row_id','target'] \n)\ndf_sample.to_csv('submission.csv',index=False)\ndf_sample.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T15:05:23.945802Z","iopub.execute_input":"2021-09-05T15:05:23.946155Z","iopub.status.idle":"2021-09-05T15:05:23.962242Z","shell.execute_reply.started":"2021-09-05T15:05:23.946116Z","shell.execute_reply":"2021-09-05T15:05:23.961149Z"},"trusted":true},"execution_count":null,"outputs":[]}]}