{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# https://www.kaggle.com/c/bengaliai-cv19/discussion/123198\n# 위와 같이 best single model에 들어가서 사람들이 어떤 식으로 모델링을 하고 점수를 어느정도 얻었는지 파악하고 \n# 대회를 진행하면 비교적 수월하게 진행가능","metadata":{"execution":{"iopub.status.busy":"2021-10-07T14:11:47.027344Z","iopub.execute_input":"2021-10-07T14:11:47.028141Z","iopub.status.idle":"2021-10-07T14:11:47.057373Z","shell.execute_reply.started":"2021-10-07T14:11:47.027996Z","shell.execute_reply":"2021-10-07T14:11:47.056320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os","metadata":{"execution":{"iopub.status.busy":"2021-10-07T14:11:47.060880Z","iopub.execute_input":"2021-10-07T14:11:47.061089Z","iopub.status.idle":"2021-10-07T14:11:47.201258Z","shell.execute_reply.started":"2021-10-07T14:11:47.061065Z","shell.execute_reply":"2021-10-07T14:11:47.199994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '../input/bengaliai-cv19'","metadata":{"execution":{"iopub.status.busy":"2021-10-07T14:11:47.203079Z","iopub.execute_input":"2021-10-07T14:11:47.203699Z","iopub.status.idle":"2021-10-07T14:11:47.212268Z","shell.execute_reply.started":"2021-10-07T14:11:47.203659Z","shell.execute_reply":"2021-10-07T14:11:47.211369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Image visualization and folding","metadata":{}},{"cell_type":"code","source":"files_train = [f'train_image_data_{fid}.parquet' for fid in range(4)]\n# for문과 f string을 이용해서 불러온다.\n\n\n### f string example ###\n# name = 'Song' sex = 'male' \n# f'Hi, I am {name}. I am {sex}.'\n# >>> 'Hi, I am song. I am male.'\n","metadata":{"execution":{"iopub.status.busy":"2021-10-07T14:11:47.215721Z","iopub.execute_input":"2021-10-07T14:11:47.216512Z","iopub.status.idle":"2021-10-07T14:11:47.224968Z","shell.execute_reply.started":"2021-10-07T14:11:47.216468Z","shell.execute_reply":"2021-10-07T14:11:47.224043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_train","metadata":{"execution":{"iopub.status.busy":"2021-10-07T14:11:47.227811Z","iopub.execute_input":"2021-10-07T14:11:47.228967Z","iopub.status.idle":"2021-10-07T14:11:47.245554Z","shell.execute_reply.started":"2021-10-07T14:11:47.228931Z","shell.execute_reply":"2021-10-07T14:11:47.244706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0 = pd.read_parquet('../input/bengaliai-cv19/train_image_data_0.parquet')","metadata":{"execution":{"iopub.status.busy":"2021-10-07T14:11:47.247203Z","iopub.execute_input":"2021-10-07T14:11:47.247819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train0.shape)\n\n# column이 많은 이유 : gray scale의 사진인데 (137 heights, 236 widths)크기의 이미지이므로 \n\nprint(137*236+1) \n\n# 여기서 +1은 'image_id'이다.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### 결국 한 row가 하나의 이미지를 뜻한다. 즉 (137 multiply 236)를 일렬로 나열한 것이므로 이미지를 보고자 하면 다시 (137 multiply 236)형태로 만들어주면 되는 것.","metadata":{}},{"cell_type":"code","source":"idx=0\nimg=train0.iloc[idx,1:].values.astype(np.uint8) # '1:'인 이유는 image_id 를 빼야하기 때문 \n\n# 'astype(np.uint8)' : 굳이 큰 용량이 필요한 datatype을 사용할 필요가 없고 효율적으로 진행하기위해","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.reshape(137,236).shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img.reshape(137,236))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img.reshape(137,236), cmap = 'gray') # 흑백이미지 스케일로 보기","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train0) \n\n# 행의 수","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx = np.random.randint(len(train0))\n\n# 이미지 대회를 할때 중요한 것은 사진을 계속 보면서 어떻게 생겼는지 대충 파악을 해야 된다.\n# 또한, 어떤 augmentation을 할지도 생각.\n# 지금 이미지를 분석해보면 어디 쏠리지 않고 나름 중심을 두고 적혀있다. -> 사진의 quality가 괜찮다\n\nimg=train0.iloc[idx,1:].values.astype(np.uint8)\n\nplt.imshow(255 - img.reshape(137,236), cmap = 'gray') # 흑백이미지 반전 스케일로 보기 like MNIST","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Multi-label stratification folding","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('../input/bengaliai-cv19/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking the Distribution of label","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,50))\ndf_train['grapheme_root'].value_counts().sort_index().plot.barh()\n\n# grapheme_root라는 label의 class가 총 168개(0~167)가 있는데\n# 각 클래스가 어떤 count를 가지는지 보여주는 그래프","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Each class in grapheme_root is very unbalanced!\n- In this condition, if we randomly sample, the distribution of classes for each fold may not be properly entered.\n- If that happens, the model may not be able to train properly.\n#### So we have to do a stratified fold","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\ndf_train['vowel_diacritic'].value_counts().sort_index().plot.barh()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Same situation as grapheme_root","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,7))\ndf_train['consonant_diacritic'].value_counts().sort_index().plot.barh()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Same situation as grapheme_root","metadata":{}},{"cell_type":"markdown","source":"#### Now we have to use Stratified fold, but The stratified fold provided by sklearn is applied to only one label.\n#### Now that we are dealing with three labels, we have to use 'iterative-stratification', a library that helps us fold while keeping the distribution of all three labels the same.","metadata":{}},{"cell_type":"code","source":"!pip install iterative-stratification\n\nfrom iterstrat.ml_stratifiers import MultilabelStratifiedKFold","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['id'] = df_train['image_id'].apply(lambda x: int(x.split('_')[1]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_train[['id', 'grapheme_root', 'vowel_diacritic', 'consonant_diacritic']].values[:, 0] # id\ny = df_train[['id', 'grapheme_root', 'vowel_diacritic', 'consonant_diacritic']].values[:, 1:] # target","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mskf = MultilabelStratifiedKFold(n_splits=6, random_state=1944, shuffle=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['fold'] = -1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, (trn_idx, vid_idx) in enumerate(mskf.split(X,y)):\n    df_train.loc[vid_idx, 'fold'] = i  ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['fold'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.to_csv('df_folds.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Efficient learning process\n- dataframe을 row by row로 잘라서 학습할 때 조금 더 빨리 할 수 있게 하는 방법\n- 본 대회는 parquet파일 4개를 학습시켜서 합쳐야 하기 때문에 메모리가 크지 않으면 자칫 학습하다가 끊길수도 있다..\n- pandas가 읽고 쓰는 속도가 생각보다 느린 편.\n- 만약에 모델을 학습하는 코드를 만들었는데 pandas에서 불러오는 식으로 만들면 전체적인 학습 시간이 길어진다.\n- So How to fix it??","metadata":{}},{"cell_type":"code","source":"import joblib\n\nfrom tqdm import tqdm\n# tqdm : 반복문이 어디까지 진행되었는지 알고싶을때 사용","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_ids = train0['image_id'].values\nimg_array = train0.iloc[:, 1:].values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for idx in tqdm(range(len(train0))):\n    break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_id = img_ids[idx]\nimg = img_array[idx]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir('/kaggle/working/train_images/')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joblib.dump(img, f'./train_images/{img_id}.pkl')\n\n# pkl로 저장하면 아주 빨리 읽을 수 있다","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimg0 = joblib.load(f'./train_images/{img_id}.pkl')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain0.iloc[0, 1:].values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"25.2/1.8\n# 14배 정도 시간차이가 난다.!!\n# RAM이 작다면 pkl로 저장해서 load하는게 훨씬 효율적이다","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Full code\n\n# for fname in files_train: \n#     F = os.path.join(data_dir, fname)\n#     train_image = pd.read_parquet(F)\n#     img_ids = train_image['image_id'].values\n#     img_array = train_image.iloc[:, 1:].values\n#     for idx in tqdm(range(len(train_image))):\n#         img_id = img_ids[idx]\n#         img = img_array[idx]\n#         joblib.dump(img, f'./train_images/{img_id}.pkl')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pytorch dataset","metadata":{}},{"cell_type":"code","source":"import torch \n\nimport warnings \nwarnings.filterwarnings('ignore')\n\nfrom torch.utils.data import Dataset\nimport matplotlib.pyplot as plt\nimport joblib","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index=0\nHEIGHT=137\nWIDTH=236","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_ids = df_train['image_id'].values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_id = img_ids[index]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_id","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = joblib.load(f'./train_images/{img_id}.pkl').astype(np.uint8)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = img.reshape(HEIGHT, WIDTH) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img, cmap='gray')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = 255-img","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img, cmap='gray')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pytorch model에 넣어주려면 channel이 있어야 함\n\nprint(img[:, :, np.newaxis].shape)\nimg = img[:, :, np.newaxis]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_1 = df_train.iloc[index].grapheme_root\nlabel_2 = df_train.iloc[index].vowel_diacritic\nlabel_3 = df_train.iloc[index].consonant_diacritic","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 특정 index에 맞는 image pkl을 불러온다\n- 그 pkl을 reshape 한다\n- channel 추가(for pytorch)\n- 그 image가 가지고 있는 각 label별 클래스를 다 가져온다","metadata":{}},{"cell_type":"code","source":"class BengaliDataset(Dataset):\n    def __init__(self, csv, img_height, img_width):\n        self.csv = csv.reset_index()\n        self.img_ids = csv['image_id'].values\n        self.img_height = img_height\n        self.img_width = img_width\n    \n    def __len__(self):\n        return len(self.csv) # Dataset 길이 ; 지금은 csv파일이 dataset이므로 csv파일 길이를 return하면 됨\n    \n    def __getitem__(self, index):\n        img_id = self.img_ids[index]\n        img = joblib.load(f'./train_images/{img_id}.pkl')\n        img = img.reshape(self.img_height, self.img_width).astype(np.uint8)\n        img = 255-img\n        img = img[:,:,np.newaxis]\n        \n        label_1 = self.csv.iloc[index].grapheme_root\n        label_2 = self.csv.iloc[index].vowel_diacritic\n        label_3 = self.csv.iloc[index].consonant_diacritic\n        \n        return (torch.tensor(img, dtype=torch.float).permute(2,0,1), torch.tensor(label_1, dtype=torch.long), \n                torch.tensor(label_2, dtype=torch.long), torch.tensor(label_3, dtype=torch.long))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### For Pytorch model\n\n- (B, W, H, C) -> (B, C, W, H)\n- B : Batch, W : Width, H : Height, C : Channel\n- (16, 137, 236, 1) -> (16, 1, 137, 236)","metadata":{}},{"cell_type":"code","source":"print(img.shape)\nprint(torch.tensor(img).shape)\nprint(torch.tensor(img).permute(2,0,1).shape) # permute : 차원을 바꿔주기 위해 사용","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['fold'] = pd.read_csv('./df_folds.csv')['fold']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['fold']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_fold = [i for i in range(6) if i not in [5]]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_fold","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vid_fold = [5]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_idx = df_train.loc[df_train['fold'].isin(trn_fold)].index\nvid_idx = df_train.loc[df_train['fold'].isin(vid_fold)].index","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_dataset = BengaliDataset(csv=df_train.loc[trn_idx], img_height = HEIGHT, img_width = WIDTH)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_dataset[0][0].shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(trn_dataset[0][0].permute(1,2,0).numpy()[:,:,0], cmap='gray')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(trn_dataset[0][0].permute(1,2,0).numpy().shape) # channel을 없애주어야 사진을 볼 수 있다.\nprint(trn_dataset[0][0].permute(1,2,0).numpy()[:,:,0].shape) # '[:,:,0]'을 통해 channel 삭제","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# idx = 0\n# idx += 1\n# plt.imshow(trn_dataset[idx][0].permute(1,2,0).numpy()[:,:,0], cmap='gray')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}