{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport tensorflow as tf\nprint(tf.__version__)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","id":"JHDaD6UBDQv_","outputId":"47dbace9-acb0-4645-9d91-eece7c9357b5","execution":{"iopub.status.busy":"2022-09-12T18:48:36.4611Z","iopub.execute_input":"2022-09-12T18:48:36.461487Z","iopub.status.idle":"2022-09-12T18:48:36.467053Z","shell.execute_reply.started":"2022-09-12T18:48:36.461453Z","shell.execute_reply":"2022-09-12T18:48:36.466038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport os\nimport keras\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nimport cv2\nfrom keras import applications\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense,PReLU ,Dropout, Input,Activation, Dropout, Flatten, Dense, Input, Conv2D, MaxPooling2D, BatchNormalization, Concatenate, ReLU, LeakyReLU, GlobalAveragePooling2D\nfrom keras.models import Model\n#from keras.optimizers import Adam\nimport os, sys, math\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nfrom tensorflow.keras import optimizers\nfrom tensorflow.keras import regularizers\noptimizers.RMSprop\noptimizers.Adam\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\nfrom tensorflow.keras.activations import selu\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport os\nimport keras\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nimport cv2\nfrom keras import applications\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, Input,Activation, Dropout, Flatten, Dense, Input, Conv2D, MaxPooling2D, BatchNormalization, Concatenate, ReLU, LeakyReLU, GlobalAveragePooling2D\nfrom keras.models import Model\n#from keras.optimizers import Adam\nimport os, sys, math\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nfrom tensorflow.keras import optimizers\nfrom tensorflow.keras import regularizers\noptimizers.RMSprop\noptimizers.Adam\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\nfrom tensorflow.keras.activations import selu\nfrom tensorflow import keras \nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport os\nfrom tensorflow import keras \nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\n#import cv2\nfrom keras import applications\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, Input,Activation, Dropout, Flatten, Dense, Input, Conv2D, MaxPooling2D, BatchNormalization, Concatenate, PReLU, LeakyReLU, GlobalAveragePooling2D\nfrom keras.models import Model\n#from keras.optimizers import Adam\nimport os, sys, math\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nfrom tensorflow.keras import optimizers\noptimizers.RMSprop\noptimizers.Adam\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau","metadata":{"id":"07E40Xs3DQwE","execution":{"iopub.status.busy":"2022-09-12T18:48:36.579608Z","iopub.execute_input":"2022-09-12T18:48:36.579944Z","iopub.status.idle":"2022-09-12T18:48:36.600334Z","shell.execute_reply.started":"2022-09-12T18:48:36.579916Z","shell.execute_reply":"2022-09-12T18:48:36.599392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data= pd.read_csv('../input/human-protein-atlas-image-classification/train.csv')\npath_to_train = '../input/human-protein-atlas-image-classification/train/'","metadata":{"id":"W4F_yNShDQwG","execution":{"iopub.status.busy":"2022-09-12T18:48:36.601959Z","iopub.execute_input":"2022-09-12T18:48:36.602906Z","iopub.status.idle":"2022-09-12T18:48:36.642763Z","shell.execute_reply.started":"2022-09-12T18:48:36.602868Z","shell.execute_reply":"2022-09-12T18:48:36.641975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe, _ = train_test_split(all_data, test_size = 0.5, random_state=42)","metadata":{"id":"efTyHQQlDQwH","execution":{"iopub.status.busy":"2022-09-12T18:48:36.765088Z","iopub.execute_input":"2022-09-12T18:48:36.765742Z","iopub.status.idle":"2022-09-12T18:48:36.776681Z","shell.execute_reply.started":"2022-09-12T18:48:36.765705Z","shell.execute_reply":"2022-09-12T18:48:36.775789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe.shape","metadata":{"execution":{"iopub.status.busy":"2022-09-12T18:48:36.921247Z","iopub.execute_input":"2022-09-12T18:48:36.921905Z","iopub.status.idle":"2022-09-12T18:48:36.928973Z","shell.execute_reply.started":"2022-09-12T18:48:36.921865Z","shell.execute_reply":"2022-09-12T18:48:36.928026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe, datatest = train_test_split(dataframe, test_size = 0.1, random_state=None)","metadata":{"id":"zbW69-1JE9ed","execution":{"iopub.status.busy":"2022-09-12T18:48:37.009808Z","iopub.execute_input":"2022-09-12T18:48:37.010485Z","iopub.status.idle":"2022-09-12T18:48:37.018858Z","shell.execute_reply.started":"2022-09-12T18:48:37.010453Z","shell.execute_reply":"2022-09-12T18:48:37.018074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"shtLDmA-DQwH"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe = dataframe.reset_index()","metadata":{"id":"bazsCRn1DQwH","execution":{"iopub.status.busy":"2022-09-12T18:48:37.161224Z","iopub.execute_input":"2022-09-12T18:48:37.161852Z","iopub.status.idle":"2022-09-12T18:48:37.167602Z","shell.execute_reply.started":"2022-09-12T18:48:37.161819Z","shell.execute_reply":"2022-09-12T18:48:37.166667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe['aug'] = 0\ndatatest['aug'] = 0","metadata":{"id":"OEQgm70HDQwI","execution":{"iopub.status.busy":"2022-09-12T18:48:37.271889Z","iopub.execute_input":"2022-09-12T18:48:37.272516Z","iopub.status.idle":"2022-09-12T18:48:37.278577Z","shell.execute_reply.started":"2022-09-12T18:48:37.272484Z","shell.execute_reply":"2022-09-12T18:48:37.277494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datatest = datatest[['Id','Target','aug']].reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-09-12T18:48:37.370421Z","iopub.execute_input":"2022-09-12T18:48:37.371092Z","iopub.status.idle":"2022-09-12T18:48:37.379165Z","shell.execute_reply.started":"2022-09-12T18:48:37.371045Z","shell.execute_reply":"2022-09-12T18:48:37.378354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe.head()","metadata":{"id":"Pe2RKsEiDQwI","outputId":"635b1ed9-6f6e-41a6-dd40-e79c212d184a","execution":{"iopub.status.busy":"2022-09-12T18:48:39.074477Z","iopub.execute_input":"2022-09-12T18:48:39.075064Z","iopub.status.idle":"2022-09-12T18:48:39.086427Z","shell.execute_reply.started":"2022-09-12T18:48:39.075028Z","shell.execute_reply":"2022-09-12T18:48:39.085509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import itertools\nlabels = [x.split() for x in dataframe.Target.tolist()]\nlabels = list(itertools.chain(*labels))\nflat_labels =[int(l) for l in labels]\n\nflat_labels =np.array(flat_labels)\n(label, counts) = np.unique(flat_labels, return_counts=True)\nlist(zip(label, counts))","metadata":{"id":"xcO_oqlKDQwJ","outputId":"f9c20c7f-e594-4fc8-b396-0b531a9479f5","execution":{"iopub.status.busy":"2022-09-12T18:48:39.301572Z","iopub.execute_input":"2022-09-12T18:48:39.302337Z","iopub.status.idle":"2022-09-12T18:48:39.328382Z","shell.execute_reply.started":"2022-09-12T18:48:39.302298Z","shell.execute_reply":"2022-09-12T18:48:39.327505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for elm in flat_labels:\n#    print ( elm )\nimport numpy as np\narr = np.array(label)\nelement = np.where(arr == 25)\nprint(element)\nprint(element[0][0])\n#print(label[element[0][0]])","metadata":{"id":"HHTJs_jfDQwK","outputId":"8b3b6e1c-0e40-4d1e-cf57-c6ce9fd2ec33","execution":{"iopub.status.busy":"2022-09-12T18:48:39.443515Z","iopub.execute_input":"2022-09-12T18:48:39.444227Z","iopub.status.idle":"2022-09-12T18:48:39.450188Z","shell.execute_reply.started":"2022-09-12T18:48:39.444195Z","shell.execute_reply":"2022-09-12T18:48:39.449057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[10,8])\nplt.bar(label, height=counts)\n\n\nplt.grid(axis='y', alpha=0.75)\nplt.xlabel('Label',fontsize=15)\nplt.ylabel('Frequency',fontsize=15)\nplt.xticks(fontsize=15)\nplt.yticks(label,fontsize=15) \nplt.ylabel('Frequency',fontsize=15)\nplt.title('Labels Distribution Histogram',fontsize=15)\nplt.show()","metadata":{"id":"ZKn2C9wQDQwK","outputId":"a41db2f0-4684-417c-fbce-5b873ee36ba1","execution":{"iopub.status.busy":"2022-09-12T18:48:39.540028Z","iopub.execute_input":"2022-09-12T18:48:39.540706Z","iopub.status.idle":"2022-09-12T18:48:39.845005Z","shell.execute_reply.started":"2022-09-12T18:48:39.540673Z","shell.execute_reply":"2022-09-12T18:48:39.843305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lb_to_balanced = [1,2, 3, 4,5,6, 7 , 8,  9,  10, 11 ,12, 13,14, 15, 16, 17,  18, 19 , 20, 21,22, 23,24,25,26,27]\n\nt = [counts[0] for _ in range(27)]\nw = list(counts)\nw.remove(counts[0])\n\nlb_to_added    = [x-y for x,y in zip(t,w) ]\n\n\n\n#lb_to_balanced = [1,2, 3, 4,5,6, 7  ,    8,    9,  10, 11 ,12,   13,14,  15,  16,  17,  18, 19 , 20, 21,22, 23,24\n              #    , 26, 27]\n#lb_to_added =    [5000,5000,5000,5000\n                 # ,5000,5000,5000, 5000\n                 # , 5000,5000 ,5000, 5000\n                 # ,6000,5000 ,5000,5000 , 5000\n                 # ,5000 ,5000,5000, 5000\n                 # , 5000, 5000, 5000,5000, 5000]\n\n\n\ndef generate_large_list(ls, size):\n    n_copies = size // len(ls)\n    excess = size % len(ls)\n\n    result = sorted([element \n                     for i in range(n_copies) \n                     for element in ls] + ls[:excess]) \n    print(len(result))\n    return result\n\ndf = dataframe.copy()\ndf['labels'] = [[ int(l) for l in x.split()] for x in df.Target.tolist()]\nminority_files = []\ntargets = []\nfor label in lb_to_balanced:\n    files = []\n    for i in range(len(df)):\n        if label in df.iloc[i,4]:\n            files.append( df.iloc[i,1])\n    size = lb_to_added[lb_to_balanced.index(label)]\n    minority_files.extend( generate_large_list(files, size) )\n    targets.extend([str(label) for _ in range(size)])\n#     targets.extend(np.array([np.array([label]) for _ in range(size)]))\n\n\n# targets = np.array(targets)\nminority_dataframe = pd.DataFrame({'Id': minority_files, 'Target': targets})\nminority_dataframe['aug'] = 1\n\ndataframe = dataframe[['Id', 'Target', 'aug']]\ndataframe = pd.concat([dataframe, minority_dataframe], ignore_index=True)\ndataframe.shape","metadata":{"id":"MVJoJ15mDQwL","outputId":"8280911d-7ab1-4f9d-b183-ca48a7e3e1ae","execution":{"iopub.status.busy":"2022-09-12T18:48:39.846656Z","iopub.execute_input":"2022-09-12T18:48:39.847585Z","iopub.status.idle":"2022-09-12T18:48:50.794075Z","shell.execute_reply.started":"2022-09-12T18:48:39.847532Z","shell.execute_reply":"2022-09-12T18:48:50.79306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"id":"yrTzLP_9DQwL","outputId":"70fafed1-11e4-4cb6-af94-498b44852b6c","execution":{"iopub.status.busy":"2022-09-12T18:48:50.795786Z","iopub.execute_input":"2022-09-12T18:48:50.796233Z","iopub.status.idle":"2022-09-12T18:48:50.811229Z","shell.execute_reply.started":"2022-09-12T18:48:50.796196Z","shell.execute_reply":"2022-09-12T18:48:50.810471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe.head()","metadata":{"id":"Dj_1EXs3DQwM","outputId":"1f54244b-0a86-486b-df57-105b390e83a2","execution":{"iopub.status.busy":"2022-09-12T18:48:50.812233Z","iopub.execute_input":"2022-09-12T18:48:50.812509Z","iopub.status.idle":"2022-09-12T18:48:50.848385Z","shell.execute_reply.started":"2022-09-12T18:48:50.812484Z","shell.execute_reply":"2022-09-12T18:48:50.847567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = [x.split() for x in dataframe.Target.tolist()]\nlabels = list(itertools.chain(*labels))\nflat_labels = [int(l) for l in labels]\n\nflat_labels = np.array(flat_labels)\n(label, counts) = np.unique(flat_labels, return_counts=True)\nlist(zip(label, counts))","metadata":{"id":"OUiKUjimDQwM","outputId":"5e9d68c3-5301-4e5a-e2d8-2e7e416bbe4e","execution":{"iopub.status.busy":"2022-09-12T18:48:50.857958Z","iopub.execute_input":"2022-09-12T18:48:50.861513Z","iopub.status.idle":"2022-09-12T18:48:51.303529Z","shell.execute_reply.started":"2022-09-12T18:48:50.861469Z","shell.execute_reply":"2022-09-12T18:48:51.302713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=[10,8])\nplt.bar(label, height=counts)\n\n\nplt.grid(axis='y', alpha=0.75)\nplt.xlabel('Label',fontsize=15)\nplt.ylabel('Frequency',fontsize=15)\nplt.xticks(fontsize=15)\nplt.yticks(label,fontsize=15) \nplt.ylabel('Frequency',fontsize=15)\nplt.title('Labels Distribution Histogram',fontsize=15)\nplt.show()","metadata":{"id":"MlziXAUuDQwN","outputId":"49a6ad72-b372-4eec-83fd-77856874d4a6","execution":{"iopub.status.busy":"2022-09-12T18:48:51.307827Z","iopub.execute_input":"2022-09-12T18:48:51.309914Z","iopub.status.idle":"2022-09-12T18:48:51.675072Z","shell.execute_reply.started":"2022-09-12T18:48:51.309876Z","shell.execute_reply":"2022-09-12T18:48:51.674173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset_info = []\nfor name, aug, labels in zip(dataframe['Id'], dataframe['aug'],\n                             dataframe['Target'].str.split(' ')):\n    train_dataset_info.append({\n        'path':os.path.join(path_to_train, name),\n        'labels':np.array([int(label) for label in labels]),\n        'aug': aug})\ntrain_dataset_info = np.array(train_dataset_info)","metadata":{"id":"rvgci20eDQwN","execution":{"iopub.status.busy":"2022-09-12T18:48:51.679011Z","iopub.execute_input":"2022-09-12T18:48:51.681084Z","iopub.status.idle":"2022-09-12T18:48:53.309704Z","shell.execute_reply.started":"2022-09-12T18:48:51.681043Z","shell.execute_reply":"2022-09-12T18:48:53.308785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset_info = []\nfor name, aug, labels in zip(datatest['Id'], datatest['aug'],\n                             datatest['Target'].str.split(' ')):\n    test_dataset_info.append({\n        'path':os.path.join(path_to_train, name),\n        'labels':np.array([int(label) for label in labels]),\n        'aug': aug})\ntest_dataset_info = np.array(test_dataset_info)","metadata":{"id":"WRrzwL77GA57","execution":{"iopub.status.busy":"2022-09-12T18:48:53.314242Z","iopub.execute_input":"2022-09-12T18:48:53.316399Z","iopub.status.idle":"2022-09-12T18:48:53.339154Z","shell.execute_reply.started":"2022-09-12T18:48:53.31636Z","shell.execute_reply":"2022-09-12T18:48:53.338371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_SHAPE = (299,299,4)\n#BATCH_SIZE = 10\n","metadata":{"id":"3Vv5F_FSDQwN","execution":{"iopub.status.busy":"2022-09-12T18:48:53.342836Z","iopub.execute_input":"2022-09-12T18:48:53.34487Z","iopub.status.idle":"2022-09-12T18:48:53.350332Z","shell.execute_reply.started":"2022-09-12T18:48:53.344833Z","shell.execute_reply":"2022-09-12T18:48:53.349673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_ids, valid_ids, train_targets, valid_target = train_test_split(\n    dataframe['Id'], dataframe['Target'], test_size=0.1, random_state=None)\n\ntest_ids, test_target = datatest['Id'], datatest['Target']","metadata":{"id":"8xbqpcX7DQwO","execution":{"iopub.status.busy":"2022-09-12T18:48:53.354329Z","iopub.execute_input":"2022-09-12T18:48:53.356564Z","iopub.status.idle":"2022-09-12T18:48:53.408312Z","shell.execute_reply.started":"2022-09-12T18:48:53.356509Z","shell.execute_reply":"2022-09-12T18:48:53.407395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_ids.shape, valid_ids.shape, test_ids.shape)","metadata":{"id":"VtmbSCkqDQwO","outputId":"03a33ca7-b402-4a1b-f794-ef4597ce0b6c","execution":{"iopub.status.busy":"2022-09-12T18:48:53.414297Z","iopub.execute_input":"2022-09-12T18:48:53.416453Z","iopub.status.idle":"2022-09-12T18:48:53.42489Z","shell.execute_reply.started":"2022-09-12T18:48:53.416415Z","shell.execute_reply":"2022-09-12T18:48:53.424114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class data_generator:\n    def create_train(dataset_info, batch_size, shape, augument=True):\n        assert shape[2] == 4\n        while True:\n            random_indexes = np.random.choice(len(dataset_info), batch_size)\n            batch_images = np.empty((batch_size, shape[0], shape[1], shape[2]))\n            batch_labels = np.zeros((batch_size, 28))\n            for i, idx in enumerate(random_indexes):\n                image = data_generator.load_image(\n                    dataset_info[idx]['path'], shape)   \n                if dataset_info[idx]['aug']==1:\n                    image = data_generator.augment(image)\n                batch_images[i] = image\n                batch_labels[i][dataset_info[idx]['labels']] = 1\n            yield batch_images, batch_labels\n    def load_image(path, shape):\n        R = np.array(Image.open(path+'_red.png'))\n        G = np.array(Image.open(path+'_green.png'))\n        B = np.array(Image.open(path+'_blue.png'))\n        Y = np.array(Image.open(path+'_yellow.png'))\n        #image = np.stack((\n            #R/2 + Y/2, \n            #G/2 + Y/2, \n            #B),-1)\n        image = np.stack((R,G,B,Y),-1)\n        image = cv2.resize(image, (shape[0], shape[1]))\n        image = np.divide(image, 255)\n        return image  \n                \n            \n    def augment(image):\n        augment_img = iaa.Sequential([\n            iaa.OneOf([\n                iaa.Affine(rotate=0),\n                iaa.Affine(rotate=90),\n                iaa.Affine(rotate=180),\n                iaa.Affine(rotate=270),\n                iaa.Fliplr(0.5),\n                iaa.Flipud(0.5),\n            ])], random_order=True)\n        \n        image_aug = augment_img.augment_image(image)\n        return image_aug\n    \n   ","metadata":{"id":"HQn_11_sDQwO","execution":{"iopub.status.busy":"2022-09-12T18:48:53.429016Z","iopub.execute_input":"2022-09-12T18:48:53.431399Z","iopub.status.idle":"2022-09-12T18:48:53.447482Z","shell.execute_reply.started":"2022-09-12T18:48:53.431364Z","shell.execute_reply":"2022-09-12T18:48:53.446764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras import backend as K\ndef f1_m(y_true, y_pred):\n    precision = precision_m(y_true, y_pred)\n    recall = recall_m(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))\ndef recall_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n    recall = true_positives / (possible_positives + K.epsilon())\n    return recall\ndef precision_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    precision = true_positives / (predicted_positives + K.epsilon())\n    return precision","metadata":{"id":"AMCQzK29DQwP","execution":{"iopub.status.busy":"2022-09-12T18:48:53.451526Z","iopub.execute_input":"2022-09-12T18:48:53.453977Z","iopub.status.idle":"2022-09-12T18:48:53.46479Z","shell.execute_reply.started":"2022-09-12T18:48:53.453941Z","shell.execute_reply":"2022-09-12T18:48:53.46385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model(input_shape):\n    \n    dropRate = 0.25\n\n    init = Input(input_shape)\n    #x = BatchNormalization(axis=-1)(init)\n    x = Conv2D(16,3,2,padding='same',activation=\"PReLU\")(init)\n    x = MaxPooling2D(2,2,padding='valid')(x)\n    x = Dropout(dropRate)(x)\n    x = Conv2D(32, 3,2,padding='same',activation=\"PReLU\")(x)\n    x = MaxPooling2D(2,2,padding='valid')(x)\n    x = Dropout(dropRate)(x)\n    x = Conv2D(64, 3,2,padding='same',activation=\"PReLU\")(x)\n    x = MaxPooling2D(2,2,padding='valid')(x)\n    x = Flatten()(x)\n    x = Dropout(dropRate)(x)\n\n    x=Dense(128, activation='PReLU')(x)\n    x=BatchNormalization()(x)\n    x=Dense(256, activation='PReLU')(x)\n    x=BatchNormalization()(x)     \n    x=Dense(28, activation='sigmoid')(x)\n    \n    model = Model(init, x)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-09-12T18:50:45.042777Z","iopub.execute_input":"2022-09-12T18:50:45.043264Z","iopub.status.idle":"2022-09-12T18:50:45.057679Z","shell.execute_reply.started":"2022-09-12T18:50:45.043213Z","shell.execute_reply":"2022-09-12T18:50:45.056344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SHAPE=(299,299,4)\nmodel = create_model(SHAPE)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-09-12T18:50:47.854743Z","iopub.execute_input":"2022-09-12T18:50:47.855129Z","iopub.status.idle":"2022-09-12T18:50:47.964584Z","shell.execute_reply.started":"2022-09-12T18:50:47.855097Z","shell.execute_reply":"2022-09-12T18:50:47.96309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = data_generator.create_train(\n    train_dataset_info[train_ids.index], 128, INPUT_SHAPE)\nvalidation_generator = data_generator.create_train(\n    train_dataset_info[valid_ids.index], 32, INPUT_SHAPE)\ntest_generator = data_generator.create_train(\n    test_dataset_info[test_ids.index], 32, INPUT_SHAPE)","metadata":{"id":"3MLyO0EMDQwR","execution":{"iopub.status.busy":"2022-09-12T18:50:57.99092Z","iopub.execute_input":"2022-09-12T18:50:57.991692Z","iopub.status.idle":"2022-09-12T18:50:58.005342Z","shell.execute_reply.started":"2022-09-12T18:50:57.991654Z","shell.execute_reply":"2022-09-12T18:50:58.004406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN = len(train_ids) // 128\nSTEP_SIZE_VALID = len(valid_ids) // 32\nSTEP_SIZE_TEST = len(test_ids) // 32","metadata":{"id":"zhq30hwfDQwR","execution":{"iopub.status.busy":"2022-09-12T18:51:02.083699Z","iopub.execute_input":"2022-09-12T18:51:02.084305Z","iopub.status.idle":"2022-09-12T18:51:02.089044Z","shell.execute_reply.started":"2022-09-12T18:51:02.084269Z","shell.execute_reply":"2022-09-12T18:51:02.088033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN, STEP_SIZE_VALID, STEP_SIZE_TEST","metadata":{"id":"KPxLK4eKDQwR","outputId":"127b5caa-78b8-49ab-9381-3dc015908b55","execution":{"iopub.status.busy":"2022-09-12T18:48:53.837655Z","iopub.execute_input":"2022-09-12T18:48:53.840035Z","iopub.status.idle":"2022-09-12T18:48:53.850097Z","shell.execute_reply.started":"2022-09-12T18:48:53.839997Z","shell.execute_reply":"2022-09-12T18:48:53.848992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"VRLHH5sgDQwR"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n#first one\nreduce_lr = keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5,\n            patience=5, min_lr=0.001, mode='min')\n\n\n# reduce_lr = ReduceLROnPlateau(monitor='acc_loss', factor=0.5,\n#                              patience=5, min_lr=0.001, mode='max')\n\n\n\nearlystopper = tf.keras.callbacks.EarlyStopping(verbose=0,\n   monitor='val_loss', mode='min')\n\nopt = tf.keras.optimizers.SGD(learning_rate=0.1,momentum=0.9,clipnorm=5.0)\n\n#no\n# patience=10, verbose=0,\n\n\n\nmodel.compile(\n   loss='binary_crossentropy', \n   optimizer=opt,\n    metrics=[tf.keras.metrics.AUC(multi_label=True),'accuracy',f1_m,recall_m,precision_m])\n\nhist = model.fit(train_generator, \n          epochs=20,\n          steps_per_epoch=STEP_SIZE_TRAIN,\n          validation_data=next(validation_generator),\n        \n          callbacks=[reduce_lr]\n               )","metadata":{"id":"FuXVfFdUDQwS","outputId":"0e5abc93-7b8a-40c2-e4f0-2597c7b2df91","execution":{"iopub.status.busy":"2022-09-12T20:50:26.477486Z","iopub.execute_input":"2022-09-12T20:50:26.478266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#steps_by_epochs = len(test_ids) //80","metadata":{"id":"GnHnn8yKDQwS","execution":{"iopub.status.busy":"2022-09-12T18:49:04.367111Z","iopub.status.idle":"2022-09-12T18:49:04.368929Z","shell.execute_reply.started":"2022-09-12T18:49:04.36868Z","shell.execute_reply":"2022-09-12T18:49:04.368705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.evaluate(test_generator, batch_size=32, steps=STEP_SIZE_TEST)","metadata":{"id":"DtC5qf0mDQwS","outputId":"994a9122-b4fe-4df7-e033-2272a78fe029","execution":{"iopub.status.busy":"2022-09-12T18:49:04.370237Z","iopub.status.idle":"2022-09-12T18:49:04.370955Z","shell.execute_reply.started":"2022-09-12T18:49:04.370707Z","shell.execute_reply":"2022-09-12T18:49:04.370732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('gapnet1prelumulticopiesActivationadd.h5')","metadata":{"id":"Hc76KiZqDQwT","execution":{"iopub.status.busy":"2022-09-12T18:49:04.372241Z","iopub.status.idle":"2022-09-12T18:49:04.372954Z","shell.execute_reply.started":"2022-09-12T18:49:04.372715Z","shell.execute_reply":"2022-09-12T18:49:04.372739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nhist_df=pd.DataFrame(hist.history)\nhist_csv_file='historygapnet1multipreluuActivationadd.csv'\nwith open(hist_csv_file, mode='w')as f:\n    hist_df.to_csv(f)","metadata":{"id":"RflkkcgEDQwT","execution":{"iopub.status.busy":"2022-09-12T18:49:04.374199Z","iopub.status.idle":"2022-09-12T18:49:04.374903Z","shell.execute_reply.started":"2022-09-12T18:49:04.374662Z","shell.execute_reply":"2022-09-12T18:49:04.374686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import HTML\ndef create_download_link(title = \"Download CSV file\", filename = \"historygapnet1seluActivationadd.csv\"):  \n    html = '<a href={filename}>{title}</a>'\n    html = html.format(title=title,filename=filename)\n    return HTML(html)\n\n# create a link to download the dataframe which was saved with .to_csv method\ncreate_download_link(filename='historygapnet1multipreluuActivationadd.csv')","metadata":{"id":"GvqskKdvDQwT","outputId":"7a34c677-2afc-4555-f90d-6fd26a043a79","execution":{"iopub.status.busy":"2022-09-12T18:49:04.37615Z","iopub.status.idle":"2022-09-12T18:49:04.376849Z","shell.execute_reply.started":"2022-09-12T18:49:04.376621Z","shell.execute_reply":"2022-09-12T18:49:04.376644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"ssMbL5GXDQwT"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get results\nacc = hist.history['accuracy']\nval_acc = hist.history['val_accuracy']\nloss = hist.history['loss']\nval_loss = hist.history['val_loss']\n\n#plot results\n#accuracy\nplt.figure(figsize=(8, 12))\nplt.rcParams['figure.figsize'] = [16, 16]\nplt.rcParams['font.size'] = 14\nplt.rcParams['axes.grid'] = True\nplt.rcParams['figure.facecolor'] = 'white'\nplt.subplot(2, 1, 1)\nplt.plot(acc, label='Training Accuracy')\nplt.plot(val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.ylabel('Accuracy')\nplt.title(f'Gapnet1 \\nTraining and Validation Accuracy. \\nTrain Accuracy: {str(acc[-1])}\\nValidation Accuracy: {str(val_acc[-1])}')\n\n#loss\nplt.subplot(2, 1, 2)\nplt.plot(loss, label='Training Loss')\nplt.plot(val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.ylabel('Cross Entropy')\nplt.title(f'Training and Validation Loss. \\nTrain Loss using noisey labels: {str(loss[-1])}\\nValidation Loss: {str(val_loss[-1])}')\nplt.xlabel('epoch')\nplt.tight_layout(pad=3.0)\nplt.show()","metadata":{"id":"fm16hUMQDQwT","outputId":"090788a1-fd4a-4ab0-d5cb-0a8d6b5936d2","execution":{"iopub.status.busy":"2022-09-12T18:49:04.37809Z","iopub.status.idle":"2022-09-12T18:49:04.37879Z","shell.execute_reply.started":"2022-09-12T18:49:04.378538Z","shell.execute_reply":"2022-09-12T18:49:04.37858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"khC_2VtvDQwU"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"4KqaNHwjDQwU"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"kYxASr9dDQwU"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"_93qUGiLDQwU"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"yOF9ZdcRDQwU"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"wn9peor5DQwU"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"jOtOQWL6DQwU"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"fPv_WHsbDQwU"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"YBDEwxw1DQwV"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"h1utcUv8DQwV"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"aTZz-svtDQwV"},"execution_count":null,"outputs":[]}]}