{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-29T16:29:19.801361Z","iopub.execute_input":"2021-07-29T16:29:19.801744Z","iopub.status.idle":"2021-07-29T16:29:19.805819Z","shell.execute_reply.started":"2021-07-29T16:29:19.801711Z","shell.execute_reply":"2021-07-29T16:29:19.804762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten\nfrom keras.layers import Conv2D, MaxPooling2D,MaxPool3D\nfrom keras.preprocessing import image\nfrom keras.utils import to_categorical\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nfrom keras.layers import BatchNormalization\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:39:20.516451Z","iopub.execute_input":"2021-07-29T16:39:20.516871Z","iopub.status.idle":"2021-07-29T16:39:20.522754Z","shell.execute_reply.started":"2021-07-29T16:39:20.516835Z","shell.execute_reply":"2021-07-29T16:39:20.521713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"work_dir = \"/kaggle/input/human-protein-atlas-image-classification/\"\n! ls $work_dir","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:29:19.827921Z","iopub.execute_input":"2021-07-29T16:29:19.82826Z","iopub.status.idle":"2021-07-29T16:29:20.509744Z","shell.execute_reply.started":"2021-07-29T16:29:19.82823Z","shell.execute_reply":"2021-07-29T16:29:20.508743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(work_dir +\"train.csv\")\nprint(train_labels.head(10))\nprint(train_labels.columns)\ntrain_labels = train_labels.loc[:100]\n","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:29:20.511338Z","iopub.execute_input":"2021-07-29T16:29:20.511621Z","iopub.status.idle":"2021-07-29T16:29:20.556542Z","shell.execute_reply.started":"2021-07-29T16:29:20.511588Z","shell.execute_reply":"2021-07-29T16:29:20.555688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_img(train_labels , i , color) :\n    img = image.load_img(work_dir+\"train/\" +train_labels['Id'][i]+'_'+color +'.png' , target_size = (512,512))\n    img = image.img_to_array(img)\n    img = img/255\n    img = img.sum(axis=-1)\n    return img\n","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:29:20.559123Z","iopub.execute_input":"2021-07-29T16:29:20.559473Z","iopub.status.idle":"2021-07-29T16:29:20.566095Z","shell.execute_reply.started":"2021-07-29T16:29:20.559443Z","shell.execute_reply":"2021-07-29T16:29:20.564881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n    \n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_dataset = []  \nfor i in tqdm(range(train_labels.shape[0])):\n    img_b = get_img(train_labels , i , 'blue')\n    img_g = get_img(train_labels , i , 'green')\n    img_r = get_img(train_labels , i , 'red')\n    img_y = get_img(train_labels , i , 'yellow')\n    img_gb = np.dstack((img_g , img_b))\n    img_gr = np.dstack((img_g , img_r))\n    img_gy = np.dstack((img_g , img_y))\n    img = [img_gr , img_gy , img_gb]\n    X_dataset.append(img)\n    \nX = np.array(X_dataset)\nX.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:35:00.951396Z","iopub.execute_input":"2021-07-29T16:35:00.951765Z","iopub.status.idle":"2021-07-29T16:35:05.590058Z","shell.execute_reply.started":"2021-07-29T16:35:00.951732Z","shell.execute_reply":"2021-07-29T16:35:05.588768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:29:24.95662Z","iopub.execute_input":"2021-07-29T16:29:24.957013Z","iopub.status.idle":"2021-07-29T16:29:24.965115Z","shell.execute_reply.started":"2021-07-29T16:29:24.95697Z","shell.execute_reply":"2021-07-29T16:29:24.964032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_size = 28\n\n\ndef one_hot_encode() :\n    coded_labels =[]\n    raw = list(train_labels['Target'])\n    raw  =[(item.split()) for item in raw]\n    raw =[[int(i) for i in item ] for item in raw]\n    for index,  item in enumerate(raw) : \n        coded_row = np.zeros((cat_size,), dtype=int)\n        for i in item :\n            coded_row[i] =1\n        coded_labels.append( coded_row)\n        \n    return np.array(coded_labels , int)\n            \n            \n    \nY = one_hot_encode()\n\nprint (X.shape) \nprint (Y.shape)","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:29:24.966701Z","iopub.execute_input":"2021-07-29T16:29:24.967121Z","iopub.status.idle":"2021-07-29T16:29:24.976872Z","shell.execute_reply.started":"2021-07-29T16:29:24.967083Z","shell.execute_reply":"2021-07-29T16:29:24.975955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, Y, random_state=20, test_size=0.3)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:33:41.492145Z","iopub.execute_input":"2021-07-29T16:33:41.492487Z","iopub.status.idle":"2021-07-29T16:33:41.675739Z","shell.execute_reply.started":"2021-07-29T16:33:41.492458Z","shell.execute_reply":"2021-07-29T16:33:41.674832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (X_train.shape)\nprint (y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:33:46.233765Z","iopub.execute_input":"2021-07-29T16:33:46.234138Z","iopub.status.idle":"2021-07-29T16:33:46.239814Z","shell.execute_reply.started":"2021-07-29T16:33:46.234104Z","shell.execute_reply":"2021-07-29T16:33:46.238529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nmodel = Sequential()\n\nmodel.add(Conv2D(filters=16, kernel_size=(5, 5), activation=\"relu\", input_shape=(3,512,512,2)))\nmodel.add(BatchNormalization())\nmodel.add(MaxPool3D(pool_size=(2, 2, 2))\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(filters=16, kernel_size=(5, 5), activation='relu'))\nmodel.add(MaxPool3D(pool_size=(2, 2, 2))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\n\n \nmodel.add(Conv2D(filters=8, kernel_size=(5, 5), activation='relu'))\nmodel.add(MaxPool3D(pool_size=(2, 2, 2))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(filters=8, kernel_size=(5, 5), activation='relu'))\nmodel.add(MaxPool3D(pool_size=(2, 2, 2))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(filters=8, kernel_size=(5, 5), activation='relu'))\nmodel.add(MaxPool3D(pool_size=(2, 2, 2))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\n\n\n\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense( cat_size , activation='sigmoid'))\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:33:49.356369Z","iopub.execute_input":"2021-07-29T16:33:49.356731Z","iopub.status.idle":"2021-07-29T16:33:49.539732Z","shell.execute_reply.started":"2021-07-29T16:33:49.356698Z","shell.execute_reply":"2021-07-29T16:33:49.53869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()\n","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:29:25.668011Z","iopub.execute_input":"2021-07-29T16:29:25.668354Z","iopub.status.idle":"2021-07-29T16:29:25.686095Z","shell.execute_reply.started":"2021-07-29T16:29:25.668316Z","shell.execute_reply":"2021-07-29T16:29:25.685341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:34:00.48454Z","iopub.execute_input":"2021-07-29T16:34:00.484908Z","iopub.status.idle":"2021-07-29T16:34:00.499393Z","shell.execute_reply.started":"2021-07-29T16:34:00.484876Z","shell.execute_reply":"2021-07-29T16:34:00.498146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=15, validation_data=(X_test, y_test), batch_size=64)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:34:02.986075Z","iopub.execute_input":"2021-07-29T16:34:02.986461Z","iopub.status.idle":"2021-07-29T16:34:22.820105Z","shell.execute_reply.started":"2021-07-29T16:34:02.986426Z","shell.execute_reply":"2021-07-29T16:34:22.819145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot the training and validation accuracy and loss at each epoch\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs = range(1, len(loss) + 1)\nplt.plot(epochs, loss, 'y', label='Training loss')\nplt.plot(epochs, val_loss, 'r', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\nprint(history.history)\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nplt.plot(epochs, acc, 'y', label='Training acc')\nplt.plot(epochs, val_acc, 'r', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:29:26.386507Z","iopub.status.idle":"2021-07-29T16:29:26.387452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n_, acc = model.evaluate(X_test, y_test)\nprint(\"Accuracy = \", (acc * 100.0), \"%\")\n","metadata":{"execution":{"iopub.status.busy":"2021-07-29T16:29:26.388759Z","iopub.status.idle":"2021-07-29T16:29:26.38972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}