{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Imports"},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt\nimport re\nimport numpy as np\nfrom skimage import io\n\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\n\nfrom tensorflow.keras.losses import categorical_crossentropy\nfrom tensorflow.keras.optimizers import Adam","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Directories"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"ROOT_DIR = '/kaggle/input'\n\nDATASET_DIR = os.path.join(ROOT_DIR, 'hpa-single-label-cell-level-dataset/single-label-cell-level')\nRGB_DATASET_DIR = os.path.join(DATASET_DIR, 'rgb')\nPROTEIN_DATASET_DIR = os.path.join(DATASET_DIR, 'protein')\n\nTRAIN_CSV_PATH = os.path.join(ROOT_DIR, \"hpa-single-cell-image-classification/train.csv\")\ntrain_df = pd.read_csv(TRAIN_CSV_PATH)\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def extract_image_level_id(file_name, df):\n    return int(df.loc[df['ID'] == file_name.split('_')[0]]['Label'].item())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"PROTEIN_IDS = os.listdir(PROTEIN_DATASET_DIR)[:1000]\nPROTEIN_LABELS = list(extract_image_level_id(protein_id, train_df) for protein_id in PROTEIN_IDS)\nPROTEIN_LABELS[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = cv2.imread(os.path.join(PROTEIN_DATASET_DIR, PROTEIN_IDS[2]), cv2.IMREAD_UNCHANGED)\nplt.imshow(img, cmap='gray')\nplt.show()\nimg.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class ImageGenerator():\n\n    def __init__(self, images_ids, labels, num_of_classes, batch_size):        \n        \n        self.images_ids = images_ids\n        self.labels = labels\n        self.num_of_classes = num_of_classes\n        self.batch_size = batch_size\n         \n    def _load_image(self, img_path):\n        \n        #load image from path and convert to array\n        img = io.imread(img_path)\n        img = np.array(img)\n        \n        return img.reshape((img.shape[0], img.shape[1], 1))\n    \n    def __len__(self):\n\n        #number of batches in an epoch\n        return int(np.ceil(len(self.images_ids) / float(self.batch_size)))\n\n    def __call__(self):\n        #shuffle index\n        idx = list(range(len(self.images_ids)))\n        np.random.shuffle(idx)\n\n        #generate batches\n        for batch in range(len(self)):\n            \n            batch_range = idx[batch*self.batch_size:(batch+1)*self.batch_size]\n            batch_image_ids = [self.images_ids[i] for i in batch_range]\n            batch_images = [self._load_image(os.path.join(PROTEIN_DATASET_DIR, img_id)) for img_id in batch_image_ids]\n            batch_labels = [self.labels[i] for i in batch_range]\n            \n#             print(batch_images[0].dtype)\n#             print(tf.one_hot(batch_labels, self.num_of_classes).dtype)\n            \n            yield batch_images, tf.one_hot(batch_labels, self.num_of_classes)\n            \n\n\n#print(np.asarray(PROTEIN_LABELS))\n            \n            \ntrain_generator = ImageGenerator(PROTEIN_IDS, PROTEIN_LABELS, 19, 16)\ntrain_generator()\n\ntrain_dataset = tf.data.Dataset.from_generator(train_generator, (tf.uint8, tf.float32))\n\n\nefficient_net = EfficientNetB0(\n    include_top=False,\n    weights=None,\n    input_shape=(128,128,1),\n    pooling='max'\n)\n\nmodel = Sequential()\nmodel.add(efficient_net)\nmodel.add(Dense(units = 120, activation='relu'))\nmodel.add(Dense(units = 120, activation = 'relu'))\nmodel.add(Dense(units = 19, activation='softmax'))\nmodel.summary()\n\nmodel.compile(optimizer=Adam(lr=0.0001), loss='categorical_crossentropy')\nhistory = model.fit(train_dataset, epochs=10)\n\n# for element in train_dataset:\n#   print(element[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = io.imread(os.path.join(PROTEIN_DATASET_DIR, PROTEIN_IDS[2]), cv2.IMREAD_UNCHANGED)\nimg = np.array(img)\nimg = img.reshape((1, img.shape[0], img.shape[1], 1))\nprint(img.shape)\n\npred = model.predict(img)\npred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"PROTEIN_LABELS[2]","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}