{"cells":[{"metadata":{},"cell_type":"markdown","source":"Image channels\n\nAll images have 4 channels: Red (Microtubules), Green (Protein of interest), Blue (Nucleus), Yellow (Endoplasmic reticulum) as followed:\n\n![](https://images.proteinatlas.org/115/663_E2_1_red_medium.jpg) ![](https://images.proteinatlas.org/115/663_E2_1_green_medium.jpg) ![](https://images.proteinatlas.org/115/663_E2_1_blue_medium.jpg) ![](https://images.proteinatlas.org/115/663_E2_1_yellow_medium.jpg)"},{"metadata":{"trusted":true},"cell_type":"code","source":"specified_class_names = \"\"\"0. Nucleoplasm\n1. Nuclear membrane\n2. Nucleoli\n3. Nucleoli fibrillar center\n4. Nuclear speckles\n5. Nuclear bodies\n6. Endoplasmic reticulum\n7. Golgi apparatus\n8. Intermediate filaments\n9. Actin filaments \n10. Microtubules\n11. Mitotic spindle\n12. Centrosome\n13. Plasma membrane\n14. Mitochondria\n15. Aggresome\n16. Cytosol\n17. Vesicles and punctate cytosolic patterns\n18. Negative\"\"\"\n\nclass_names = [class_name.split('. ')[1] for class_name in specified_class_names.split('\\n')]\nclass_names","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n#import math, re, os\nimport tensorflow as tf\nimport pandas as pd\nimport matplotlib.pyplot as plt\n#from kaggle_datasets import KaggleDatasets\nfrom tensorflow import keras\n#from functools import partial\nfrom sklearn.model_selection import train_test_split\nimport seaborn as sn\nprint(\"Tensorflow version \" + tf.__version__)\nimport cv2\nfrom PIL import Image\n#from tensorflow_examples.models.pix2pix import pix2pix\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAINING_FILENAMES, VALID_FILENAMES = train_test_split(\n    tf.io.gfile.glob('../input/hpa-write-mask/*.npz'),\n    test_size=0.35, random_state=28 ## was 0.35\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!pip install  git+'https://github.com/HasnainRaz/SemSegPipeline.git'\n\n!pip install -q git+'https://github.com/tensorflow/examples.git'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow_examples.models.pix2pix import pix2pix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#import hpacellseg.cellsegmentator as cellsegmentator\n#from hpacellseg.utils import label_cell,label_nuclei","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Set up Variables"},{"metadata":{"trusted":true},"cell_type":"code","source":"#AUTOTUNE = tf.data.experimental.AUTOTUNE\n#GCS_PATH = KaggleDatasets().get_gcs_path()\n#BATCH_SIZE = 1# 8 * strategy.num_replicas_in_sync\n#IMAGE_SIZE = [1024, 1024] # was 512\n#CLASSES = ['0', '1', '2', '3', '4']\n#EPOCHS = 25","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"filename = '../input/hpa-single-cell-image-classification/train.csv'\n\ndf_train = pd.read_csv(filename)\n\n#df_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#df_train[~df_train[\"Label\"].str.contains('\\|', na = False)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df = df_train[~df_train[\"Label\"].str.contains('\\|', na = False)].head(500)\nnew_df['images'] = '../input/write-data-one-label/' + new_df['ID'] + '_blend.png'\nnew_df['masks'] =  '../input/hpa-write-mask/' + new_df['ID'] + '.npz'\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.set_option('display.max_colwidth', None)\n#new_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#filename = ['../input/hpa-write-mask/5e22a522-bb99-11e8-b2b9-ac1f6b6435d0.npz',\n#           '../input/hpa-write-mask/5c801c04-bb99-11e8-b2b9-ac1f6b6435d0.npz']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"IMG_SIZE, BATCH_SIZE = [512,512], 4\nnum_classes = 20","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_data_from_filename(filename):\n    npdata = np.load(filename)\n    label = npdata['label']\n    label[label >= 0.5] = 1\n    label[label < 0.5] = 0\n    label = cv2.resize(label, dsize=(512,512), interpolation=cv2.INTER_AREA)\n    image = npdata['image']\n    image = cv2.resize(image, dsize=(512,512), interpolation=cv2.INTER_AREA)\n    return tf.convert_to_tensor(image), tf.convert_to_tensor(label)\n    #return \n\n#def get_data_wrapper(filename):\n#   # Assuming here that both your data and label is float type.\n#    features,labels = tf.compat.v1.py_func(\n#       get_data_from_filename, [filename], (tf.float64,tf.float64)) \n#    return tf.data.Dataset.from_tensor_slices((features,labels))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def data_setshape(image, label):\n    image.set_shape([*IMG_SIZE, 3])\n    label.set_shape([*IMG_SIZE, 20])\n    return image,label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#a = get_data_from_filename('../input/hpa-write-mask/5e22a522-bb99-11e8-b2b9-ac1f6b6435d0.npz')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tf.config.run_functions_eagerly(True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_training_dataset():\n    dataset_train = tf.data.Dataset.list_files(TRAINING_FILENAMES)\n    #dataset_train = dataset_train.shuffle(100)\n    dataset_train = dataset_train.repeat()\n    dataset_train = dataset_train.map(lambda name: tf.compat.v1.py_func(get_data_from_filename, [name],\n                                                        (tf.float32,\n                                                         tf.float64)))\n    dataset_train = dataset_train.map(data_setshape)\n    #\n    \n    dataset_train = dataset_train.batch(BATCH_SIZE)\n    dataset_train = dataset_train.prefetch(AUTOTUNE)\n    return dataset_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_validation_dataset():\n    dataset_val = tf.data.Dataset.list_files(VALID_FILENAMES)\n    dataset_val = dataset_val.map(lambda name: tf.compat.v1.py_func(get_data_from_filename, [name],\n                                                        (tf.float32,\n                                                         tf.float64)))\n    dataset_val = dataset_val.map(data_setshape)\n    dataset_val = dataset_val.batch(BATCH_SIZE)\n    dataset_val = dataset_val.cache()\n    dataset_val = dataset_val.prefetch(AUTOTUNE)\n    \n    return dataset_val","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traning_data = get_training_dataset()\nvalidation_data = get_validation_dataset()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in traning_data.take(1):\n    print(i[1][0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df1 = df_train[~df_train[\"Label\"].str.contains('\\|', na = False)].head(100)\nnew_df1['images'] = '../input/write-data-one-label/' + new_df1['ID'] + '_blend.png'\nnew_df1['masks'] =  '../input/hpa-write-mask/' + new_df1['ID'] + '.npz'\nnew_df1['masks']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df['Label'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#datagen_im = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1/255)\n#datagen_mas = tf.keras.preprocessing.image.ImageDataGenerator()\n\n#new_df\n# seed, so that image_generator and mask_generator will rotate and shuffle equivalently\n#seed = 42\n#image_generator = datagen_im.flow_from_dataframe(new_df, \n#                                      directory='.', \n#                                      x_col='images', \n                                      #y_col='masks', \n #                                     batch_size=BATCH_SIZE, \n #                                     target_size = (IMG_SIZE, IMG_SIZE),\n #                                     class_mode=None,                                       \n #                                     seed=seed)\n#mask_generator = datagen_mas.flow_from_dataframe(new_df, \n#                                     directory='.', \n#                                     x_col='masks', \n                                     #y_col='images', # Or whatever \n#                                     batch_size=BATCH_SIZE,\n#                                     target_size = (IMG_SIZE, IMG_SIZE),\n#                                     class_mode=None, \n#                                     seed=seed)\n\n#train_generator = zip(image_generator, ds_mask)\n\n# same for validation data generator\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"boolean_array = np.logical_and(i[1][0] > 0, i[1][0] < 1)\nin_range_indices = np.where(boolean_array)\n#boolean_array\nin_range_indices\n#print(in_range_indices)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(30,30),constrained_layout = True)\nax = fig.add_subplot(1, 2, 1,)\nplt.imshow(i[0][0])\nax = fig.add_subplot(1, 2, 2,)\nplt.imshow(i[1][0][:,:,0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"my_callbacks = [\n    tf.keras.callbacks.EarlyStopping(patience=3,verbose=0,monitor='val_categorical_accuracy'),\n    tf.keras.callbacks.ModelCheckpoint(filepath='HPA_model_eff3.h5',verbose=0,monitor='val_loss',save_best_only=True),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor='val_loss',\n        factor=0.2,\n        patience=2,\n        min_lr=1e-7,\n        mode='min',\n        verbose=0,\n    )\n    \n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"loss_func = tf.keras.losses.CategoricalCrossentropy(\n    from_logits=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base_model = tf.keras.applications.MobileNetV2(input_shape=[*IMG_SIZE, 3],weights='imagenet', include_top=False) #imagenet(input_shape=[1024, 1024, 3], include_top=False)\n\n# Use the activations of these layers\n\n#layer_names [\n#    'block6f_expand_activation',\n#    'block6f_se_expand',\n#    'block6e_expand_conv'\n#]\n\nlayer_names = [\n    'block_1_expand_relu',   # 64x64\n    'block_3_expand_relu',   # 32x32\n    'block_6_expand_relu',   # 16x16\n    'block_13_expand_relu',  # 8x8\n    'block_16_project',      # 4x4\n]\nlayers = [base_model.get_layer(name).output for name in layer_names]\n\n# Create the feature extraction model\ndown_stack = tf.keras.Model(inputs=base_model.input, outputs=layers)\n\ndown_stack.trainable = False\n#base_model.trainable = False\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"up_stack = [\n    pix2pix.upsample(1024, 3),\n    pix2pix.upsample(512, 3),  # 4x4 -> 8x8\n    pix2pix.upsample(256, 3),  # 8x8 -> 16x16\n    pix2pix.upsample(128, 3),  # 16x16 -> 32x32\n    pix2pix.upsample(64, 3),   # 32x32 -> 64x64\n]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def unet_model(output_channels):\n    inputs = tf.keras.layers.Input(shape=[*IMG_SIZE, 3])\n    x = inputs\n\n  # Downsampling through the model\n    skips = down_stack(x)\n    x = skips[-1]\n    skips = reversed(skips[:-1])\n\n  # Upsampling and establishing the skip connections\n    for up, skip in zip(up_stack, skips):\n        x = up(x)\n        concat = tf.keras.layers.Concatenate()\n        x = concat([x, skip])\n\n  # This is the last layer of the model\n    last = tf.keras.layers.Conv2DTranspose(\n      output_channels, 3, strides=2,\n      padding='same')  #64x64 -> 128x128\n\n    x = last(x)\n\n    return tf.keras.Model(inputs=inputs, outputs=x)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"OUTPUT_CHANNELS = 20\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = unet_model(OUTPUT_CHANNELS)\n\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=1e-3),\n              loss=tf.keras.losses.CategoricalCrossentropy(from_logits=True),\n              metrics=['categorical_accuracy'])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tf.keras.utils.plot_model(model, show_shapes=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_mask(pred_mask):\n    pred_mask = tf.argmax(pred_mask, axis=-1)\n    pred_mask = pred_mask[..., tf.newaxis]\n    return pred_mask[0]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#model.predict('../input/write-data-one-label/008a8630-bb9b-11e8-b2b9-ac1f6b6435d0_blend.png')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#NUM_TRAINING_IMAGES = count_data_items(TRAINING_FILENAMES)\n#NUM_VALIDATION_IMAGES = count_data_items(VALID_FILENAMES)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"STEPS_PER_EPOCH = 650 // BATCH_SIZE\nVALID_STEPS = 350 // BATCH_SIZE","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS = 20\nVAL_SUBSPLITS = 5\n#VALIDATION_STEPS = 100\n\nmodel_history = model.fit(traning_data,\n                          epochs=EPOCHS,\n                          steps_per_epoch=STEPS_PER_EPOCH,\n                          #validation_split=0.2,\n                          validation_steps=VALID_STEPS,\n                          validation_data=validation_data,\n                          callbacks=my_callbacks\n                         )\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_frame = pd.DataFrame(model_history.history)\nhistory_frame.loc[:, ['loss', 'val_loss']].plot()\nhistory_frame.loc[:, ['accuracy', 'val_accuracy']].plot();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\ndef unfreeze_model(model):\n    # We unfreeze the top 20 layers while leaving BatchNorm layers frozen\n    for layer in model.layers[-18:]:\n        if not isinstance(layer, tf.keras.layers.BatchNormalization):\n             layer.trainable = True\n\n    optimizer = tf.keras.optimizers.Adam(learning_rate=1e-3)\n    model.compile(\n        optimizer=optimizer, \n        #loss='sparse_categorical_crossentropy',  \n        loss=loss_func,\n        metrics=['categorical_accuracy']\n    )\n\n\nunfreeze_model(model)\n\nepochs = 10  # @param {type: \"slider\", min:8, max:50}\n\nhistory = model.fit(dataset_train, \n                    steps_per_epoch=100, \n                    epochs=15,\n                    validation_data=dataset_val,\n                    #validation_steps=VALID_STEPS,\n                   callbacks=my_callbacks)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history_frame = pd.DataFrame(history.history)\nhistory_frame.loc[:, ['loss', 'val_loss']].plot()\nhistory_frame.loc[:, ['accuracy', 'val_accuracy']].plot();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_mask(pred_mask):\n    pred_mask = tf.argmax(pred_mask, axis=-1)\n    pred_mask = pred_mask[..., tf.newaxis]\n    return pred_mask[0]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"create_mask(a)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"create_mask(a)\nplt.imshow(np.asarray(create_mask(a)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(np.asarray(b[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bild = cv2.imread('../input/write-data-one-label/5c801c04-bb99-11e8-b2b9-ac1f6b6435d0_blend.png', cv2.IMREAD_UNCHANGED)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"bild = bild / 255","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# EDA"},{"metadata":{"trusted":true},"cell_type":"code","source":"filename = '../input/hpa-single-cell-image-classification/train.csv'\n\ndf_train = pd.read_csv(filename)\n\ndf_train","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Helper Functions"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}