{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#_______________________________________________\n# Charegement des libraries Keras \n#\nfrom tensorflow.keras.models import Sequential\n### Conv2d pour des photos - 2D si Video ou coleur alors 3D \nfrom  tensorflow.keras.layers import Conv2D\n### Pour la phse de Pooling 2D si  couleur ou Videos alors 3D\nfrom  tensorflow.keras.layers import MaxPooling2D\n### Etape 3 - Applatir dans un vector Vertical \nfrom  tensorflow.keras.layers import Flatten\n### Dense pour ajouter des couches connectées.\nfrom  tensorflow.keras.layers import Dense\n### Droput si necessaire\nfrom tensorflow.keras.layers import MaxPool2D,Dropout\n### import Metrics\nfrom  tensorflow.keras.metrics import *\n#import keras","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#from tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras import backend\nfrom tensorflow.keras.layers import Dropout\nfrom tensorflow.keras.optimizers import Adam, Adagrad, Adadelta, Adamax, RMSprop\nfrom  tensorflow.keras.metrics import *\nfrom tensorflow.keras import Model\n#from keras import backend as \nfrom tensorflow.keras.callbacks import Callback, ReduceLROnPlateau, ModelCheckpoint, EarlyStopping\n\n## With Regularization\nfrom tensorflow.keras import regularizers, optimizers","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras import Input\nfrom tensorflow.keras.layers import Flatten\nfrom tensorflow.keras.layers import BatchNormalization\nfrom keras.models import Model\nfrom tensorflow.keras.layers import Activation","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import keras \nfrom keras import backend as K\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"keras.__version__","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.python.client import device_lib\nprint(device_lib.list_local_devices())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Prepare Dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Name Label Dictionary\nname_label_dict = {\n0:  \"Nucleoplasm\", \n1:  \"Nuclear membrane\",   \n2:  \"Nucleoli\",   \n3:  \"Nucleoli fibrillar center\" ,  \n4:  \"Nuclear speckles\"   ,\n5:  \"Nuclear bodies\"   ,\n6:  \"Endoplasmic reticulum\",   \n7:  \"Golgi apparatus\"   ,\n8:  \"Intermediate filaments\",   \n9:  \"Actin filaments\"   ,\n10:  \"Microtubules\"   ,\n11:  \"Mitotic spindle\"   ,\n12:  \"Centrosome\" , \n13:  \"Plasma membrane\",   \n14:  \"Mitochondria\"   ,\n15:  \"Aggresome\"   ,\n16:  \"Cytosol\",  \n17:  \"Vesicles and punctate cytosolic patterns\",   \n18:  \"Negative\" \n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Load Train Dataset \npath_to_train = '../input/hpa-single-cell-image-classification/train/'\ndata = pd.read_csv('../input/hpa-single-cell-image-classification/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.head(), data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_labels(df):\n    # dataframe with column for each class\n    labels = list()\n    \n    for index, sample in df.iterrows():\n        # zero out class array\n        label = [0] * 19\n        \n        # for each class found in training sample, flip lablel value to one\n        for cls in sample['Label'].split('|'):\n            label[int(cls)] = 1\n\n        # Append label to list\n        labels.append( np.array(label) )\n\n    return np.vstack(labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train_labels = build_labels(data)\ndf_train_labels.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#____________________________________________________\n# Add full path to images files\ndata['image_path']=path_to_train + data['ID']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Join train_labels \n## Just Testing Green \"Cells\"\ndf_train_green = pd.DataFrame(data['image_path']).join(pd.DataFrame(df_train_labels))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train_green.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train_green['image_path'] = df_train_green['image_path']+'_green.png'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.set_option('display.max_colwidth', None)\ndf_train_green.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"classes = []\nfor key, value in name_label_dict.items():\n    print(value)\n    classes.append(value)\nprint(classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train_green.columns = ['path']+classes\ndf_train_green.head(2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Save dataset\n## df_train_green.to_csv('./df_train_green.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Load dataset\ndf_train_green = pd.read_csv('../input/df-train-green-csv/df_train_green.csv') ## Multioutpt Classes 19 Columns\ndf_train_green.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"classes,type(classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"classes = ['Nucleoplasm', 'Nuclear membrane', 'Nucleoli', 'Nucleoli fibrillar center', \n           'Nuclear speckles', 'Nuclear bodies', 'Endoplasmic reticulum', 'Golgi apparatus', \n           'Intermediate filaments', 'Actin filaments', 'Microtubules', 'Mitotic spindle', \n           'Centrosome', 'Plasma membrane', 'Mitochondria', 'Aggresome', 'Cytosol', \n           'Vesicles and punctate cytosolic patterns', 'Negative']\ntype(classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"output_list=[]\nloss_list = []\nfor p in enumerate(classes):\n    #print(p[0])\n    print('output'+str(p[0])+',')\n    output_list.append('output'+str(p[0]))\n    loss_list.append(\"binary_crossentropy\")\noutput_list, loss_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nfrom keras.preprocessing.image import ImageDataGenerator\n\ntrain_datagen = ImageDataGenerator(rescale = 1./255,\n                                   shear_range = 0.2,\n                                   zoom_range = 0.2,\n                                   #validation_split=0.3, ## Not Yet Compiled - \n                                   horizontal_flip = True)\n\n### pour le Test set\ntest_datagen = ImageDataGenerator(rescale = 1./255)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"training_set = train_datagen.flow_from_dataframe(dataframe=df_train_green[:200],\n                                                 x_col='path',\n                                                 y_col= classes,\n                                                 #directory='../input/hpa-single-cell-image-classification/train',\n                                                 target_size = (128, 128),\n                                                 batch_size = 256,\n                                                 validation_split = .3,\n                                                 seed = 7,\n                                                 class_mode = 'multi_output')  ## if \"raw\" need wrapper_generator function ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"### \n\ntest_set = test_datagen.flow_from_dataframe(dataframe=df_train_green[200:300],\n                                                 x_col='path',\n                                                 y_col= classes,\n                                            #directory='../input/hpa-single-cell-image-classification/train'\n                                            #imagepath_test,\n                                            target_size = (128, 128),\n                                            batch_size = 256,\n                                            seed = 7,\n                                            shuffle=False,  ## Si nous voulons utiliser X_test dans Matrice Conf.\n                                            #index_array = None,  ## Si nous voulons utiliser X_test dans Matrice Conf.\n                                            class_mode = 'multi_output')  ## if \"raw\" need wrapper_generator function ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"inp = Input(shape = (128,128,3))\nx = Conv2D(32, (3, 3), padding = 'valid')(inp)\nx = Activation('relu')(x)\nx = Conv2D(32, (3, 3))(x)\nx = Activation('relu')(x)\nx = MaxPooling2D(pool_size = (2, 2))(x)\nx = Dropout(0.25)(x)\nx = Conv2D(64, (3, 3), padding = 'valid')(x)\nx = Activation('relu')(x)\nx = Conv2D(64, (3, 3))(x)\nx = Activation('relu')(x)\nx = MaxPooling2D(pool_size = (2, 2))(x)\nx = Dropout(0.25)(x)\nx = Flatten()(x)\nx = Dense(32)(x)\nx = Activation('relu')(x)\nx = Dropout(0.5)(x)\noutput0 = Dense(1, activation = 'sigmoid')(x)\noutput1 = Dense(1, activation = 'sigmoid')(x)\noutput2 = Dense(1, activation = 'sigmoid')(x)\noutput3 = Dense(1, activation = 'sigmoid')(x)\noutput4 = Dense(1, activation = 'sigmoid')(x)\noutput5 = Dense(1, activation = 'sigmoid')(x)\noutput6 = Dense(1, activation = 'sigmoid')(x)\noutput7 = Dense(1, activation = 'sigmoid')(x)\noutput8 = Dense(1, activation = 'sigmoid')(x)\noutput9 = Dense(1, activation = 'sigmoid')(x)\noutput10 = Dense(1, activation = 'sigmoid')(x)\noutput11 = Dense(1, activation = 'sigmoid')(x)\noutput12 = Dense(1, activation = 'sigmoid')(x)\noutput13 = Dense(1, activation = 'sigmoid')(x)\noutput14 = Dense(1, activation = 'sigmoid')(x)\noutput15 = Dense(1, activation = 'sigmoid')(x)\noutput16 = Dense(1, activation = 'sigmoid')(x)\noutput17 = Dense(1, activation = 'sigmoid')(x)\noutput18 = Dense(1, activation = 'sigmoid')(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Model(inp,[output0,\noutput1,\noutput2,\noutput3,\noutput4,\noutput5,\noutput6,\noutput7,\noutput8,\noutput9,\noutput10,\noutput11,\noutput12,\noutput13,\noutput14,\noutput15,\noutput16,\noutput17,\noutput18])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rlr = ReduceLROnPlateau(monitor = 'val_AUC', factor = 0.1, patience = 2, verbose = 0, \n                            min_delta = 1e-4, mode = 'max')\nes = EarlyStopping(monitor = 'val_AUC', min_delta = 1e-4, patience = 2, mode = 'max', \n                       baseline = None, restore_best_weights = True, verbose = 0)\nckp = ModelCheckpoint('./model_1.hdf5', monitor = 'val_AUC', verbose = 0, \n                        save_best_only = True, save_weights_only = False, mode = 'max')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(optimizer = Adam(lr = 0.001, decay = 1e-6),\n              loss = tf.keras.losses.BinaryCrossentropy(label_smoothing = 1e-3), \n              #metrics = [BinaryAccuracy(name='binary_accuracy', dtype=None, threshold=0.5), Precision(name='precision'), Recall(name='recall')] \n              metrics = [tf.keras.metrics.CategoricalCrossentropy(name='categorical_crossentropy'), tf.keras.metrics.CategoricalAccuracy(name='categorical_accuracy'), \\\n                         Precision(name='precision'), Recall(name='recall') ] \n              #metrics = ['categorical_accuracy', 'accuracy'],\n             )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Show a summary of the model. Check the number of trainable parameters\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.utils import plot_model\n# plot the autoencoder\nplot_model(model, 'HPA-model1.png', show_shapes=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## IF \"raw\" in \ndef generator_wrapper(generator):\n    for batch_x,batch_y in generator:\n        yield (batch_x,[batch_y[:,i] for i in range(19)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nSTEP_SIZE_TRAIN=training_set.n//training_set.batch_size\n#STEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\n#STEP_SIZE_TEST=test_set.n//test_set.batch_size\nmodel.fit(generator_wrapper(training_set),\n                  #(np.asarray(training_set).astype(\"float32\")),\n                    #steps_per_epoch=STEP_SIZE_TRAIN,\n                    validation_data=generator_wrapper(test_set),\n                    #batch_size=4096, \n                    #validation_steps=STEP_SIZE_VALID,\n                    epochs=3,\n                    callbacks = [rlr,ckp,es],\n                    #verbose=2\n         )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"STEP_SIZE_TRAIN=training_set.n//training_set.batch_size\n#STEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\n#STEP_SIZE_TEST=test_set.n//test_set.batch_size\nmodel.fit(training_set,\n                  #(np.asarray(training_set).astype(\"float32\")),\n                    #steps_per_epoch=STEP_SIZE_TRAIN,\n                    validation_data=test_set,\n                    #batch_size=4096, \n                    #validation_steps=STEP_SIZE_VALID,\n                    epochs=10,\n                    callbacks = [rlr,ckp,es],\n                    #verbose=2\n         )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_name = 'HPA-model1-green-data'\nprint(model_name)\nmodel.save('./'+model_name+'.hdf5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ntest_set.reset() ## To get the output in same oder as y_test\npred=model.predict(test_set,\nverbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred[:1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.DataFrame(np.asarray(pred).reshape(100,-1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train_green.iloc[200:206,1:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"(np.asarray(pred).reshape(100,-1)==).all()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len([[0.24326178],\n[0.24545057],\n[0.18905358],\n[0.22454323],\n[0.17228684],\n[0.23446988],\n[0.23484272],\n[0.19753459],\n[0.26074097],\n[0.2010675 ],\n[0.10761576],\n[0.22031462],\n[0.23193006],\n[0.22713315],\n[0.11388983],\n[0.23554248],\n[0.1785873 ],\n[0.25215697],\n[0.18093413],\n[0.1990953 ],\n[0.20176446],\n[0.19727392],\n[0.2491296 ],\n[0.11087943],\n[0.22272176],\n[0.21852216],\n[0.207871  ],\n[0.1856019 ],\n[0.2678131 ],\n[0.20906556],\n[0.27924448],\n[0.27829096],\n[0.13388641],\n[0.2648097 ],\n[0.24399373],\n[0.24952464],\n[0.157573  ],\n[0.23329581],\n[0.23566657],\n[0.27109227],\n[0.22023298],\n[0.21330912],\n[0.28557703],\n[0.17392349],\n[0.17297237],\n[0.24998474],\n[0.24437207],\n[0.06390683],\n[0.20857255],\n[0.23521458]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def clean_prediction(prediction):\n    prediction = prediction.copy()\n    for batch_index in range(len(prediction)):\n        for class_index in range(len(prediction[batch_index])):\n            prediction[batch_index][class_index] = 1 if prediction[batch_index][class_index] >= 0.4 else 0\n    return np.array(prediction).astype(np.int) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"clean_pred = clean_prediction(pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.set_option('display.max_colwidth', None)\npd.DataFrame(clean_pred.reshape(100,19))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"(clean_pred.reshape(100,19) == df_train_green.iloc[200:300,1:].values).all()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## #Only 1 Output"},{"metadata":{"trusted":true},"cell_type":"code","source":"## testing 1 output for change in classe mode from raw to multi_output\ninp = Input(shape = (128,128,3))\nx = Conv2D(32, (3, 3), padding = 'valid')(inp)\nx = Activation('relu')(x)\nx = Conv2D(32, (3, 3))(x)\nx = Activation('relu')(x)\nx = MaxPooling2D(pool_size = (2, 2))(x)\nx = Dropout(0.25)(x)\nx = Conv2D(64, (3, 3), padding = 'valid')(x)\nx = Activation('relu')(x)\nx = Conv2D(64, (3, 3))(x)\nx = Activation('relu')(x)\nx = MaxPooling2D(pool_size = (2, 2))(x)\nx = Dropout(0.25)(x)\nx = Flatten()(x)\nx = Dense(32)(x)\nx = Activation('relu')(x)\nx = Dropout(0.5)(x)\noutput0 = Dense(1, activation = 'sigmoid')(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## testing for change in classe mode from raw to multi_output\nmodel = Model(inp,output0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"STEP_SIZE_TRAIN=training_set.n//training_set.batch_size\n#STEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\n#STEP_SIZE_TEST=test_set.n//test_set.batch_size\nmodel.fit(training_set,\n                  #(np.asarray(training_set).astype(\"float32\")),\n                    #steps_per_epoch=STEP_SIZE_TRAIN,\n                    validation_data=test_set,\n                    #batch_size=4096, \n                    #validation_steps=STEP_SIZE_VALID,\n                    epochs=10,\n                    callbacks = [rlr,ckp,es],\n                    #verbose=2\n         )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_set.reset()\npred=model.predict(test_set,\nverbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}