{"cells":[{"metadata":{"_uuid":"f140b1cb58d54a3e06e57572678a725e275a8cdd"},"cell_type":"markdown","source":"# ** Réseaux convolutif - Détection d'un cancer **"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nfrom pandas import read_csv\nimport matplotlib.pyplot as plt\nimport glob\nfrom PIL import Image\nimport os\n\n#Afficher liste des dossiers\nprint(os.listdir(\"../input/histopathologic-cancer-detection/\"))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5d0625cf9026f8cae6e66c1ce6e57ce550f3db30"},"cell_type":"markdown","source":"# **Importation des labels**"},{"metadata":{"trusted":true,"_uuid":"739e7f674edea9b2cd50863da6742a16502a1c5e"},"cell_type":"code","source":"test = read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\n\ntest = np.array(test)\nid = test[:,0]\ntargets = test[:,1]\n\nprint(id[:5])\nprint(targets[:5])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e1e4f85ce44d537b9fdfafee61548a8be9ba4a0c"},"cell_type":"markdown","source":"# **Importation des images**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"files = os.listdir(\"../input/histopathologic-cancer-detection/train/\")\n\nfeatures = []\n\nfor i in range(0,len(id)):\n    features.append(np.array(Image.open(\"../input/histopathologic-cancer-detection/train/\" + id[i] + \".tif\").resize((90, 90))))\n\nfeatures = np.array(features)\n\nprint(\"Taille des features :\",features.shape)\nprint(\"Taille des targets :\",targets.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"81a2e6d56b9493255044e77700a722ec9bcade67"},"cell_type":"markdown","source":"# **Visualisation des images**"},{"metadata":{"trusted":true,"_uuid":"9bf378e3709d15f653c89e4b4e3a55bd13babc1b"},"cell_type":"code","source":"index = np.arange(len(features))\nnp.random.shuffle(index)\nfeatures = features[index]\n\nfig = plt.figure(figsize=(17,17))\nplt.gcf().subplots_adjust( wspace = 0, hspace = 0.2)\nfor i in range(1,26):\n    plt.subplot(5, 5, i)\n    img = plt.imshow(features[i])\n    img.axes.get_xaxis().set_visible(False)\n    img.axes.get_yaxis().set_visible(False)\n    if targets[i]==1:\n        label = \"Cancer\"\n    else:\n        label = \"Sain\"\n    plt.title((label))\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5be9c1537c07144ee923b1b4ef43bb670a8ef4b"},"cell_type":"markdown","source":"# **Echantillonage**"},{"metadata":{"trusted":true,"_uuid":"92e723a7c5d571c74f1b73b8c2167d96fb27bbc6"},"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nonehot_encoder = OneHotEncoder(sparse=False)\ntargets = onehot_encoder.fit_transform(targets.reshape(len(targets), 1))\nprint(targets)\n\nfrom sklearn.model_selection import train_test_split\n\nx_train, x_valid, y_train, y_valid = train_test_split(features, targets, test_size=0.1, random_state=123)\n\nprint(\"x_train =\",x_train.shape,\"  |  y_train =\",y_train.shape)\nprint(\"x_valid =\",x_valid.shape,\"  |  y_valid =\",y_valid.shape)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f76e72412eaacf6f80290c4a3ada2fa7959b0a24"},"cell_type":"markdown","source":"# **Création du modèle convolutif**"},{"metadata":{"trusted":true,"_uuid":"a2a29503cf0f1b67e31c789933b74e0101bd6624"},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense, Activation, BatchNormalization\nfrom keras import optimizers","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74e592e7def45b1b5e3f8c74ddbbe67775cde4ad"},"cell_type":"code","source":"model = Sequential()\n\nmodel.add(Conv2D(64, (3, 3), activation='relu', input_shape=(90, 90, 3)))\nmodel.add(BatchNormalization())\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.20))\n\nmodel.add(Conv2D(128, (3, 3), activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.20))\n\nmodel.add(Conv2D(256, (3, 3), activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.20))\n\nmodel.add(Conv2D(512, (3, 3), activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.20))\n\n# model.add(Conv2D(512, (3, 3), activation='relu'))\n# model.add(BatchNormalization())\n# model.add(MaxPooling2D(pool_size=(2, 2)))\n# model.add(Dropout(0.20))\n\nmodel.add(Flatten())\nmodel.add(Dense(1500, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.3))\nmodel.add(Dense(2, activation='softmax'))\n\n\nsgd = optimizers.RMSprop(lr=0.001, rho=0.9, epsilon=None, decay=0.0)\nmodel.compile(optimizer = sgd, loss = 'categorical_crossentropy', metrics = ['accuracy'])\n\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40a98fcc02459120ff4509c0c4f82daec1c05f2d"},"cell_type":"code","source":"model.fit(x=x_train,\n          y=y_train,\n          batch_size=100,\n          epochs=12,\n          verbose=1,\n          callbacks=None,\n          validation_data=(x_valid,y_valid),\n          shuffle=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c6ff6d46620d29e710c18f63030eead483a7613"},"cell_type":"markdown","source":"1er modèle : conv1 = (32,8,8)[0,25]  |  conv2 = (64,5,5)[0,25]  |   conv2 = (128,5,5)[0,25]  |  conv3 = (256,3,3)[0,25]  |   fully : 512[0,5]\nAcc = 59% (2 epochs)\n2eme modèle : conv1 = (32,5,5)[0,2]  |  conv2 = (64,5,5)[0,2]  |   conv2 = (128,5,5)[0,2]  |  conv3 = (256,3,3)[0,2]  |   fully : 512[0,3]\nAcc = \n        "},{"metadata":{"trusted":true,"_uuid":"8caea7191fe23ad95d9bb51e5a2618413fcda9be"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"000f7bf36789fbc0594ab61b53258780c9a2c780"},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\n%matplotlib inline \n\nimport cv2\n\nimport os","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f876a0eb38a2550a7293714e476974c74ecd85a"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c19110f35417c65c5d8b7a7cf98bcfd554ef3c6e"},"cell_type":"code","source":"NUM_CLASSES = 2\n\n# Fixed for Cats & Dogs color images\nCHANNELS = 3\n\nIMAGE_RESIZE = 75\nRESNET50_POOLING_AVERAGE = 'avg'\nDENSE_LAYER_ACTIVATION = 'softmax'\nOBJECTIVE_FUNCTION = 'categorical_crossentropy'\n\n# Common accuracy metric for all outputs, but can use different metrics for different output\nLOSS_METRICS = ['accuracy']\n\n# EARLY_STOP_PATIENCE must be < NUM_EPOCHS\nNUM_EPOCHS = 10\nEARLY_STOP_PATIENCE = 3\n\n# These steps value should be proper FACTOR of no.-of-images in train & valid folders respectively\n# Training images processed in each step would be no.-of-train-images / STEPS_PER_EPOCH_TRAINING\nSTEPS_PER_EPOCH_TRAINING = 10\nSTEPS_PER_EPOCH_VALIDATION = 10\n\n# These steps value should be proper FACTOR of no.-of-images in train & valid folders respectively\n# NOTE that these BATCH* are for Keras ImageDataGenerator batching to fill epoch step input\nBATCH_SIZE_TRAINING = 100\nBATCH_SIZE_VALIDATION = 100\n\n# Using 1 to easily manage mapping between test_generator & prediction for submission preparation\nBATCH_SIZE_TESTING = 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f8915995ffadcf375059934481ce8116b0b440c1"},"cell_type":"code","source":"from tensorflow.python.keras.applications import ResNet50\nfrom tensorflow.python.keras.models import Sequential\nfrom tensorflow.python.keras.layers import Dense","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"96bf8302ea5f6250463afc0623a532760befec73"},"cell_type":"code","source":"resnet_weights_path = '../input/resnet50/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5'\nprint(resnet_weights_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ffa0cd167754828106314f5cf6e3ab54d99fc245"},"cell_type":"code","source":"model = Sequential()\n\n# 1st layer as the lumpsum weights from resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5\n# NOTE that this layer will be set below as NOT TRAINABLE, i.e., use it as is\nmodel.add(ResNet50(include_top = False, pooling = RESNET50_POOLING_AVERAGE, weights = resnet_weights_path))\n\n# 2nd layer as Dense for 2-class classification, i.e., dog or cat using SoftMax activation\nmodel.add(Dense(NUM_CLASSES, activation = DENSE_LAYER_ACTIVATION))\n\n# Say not to train first layer (ResNet) model as it is already trained\nmodel.layers[0].trainable = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6781259b92b7867b7a440da5071954a6b3da4f3b"},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe68d8f8bd4d8db12e982e53c81d5b3f505d7a3c"},"cell_type":"code","source":"from tensorflow.python.keras import optimizers\n\nsgd = optimizers.SGD(lr = 0.01, decay = 1e-6, momentum = 0.9, nesterov = True)\nmodel.compile(optimizer = sgd, loss = OBJECTIVE_FUNCTION, metrics = LOSS_METRICS)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"536c2e87df4964d6c0936f334f6023196354d5e8"},"cell_type":"code","source":"from tensorflow.python.keras.callbacks import EarlyStopping, ModelCheckpoint\n\ncb_early_stopper = EarlyStopping(monitor = 'val_loss', patience = EARLY_STOP_PATIENCE)\ncb_checkpointer = ModelCheckpoint(filepath = '../working/best.hdf5', monitor = 'val_loss', save_best_only = True, mode = 'auto')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ea8818c415c742001bbe8d227c5848dc7e800ae3"},"cell_type":"code","source":"fit_history = model.fit(\n        (x_train,y_train),\n        steps_per_epoch=STEPS_PER_EPOCH_TRAINING,\n        epochs = NUM_EPOCHS,\n        validation_data=(x_valid,y_valid),\n        validation_steps=STEPS_PER_EPOCH_VALIDATION,\n        callbacks=[cb_checkpointer, cb_early_stopper]\n)\nmodel.load_weights(\"../working/best.hdf5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"391644bc17efcae31cc6b42528517f0adadd298f"},"cell_type":"code","source":"model.fit(x=x_train,\n          y=y_train,\n          batch_size=100,\n          epochs=12,\n          callbacks=[cb_checkpointer, cb_early_stopper],\n          validation_data=(x_valid,y_valid),\n          shuffle=True)\n\nmodel.load_weights(\"../working/best.hdf5\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}