{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.4"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.metrics import AUC\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport pandas as pd \nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import layers\n\n\n\nfrom PIL import Image\nfrom PIL import ImageDraw\ntrain_on_gpu = True\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DIR = './histopathologic-cancer-detection'\n\ndf_labels = pd.read_csv(f'{DIR}/train_labels.csv')\ndf_samples = pd.read_csv(f'{DIR}/sample_submission.csv')\n\n# Take a look\ndf_labels.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = f\"{DIR}/train/\"\ntest = f\"{DIR}/test\"\n\nprint(f\"Number of training images: {len(os.listdir(train))}\")\nprint(f\"Number of test images: {len(os.listdir(test))}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_train = os.listdir(train)\nimg_test = os.listdir(test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(25, 4))\n\nfor i in range(5):\n    ax = fig.add_subplot(1, 5, i + 1, xticks=[], yticks=[])\n    im = Image.open(train + img_train[i])\n    plt.imshow(im)\n    label = df_labels.loc[df_labels['id'] == img_train[i].split('.')[0], 'label'].values[0]\n    ax.set_title(f'#{i+1} - Label: {label}')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = df_labels.isnull().sum()\nmissing_values","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels[df_labels.duplicated(keep=False)]\ndf_labels['label'].value_counts()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"malignant = df_labels.loc[df_labels['label']==1]['id'].values    \nnormal = df_labels.loc[df_labels['label']==0]['id'].values      \ndf_labels['label'].hist()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_fig(ids,title,nrows=3,ncols=3):\n\n    fig,ax = plt.subplots(nrows,ncols,figsize=(7,7))\n    plt.subplots_adjust(wspace=0, hspace=0) \n    for i,j in enumerate(ids[:nrows*ncols]):\n        fname = os.path.join(train ,j +'.tif')\n        img = Image.open(fname)\n        idcol = ImageDraw.Draw(img)\n        idcol.rectangle(((0,0),(95,95)),outline='white')\n        plt.subplot(nrows, ncols, i+1) \n        plt.imshow(np.array(img))\n        plt.axis('off')\n\n    plt.suptitle(title, y=0.94)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_fig(malignant,'Malignant cases')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_fig(normal,'Normal cases')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, val_data = train_test_split(df_labels, test_size=0.2, random_state=42, stratify=df_labels['label'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.astype(str)\nval_data = val_data.astype(str)\nprint(train_data.shape,val_data.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale=1./255,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_datagen = ImageDataGenerator(rescale=1./255)\ntest_datagen = ImageDataGenerator(rescale=1./255)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['id'] += '.tif'\nval_data['id'] += '.tif'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_data,\n    directory = train,\n    x_col = 'id',\n    y_col = 'label',\n    target_size=(96,96),\n    batch_size=32,\n    class_mode='binary'\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator = val_datagen.flow_from_dataframe(\n    dataframe=val_data,\n    directory = train,\n    x_col='id',\n    y_col='label',\n    target_size=(96,96),\n    batch_size=32,\n    class_mode='binary'\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \ntest_data = df_samples.astype(str)\ntest_data['id'] += '.tif'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generator = test_datagen.flow_from_dataframe(\n    dataframe=test_data,\n    directory = test,\n    x_col='id',\n    y_col='label',\n    target_size=(96,96),\n    batch_size=32,\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the CNN function\ndef CNN(model):\n    model.add(layers.Conv2D(16, (3,3), activation='relu', input_shape=(96,96,3)))\n    model.add(layers.Conv2D(16, (3,3), activation='relu'))\n    model.add(layers.BatchNormalization())\n    model.add(layers.MaxPool2D(pool_size=(3,3)))\n\n    # Batch normalization\n    model.add(layers.Conv2D(32, (3,3), activation='relu'))\n    model.add(layers.Conv2D(32, (3,3), activation='relu'))\n    model.add(layers.BatchNormalization())\n    model.add(layers.MaxPool2D(pool_size=(3,3)))\n\n    # Batch normalization\n    model.add(layers.Conv2D(64, (3,3), activation='relu'))\n    model.add(layers.Conv2D(64, (3,3), activation='relu'))\n    model.add(layers.BatchNormalization())\n    model.add(layers.MaxPool2D(pool_size=(3,3)))\n\n    # Convert to 1D vector\n    model.add(layers.Flatten())\n\n    # Classification layers\n    model.add(layers.Dense(64, activation='sigmoid'))\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(1, activation='sigmoid'))\n\n    return model","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RMS_model = CNN(tf.keras.Sequential())\n\nRMS_model.compile(\n    optimizer = 'RMSprop',\n    loss = 'binary_crossentropy',\n    metrics = [AUC()]\n)\n\nRMS_history = RMS_model.fit(\n    train_generator,\n    validation_data = val_generator,\n    epochs = 6\n)\n\n\nprediction_labels_1 = RMS_model.predict(test_generator)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_pred_df_1 = pd.DataFrame(columns=['id', 'label'])\nimg_test=sorted(img_test)\n\n# Take a quick look\nmodel_pred_df_1['id'] = [filename.split('.')[0] for filename in img_test]\nmodel_pred_df_1['label'] = np.round(prediction_labels_1.flatten()).astype('int')\nmodel_pred_df_1","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot\nplt.plot(RMS_history.history['loss'], label='loss')\nplt.plot(RMS_history.history['val_loss'], label = 'val_loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.title('Train vs Val. Loss Per Epoch')\n\nplt.subplot(1, 2, 2)\nplt.plot(RMS_history.history['auc_4'], label='Training AUC')\nplt.plot(RMS_history.history['val_auc_4'], label='Validation AUC')\nplt.title('Train and Val AUC')\nplt.xlabel('Epochs')\nplt.ylabel('AUC')\nplt.legend()\n\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile and Fit\nSGD_model = CNN(tf.keras.Sequential())\nSGD_model.compile(\n    optimizer = 'SGD',\n    loss = 'binary_crossentropy',\n    metrics = [AUC()]\n)\n\nSGD_history= SGD_model.fit(\n    train_generator,\n    validation_data = val_generator,\n    epochs = 6\n)\n\n\nprediction_labels_2 = SGD_model.predict(test_generator)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_pred_df_2 = pd.DataFrame(columns=['id', 'label'])\nimg_test=sorted(img_test)\n\n# Take a look\nmodel_pred_df_2['id'] = [filename.split('.')[0] for filename in img_test]\nmodel_pred_df_2['label'] = np.round(prediction_labels_2.flatten()).astype('int')\nmodel_pred_df_2","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot data\nplt.subplot(1, 2, 2)\nplt.plot(RMS_history.history['auc_4'], label='Training AUC')\nplt.plot(RMS_history.history['val_auc_4'], label='Validation AUC')\nplt.title('Training and Validation AUC')\nplt.xlabel('Epochs')\nplt.ylabel('AUC')\nplt.legend()\n\nplt.plot(RMS_history.history['loss'], label='loss')\nplt.plot(RMS_history.history['val_loss'], label = 'val_loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.title('Train vs Validation Loss Per Epoch')\n\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RMS_csv = model_pred_df_1.to_csv('RMS_model_predictions.csv', index=False)\nSGD_csv = model_pred_df_2.to_csv('SGD_model_predictions.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RMS_csv","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SGD_csv","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save model.\nSGD_model.save(\"cancer_model.h5\")","metadata":{},"execution_count":null,"outputs":[]}]}