{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Histopathologic Cancer Detection","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-21T23:08:33.663518Z","iopub.execute_input":"2024-04-21T23:08:33.664038Z","iopub.status.idle":"2024-04-21T23:08:34.651368Z","shell.execute_reply.started":"2024-04-21T23:08:33.664001Z","shell.execute_reply":"2024-04-21T23:08:34.650034Z"}}},{"cell_type":"markdown","source":"#### The Histopathologic Cancer Detection project typically involves the use of machine learning models to automatically detect cancerous tissues from histopathologic scans of lymph node sections. This task is crucial for diagnosing cancer more accurately and promptly.\n#### The primary goal of the Histopathologic Cancer Detection project is to identify metastatic tissue in histopathological slides of lymph nodes. This is done by analyzing microscopic images of tissue sections to detect small structures indicative of cancer.\n","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport sklearn\nimport os\nimport gc\nimport cv2 \n\nfrom PIL import Image\nfrom PIL import ImageDraw\ntrain_on_gpu = True","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:04.307892Z","iopub.execute_input":"2024-05-01T23:30:04.308278Z","iopub.status.idle":"2024-05-01T23:30:07.661363Z","shell.execute_reply.started":"2024-05-01T23:30:04.308248Z","shell.execute_reply":"2024-05-01T23:30:07.660044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\ndf_samples = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/sample_submission.csv')\ndf_labels.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:07.664390Z","iopub.execute_input":"2024-05-01T23:30:07.665398Z","iopub.status.idle":"2024-05-01T23:30:08.255399Z","shell.execute_reply.started":"2024-05-01T23:30:07.665351Z","shell.execute_reply":"2024-05-01T23:30:08.254415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_labels.shape)\nprint(df_labels.columns)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:08.260872Z","iopub.execute_input":"2024-05-01T23:30:08.261573Z","iopub.status.idle":"2024-05-01T23:30:08.268670Z","shell.execute_reply.started":"2024-05-01T23:30:08.261539Z","shell.execute_reply":"2024-05-01T23:30:08.266974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_samples.shape)\nprint(df_samples.columns)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:08.270298Z","iopub.execute_input":"2024-05-01T23:30:08.271650Z","iopub.status.idle":"2024-05-01T23:30:08.284554Z","shell.execute_reply.started":"2024-05-01T23:30:08.271604Z","shell.execute_reply":"2024-05-01T23:30:08.282964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = \"/kaggle/input/histopathologic-cancer-detection/train/\"\ntest = \"/kaggle/input/histopathologic-cancer-detection/test\"\n\nprint(\"Number of training images: {}\".format(len(os.listdir(train))))\nprint(\"Number of test images: {}\".format(len(os.listdir(test))))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:08.286660Z","iopub.execute_input":"2024-05-01T23:30:08.287402Z","iopub.status.idle":"2024-05-01T23:30:11.799881Z","shell.execute_reply.started":"2024-05-01T23:30:08.287358Z","shell.execute_reply":"2024-05-01T23:30:11.798462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_train = os.listdir(train)\nimg_test = os.listdir(test)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:11.801428Z","iopub.execute_input":"2024-05-01T23:30:11.801926Z","iopub.status.idle":"2024-05-01T23:30:12.035983Z","shell.execute_reply.started":"2024-05-01T23:30:11.801884Z","shell.execute_reply":"2024-05-01T23:30:12.034823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(25, 4))\nfor i in range(5):\n    ax = fig.add_subplot(1, 5, i + 1, xticks=[], yticks=[])\n    im = Image.open(train + img_train[i])\n    plt.imshow(im)\n    label = df_labels.loc[df_labels['id'] == img_train[i].split('.')[0], 'label'].values[0]\n    ax.set_title(f'#{i+1} - Label: {label}')","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:12.037294Z","iopub.execute_input":"2024-05-01T23:30:12.037634Z","iopub.status.idle":"2024-05-01T23:30:13.009125Z","shell.execute_reply.started":"2024-05-01T23:30:12.037605Z","shell.execute_reply":"2024-05-01T23:30:13.007773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = df_labels.isnull().sum()\nmissing_values","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:13.010720Z","iopub.execute_input":"2024-05-01T23:30:13.011189Z","iopub.status.idle":"2024-05-01T23:30:13.051366Z","shell.execute_reply.started":"2024-05-01T23:30:13.011145Z","shell.execute_reply":"2024-05-01T23:30:13.050244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values = df_samples.isnull().sum()\nmissing_values","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:13.057655Z","iopub.execute_input":"2024-05-01T23:30:13.058161Z","iopub.status.idle":"2024-05-01T23:30:13.074285Z","shell.execute_reply.started":"2024-05-01T23:30:13.058117Z","shell.execute_reply":"2024-05-01T23:30:13.072871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels[df_labels.duplicated(keep=False)]","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:13.075930Z","iopub.execute_input":"2024-05-01T23:30:13.076467Z","iopub.status.idle":"2024-05-01T23:30:13.185087Z","shell.execute_reply.started":"2024-05-01T23:30:13.076405Z","shell.execute_reply":"2024-05-01T23:30:13.183920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:13.186851Z","iopub.execute_input":"2024-05-01T23:30:13.187284Z","iopub.status.idle":"2024-05-01T23:30:13.204788Z","shell.execute_reply.started":"2024-05-01T23:30:13.187245Z","shell.execute_reply":"2024-05-01T23:30:13.203503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_samples['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:13.206238Z","iopub.execute_input":"2024-05-01T23:30:13.206630Z","iopub.status.idle":"2024-05-01T23:30:13.217987Z","shell.execute_reply.started":"2024-05-01T23:30:13.206599Z","shell.execute_reply":"2024-05-01T23:30:13.216530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"malignant = df_labels.loc[df_labels['label']==1]['id'].values    \nnormal = df_labels.loc[df_labels['label']==0]['id'].values      \ndf_labels['label'].hist()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:13.219497Z","iopub.execute_input":"2024-05-01T23:30:13.219835Z","iopub.status.idle":"2024-05-01T23:30:13.567472Z","shell.execute_reply.started":"2024-05-01T23:30:13.219806Z","shell.execute_reply":"2024-05-01T23:30:13.566323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_fig(ids,title,nrows=3,ncols=3):\n\n    fig,ax = plt.subplots(nrows,ncols,figsize=(7,7))\n    plt.subplots_adjust(wspace=0, hspace=0) \n    for i,j in enumerate(ids[:nrows*ncols]):\n        fname = os.path.join(train ,j +'.tif')\n        img = Image.open(fname)\n        idcol = ImageDraw.Draw(img)\n        idcol.rectangle(((0,0),(95,95)),outline='white')\n        plt.subplot(nrows, ncols, i+1) \n        plt.imshow(np.array(img))\n        plt.axis('off')\n\n    plt.suptitle(title, y=0.94)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:13.569234Z","iopub.execute_input":"2024-05-01T23:30:13.569944Z","iopub.status.idle":"2024-05-01T23:30:13.580660Z","shell.execute_reply.started":"2024-05-01T23:30:13.569903Z","shell.execute_reply":"2024-05-01T23:30:13.579549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_fig(malignant,'Malignant cases')","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:13.582111Z","iopub.execute_input":"2024-05-01T23:30:13.582514Z","iopub.status.idle":"2024-05-01T23:30:14.352815Z","shell.execute_reply.started":"2024-05-01T23:30:13.582485Z","shell.execute_reply":"2024-05-01T23:30:14.351389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_fig(normal,'Normal cases')","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:14.354530Z","iopub.execute_input":"2024-05-01T23:30:14.355028Z","iopub.status.idle":"2024-05-01T23:30:15.168395Z","shell.execute_reply.started":"2024-05-01T23:30:14.354976Z","shell.execute_reply":"2024-05-01T23:30:15.167055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_data, val_data = train_test_split(df_labels, test_size=0.2, random_state=42, stratify=df_labels['label'])","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:15.170156Z","iopub.execute_input":"2024-05-01T23:30:15.170659Z","iopub.status.idle":"2024-05-01T23:30:15.459363Z","shell.execute_reply.started":"2024-05-01T23:30:15.170617Z","shell.execute_reply":"2024-05-01T23:30:15.458281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.astype(str)\nval_data = val_data.astype(str)\nprint(train_data.shape,val_data.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:15.460798Z","iopub.execute_input":"2024-05-01T23:30:15.461272Z","iopub.status.idle":"2024-05-01T23:30:15.585938Z","shell.execute_reply.started":"2024-05-01T23:30:15.461232Z","shell.execute_reply":"2024-05-01T23:30:15.584743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\n\n# Define an ImageDataGenerator for training data with augmentation\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:15.587666Z","iopub.execute_input":"2024-05-01T23:30:15.588011Z","iopub.status.idle":"2024-05-01T23:30:30.624866Z","shell.execute_reply.started":"2024-05-01T23:30:15.587980Z","shell.execute_reply":"2024-05-01T23:30:30.623516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_datagen = ImageDataGenerator(rescale=1./255)\ntest_datagen = ImageDataGenerator(rescale=1./255)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:30.626342Z","iopub.execute_input":"2024-05-01T23:30:30.627232Z","iopub.status.idle":"2024-05-01T23:30:30.634391Z","shell.execute_reply.started":"2024-05-01T23:30:30.627194Z","shell.execute_reply":"2024-05-01T23:30:30.632198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['id'] += '.tif'\nval_data['id'] += '.tif'","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:30.636218Z","iopub.execute_input":"2024-05-01T23:30:30.636750Z","iopub.status.idle":"2024-05-01T23:30:30.758894Z","shell.execute_reply.started":"2024-05-01T23:30:30.636703Z","shell.execute_reply":"2024-05-01T23:30:30.757688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_data,\n    directory = train,\n    x_col = 'id',\n    y_col = 'label',\n    target_size=(96,96),\n    batch_size=32,\n    class_mode='binary'\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:30:30.760363Z","iopub.execute_input":"2024-05-01T23:30:30.760991Z","iopub.status.idle":"2024-05-01T23:54:39.239161Z","shell.execute_reply.started":"2024-05-01T23:30:30.760957Z","shell.execute_reply":"2024-05-01T23:54:39.237793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_generator = val_datagen.flow_from_dataframe(\n    dataframe=val_data,\n    directory = train,\n    x_col='id',\n    y_col='label',\n    target_size=(96,96),\n    batch_size=32,\n    class_mode='binary'\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:54:39.241055Z","iopub.execute_input":"2024-05-01T23:54:39.241536Z","iopub.status.idle":"2024-05-01T23:59:29.553116Z","shell.execute_reply.started":"2024-05-01T23:54:39.241493Z","shell.execute_reply":"2024-05-01T23:59:29.551690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = df_samples.astype(str)\ntest_data['id'] += '.tif'","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:59:29.554750Z","iopub.execute_input":"2024-05-01T23:59:29.555148Z","iopub.status.idle":"2024-05-01T23:59:29.600223Z","shell.execute_reply.started":"2024-05-01T23:59:29.555116Z","shell.execute_reply":"2024-05-01T23:59:29.598997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generator = test_datagen.flow_from_dataframe(\n    dataframe=test_data,\n    directory = test,\n    x_col='id',\n    y_col='label',\n    target_size=(96,96),\n    batch_size=32,\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T23:59:29.601570Z","iopub.execute_input":"2024-05-01T23:59:29.601892Z","iopub.status.idle":"2024-05-02T00:01:41.346967Z","shell.execute_reply.started":"2024-05-01T23:59:29.601864Z","shell.execute_reply":"2024-05-02T00:01:41.345744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom tensorflow.keras import layers,optimizers,models\nfrom keras.metrics import AUC","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:01:41.348430Z","iopub.execute_input":"2024-05-02T00:01:41.348800Z","iopub.status.idle":"2024-05-02T00:01:41.354627Z","shell.execute_reply.started":"2024-05-02T00:01:41.348771Z","shell.execute_reply":"2024-05-02T00:01:41.353431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CNNs are the backbone of the image classification part of the project. Popular architectures used include VGG, ResNet, and Inception models. These networks are adept at extracting features from complex image data.","metadata":{}},{"cell_type":"code","source":"def CNN(model):\n    \n    model.add(layers.Conv2D(16, (3,3), activation='relu', input_shape=(96,96,3)))\n    model.add(layers.Conv2D(16, (3,3), activation='relu'))\n    model.add(layers.BatchNormalization())\n    model.add(layers.MaxPool2D(pool_size=(3,3)))\n\n    model.add(layers.Conv2D(32, (3,3), activation='relu'))\n    model.add(layers.Conv2D(32, (3,3), activation='relu'))\n    model.add(layers.BatchNormalization())\n    model.add(layers.MaxPool2D(pool_size=(3,3)))\n\n    model.add(layers.Conv2D(64, (3,3), activation='relu'))\n    model.add(layers.Conv2D(64, (3,3), activation='relu'))\n    model.add(layers.BatchNormalization())\n    model.add(layers.MaxPool2D(pool_size=(3,3)))\n\n    # Convert to 1D vector\n    model.add(layers.Flatten())\n\n    # Classification layers\n    model.add(layers.Dense(64, activation='sigmoid'))\n    model.add(layers.Dropout(0.5))\n    model.add(layers.Dense(1, activation='sigmoid'))\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:01:41.361551Z","iopub.execute_input":"2024-05-02T00:01:41.362003Z","iopub.status.idle":"2024-05-02T00:01:41.377494Z","shell.execute_reply.started":"2024-05-02T00:01:41.361953Z","shell.execute_reply":"2024-05-02T00:01:41.376570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RMS_model = CNN(tf.keras.Sequential())\nRMS_model.compile(\n    optimizer = 'RMSprop',\n    loss = 'binary_crossentropy',\n    metrics = [AUC()]\n)\n\nRMS_history = RMS_model.fit(\n    train_generator,\n    validation_data = val_generator,\n    epochs = 6\n)\n\n\nprediction_labels_1 = RMS_model.predict(test_generator)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T00:01:41.379073Z","iopub.execute_input":"2024-05-02T00:01:41.380462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_pred_df_1 = pd.DataFrame(columns=['id', 'label'])\nimg_test=sorted(img_test)\n\n\nmodel_pred_df_1['id'] = [filename.split('.')[0] for filename in img_test]\nmodel_pred_df_1['label'] = np.round(prediction_labels_1.flatten()).astype('int')\nmodel_pred_df_1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Model performance is typically evaluated using metrics such as accuracy, precision, recall, F1-score, and the Area Under the Curve (AUC) of the Receiver Operating Characteristics (ROC). These metrics help in assessing how well the model can identify metastatic cancer in unseen data.\n\n","metadata":{}},{"cell_type":"code","source":"plt.plot(RMS_history.history['loss'], label='loss')\nplt.plot(RMS_history.history['val_loss'], label = 'val_loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.title('Train vs Validation Loss Per Epoch')\n\nplt.subplot(1, 2, 2)\nplt.plot(RMS_history.history['auc'], label='Training AUC')\nplt.plot(RMS_history.history['val_auc'], label='Validation AUC')\nplt.title('Training and Validation AUC')\nplt.xlabel('Epochs')\nplt.ylabel('AUC')\nplt.legend()\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SGD_model = CNN(tf.keras.Sequential())\nSGD_model.compile(\n    optimizer = 'SGD',\n    loss = 'binary_crossentropy',\n    metrics = [AUC()]\n)\n\nSGD_history= SGD_model.fit(\n    train_generator,\n    validation_data = val_generator,\n    epochs = 6\n)\n\n\nprediction_labels_2 = SGD_model.predict(test_generator)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_pred_df_2 = pd.DataFrame(columns=['id', 'label'])\nimg_test=sorted(img_test)\n\n\nmodel_pred_df_2['id'] = [filename.split('.')[0] for filename in img_test]\nmodel_pred_df_2['label'] = np.round(prediction_labels_2.flatten()).astype('int')\nmodel_pred_df_2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplot(1, 2, 2)\nplt.plot(RMS_history.history['auc'], label='Training AUC')\nplt.plot(RMS_history.history['val_auc'], label='Validation AUC')\nplt.title('Training and Validation AUC')\nplt.xlabel('Epochs')\nplt.ylabel('AUC')\nplt.legend()\n\nplt.plot(RMS_history.history['loss'], label='loss')\nplt.plot(RMS_history.history['val_loss'], label = 'val_loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.title('Train vs Validation Loss Per Epoch')\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RMS_csv = model_pred_df_1.to_csv('RMS_model_predictions.csv', index=False)\nSGD_csv = model_pred_df_2.to_csv('SGD_model_predictions.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RMS_csv","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SGD_csv","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}