{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#import packages\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import train_test_split\nimport pickle\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train and test folder\nprint('Number of images in train set',len(os.listdir('../input/histopathologic-cancer-detection/train')))\nprint('Number of images in test set',len(os.listdir('../input/histopathologic-cancer-detection/test')))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the training data into a DataFrame. \n# Print the shape of the resulting DataFrame.\n\nhcd = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\nprint(hcd.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the first few rows of the dataframe.\nhcd.head() ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label distrobution\n(hcd.label.value_counts() / len(hcd)).to_frame()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Adding a variable for the image directory\nimg_dir = '/kaggle/input/histopathologic-cancer-detection/train'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = hcd.sample(n=9).reset_index()\n\nplt.figure(figsize=(3,3))\n\nfor i, row in sample.iterrows():\n\n    img = mpimg.imread(f'{img_dir}/{row.id}.tif')    \n    label = row.label\n\n    plt.subplot(3,3,i+1)\n    plt.imshow(img)\n    plt.text(0, -5, f'Class {label}', color='k')\n        \n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using data generators \ntrain_df, valid_df = train_test_split(hcd, test_size=0.2, random_state=39, stratify=hcd.label)\n\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scaling images \ntrain_datagen = ImageDataGenerator(rescale=1/255)\nvalid_datagen = ImageDataGenerator(rescale=1/255)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['id'] = train_df['id'] + '.tif'\nvalid_df['id'] = valid_df['id'] + '.tif'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating Data Generators for CNN\ntrain_datagen = ImageDataGenerator(\n    rescale=1/255,\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\nvalid_datagen = ImageDataGenerator(rescale=1/255)\n\ntrain_df['label'] = train_df['label'].astype(str)\nvalid_df['label'] = valid_df['label'].astype(str)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=img_dir,\n    x_col='id',\n    y_col='label',\n    target_size=(64, 64),\n    batch_size=32,\n    class_mode='categorical'\n)\n\nvalidation_generator = valid_datagen.flow_from_dataframe(\n    dataframe=valid_df,\n    directory=img_dir,\n    x_col='id',\n    y_col='label',\n    target_size=(64, 64),\n    batch_size=32,\n    class_mode='categorical'\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_generator)\nVA_STEPS = len(validation_generator)\n\nprint('Number of batches in the training set:',TR_STEPS)\nprint('Number of batches in the validation set:',VA_STEPS)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential([\n    \n    Conv2D(32, (3,3), activation = 'relu', padding = 'same', input_shape=(64, 64, 3)),\n    Conv2D(32, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.25),\n    BatchNormalization(),\n\n    Conv2D(64, (3,3), activation = 'relu', padding = 'same'),\n    Conv2D(64, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.25),\n    BatchNormalization(),\n    \n    Conv2D(128, (3,3), activation = 'relu', padding = 'same'),\n    Conv2D(128, (3,3), activation = 'relu', padding = 'same'),\n    MaxPooling2D(2,2),\n    Dropout(0.25),\n    BatchNormalization(),\n\n    Flatten(),\n    \n    Dense(128, activation='relu'),\n    Dropout(0.5),\n    Dense(64, activation='relu'),\n    Dropout(0.25),\n    BatchNormalization(),\n    Dense(2, activation='softmax')\n])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compiling the model\nopt = tf.keras.optimizers.Adam(0.001)\nmodel.compile(loss='binary_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the model\nh1 = model.fit(\n    train_generator,\n    steps_per_epoch = TR_STEPS,\n    epochs=4,\n    validation_data=validation_generator, \n    validation_steps = VA_STEPS, \n    verbose=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = h1.history\nepoch_range =range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\nplt.subplot(1,2,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,2,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('hcd_test_model.h5')\npickle.dump(history, open(f'hcd_test_history.pkl', 'wb'))","metadata":{},"execution_count":null,"outputs":[]}]}