{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30615,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%time\n#import packages\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import train_test_split\nimport pickle\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom kerastuner.tuners import RandomSearch\nfrom kerastuner.engine.hyperparameters import HyperParameters","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:37:48.687283Z","iopub.execute_input":"2023-12-08T09:37:48.688013Z","iopub.status.idle":"2023-12-08T09:38:00.682937Z","shell.execute_reply.started":"2023-12-08T09:37:48.687973Z","shell.execute_reply":"2023-12-08T09:38:00.681967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train and test folder\nprint('Number of images in train set',len(os.listdir('../input/histopathologic-cancer-detection/train')))\nprint('Number of images in test set',len(os.listdir('../input/histopathologic-cancer-detection/test')))","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:38:00.684712Z","iopub.execute_input":"2023-12-08T09:38:00.685315Z","iopub.status.idle":"2023-12-08T09:38:02.625645Z","shell.execute_reply.started":"2023-12-08T09:38:00.685286Z","shell.execute_reply":"2023-12-08T09:38:02.624509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the training data into a DataFrame. \n# Print the shape of the resulting DataFrame.\n\nhcd = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\nprint(hcd.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:38:02.626879Z","iopub.execute_input":"2023-12-08T09:38:02.627242Z","iopub.status.idle":"2023-12-08T09:38:02.969947Z","shell.execute_reply.started":"2023-12-08T09:38:02.627213Z","shell.execute_reply":"2023-12-08T09:38:02.968844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Adding a variable for the image directory\nimg_dir = '/kaggle/input/histopathologic-cancer-detection/train'","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:38:02.972938Z","iopub.execute_input":"2023-12-08T09:38:02.973352Z","iopub.status.idle":"2023-12-08T09:38:02.979888Z","shell.execute_reply.started":"2023-12-08T09:38:02.973314Z","shell.execute_reply":"2023-12-08T09:38:02.977173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the first few rows of the dataframe.\nhcd.head() ","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:38:02.987099Z","iopub.execute_input":"2023-12-08T09:38:02.987441Z","iopub.status.idle":"2023-12-08T09:38:03.058551Z","shell.execute_reply.started":"2023-12-08T09:38:02.987414Z","shell.execute_reply":"2023-12-08T09:38:03.057264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#label distrobution\n(hcd.label.value_counts() / len(hcd)).to_frame()","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:38:03.059786Z","iopub.execute_input":"2023-12-08T09:38:03.060362Z","iopub.status.idle":"2023-12-08T09:38:03.079400Z","shell.execute_reply.started":"2023-12-08T09:38:03.060333Z","shell.execute_reply":"2023-12-08T09:38:03.078337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = hcd.sample(n=9).reset_index()\n\nplt.figure(figsize=(3,3))\n\nfor i, row in sample.iterrows():\n\n    img = mpimg.imread(f'{img_dir}/{row.id}.tif')    \n    label = row.label\n\n    plt.subplot(3,3,i+1)\n    plt.imshow(img)\n    plt.text(0, -5, f'Class {label}', color='k')\n        \n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:38:03.080579Z","iopub.execute_input":"2023-12-08T09:38:03.080854Z","iopub.status.idle":"2023-12-08T09:38:03.592308Z","shell.execute_reply.started":"2023-12-08T09:38:03.080829Z","shell.execute_reply":"2023-12-08T09:38:03.591196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using data generators \ntrain_df, valid_df = train_test_split(hcd, test_size=0.2, random_state=39, stratify=hcd.label)\n\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:38:03.593333Z","iopub.execute_input":"2023-12-08T09:38:03.594729Z","iopub.status.idle":"2023-12-08T09:38:03.705636Z","shell.execute_reply.started":"2023-12-08T09:38:03.594681Z","shell.execute_reply":"2023-12-08T09:38:03.704792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['id'] = train_df['id'] + '.tif'\nvalid_df['id'] = valid_df['id'] + '.tif'\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:38:03.706718Z","iopub.execute_input":"2023-12-08T09:38:03.707001Z","iopub.status.idle":"2023-12-08T09:38:03.763849Z","shell.execute_reply.started":"2023-12-08T09:38:03.706975Z","shell.execute_reply":"2023-12-08T09:38:03.762823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating Data Generators for CNN\ntrain_datagen = ImageDataGenerator(\n    rescale=1/255,\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\n\nvalid_datagen = ImageDataGenerator(rescale=1/255)\n\ntrain_df['label'] = train_df['label'].astype(str)\nvalid_df['label'] = valid_df['label'].astype(str)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=img_dir,\n    x_col='id',\n    y_col='label',\n    target_size=(96, 96),\n    batch_size=32,\n    class_mode='binary'\n)\n\nvalidation_generator = valid_datagen.flow_from_dataframe(\n    dataframe=valid_df,\n    directory=img_dir,\n    x_col='id',\n    y_col='label',\n    target_size=(96, 96),\n    batch_size=32,\n    class_mode='binary'\n)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:38:03.764943Z","iopub.execute_input":"2023-12-08T09:38:03.765259Z","iopub.status.idle":"2023-12-08T09:47:32.765554Z","shell.execute_reply.started":"2023-12-08T09:38:03.765232Z","shell.execute_reply":"2023-12-08T09:47:32.764538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TR_STEPS = len(train_generator)\nVA_STEPS = len(validation_generator)\n\nprint('Number of batches in the training set:',TR_STEPS)\nprint('Number of batches in the validation set:',VA_STEPS)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:47:32.767025Z","iopub.execute_input":"2023-12-08T09:47:32.767897Z","iopub.status.idle":"2023-12-08T09:47:32.773296Z","shell.execute_reply.started":"2023-12-08T09:47:32.767855Z","shell.execute_reply":"2023-12-08T09:47:32.772443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndef build_model(hp):\n    model = Sequential()\n\n    # Convolutional layers\n    model.add(Conv2D(hp.Int('conv1_units', min_value=32, max_value=128, step=32), (3, 3), activation='relu', input_shape=(96, 96, 3)))\n    model.add(MaxPooling2D((2, 2)))\n\n    model.add(Conv2D(hp.Int('conv2_units', min_value=32, max_value=128, step=32), (3, 3), activation='relu'))\n    model.add(MaxPooling2D((2, 2)))\n\n    model.add(Conv2D(hp.Int('conv3_units', min_value=32, max_value=128, step=32), (3, 3), activation='relu'))\n    model.add(MaxPooling2D((2, 2)))\n\n    model.add(Flatten())\n\n    # Dense layers\n    model.add(Dense(hp.Int('dense_units', min_value=32, max_value=64, step=32), activation='relu'))\n\n    # Dropout layer\n    model.add(Dropout(hp.Float('dropout_rate', min_value=0.2, max_value=0.5, step=0.1)))\n\n    # Output layer\n    model.add(Dense(1, activation='sigmoid'))\n\n    # Compile the model\n    opt = tf.keras.optimizers.Adam(hp.Choice('learning_rate', values=[0.0001, 0.001,1.0]))\n    model.compile(loss='binary_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:47:32.774653Z","iopub.execute_input":"2023-12-08T09:47:32.775073Z","iopub.status.idle":"2023-12-08T09:47:32.798993Z","shell.execute_reply.started":"2023-12-08T09:47:32.775037Z","shell.execute_reply":"2023-12-08T09:47:32.798167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n#finding the best learning rate \ntuner = RandomSearch(\n    build_model,\n    objective='val_accuracy',\n    max_trials=3,  \n    directory='HCD_tuner_dir',  \n    project_name='HCD_SB_6'\n)\n\ntuner.search(train_generator, epochs=5, validation_data=validation_generator, validation_steps=VA_STEPS)\n\n# Get the best hyperparameters\nbest_hps = tuner.get_best_hyperparameters(num_trials=1)[0]","metadata":{"execution":{"iopub.status.busy":"2023-12-08T09:47:32.800044Z","iopub.execute_input":"2023-12-08T09:47:32.800339Z","iopub.status.idle":"2023-12-08T12:58:05.591966Z","shell.execute_reply.started":"2023-12-08T09:47:32.800315Z","shell.execute_reply":"2023-12-08T12:58:05.590961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#apply the best hyperparameters\nfinal_model = tuner.hypermodel.build(best_hps)","metadata":{"execution":{"iopub.status.busy":"2023-12-08T12:58:05.595589Z","iopub.execute_input":"2023-12-08T12:58:05.595912Z","iopub.status.idle":"2023-12-08T12:58:05.696123Z","shell.execute_reply.started":"2023-12-08T12:58:05.595881Z","shell.execute_reply":"2023-12-08T12:58:05.695157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#using the best hyperparameters and running it through epochs to train it\n\n\nh1 = final_model.fit(\n    train_generator,\n    steps_per_epoch=TR_STEPS,\n    epochs=5,\n    validation_data=validation_generator,\n    validation_steps=VA_STEPS,\n    verbose=1\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-08T12:58:05.697561Z","iopub.execute_input":"2023-12-08T12:58:05.697949Z","iopub.status.idle":"2023-12-08T13:57:50.186548Z","shell.execute_reply.started":"2023-12-08T12:58:05.697909Z","shell.execute_reply":"2023-12-08T13:57:50.185454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#displaying the data \n\nhistory = h1.history\nepoch_range =range(1, len(history['loss'])+1)\n\nplt.figure(figsize=[14,4])\nplt.subplot(1,2,1)\nplt.plot(epoch_range, history['loss'], label='Training')\nplt.plot(epoch_range, history['val_loss'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Loss'); plt.title('Loss')\nplt.legend()\nplt.subplot(1,2,2)\nplt.plot(epoch_range, history['accuracy'], label='Training')\nplt.plot(epoch_range, history['val_accuracy'], label='Validation')\nplt.xlabel('Epoch'); plt.ylabel('Accuracy'); plt.title('Accuracy')\nplt.legend()\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-08T13:57:50.188195Z","iopub.execute_input":"2023-12-08T13:57:50.188966Z","iopub.status.idle":"2023-12-08T13:57:50.820007Z","shell.execute_reply.started":"2023-12-08T13:57:50.188923Z","shell.execute_reply":"2023-12-08T13:57:50.819149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the model\nfinal_model.save('HCD_Test_1_Model.h5')\n# Save the history\npickle.dump(history, open(f'HCD_Test_1_Model_history.pkl', 'wb'))","metadata":{"execution":{"iopub.status.busy":"2023-12-08T13:57:50.821316Z","iopub.execute_input":"2023-12-08T13:57:50.821616Z","iopub.status.idle":"2023-12-08T13:57:50.868809Z","shell.execute_reply.started":"2023-12-08T13:57:50.821588Z","shell.execute_reply":"2023-12-08T13:57:50.867968Z"},"trusted":true},"execution_count":null,"outputs":[]}]}