{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import necessary libraries\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nimport seaborn as sns\nfrom PIL import Image\nfrom tqdm.notebook import tqdm\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, Input","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:28.453684Z","iopub.execute_input":"2024-11-05T08:36:28.454051Z","iopub.status.idle":"2024-11-05T08:36:28.461991Z","shell.execute_reply.started":"2024-11-05T08:36:28.454017Z","shell.execute_reply":"2024-11-05T08:36:28.461158Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = pd.read_csv(\"/kaggle/input/histopathologic-cancer-detection/train_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:28.462870Z","iopub.execute_input":"2024-11-05T08:36:28.463177Z","iopub.status.idle":"2024-11-05T08:36:28.708065Z","shell.execute_reply.started":"2024-11-05T08:36:28.463146Z","shell.execute_reply":"2024-11-05T08:36:28.707066Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = '/kaggle/input/histopathologic-cancer-detection/train'","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:28.709245Z","iopub.execute_input":"2024-11-05T08:36:28.709568Z","iopub.status.idle":"2024-11-05T08:36:28.713567Z","shell.execute_reply.started":"2024-11-05T08:36:28.709533Z","shell.execute_reply":"2024-11-05T08:36:28.712664Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\ndef show_samples(label, num_samples=7):\n    # Mapping from numeric labels to descriptive strings\n    label_mapping = {0: 'No Cancer', 1: 'Cancer'}\n    label_name = label_mapping.get(label, 'Unknown')  # Get the descriptive label name\n\n    # Get a random sample of images with the specified label\n    sample_images = labels[labels['label'] == label].sample(num_samples)\n    plt.figure(figsize=(14, 2))\n    \n    for i, img_name in enumerate(sample_images['id']):\n        img_path = os.path.join(train_dir, img_name + '.tif')  # Construct the image path\n        img = Image.open(img_path)  # Open the image\n        plt.subplot(1, num_samples, i + 1)  \n        plt.imshow(img) \n        plt.axis('off')  \n        \n    plt.suptitle(f'Sample Images with : {label_name}') \n    plt.show()\n\n\nshow_samples(label=0)  # Show samples for 'No Cancer'\nshow_samples(label=1)  # Show samples for 'Cancer'\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:28.714699Z","iopub.execute_input":"2024-11-05T08:36:28.714981Z","iopub.status.idle":"2024-11-05T08:36:29.566632Z","shell.execute_reply.started":"2024-11-05T08:36:28.714939Z","shell.execute_reply":"2024-11-05T08:36:29.565738Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(x='label', data=labels)\nplt.title('Class Distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:29.570573Z","iopub.execute_input":"2024-11-05T08:36:29.571245Z","iopub.status.idle":"2024-11-05T08:36:29.801531Z","shell.execute_reply.started":"2024-11-05T08:36:29.571197Z","shell.execute_reply":"2024-11-05T08:36:29.800659Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:29.803037Z","iopub.execute_input":"2024-11-05T08:36:29.803374Z","iopub.status.idle":"2024-11-05T08:36:29.831190Z","shell.execute_reply.started":"2024-11-05T08:36:29.803338Z","shell.execute_reply":"2024-11-05T08:36:29.830341Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_train_images = len(os.listdir(train_dir))\n\nprint(f'{num_train_images} pictures in train.')\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:29.832270Z","iopub.execute_input":"2024-11-05T08:36:29.832558Z","iopub.status.idle":"2024-11-05T08:36:31.926976Z","shell.execute_reply.started":"2024-11-05T08:36:29.832526Z","shell.execute_reply":"2024-11-05T08:36:31.925841Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"countzero = 0\ncountone = 0\n\nfor i in range(220025):\n    if labels['label'][i] == 1:\n        countone += 1\n    else:\n        countzero += 1","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:31.928389Z","iopub.execute_input":"2024-11-05T08:36:31.928805Z","iopub.status.idle":"2024-11-05T08:36:34.133249Z","shell.execute_reply.started":"2024-11-05T08:36:31.928754Z","shell.execute_reply":"2024-11-05T08:36:34.132445Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(countone,countzero)","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:34.134326Z","iopub.execute_input":"2024-11-05T08:36:34.134636Z","iopub.status.idle":"2024-11-05T08:36:34.139265Z","shell.execute_reply.started":"2024-11-05T08:36:34.134603Z","shell.execute_reply":"2024-11-05T08:36:34.138421Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dir = \"/kaggle/input/histopathologic-cancer-detection/test\"","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:34.140372Z","iopub.execute_input":"2024-11-05T08:36:34.140656Z","iopub.status.idle":"2024-11-05T08:36:34.148745Z","shell.execute_reply.started":"2024-11-05T08:36:34.140626Z","shell.execute_reply":"2024-11-05T08:36:34.147975Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:34.149784Z","iopub.execute_input":"2024-11-05T08:36:34.150076Z","iopub.status.idle":"2024-11-05T08:36:34.166759Z","shell.execute_reply.started":"2024-11-05T08:36:34.150044Z","shell.execute_reply":"2024-11-05T08:36:34.165907Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_baseline_model():\n    model = Sequential([\n        Input(shape=(96, 96, 3)),  # Input layer with specified input shape\n        Conv2D(32, (3,3), activation='relu'),\n        MaxPooling2D(2,2),\n        Conv2D(64, (3,3), activation='relu'),\n        MaxPooling2D(2,2),\n        Flatten(),\n        Dense(128, activation='relu'),\n        Dense(1, activation='sigmoid')\n    ])\n    return model\n\nbaseline_model = create_baseline_model()\nbaseline_model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:34.167800Z","iopub.execute_input":"2024-11-05T08:36:34.168080Z","iopub.status.idle":"2024-11-05T08:36:34.292372Z","shell.execute_reply.started":"2024-11-05T08:36:34.168049Z","shell.execute_reply":"2024-11-05T08:36:34.291512Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the advanced model with the same change\ndef create_advanced_model():\n    model = Sequential([\n        Input(shape=(96, 96, 3)),  # Input layer with specified input shape\n        Conv2D(32, (3,3), activation='relu'),\n        MaxPooling2D(2,2),\n        Dropout(0.2),\n        Conv2D(64, (3,3), activation='relu'),\n        MaxPooling2D(2,2),\n        Dropout(0.2),\n        Conv2D(128, (3,3), activation='relu'),\n        MaxPooling2D(2,2),\n        Flatten(),\n        Dense(256, activation='relu'),\n        Dropout(0.5),\n        Dense(1, activation='sigmoid')\n    ])\n    return model\n\nadvanced_model = create_advanced_model()\nadvanced_model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:34.293660Z","iopub.execute_input":"2024-11-05T08:36:34.293998Z","iopub.status.idle":"2024-11-05T08:36:34.392897Z","shell.execute_reply.started":"2024-11-05T08:36:34.293962Z","shell.execute_reply":"2024-11-05T08:36:34.391974Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Splitting labels into training and validation\n\n\nsubset_0 = labels[labels['label'] == 0].sample(frac=0.01, random_state=42)\nsubset_1 = labels[labels['label'] == 1].sample(frac=0.01, random_state=42)\n\n# Split each label subset separately to maintain equal distribution in train and validation sets\ntrain_0, val_0 = train_test_split(subset_0, test_size=0.2, random_state=42)\ntrain_1, val_1 = train_test_split(subset_1, test_size=0.2, random_state=42)\n\n# Combine the separate train and validation splits\ntrain_labels = pd.concat([train_0, train_1]).sample(frac=1, random_state=42)  # Shuffle to mix labels\nval_labels = pd.concat([val_0, val_1]).sample(frac=1, random_state=42)\n\n\ntrain_labels['label'] = train_labels['label'].astype(str)\nval_labels['label'] = val_labels['label'].astype(str)\n\n# Adding .tif extension to match file names\ntrain_labels['filename'] = train_labels['id'] + '.tif'\nval_labels['filename'] = val_labels['id'] + '.tif'","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:34.394156Z","iopub.execute_input":"2024-11-05T08:36:34.394471Z","iopub.status.idle":"2024-11-05T08:36:34.429577Z","shell.execute_reply.started":"2024-11-05T08:36:34.394438Z","shell.execute_reply":"2024-11-05T08:36:34.428602Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:34.430903Z","iopub.execute_input":"2024-11-05T08:36:34.431653Z","iopub.status.idle":"2024-11-05T08:36:34.440691Z","shell.execute_reply.started":"2024-11-05T08:36:34.431607Z","shell.execute_reply":"2024-11-05T08:36:34.439770Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:34.442047Z","iopub.execute_input":"2024-11-05T08:36:34.442351Z","iopub.status.idle":"2024-11-05T08:36:34.450747Z","shell.execute_reply.started":"2024-11-05T08:36:34.442317Z","shell.execute_reply":"2024-11-05T08:36:34.449900Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(val_labels)","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:34.451999Z","iopub.execute_input":"2024-11-05T08:36:34.452522Z","iopub.status.idle":"2024-11-05T08:36:34.461876Z","shell.execute_reply.started":"2024-11-05T08:36:34.452489Z","shell.execute_reply":"2024-11-05T08:36:34.460980Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. Create ImageDataGenerator for data augmentation\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,  # Normalize pixel values\n    horizontal_flip=True,  # Randomly flip images horizontally\n    vertical_flip=True  # Randomly flip images vertically\n)\nval_datagen = ImageDataGenerator(rescale=1./255)  # Validation data is only rescaled, no augmentation\n\n# Flow images from dataframe for training and validation\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_labels,\n    directory=train_dir,\n    x_col='filename',\n    y_col='label',\n    target_size=(96, 96),\n    batch_size=32,\n    class_mode='binary'\n)\n\nval_generator = val_datagen.flow_from_dataframe(\n    dataframe=val_labels,\n    directory=train_dir,\n    x_col='filename',\n    y_col='label',\n    target_size=(96, 96),\n    batch_size=32,\n    class_mode='binary'\n)","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:34.465528Z","iopub.execute_input":"2024-11-05T08:36:34.466072Z","iopub.status.idle":"2024-11-05T08:36:38.449525Z","shell.execute_reply.started":"2024-11-05T08:36:34.466039Z","shell.execute_reply":"2024-11-05T08:36:38.448491Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 5. Compile models to define optimizer, loss, and evaluation metrics\nbaseline_model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\nadvanced_model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:36:38.454252Z","iopub.execute_input":"2024-11-05T08:36:38.457139Z","iopub.status.idle":"2024-11-05T08:36:38.476609Z","shell.execute_reply.started":"2024-11-05T08:36:38.457088Z","shell.execute_reply":"2024-11-05T08:36:38.475684Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\n\n# Set your desired learning rate, batch size, and steps per epoch\nlearning_rate = 0.001  # Adjust this value as needed\nbatch_size = 32  # Set your batch size here\nsteps_per_epoch = 500  # Adjust this to control how many batches per epoch\n\n# Compile the model with the specified learning rate\noptimizer = Adam(learning_rate=learning_rate)  # Set the learning rate here\nbaseline_model.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory_baseline = baseline_model.fit(\n    train_generator,\n    epochs=5,\n    steps_per_epoch=steps_per_epoch,  # Control the number of batches per epoch\n    validation_data=val_generator\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T07:32:41.563970Z","iopub.execute_input":"2024-11-05T07:32:41.564371Z","iopub.status.idle":"2024-11-05T07:34:07.765608Z","shell.execute_reply.started":"2024-11-05T07:32:41.564334Z","shell.execute_reply":"2024-11-05T07:34:07.764794Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam\nimport tensorflow as tf\n\n# Set your desired learning rate, batch size, and steps per epoch\nlearning_rate = 0.0001  # Adjust this value as needed\nbatch_size = 32  # Set your batch size here\nsteps_per_epoch = 100  # Adjust this to control the number of batches per epoch\n\n# Compile and Train Advanced Model on GPU\nwith tf.device('/GPU:0'):  # Use GPU if available\n    optimizer = Adam(learning_rate=learning_rate)  # Set the learning rate here\n    advanced_model.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n    \n    history_advanced = advanced_model.fit(\n        train_generator,\n        epochs=10,\n        steps_per_epoch=steps_per_epoch,  # Control the number of batches per epoch\n        validation_data=val_generator\n    )\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T07:28:28.670299Z","iopub.execute_input":"2024-11-05T07:28:28.670632Z","iopub.status.idle":"2024-11-05T07:31:26.745239Z","shell.execute_reply.started":"2024-11-05T07:28:28.670596Z","shell.execute_reply":"2024-11-05T07:31:26.744423Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.optimizers import Adam\nimport matplotlib.pyplot as plt\nimport itertools\n\n# Define values for learning rate, constant steps per epoch, and batch size\nlearning_rates = [0.01, 0.001, 0.0001]\nsteps_per_epoch = 150  # Constant step size\nbatch_sizes = [32, 64]\n\n# Initialize a dictionary to store history for each configuration\nhistories = {}\n\n# Train model for each combination of learning rate and batch size\nfor lr, batch_size in itertools.product(learning_rates, batch_sizes):\n    with tf.device('/GPU:0'):  # Use GPU if available\n        # Configure optimizer and compile model\n        optimizer = Adam(learning_rate=lr)\n        advanced_model.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n        \n        # Train the model and store the history\n        history = advanced_model.fit(\n            train_generator,\n            epochs=5,\n            steps_per_epoch=steps_per_epoch,\n            validation_data=val_generator,\n            batch_size=batch_size,\n            validation_steps=50  # Adjust validation steps as necessary\n        )\n        \n        # Store the history with configuration details as key\n        key = f\"lr={lr}_steps={steps_per_epoch}_batch={batch_size}\"\n        histories[key] = history.history\n\n# Create a 3x2 grid layout for the configurations\nfig, axs = plt.subplots(3, 2, figsize=(12, 18))  # Adjust the figsize as needed\nfig.suptitle(\"Epoch vs. Accuracy and Loss for Different Configurations\", fontsize=16)\n\n# Plot the accuracy and loss for each configuration\nfor idx, (key, history) in enumerate(histories.items()):\n    # Determine subplot position\n    row = idx // 2  # Change to 2 columns\n    col = idx % 2\n\n    # Plot accuracy vs. epoch\n    axs[row, col].plot(history['accuracy'], label=\"Training Accuracy\")\n    axs[row, col].plot(history['val_accuracy'], label=\"Validation Accuracy\")\n    axs[row, col].set_title(f\"{key} (Accuracy)\")\n    axs[row, col].set_xlabel(\"Epoch\")\n    axs[row, col].set_ylabel(\"Accuracy\")\n    axs[row, col].legend()\n\n    # Plot loss vs. epoch in the same subplot\n    ax2 = axs[row, col].twinx()  # Create a second y-axis for the loss\n    ax2.plot(history['loss'], label=\"Training Loss\", linestyle='--', color=\"red\")\n    ax2.plot(history['val_loss'], label=\"Validation Loss\", linestyle='--', color=\"purple\")\n    ax2.set_ylabel(\"Loss\")\n    ax2.legend(loc=\"upper right\")\n\nplt.tight_layout(rect=[0, 0.03, 1, 0.95])\nplt.subplots_adjust(hspace=0.5)  # Increase vertical space between rows\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:39:56.085046Z","iopub.execute_input":"2024-11-05T08:39:56.085466Z","iopub.status.idle":"2024-11-05T08:48:11.197609Z","shell.execute_reply.started":"2024-11-05T08:39:56.085428Z","shell.execute_reply":"2024-11-05T08:48:11.196735Z"}},"outputs":[],"execution_count":null}]}