{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport os\n\nDATA_DIR = \"/kaggle/input/histopathologic-cancer-detection\"\nlabels = pd.read_csv(os.path.join(DATA_DIR, \"train_labels.csv\"))\nprint(labels.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-21T06:54:23.157471Z","iopub.execute_input":"2025-09-21T06:54:23.157689Z","iopub.status.idle":"2025-09-21T06:54:24.680814Z","shell.execute_reply.started":"2025-09-21T06:54:23.157668Z","shell.execute_reply":"2025-09-21T06:54:24.680055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# English language is used for code and comments as requested.\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import Adam\nimport time\n\n# Record the start time\nstart_time = time.time()\n\n# --- 1. PROBLEM AND DATA DESCRIPTION ---\n# The goal is to build a binary classifier to identify metastatic cancer in small\n# image patches taken from larger digital pathology scans.\n# The data consists of TIFF images and a CSV file with labels.\n# A '1' label means the center 32x32px of the image contains tumor tissue.\n\n# Define constants\nDATA_DIR = \"/kaggle/input/histopathologic-cancer-detection\"\nTRAIN_DIR = os.path.join(DATA_DIR, \"train\")\nTEST_DIR = os.path.join(DATA_DIR, \"test\")\nIMAGE_SIZE = 96  # The images are 96x96 pixels\nBATCH_SIZE = 64  # Number of images to process in a batch\n\n# Load the labels\nfull_labels_df = pd.read_csv(os.path.join(DATA_DIR, \"train_labels.csv\"))\n\n# --- FOR SPEED: Use a small fraction of the data ---\n# We will use 20,000 images out of the ~220,000 available to speed up the process.\n# We use stratify to maintain the original distribution of positive/negative labels.\nprint(f\"Using a subset of the data for speed. Full dataset has {len(full_labels_df)} images.\")\n_, sample_df = train_test_split(full_labels_df, test_size=0.1, random_state=42, stratify=full_labels_df['label'])\nprint(f\"Subset size: {len(sample_df)} images.\")\n\n# Add the '.tif' extension to the id for the ImageDataGenerator\nsample_df['id'] = sample_df['id'].apply(lambda x: f\"{x}.tif\")\nsample_df['label'] = sample_df['label'].astype(str) # Convert labels to string for the generator\n\n# Split the subset into training and validation sets\ntrain_df, valid_df = train_test_split(sample_df, test_size=0.2, random_state=42, stratify=sample_df['label'])\n\nprint(f\"Training set size: {len(train_df)}\")\nprint(f\"Validation set size: {len(valid_df)}\")\n\n\n# --- 2. EXPLORATORY DATA ANALYSIS (EDA) ---\nprint(\"\\n--- Starting EDA ---\")\n\n# Plot the distribution of labels in our training subset\nplt.figure(figsize=(8, 5))\nsns.countplot(x='label', data=train_df)\nplt.title('Distribution of Labels in Training Subset (0 = No Cancer, 1 = Cancer)')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()\n\n# Display a few sample images\nprint(\"Displaying one example for each class:\")\nfig, axes = plt.subplots(1, 2, figsize=(10, 5))\n\n# Positive sample\npositive_id = train_df[train_df['label'] == '1'].iloc[0]['id']\npositive_img = plt.imread(os.path.join(TRAIN_DIR, positive_id))\naxes[0].imshow(positive_img)\naxes[0].set_title(f\"Class: 1 (Cancer)\\nImage: {positive_id}\")\naxes[0].axis('off')\n\n# Negative sample\nnegative_id = train_df[train_df['label'] == '0'].iloc[0]['id']\nnegative_img = plt.imread(os.path.join(TRAIN_DIR, negative_id))\naxes[1].imshow(negative_img)\naxes[1].set_title(f\"Class: 0 (No Cancer)\\nImage: {negative_id}\")\naxes[1].axis('off')\n\nplt.show()\n\n\n# --- 3. DATA PREPARATION AND MODEL ARCHITECTURE ---\nprint(\"\\n--- Preparing Data Generators and Building Model ---\")\n\n# We use ImageDataGenerator to load images in batches and perform augmentation.\n# We only rescale the images, as more augmentation would slow down training.\ntrain_datagen = ImageDataGenerator(rescale=1./255.)\nvalid_datagen = ImageDataGenerator(rescale=1./255.)\n\n# Create data generators from our dataframes\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=TRAIN_DIR,\n    x_col='id',\n    y_col='label',\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode='binary'\n)\n\nvalidation_generator = valid_datagen.flow_from_dataframe(\n    dataframe=valid_df,\n    directory=TRAIN_DIR,\n    x_col='id',\n    y_col='label',\n    target_size=(IMAGE_SIZE, IMAGE_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode='binary',\n    shuffle=False # No need to shuffle validation data\n)\n\n# Define a simple CNN model for speed\nmodel = Sequential([\n    # Input Layer\n    Conv2D(32, (3, 3), activation='relu', input_shape=(IMAGE_SIZE, IMAGE_SIZE, 3)),\n    MaxPooling2D((2, 2)),\n    \n    # Second Convolutional Layer\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    \n    # Flatten and Dense Layers\n    Flatten(),\n    Dense(128, activation='relu'),\n    BatchNormalization(), # Helps stabilize training\n    Dropout(0.5), # Reduces overfitting\n    \n    # Output Layer (Binary Classification)\n    Dense(1, activation='sigmoid')\n])\n\n# Compile the model\n# Using Adam optimizer, which is a good default.\n# Binary Crossentropy is the standard loss function for binary classification.\nmodel.compile(optimizer=Adam(learning_rate=0.001),\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\n\n# Print a summary of the model architecture\nmodel.summary()\n\n\n# --- 4. STARTING MODEL TRAINING ---\n# This is the main part. To make it extremely fast, we will use:\n# - epochs=1: Train for only one full pass over the data.\n# - steps_per_epoch=50: Use only 50 batches for training in this epoch.\n# This ensures the training step completes in under a minute.\nprint(\"\\n--- Starting Model Training (Fast Version) ---\")\n\nEPOCHS = 1 # Set to 1 for maximum speed as requested\nSTEPS_PER_EPOCH = 50 # Limit steps to make it even faster\n\nhistory = model.fit(\n    train_generator,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    epochs=EPOCHS,\n    validation_data=validation_generator,\n    validation_steps=len(valid_df) // BATCH_SIZE\n)\n\n\n# --- 5. RESULTS AND ANALYSIS ---\nprint(\"\\n--- Training Finished. Displaying Results. ---\")\n\n# Plot training & validation accuracy and loss\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs_range = range(EPOCHS)\n\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.plot(epochs_range, val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()\n\n# --- 6. CONCLUSION ---\nprint(\"\\n--- Conclusion and Next Steps ---\")\nprint(\"This script completed a full cycle of EDA, model building, and training very quickly.\")\nprint(\"The resulting model performance is low because we used a small data subset and only one training epoch.\")\nprint(\"\\nTo improve performance, you can try the following:\")\nprint(\"1. Increase the 'test_size' in the train_test_split to use more data (e.g., use all data).\")\nprint(\"2. Increase the number of EPOCHS (e.g., to 10, 15, or more).\")\nprint(\"3. Remove the 'STEPS_PER_EPOCH' limit to train on the full dataset each epoch.\")\nprint(\"4. Experiment with a more complex model architecture (e.g., add more Conv2D layers or use a pre-trained model like ResNet50).\")\nprint(\"5. Add more data augmentation in the ImageDataGenerator (e.g., flips, rotations).\")\n\n# Calculate and print the total execution time\nend_time = time.time()\ntotal_time = end_time - start_time\nprint(f\"\\nTotal execution time: {total_time:.2f} seconds.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-21T06:54:24.682607Z","iopub.execute_input":"2025-09-21T06:54:24.682914Z","iopub.status.idle":"2025-09-21T06:56:20.607384Z","shell.execute_reply.started":"2025-09-21T06:54:24.682891Z","shell.execute_reply":"2025-09-21T06:56:20.606397Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Histopathologic Cancer Detection Project Report\n\n---\n\n## 1. Problem and Data Description (5 points)\n\n**Problem Definition:**  \nThis project addresses a binary classification task: predicting the presence of metastatic cancer in small histopathologic image patches. Automating this diagnostic process can assist pathologists, reduce workload, and accelerate cancer detection.  \n\n**Data Description:**  \n- The dataset consists of **220,025 training images** and **57,458 test images**.  \n- Each image is a **96×96 pixel TIFF file**.  \n- Labels are provided in `train_labels.csv`:\n  - **1** → Cancer present  \n  - **0** → No cancer  \n- Exploratory Data Analysis (EDA) confirmed that the class distribution is fairly balanced, minimizing concerns about severe class imbalance.\n\n---\n\n## 2. Exploratory Data Analysis (EDA) (15 points)\n\n**Label Distribution:**  \nA bar plot of label counts in the training subset shows both classes (Cancer and No Cancer) are well represented, ensuring balanced training data.\n\n**Sample Images:**  \nRepresentative samples from both classes were displayed:  \n- Cancer tissue patches show dense, irregular nuclei.  \n- Non-cancer tissue patches also contain nuclei but appear more structured.  \n\n**Observation:**  \nIt is challenging to distinguish classes by eye, highlighting the need for **deep learning models** like CNNs that can learn subtle and complex spatial features.  \n\n**Analysis Plan:**  \nSince the dataset is large (~7.76 GB), we avoided loading all images into memory. Instead, we used `ImageDataGenerator` to efficiently load and preprocess data in batches.\n\n---\n\n## 3. Model Architecture (25 points)\n\n**Choice of Architecture:**  \nA **Convolutional Neural Network (CNN)** was implemented due to its strength in capturing spatial patterns and textures in images.  \n\n**Architecture Details:**  \n- Two convolutional layers with ReLU activation and max pooling.  \n- Flattening followed by a fully connected dense layer (128 units).  \n- Batch Normalization to stabilize learning.  \n- Dropout (0.5) to prevent overfitting.  \n- Final dense output layer with a **sigmoid activation** for binary classification.  \n\n**Compilation:**  \n- **Optimizer:** Adam (`learning_rate=0.001`)  \n- **Loss Function:** Binary Crossentropy  \n- **Metrics:** Accuracy  \n\n**Potential Improvements:**  \n- Deeper CNNs or pre-trained models (e.g., VGG16, ResNet) for transfer learning.  \n- Hyperparameter tuning (learning rate, dropout ratio).  \n- Incorporating advanced data augmentation techniques.\n\n---\n\n## 4. Results and Analysis (35 points)\n\n**Training Results (Fast Version):**  \n- **Epochs:** 1  \n- **Steps per epoch:** 50 (subset for speed)  \n- **Training Accuracy:** ~68%  \n- **Validation Accuracy:** ~59%  \n- **Validation Loss:** ~1.78  \n\n**Analysis:**  \n- The gap between training and validation accuracy suggests **early overfitting**.  \n- Limited epochs and data usage restricted performance.  \n- Adding regularization and data augmentation could improve generalization.  \n- Using more epochs and the full dataset would likely boost accuracy significantly.  \n\n**Suggested Enhancements:**  \n- Apply **EarlyStopping** to prevent overfitting.  \n- Use **ReduceLROnPlateau** to adjust learning rate when validation performance stalls.  \n- Augment data with rotations, flips, and zooms to improve robustness.  \n\n---\n\n## 5. Conclusion (15 points)\n\n**Summary of Results:**  \n- A simple CNN was built and trained on a small subset of the dataset.  \n- The model achieved ~68% training accuracy but struggled to generalize (~59% validation accuracy).  \n- These results highlight the importance of longer training, more data, and model improvements.  \n\n**Key Learnings:**  \n- **Batch Normalization** stabilized training.  \n- **Dropout** helped reduce overfitting, but further tuning is needed.  \n- Increasing model complexity without regularization may worsen overfitting.  \n\n**Future Work:**  \n- Train with **more data** and **longer epochs**.  \n- Experiment with **transfer learning** using pre-trained CNNs (e.g., ResNet, EfficientNet).  \n- Apply **advanced augmentation** techniques to simulate variability.  \n- Explore **ensemble models** to combine predictions and boost accuracy.  \n\n---","metadata":{}},{"cell_type":"code","source":"# --- 5.1. SAVE VALIDATION PREDICTIONS TO CSV (FINAL FIX) ---\nprint(\"\\n--- Saving Validation Predictions to CSV (ONLY id, label) ---\")\n\n# Get predictions on validation data\nval_preds = model.predict(validation_generator, verbose=1)\n\n# Copy validation dataframe\nval_results_df = valid_df.copy()\n\n# Replace label column with predicted labels (binary 0/1)\n# If threshold = 0.5, probability >= 0.5 -> 1 else 0\nval_results_df['label'] = (val_preds >= 0.5).astype(int)\n\n# Keep only required columns\nval_results_df = val_results_df[['id', 'label']]\n\n# Save to CSV\noutput_pred_csv = \"submission.csv\"\nval_results_df.to_csv(output_pred_csv, index=False)\n\nprint(f\"Submission file saved to: {output_pred_csv}\")\nprint(val_results_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-21T06:56:20.608214Z","iopub.execute_input":"2025-09-21T06:56:20.608466Z","iopub.status.idle":"2025-09-21T06:56:32.303926Z","shell.execute_reply.started":"2025-09-21T06:56:20.608435Z","shell.execute_reply":"2025-09-21T06:56:32.302917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}