{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:17:28.853437Z","iopub.execute_input":"2025-03-15T09:17:28.853733Z","iopub.status.idle":"2025-03-15T09:17:28.857421Z","shell.execute_reply.started":"2025-03-15T09:17:28.853708Z","shell.execute_reply":"2025-03-15T09:17:28.856513Z"}},"outputs":[],"execution_count":1},{"cell_type":"code","source":"import os\nimport json\nfrom pathlib import Path\nfrom tqdm.notebook import trange, tqdm\nimport copy\nimport random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score, confusion_matrix, classification_report\n\nimport torch\nfrom torch.utils.data import Dataset, DataLoader, Subset, random_split\nfrom torch.optim import Adam\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision.transforms as transforms\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nrandom.seed(2025)\ntorch.manual_seed(2025)\n\nsns.set_context('notebook')\nsns.set_style('white')\n\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:17:28.864356Z","iopub.execute_input":"2025-03-15T09:17:28.864657Z","iopub.status.idle":"2025-03-15T09:17:35.51049Z","shell.execute_reply.started":"2025-03-15T09:17:28.864626Z","shell.execute_reply":"2025-03-15T09:17:35.509783Z"}},"outputs":[],"execution_count":2},{"cell_type":"code","source":"if torch.cuda.is_available():\n    print(f\"Compatible GPU ({torch.cuda.get_device_name()}) found\")\nelse:\n    print(f\"No compatible GPU found.\")\n    \ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:17:35.5114Z","iopub.execute_input":"2025-03-15T09:17:35.511761Z","iopub.status.idle":"2025-03-15T09:17:35.601726Z","shell.execute_reply.started":"2025-03-15T09:17:35.51174Z","shell.execute_reply":"2025-03-15T09:17:35.601014Z"}},"outputs":[{"name":"stdout","text":"Compatible GPU (Tesla P100-PCIE-16GB) found\n","output_type":"stream"}],"execution_count":3},{"cell_type":"code","source":"class CancerDataset(Dataset):\n    \"\"\"\n    Custom Dataset for loading histopathologic cancer detection images.\n\n    Args:\n        data_dir (str or Path): Root directory containing the image data and labels.\n        transform (callable, optional): Transformation function to apply to the images.\n        data_type (str): Directory selection - \"train\", \"test\", or \"val\".\n        num_samples (int, optional): Number of samples to randomly select from the dataset.\n    \"\"\"\n    \n    def __init__(self, data_dir, transform=None, data_type=\"train\"):\n        self.data_dir = Path(data_dir)\n        self.data_type = data_type\n        self.transform = transform\n\n        # Define the image directory based on the data_type (e.g., train, test)\n        image_dir = self.data_dir / data_type\n        if not image_dir.exists():\n            raise FileNotFoundError(f\"Directory '{image_dir}' not found.\")        \n        \n        # Load all valid image files (.tif) in the directory\n        all_files = list(image_dir.glob(\"*.tif\"))\n\n        # No sample\n        num_samples = len(all_files)\n        \n        if num_samples > len(all_files):\n            raise ValueError(f\"num_samples ({num_samples}) exceeds available images ({len(all_files)}).\")\n        \n        # Randomly select a subset of image files\n        self.full_filenames = np.random.choice(all_files, num_samples, replace=False).tolist()\n\n        # Load labels from a CSV file (expects columns 'id' and 'label')\n        labels_file = self.data_dir / \"train_labels.csv\"\n        if not labels_file.exists():\n            raise FileNotFoundError(f\"Labels file '{labels_file}' not found.\")\n        \n        labels_df = pd.read_csv(labels_file).set_index(\"id\")\n        self.labels = [labels_df.loc[img.stem].values[0] for img in self.full_filenames]\n\n    def __len__(self):\n        # Return the number of samples in the dataset\n        return len(self.full_filenames)\n\n    def __getitem__(self, idx):\n        # Retrieve the image path and corresponding label using the index\n        img_path = self.full_filenames[idx]\n        image = Image.open(img_path).convert(\"RGB\")  # Ensure image is in RGB format\n    \n        if self.transform:\n            image = self.transform(image)\n    \n        label = self.labels[idx]\n        img_id = img_path.stem  # Extract the image ID (filename without extension)\n        return image, label, img_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:17:35.603353Z","iopub.execute_input":"2025-03-15T09:17:35.603576Z","iopub.status.idle":"2025-03-15T09:17:35.61053Z","shell.execute_reply.started":"2025-03-15T09:17:35.603555Z","shell.execute_reply":"2025-03-15T09:17:35.609808Z"}},"outputs":[],"execution_count":4},{"cell_type":"code","source":"# Define transformations for training and validation datasets\ntrain_transforms = transforms.Compose([\n    transforms.RandomHorizontalFlip(p=0.5),\n    transforms.RandomVerticalFlip(p=0.5),\n    transforms.RandomRotation(45),\n    transforms.ToTensor()\n])\n\nval_transforms = transforms.Compose([\n    transforms.ToTensor()\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:17:35.611576Z","iopub.execute_input":"2025-03-15T09:17:35.611882Z","iopub.status.idle":"2025-03-15T09:17:35.629151Z","shell.execute_reply.started":"2025-03-15T09:17:35.611822Z","shell.execute_reply":"2025-03-15T09:17:35.628389Z"}},"outputs":[],"execution_count":5},{"cell_type":"code","source":"# Set the path to the dataset directory\ndata_dir = '/kaggle/input/histopathologic-cancer-detection/'\n\n# Initialize the CancerDataset for training data\ndataset = CancerDataset(data_dir, transform=train_transforms, data_type=\"train\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:17:35.629983Z","iopub.execute_input":"2025-03-15T09:17:35.630262Z","iopub.status.idle":"2025-03-15T09:18:03.613002Z","shell.execute_reply.started":"2025-03-15T09:17:35.630234Z","shell.execute_reply":"2025-03-15T09:18:03.612098Z"}},"outputs":[],"execution_count":6},{"cell_type":"code","source":"# Create an array of indices for the entire dataset\nindices = np.arange(len(dataset))\n\n# First split: Separate 20% (temporary set) from the 80% training set\ntrain_indices, temp_indices = train_test_split(indices, test_size=0.2, random_state=2025)\n\n# Second split: Divide the temporary set into two halves for validation and test (10% each)\nval_indices, test_indices = train_test_split(temp_indices, test_size=0.5, random_state=2025)\n\n# Create subsets for training, validation, and testing using the indices\ntrain_dataset = Subset(dataset, train_indices)\nval_dataset = Subset(dataset, val_indices)\ntest_dataset = Subset(dataset, test_indices)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:18:03.613866Z","iopub.execute_input":"2025-03-15T09:18:03.614135Z","iopub.status.idle":"2025-03-15T09:18:03.629047Z","shell.execute_reply.started":"2025-03-15T09:18:03.614104Z","shell.execute_reply":"2025-03-15T09:18:03.628241Z"}},"outputs":[],"execution_count":7},{"cell_type":"code","source":"# Apply appropriate transformations\ntrain_dataset.dataset.transform = train_transforms\nval_dataset.dataset.transform = val_transforms\ntest_dataset.dataset.transform = val_transforms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:18:03.629887Z","iopub.execute_input":"2025-03-15T09:18:03.63015Z","iopub.status.idle":"2025-03-15T09:18:03.633738Z","shell.execute_reply.started":"2025-03-15T09:18:03.630131Z","shell.execute_reply":"2025-03-15T09:18:03.633053Z"}},"outputs":[],"execution_count":8},{"cell_type":"code","source":"# Print dataset sizes for verification\nprint(f\"Training dataset size: {len(train_dataset)}\")\nprint(f\"Validation dataset size: {len(val_dataset)}\")\nprint(f\"Test dataset size: {len(test_dataset)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:18:03.63458Z","iopub.execute_input":"2025-03-15T09:18:03.634909Z","iopub.status.idle":"2025-03-15T09:18:03.650806Z","shell.execute_reply.started":"2025-03-15T09:18:03.63488Z","shell.execute_reply":"2025-03-15T09:18:03.649928Z"}},"outputs":[{"name":"stdout","text":"Training dataset size: 176020\nValidation dataset size: 22002\nTest dataset size: 22003\n","output_type":"stream"}],"execution_count":9},{"cell_type":"code","source":"# Define DataLoaders for training and validation\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True, num_workers=4)\nval_loader = DataLoader(val_dataset, batch_size=32, shuffle=False, num_workers=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:18:03.652633Z","iopub.execute_input":"2025-03-15T09:18:03.652847Z","iopub.status.idle":"2025-03-15T09:18:03.669303Z","shell.execute_reply.started":"2025-03-15T09:18:03.652828Z","shell.execute_reply":"2025-03-15T09:18:03.668557Z"}},"outputs":[],"execution_count":10},{"cell_type":"code","source":"def plot_sample_images(dataset, data_dir, data_type=\"train\", num_images=12, title=\"Sample Images\"):\n    \"\"\"\n    Plots a grid of sample images from the dataset to visualize their appearance.\n\n    Args:\n        dataset (Dataset or Subset): The dataset object (train, validation, or test).\n        data_dir (str or Path): Root directory containing image data and labels.\n        data_type (str): Directory selection - \"train\", \"test\", or \"val\".\n        num_images (int): Number of images to display in the grid.\n        title (str): Title of the plot.\n    \n    Displays:\n        A matplotlib figure with sample images and their corresponding labels.\n    \"\"\"\n\n    # Define the grid layout (3 rows, num_images/3 columns)\n    fig, axes = plt.subplots(nrows=3, ncols=num_images // 3, figsize=(15, 7))\n    fig.suptitle(title, fontsize=16, fontweight='bold')\n\n    for ax in axes.flat:\n        # Select a random image index\n        idx = random.randint(0, len(dataset) - 1)\n\n        # Retrieve image and label (handling test set separately)\n        if data_type == \"test\":\n            filename, label = dataset[idx]\n        else:\n            img, label, filename = dataset[idx]        \n\n        # Load and display the image\n        img_path = os.path.join(data_dir, data_type, filename + '.tif')\n        ax.imshow(Image.open(img_path))\n        ax.set_title(f\"{'Cancer' if label == 1 else 'Normal'}\", fontsize=10)\n        ax.axis(\"off\")\n\n    # Adjust layout to prevent overlap\n    plt.tight_layout()\n    plt.subplots_adjust(top=0.9)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:18:03.670067Z","iopub.execute_input":"2025-03-15T09:18:03.670332Z","iopub.status.idle":"2025-03-15T09:18:03.684693Z","shell.execute_reply.started":"2025-03-15T09:18:03.670302Z","shell.execute_reply":"2025-03-15T09:18:03.68387Z"}},"outputs":[],"execution_count":11},{"cell_type":"code","source":"# Plot sample images from training and validation sets\nplot_sample_images(train_dataset, data_dir, data_type=\"train\", title=\"Training Set Samples\")\nplot_sample_images(val_dataset, data_dir, data_type=\"train\", title=\"Validation Set Samples\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T09:18:03.685469Z","iopub.execute_input":"2025-03-15T09:18:03.685721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}