{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import of the required libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Import PyTorch\nimport torch\nimport torchvision.transforms as transforms #Image transformations in tensors\nfrom torch.utils.data import Subset #Dimensionality reduction\nimport torch.nn as nn  #Define neural network layers\nimport torch.nn.functional as F #Activation functions\nfrom torch.utils.data import TensorDataset, DataLoader, Dataset #Creation of training and validation datasets and dataloaders.\nimport torch.optim as optim #Definition of optimizers\n\n# Import sklearn\nfrom sklearn.model_selection import train_test_split #Division of data in training and validation\n\n# Matplotlib for visualization\nimport matplotlib.pyplot as plt\nplt.style.use(\"ggplot\")\n\n#Import PIL\nfrom PIL import Image #Image processing","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-01T11:00:26.599768Z","iopub.execute_input":"2024-10-01T11:00:26.600232Z","iopub.status.idle":"2024-10-01T11:00:26.610298Z","shell.execute_reply.started":"2024-10-01T11:00:26.600184Z","shell.execute_reply":"2024-10-01T11:00:26.608831Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Loading of training and validation datasets","metadata":{}},{"cell_type":"code","source":"print(os.listdir(\"../input/histopathologic-cancer-detection\"))","metadata":{"execution":{"iopub.status.busy":"2024-10-01T10:58:51.996142Z","iopub.execute_input":"2024-10-01T10:58:51.997008Z","iopub.status.idle":"2024-10-01T10:58:52.007916Z","shell.execute_reply.started":"2024-10-01T10:58:51.996955Z","shell.execute_reply":"2024-10-01T10:58:52.005621Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the file train_labels.csv in a DataFrame\ntrain_labels = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv')\n\n# Check the first rows\nprint(train_labels.head())\n\n#  Divide the train DataFrame into training and validation sets.\ntrain_data, val_data = train_test_split(train_labels, test_size=0.2, random_state=42, stratify=train_labels['label'])\n\n\nprint(f\"Training set size: {len(train_data)}\")\nprint(f\"Validation set size: {len(val_data)}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-01T10:58:55.155727Z","iopub.execute_input":"2024-10-01T10:58:55.156259Z","iopub.status.idle":"2024-10-01T10:58:55.803857Z","shell.execute_reply.started":"2024-10-01T10:58:55.156200Z","shell.execute_reply":"2024-10-01T10:58:55.802594Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data preview","metadata":{}},{"cell_type":"code","source":"# Distribution of training dataset values\nprint(train_data['label'].value_counts())\ntrain_data['label'].value_counts().plot(kind='bar',color=['green', 'red'])\nplt.title('Distribution of labels in the training set')\nplt.xlabel('Label')\nplt.ylabel('Number of images')\nplt.show()\n\n# Distribution of validation dataset values\nprint(val_data['label'].value_counts())\nval_data['label'].value_counts().plot(kind='bar',color=['green', 'red'])\nplt.title('Distribution of labels in the validation set')\nplt.xlabel('Label')\nplt.ylabel('Number of images')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T10:58:58.085434Z","iopub.execute_input":"2024-10-01T10:58:58.085938Z","iopub.status.idle":"2024-10-01T10:58:58.691923Z","shell.execute_reply.started":"2024-10-01T10:58:58.085888Z","shell.execute_reply":"2024-10-01T10:58:58.690146Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Directory where the images are located\ntrain_dir = '../input/histopathologic-cancer-detection/train/'\nimage_files = os.listdir(train_dir)\n\n# Show first 5 images\nfor img_file in image_files[:5]:  \n    img_path = os.path.join(train_dir, img_file)\n    \n    img = Image.open(img_path)\n    \n    plt.figure(figsize=(4, 2))\n    plt.imshow(img)\n    plt.title(f'Image: {img_file}')\n    plt.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T10:59:02.117172Z","iopub.execute_input":"2024-10-01T10:59:02.117858Z","iopub.status.idle":"2024-10-01T10:59:05.540001Z","shell.execute_reply.started":"2024-10-01T10:59:02.117794Z","shell.execute_reply":"2024-10-01T10:59:05.538645Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset and DataLoader Creation in Pytorch","metadata":{}},{"cell_type":"markdown","source":"### Dataset creation","metadata":{}},{"cell_type":"markdown","source":"Creating a dataset, we associate each image with its ID, which is then linked to its label (cancerous or non-cancerous). The ID serves as a unique identifier for each image, allowing us to match it with the correct label. The dataset is composed of two tensors: one is the result of applying transformations to the image (such as resizing, normalization, etc.), and the other provides the label associated with that image. These transformations ensure that the images are in the right format for the model, while the label tensor indicates the correct category, allowing the model to learn from the data during training.","metadata":{}},{"cell_type":"code","source":"# Define transformations: resize to 96x96, convert to tensor and normalize\ntransform = transforms.Compose([\n    transforms.Resize((96, 96)),  \n    transforms.ToTensor(),        \n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]) \n    #We could add more features to use the technique of Data augmentation.\n])\n\n# Create dataset to associate images and tags\nclass CancerImageDataset(Dataset):\n    def __init__(self, dataframe, img_dir, transform=None):\n        self.dataframe = dataframe  # DataFrame containing the IDs and the tags\n        self.img_dir = img_dir      # Directory where the images are located\n        self.transform = transform  # Transformations to be applied\n\n    def __len__(self):\n        # Returns the total number of images in the dataset\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        # Get the image ID and tag from the DataFrame\n        img_id = self.dataframe.iloc[idx]['id']  \n        label = self.dataframe.iloc[idx]['label']  \n        \n        # Build the complete image path\n        img_path = os.path.join(self.img_dir, img_id + '.tif')\n        \n        # Load image using PIL\n        image = Image.open(img_path)\n        \n        # Apply transformations \n        if self.transform:\n            image = self.transform(image)\n        \n        # Return the image and label as tensor\n        return image, torch.tensor(label, dtype=torch.long)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T10:59:10.702726Z","iopub.execute_input":"2024-10-01T10:59:10.703225Z","iopub.status.idle":"2024-10-01T10:59:10.716339Z","shell.execute_reply.started":"2024-10-01T10:59:10.703180Z","shell.execute_reply":"2024-10-01T10:59:10.714672Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create an instance of the dataset\nimg_dir = \"../input/histopathologic-cancer-detection/train\" \ntrain_dataset = CancerImageDataset(dataframe=train_data, img_dir=img_dir, transform=transform)\nval_dataset= CancerImageDataset(dataframe=val_data, img_dir=img_dir, transform=transform)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-01T10:59:13.834315Z","iopub.execute_input":"2024-10-01T10:59:13.834794Z","iopub.status.idle":"2024-10-01T10:59:13.842159Z","shell.execute_reply.started":"2024-10-01T10:59:13.834753Z","shell.execute_reply":"2024-10-01T10:59:13.840685Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First 5 images and labels\nfor i in range(5):\n    img, label = train_dataset[i]\n    print(f\"Image ID: {train_labels.iloc[i]['id']}, Label: {label.item()}\")\n    \n    # Visualization\n    plt.figure(figsize=(4, 2))\n    plt.imshow(img.permute(1, 2, 0))  # Change the order of the channels to display correctly\n    plt.title(f\"Label: {label.item()}\")\n    plt.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T10:59:16.232867Z","iopub.execute_input":"2024-10-01T10:59:16.234152Z","iopub.status.idle":"2024-10-01T10:59:17.439894Z","shell.execute_reply.started":"2024-10-01T10:59:16.234084Z","shell.execute_reply":"2024-10-01T10:59:17.438373Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Reduce dataset size","metadata":{}},{"cell_type":"markdown","source":"Due to compilation problems with the CPU, I have decided to reduce the size of the training and validation datasets (and later the test dataset), so that the code cells run smoothly, without the need for a more powerful CPU. Due to compilation problems with the CPU, I have decided to reduce the size of the training and validation datasets (and later the test dataset), so that the code cells run smoothly, without the need for a more powerful CPU. Therefore, our training and test results are biased by this condition. The model should be trained with all available images and in many more epochs to really evaluate its performance. ","metadata":{}},{"cell_type":"code","source":"import random\n\n# Initializing a random seed\nrandom.seed(42)\n\n# Get datasets size\ntrain_size = len(train_dataset)\nval_size = len(val_dataset)\n\n# Reduce each dataset to 50%.\ntrain_sample_size = int(train_size * 0.5)\nval_sample_size = int(val_size * 0.5)\n\n# Generate random indexes\ntrain_indexs = random.sample(range(train_size), train_sample_size)\nval_indexs = random.sample(range(val_size), val_sample_size)\n\n# Create subsets\ntrain_subset = Subset(train_dataset, train_indexs)\nval_subset = Subset(val_dataset, val_indexs)\n\n# Assign subsets to the original datasets\ntrain_dataset = train_subset\nval_dataset = val_subset\n","metadata":{"execution":{"iopub.status.busy":"2024-10-01T11:00:32.618715Z","iopub.execute_input":"2024-10-01T11:00:32.619174Z","iopub.status.idle":"2024-10-01T11:00:32.809347Z","shell.execute_reply.started":"2024-10-01T11:00:32.619132Z","shell.execute_reply":"2024-10-01T11:00:32.807845Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### DataLoaders creation","metadata":{}},{"cell_type":"markdown","source":"A dataloader helps us take the data from the dataset in small groups or batches (defined by batch_size), instead of loading all the data at once. This makes the training faster and uses less memory, as we don't need to load the entire dataset into memory all at once.","metadata":{}},{"cell_type":"code","source":"train_dataloader = DataLoader(train_dataset, batch_size=32, shuffle=True)\nval_dataloader = DataLoader(val_dataset, batch_size=32, shuffle=False)\n\n# Verify data upload\nfor imgs, labels in train_dataloader:\n    print(imgs.shape)  # Should show: (32, 3, 96, 96)\n    print(labels.shape)  # Should show: (32)\n    break  ","metadata":{"execution":{"iopub.status.busy":"2024-10-01T11:00:37.696893Z","iopub.execute_input":"2024-10-01T11:00:37.697395Z","iopub.status.idle":"2024-10-01T11:00:38.080310Z","shell.execute_reply.started":"2024-10-01T11:00:37.697355Z","shell.execute_reply":"2024-10-01T11:00:38.078724Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Convolutional neural network implementation","metadata":{}},{"cell_type":"markdown","source":"We are creating a Convolutional Neural Network (CNN) using PyTorch for a binary classification task. The CNN is composed of two convolutional layers followed by max-pooling layers, and two fully-connected layers. The convolutional layers extract important features from the input images, while the fully-connected layers use these features to make predictions. The number of layers and the parameters of each layer are adjustable and must be tuned to train different models. As well as the loss function and the optimizer.","metadata":{}},{"cell_type":"code","source":"# Define convolutional neural network model\n\nclass CNN(nn.Module):\n    def __init__(self):\n        super(CNN, self).__init__()\n        self.conv1 = nn.Conv2d(3, 16, kernel_size=3, stride=1, padding=1)  # Convolutional layer 1\n        self.pool = nn.MaxPool2d(kernel_size=2, stride=2, padding=0)  # Pooling layer\n        self.conv2 = nn.Conv2d(16, 32, kernel_size=3, stride=1, padding=1)  # Convolutional layer 2\n        self.fc1 = nn.Linear(32 * 24 * 24, 128)  # Fully-connected layer\n        self.dropout = nn.Dropout(p=0.5)  # Dropout layer with a probability of 0.5\n        self.fc2 = nn.Linear(128, 2)  # Output layer (2 labels: with and without cancer)\n\n    def forward(self, x):\n        x = self.pool(F.relu(self.conv1(x)))  # Apply conv1 and pooling\n        x = self.pool(F.relu(self.conv2(x)))  # Apply conv2 and pooling\n        x = x.view(-1, 32 * 24 * 24)  # Flatten the tensor\n        x = F.relu(self.fc1(x))  # Fully-connected layer\n        x = self.dropout(x)  # Apply dropout\n        x = self.fc2(x)  # Output layer\n        return x\n\n# Instantiate the model\nmodel = CNN()","metadata":{"execution":{"iopub.status.busy":"2024-10-01T11:01:18.021630Z","iopub.execute_input":"2024-10-01T11:01:18.023226Z","iopub.status.idle":"2024-10-01T11:01:18.077145Z","shell.execute_reply.started":"2024-10-01T11:01:18.023165Z","shell.execute_reply":"2024-10-01T11:01:18.075489Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#  Define the loss function and the optimizer\n\ncriterion = nn.CrossEntropyLoss()  \n\noptimizer = torch.optim.Adam(model.parameters(), lr=0.001)  ","metadata":{"execution":{"iopub.status.busy":"2024-10-01T11:01:21.009217Z","iopub.execute_input":"2024-10-01T11:01:21.009770Z","iopub.status.idle":"2024-10-01T11:01:21.017277Z","shell.execute_reply.started":"2024-10-01T11:01:21.009723Z","shell.execute_reply":"2024-10-01T11:01:21.015569Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model training","metadata":{}},{"cell_type":"code","source":"# Initialization of lists to store losses and precision\nloss_list = []\naccuracy_list = []\n\n# Model training\nnum_epochs = 10 \n\nfor epoch in range(num_epochs):\n    running_loss = 0.0  \n    for i, (images, labels) in enumerate(train_dataloader):\n        # Clean gradients\n        optimizer.zero_grad()\n\n        # Forward propagation\n        outputs = model(images)\n\n        # Calculate the loss\n        loss = criterion(outputs, labels)\n\n        # Calculate gradients\n        loss.backward()\n\n        # Update parameters\n        optimizer.step()\n        \n        # Accumulate loss\n        running_loss += loss.item()\n            \n    # Calculate accuracy\n    correct = 0\n    total = 0\n\n    # Iterate through the validation set\n    with torch.no_grad():\n        for val_images, val_labels in val_dataloader:\n            # Forward propagation in the validation set\n            val_outputs = model(val_images)\n\n            # Obtain predictions\n            predicted = torch.argmax(val_outputs.data, dim=1)\n\n            # Total number of correct labels\n            total += val_labels.size(0)\n            correct += (predicted == val_labels).sum().item() \n\n    accuracy = 100 * correct / float(total)\n\n    # Storing loss and accuracy\n    loss_list.append(running_loss / len(train_dataloader))  # Store average training loss\n    accuracy_list.append(accuracy)  \n    print(f'Epoch [{epoch + 1}/{num_epochs}], Training Loss: {running_loss / len(train_dataloader):.4f}, Accuracy: {accuracy:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2024-10-01T11:05:41.235390Z","iopub.execute_input":"2024-10-01T11:05:41.237069Z","iopub.status.idle":"2024-10-01T13:25:28.316996Z","shell.execute_reply.started":"2024-10-01T11:05:41.237001Z","shell.execute_reply":"2024-10-01T13:25:28.315766Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model testing","metadata":{}},{"cell_type":"markdown","source":"First of all, we have to re-create a dataset in Pytorch. In this case, our images are not labeled, so our dataset will now contain the id of each image and the tensor transformed image.","metadata":{}},{"cell_type":"code","source":"class CancerTestDataset(Dataset):\n    def __init__(self, data_folder, transform=None):\n        self.data_folder = data_folder\n        self.image_files_list = sorted(os.listdir(data_folder))\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.image_files_list)\n\n    def __getitem__(self, idx):\n        img_name = self.image_files_list[idx]\n        img_path = os.path.join(self.data_folder, img_name)\n        image = Image.open(img_path)\n        \n        # Aplicar las transformaciones\n        if self.transform:\n            image = self.transform(image)\n        \n        img_id = img_name.split('.')[0] \n        return image, img_id\n\n# Instantiate the Dataset and the DataLoader for the test set\ntest_dataset = CancerTestDataset(\n    data_folder='../input/histopathologic-cancer-detection/test/',\n    transform=transform\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T13:41:28.032117Z","iopub.execute_input":"2024-10-01T13:41:28.032723Z","iopub.status.idle":"2024-10-01T13:41:28.107287Z","shell.execute_reply.started":"2024-10-01T13:41:28.032674Z","shell.execute_reply":"2024-10-01T13:41:28.105972Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Create test DataLoader\ntest_dataloader = DataLoader(test_dataset, batch_size=32, shuffle=False, num_workers=4)","metadata":{"execution":{"iopub.status.busy":"2024-10-01T13:41:30.831377Z","iopub.execute_input":"2024-10-01T13:41:30.831898Z","iopub.status.idle":"2024-10-01T13:41:30.838887Z","shell.execute_reply.started":"2024-10-01T13:41:30.831853Z","shell.execute_reply":"2024-10-01T13:41:30.837401Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate predictions on the test set\nmodel.eval()\npreds = []\nimage_ids = []\nwith torch.no_grad():\n    for inputs, img_ids in test_dataloader:\n        outputs = model(inputs)\n        # Obtain the probabilities only for the neuron corresponding to “with cancer”.\n        probabilities = torch.sigmoid(outputs[:, 1]) \n        probabilities = probabilities.numpy()\n        preds.extend(probabilities)\n        image_ids.extend(img_ids)\n\n# Create a DataFrame with image IDs and predictions\nsubmission = pd.DataFrame({'id': image_ids, 'label': preds})\n\n# Make sure that the predictions are within the range [0,1].\nsubmission['label'] = submission['label'].clip(0, 1)\n\n# Save the sending file\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\n\n# Show first rows\nprint(submission.head())","metadata":{"execution":{"iopub.status.busy":"2024-10-01T13:41:33.435102Z","iopub.execute_input":"2024-10-01T13:41:33.435646Z","iopub.status.idle":"2024-10-01T13:43:39.011962Z","shell.execute_reply.started":"2024-10-01T13:41:33.435600Z","shell.execute_reply":"2024-10-01T13:43:39.010647Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Use of data augmentation technique","metadata":{}},{"cell_type":"markdown","source":"In addition, we could also use the data augmentation technique to increase the number and diversity of images in the training set,through transformations of different types applied to the original images (geometric changes, color/image alterations, distortions, etc.). ","metadata":{}},{"cell_type":"code","source":"'''\ndata_transforms = transforms.Compose([\n    transforms.RandomHorizontalFlip(),  # Randomly flip the image horizontally\n    transforms.RandomRotation(10),      # Rotate the image in a range of -10 to +10 degrees\n    transforms.RandomResizedCrop(96, scale=(0.8, 1.0)),  # Random crop and resize\n    transforms.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2, hue=0.1),  # Change the color\n    transforms.ToTensor(),              # Convert the image to a tensor\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])  # Normalization\n])\n'''","metadata":{},"outputs":[],"execution_count":null}]}