{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":1243687,"sourceType":"datasetVersion","datasetId":690737}],"dockerImageVersionId":30665,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Kaggle notebook:https://www.kaggle.com/hemalathaaae/resnet50-skin-melanoma","metadata":{"execution":{"iopub.status.busy":"2024-03-02T17:54:52.002373Z","iopub.execute_input":"2024-03-02T17:54:52.003213Z","iopub.status.idle":"2024-03-02T17:54:59.992024Z","shell.execute_reply.started":"2024-03-02T17:54:52.003181Z","shell.execute_reply":"2024-03-02T17:54:59.991004Z"}}},{"cell_type":"markdown","source":"Wandb Link: https://wandb.ai/hemalathaa/Skin%20Melanoma%20Project%20?nw=nwuserhemalathaaelumalai","metadata":{}},{"cell_type":"markdown","source":"# Importing Necessary Libraries","metadata":{}},{"cell_type":"code","source":"import torch\nimport torchvision\nimport torch\nimport torchvision\nimport torchvision.transforms as transforms\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import resample\nimport numpy as np\nimport cv2\nimport pandas as pd\nimport time\nimport wandb","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:23:08.805583Z","iopub.execute_input":"2024-03-02T19:23:08.806047Z","iopub.status.idle":"2024-03-02T19:23:08.814979Z","shell.execute_reply.started":"2024-03-02T19:23:08.805992Z","shell.execute_reply":"2024-03-02T19:23:08.814074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Image Dataset","metadata":{}},{"cell_type":"code","source":"image_path = '/kaggle/input/melanoma-merged-external-data-512x512-jpeg/512x512-dataset-melanoma/512x512-dataset-melanoma/'\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:23:27.533028Z","iopub.execute_input":"2024-03-02T19:23:27.533822Z","iopub.status.idle":"2024-03-02T19:23:27.538528Z","shell.execute_reply.started":"2024-03-02T19:23:27.533785Z","shell.execute_reply":"2024-03-02T19:23:27.537507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the function to load and preprocess images\ndef load_and_preprocess_images(image_paths):\n    images = []\n    for path in image_paths:\n        # Load image using OpenCV\n        img = cv2.imread(image_path + path + '.jpg')\n        # Resize image to (512, 512)\n        img = cv2.resize(img, (512, 512))\n        # Convert image to RGB\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        images.append(img)\n    return np.array(images)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:23:35.422415Z","iopub.execute_input":"2024-03-02T19:23:35.42312Z","iopub.status.idle":"2024-03-02T19:23:35.428838Z","shell.execute_reply.started":"2024-03-02T19:23:35.423087Z","shell.execute_reply":"2024-03-02T19:23:35.427789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:23:38.074922Z","iopub.execute_input":"2024-03-02T19:23:38.07531Z","iopub.status.idle":"2024-03-02T19:23:38.187946Z","shell.execute_reply.started":"2024-03-02T19:23:38.075281Z","shell.execute_reply":"2024-03-02T19:23:38.186961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_distribution = train_df['target'].value_counts()\nclass_distribution","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:23:39.599519Z","iopub.execute_input":"2024-03-02T19:23:39.60044Z","iopub.status.idle":"2024-03-02T19:23:39.614922Z","shell.execute_reply.started":"2024-03-02T19:23:39.600401Z","shell.execute_reply":"2024-03-02T19:23:39.613839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df['image_name'] #images\ny = train_df['target'] #target\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:23:40.884735Z","iopub.execute_input":"2024-03-02T19:23:40.885424Z","iopub.status.idle":"2024-03-02T19:23:40.890008Z","shell.execute_reply.started":"2024-03-02T19:23:40.885389Z","shell.execute_reply":"2024-03-02T19:23:40.889052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Splitting dataset into train and validation","metadata":{}},{"cell_type":"code","source":"# Assuming X and y are your original data and labels\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:23:42.572Z","iopub.execute_input":"2024-03-02T19:23:42.572396Z","iopub.status.idle":"2024-03-02T19:23:42.586215Z","shell.execute_reply.started":"2024-03-02T19:23:42.572368Z","shell.execute_reply":"2024-03-02T19:23:42.585283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Downsamling for Traning dataset","metadata":{}},{"cell_type":"code","source":"# Separate majority and minority classes in training data\nmajority_class = X_train[y_train == 0]\nminority_class = X_train[y_train == 1]\n\nprint(\"Length of Majority \",len(majority_class),\"Minority\",len(minority_class))\n\nundersampled_majority_class = resample(majority_class,\n                                        replace=False,\n                                        n_samples=len(minority_class),\n                                        random_state=42)\n\t\t\t\t\t\t\t\t\t\t\nprint(\"Length of undersampled_majority \",len(undersampled_majority_class))\n\n# Load and preprocess images for majority class\nundersampled_majority_images = load_and_preprocess_images(undersampled_majority_class)\n# Convert labels to float\nundersampled_majority_labels = np.zeros(len(undersampled_majority_images))\n\n# Load and preprocess images for minority class\nminority_images = load_and_preprocess_images(minority_class)\n# Convert labels to float\nminority_labels = np.ones(len(minority_images))\n\n# Combine minority class with undersampled majority class\nundersampled_X_train = np.concatenate([undersampled_majority_images, minority_images])\nundersampled_y_train = np.concatenate([undersampled_majority_labels, minority_labels])\n\nundersampled_X_train_tensor = torch.tensor(undersampled_X_train, dtype=torch.float32)\nundersampled_y_train_tensor = torch.tensor(undersampled_y_train, dtype=torch.float32)\n\n# Shuffle the data\nshuffled_indices = np.random.permutation(len(undersampled_y_train))\nundersampled_X_train = undersampled_X_train_tensor[shuffled_indices]\nundersampled_y_train = undersampled_y_train_tensor[shuffled_indices]","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:23:48.137303Z","iopub.execute_input":"2024-03-02T19:23:48.137695Z","iopub.status.idle":"2024-03-02T19:24:01.484156Z","shell.execute_reply.started":"2024-03-02T19:23:48.137665Z","shell.execute_reply":"2024-03-02T19:24:01.483191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transformation","metadata":{}},{"cell_type":"code","source":"# Define transformations\ntransform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))\n])\n\n# Create a custom dataset\nundersampled_dataset = TensorDataset(undersampled_X_train_tensor, undersampled_y_train_tensor.long())\n\n\n# Define data loader\nbatch_size = 32\nundersampled_dataset.transform =  transform\nundersampled_dataloader = DataLoader(undersampled_dataset, batch_size=batch_size, shuffle=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:24:01.486389Z","iopub.execute_input":"2024-03-02T19:24:01.48678Z","iopub.status.idle":"2024-03-02T19:24:01.496534Z","shell.execute_reply.started":"2024-03-02T19:24:01.486745Z","shell.execute_reply":"2024-03-02T19:24:01.495502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataloader for validation ","metadata":{}},{"cell_type":"code","source":"# Assuming X_test contains the image pixel data\nX_test_processed = load_and_preprocess_images(X_test)\ny_test_numpy = y_test.astype(np.float32).to_numpy()\n\n# Define the validation dataset\nvalidation_dataset = TensorDataset(torch.tensor(X_test_processed), torch.tensor(y_test_numpy))\n\n# Define the batch size\nbatch_size = 32\n\n# Define the validation dataloader\nvalidation_dataloader = DataLoader(validation_dataset, batch_size=batch_size, shuffle=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:24:01.497736Z","iopub.execute_input":"2024-03-02T19:24:01.498051Z","iopub.status.idle":"2024-03-02T19:25:12.169805Z","shell.execute_reply.started":"2024-03-02T19:24:01.498002Z","shell.execute_reply":"2024-03-02T19:25:12.168612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:25:12.172201Z","iopub.execute_input":"2024-03-02T19:25:12.172547Z","iopub.status.idle":"2024-03-02T19:25:12.237491Z","shell.execute_reply.started":"2024-03-02T19:25:12.172519Z","shell.execute_reply":"2024-03-02T19:25:12.236537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wandb\nwandb.init(project='Skin Melanoma Project ', save_code=True,name=\"Resnet50\")","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:25:23.738475Z","iopub.execute_input":"2024-03-02T19:25:23.738842Z","iopub.status.idle":"2024-03-02T19:26:00.235591Z","shell.execute_reply.started":"2024-03-02T19:25:23.738814Z","shell.execute_reply":"2024-03-02T19:26:00.23465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Building","metadata":{}},{"cell_type":"code","source":"\n# Load pre-trained ResNet-50 model\nmodel = torchvision.models.resnet50(pretrained=True)\nnum_ftrs = model.fc.in_features\nmodel.fc = nn.Linear(num_ftrs, 2)\n# Move the model to the appropriate device\nmodel.to(device)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:27:56.377875Z","iopub.execute_input":"2024-03-02T19:27:56.378618Z","iopub.status.idle":"2024-03-02T19:27:58.87946Z","shell.execute_reply.started":"2024-03-02T19:27:56.378582Z","shell.execute_reply":"2024-03-02T19:27:58.8786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Initializing Loss Function and Criterion","metadata":{}},{"cell_type":"code","source":"\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.SGD(model.parameters(), lr=0.001, momentum=0.9)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:28:37.335554Z","iopub.execute_input":"2024-03-02T19:28:37.335981Z","iopub.status.idle":"2024-03-02T19:28:38.47981Z","shell.execute_reply.started":"2024-03-02T19:28:37.335949Z","shell.execute_reply":"2024-03-02T19:28:38.478855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"# Train the model\nnum_epochs = 10\nfor epoch in range(num_epochs):\n    model.train()\n    running_loss = 0.0\n    correct = 0\n    total = 0\n    start_time = time.time()\n    \n    for inputs, labels in undersampled_dataloader:\n        optimizer.zero_grad()                \n        inputs = inputs.permute(0, 3, 1, 2)  # Rearrange dimensions\n        inputs, labels = inputs.to(device), labels.to(device)\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()    \n        \n        # Calculate accuracy\n        _, predicted = torch.max(outputs, 1)\n        total += labels.size(0)\n        correct += (predicted == labels).sum().item()\n    \n    end_time = time.time()\n    epoch_time = end_time - start_time\n    \n    epoch_loss = running_loss / len(undersampled_dataloader)\n    epoch_accuracy = 100 * correct / total\n    \n    # Log training loss and accuracy to Wandb\n    wandb.log({\"loss\": epoch_loss, \"accuracy\": epoch_accuracy, \"epoch\": epoch+1})\n    \n    print(f\"Epoch {epoch+1}, Train Loss: {epoch_loss}, Train Accuracy: {epoch_accuracy}%, Time: {epoch_time} seconds\")\n    \n    # Validation loop\n    model.eval()\n    val_running_loss = 0.0\n    val_correct = 0\n    val_total = 0\n\n    with torch.no_grad():\n        for val_inputs, val_labels in validation_dataloader:\n            # Ensure correct data type and device for input data\n            val_inputs = val_inputs.to(device, dtype=torch.float32)\n            val_labels = val_labels.to(device, dtype=torch.long)\n\n            # Rearrange dimensions if necessary\n            val_inputs = val_inputs.permute(0, 3, 1, 2)\n\n            # Forward pass\n            val_outputs = model(val_inputs)\n\n            # Calculate loss\n            val_loss = criterion(val_outputs, val_labels)\n            val_running_loss += val_loss.item()\n\n            # Calculate accuracy\n            _, val_predicted = torch.max(val_outputs, 1)\n            val_total += val_labels.size(0)\n            val_correct += (val_predicted == val_labels).sum().item()\n\n    # Calculate validation loss and accuracy\n    val_epoch_loss = val_running_loss / len(validation_dataloader)\n    val_epoch_accuracy = 100 * val_correct / val_total\n\n    # Log validation loss and accuracy to Wandb\n    wandb.log({\"val_loss\": val_epoch_loss, \"val_accuracy\": val_epoch_accuracy, \"epoch\": epoch+1})\n\n    print(f\"Epoch {epoch+1}, Val Loss: {val_epoch_loss}, Val Accuracy: {val_epoch_accuracy}%\")\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T19:28:48.867752Z","iopub.execute_input":"2024-03-02T19:28:48.868347Z","iopub.status.idle":"2024-03-02T20:13:17.842041Z","shell.execute_reply.started":"2024-03-02T19:28:48.868315Z","shell.execute_reply":"2024-03-02T20:13:17.841072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculating Performance Metrics","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import precision_score, recall_score, f1_score\n\n# Move model to the same device as the input data\nmodel.to(device)\n\n# Validation loop\nmodel.eval()\nval_running_loss = 0.0\nval_correct = 0\nval_total = 0\nval_preds = []\nval_labels = []\n\nwith torch.no_grad():\n    for val_inputs, val_labels_batch in validation_dataloader:\n        # Ensure correct data type and device for input data\n        val_inputs = val_inputs.to(device, dtype=torch.float32)\n        val_labels_batch = val_labels_batch.to(device, dtype=torch.long)\n\n        # Rearrange dimensions if necessary\n        val_inputs = val_inputs.permute(0, 3, 1, 2)\n\n        # Forward pass\n        val_outputs = model(val_inputs)\n\n        # Calculate loss\n        val_loss = criterion(val_outputs, val_labels_batch)\n        val_running_loss += val_loss.item()\n\n        # Append predictions and true labels\n        val_preds.extend(torch.argmax(val_outputs, axis=1).cpu().numpy())\n        val_labels.extend(val_labels_batch.cpu().numpy())\n\n        # Calculate accuracy\n        val_total += val_labels_batch.size(0)\n        val_correct += (torch.argmax(val_outputs, axis=1) == val_labels_batch).sum().item()\n\n# Calculate validation loss\nval_epoch_loss = val_running_loss / len(validation_dataloader)\n\n# Calculate validation accuracy\nval_epoch_accuracy = 100 * val_correct / val_total\n\n# Calculate precision, recall, and F1 score\nprecision = precision_score(val_labels, val_preds, average='weighted')\nrecall = recall_score(val_labels, val_preds, average='weighted')\nf1 = f1_score(val_labels, val_preds, average='weighted')\n\n# Log validation loss, accuracy, precision, recall, and F1 score to Wandb\nwandb.log({\"val_loss\": val_epoch_loss, \"val_accuracy\": val_epoch_accuracy, \"precision\": precision, \"recall\": recall, \"f1_score\": f1, \"epoch\": epoch+1})\n\nprint(f\"Epoch {epoch+1}, Val Loss: {val_epoch_loss}, Val Accuracy: {val_epoch_accuracy}%, Precision: {precision}, Recall: {recall}, F1 Score: {f1}\")\n\nprint(\"Precision:\", precision)\nprint(\"Recall:\", recall)\nprint(\"F1 Score:\", f1)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T21:25:18.879733Z","iopub.execute_input":"2024-03-02T21:25:18.880799Z","iopub.status.idle":"2024-03-02T21:26:45.640793Z","shell.execute_reply.started":"2024-03-02T21:25:18.880764Z","shell.execute_reply":"2024-03-02T21:26:45.639809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Confusion Matrix","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Calculate confusion matrix\nconf_matrix = confusion_matrix(val_labels, val_preds)\n\n# Log confusion matrix to Wandb\nwandb.log({\"confusion_matrix\": wandb.plot.confusion_matrix(probs=None,\n                                                           y_true=val_labels,\n                                                           preds=val_preds,\n                                                           class_names=[0, 1],\n                                                           title=\"Confusion Matrix\")})\n\n# Plot confusion matrix\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\")\nplt.xlabel(\"Predicted labels\")\nplt.ylabel(\"True labels\")\nplt.title(\"Confusion Matrix\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T21:29:43.510305Z","iopub.execute_input":"2024-03-02T21:29:43.511181Z","iopub.status.idle":"2024-03-02T21:29:45.688556Z","shell.execute_reply.started":"2024-03-02T21:29:43.511144Z","shell.execute_reply":"2024-03-02T21:29:45.687418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission file","metadata":{}},{"cell_type":"code","source":"# Load and preprocess test images\ntest_image_paths = ['/kaggle/input/melanoma-resized-images-512512/test/test/' + img_name + '.jpg' for img_name in test_df['image_name']]\ntest_images = load_and_preprocess_images(test_image_paths)\n\n# Convert test images to tensor\ntest_images_tensor = torch.tensor(test_images, dtype=torch.float32)\n# Create test dataset\ntest_dataset = TensorDataset(test_images_tensor)\n\n# Define test data loader\ntest_dataloader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Define test data loader\ntest_dataloader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)\n\ndef test_model(test_loader, model, device, test_df):\n    model.eval()\n    predictions = []\n    image_names = []\n\n    test_start_time = time.time()\n    with torch.no_grad():\n        for i, data in enumerate(test_loader):\n            data = data.to(device)\n            outputs = model(data)\n\n            probabilities = (torch.sigmoid(outputs) >= 0.5).cpu().numpy()\n            predictions.extend(probabilities.flatten())\n\n            image_names.extend(test_df['image_name'][i * test_loader.batch_size:(i + 1) * test_loader.batch_size])\n\n    test_end_time = time.time()\n    test_time = test_end_time - test_start_time\n\n    submission_df = pd.DataFrame({'image_name': image_names, 'target': predictions})\n    submission_df.to_csv('submission.csv', index=False)\n\n    print(f'Test Evaluation and Submission CSV is generated in: {test_time} seconds')","metadata":{},"execution_count":null,"outputs":[]}]}