{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":1243687,"sourceType":"datasetVersion","datasetId":690737}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport torch\nimport torchvision\nimport torchvision.transforms as transforms\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import resample\nimport numpy as np\nimport cv2\nimport pandas as pd\nimport wandb\nimport time","metadata":{"execution":{"iopub.status.busy":"2024-03-02T11:50:25.750875Z","iopub.execute_input":"2024-03-02T11:50:25.751465Z","iopub.status.idle":"2024-03-02T11:50:34.266988Z","shell.execute_reply.started":"2024-03-02T11:50:25.751437Z","shell.execute_reply":"2024-03-02T11:50:34.266024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = '/kaggle/input/melanoma-merged-external-data-512x512-jpeg/512x512-dataset-melanoma/512x512-dataset-melanoma/'\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T11:50:39.593736Z","iopub.execute_input":"2024-03-02T11:50:39.594225Z","iopub.status.idle":"2024-03-02T11:50:39.600582Z","shell.execute_reply.started":"2024-03-02T11:50:39.59419Z","shell.execute_reply":"2024-03-02T11:50:39.599129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the function to load and preprocess images\ndef load_and_preprocess_images(image_paths):\n    images = []\n    for path in image_paths:\n        # Load image using OpenCV\n        img = cv2.imread(image_path + path + '.jpg')\n        # Convert image to grayscale\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        images.append(img)\n    return np.array(images)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T11:51:34.473052Z","iopub.execute_input":"2024-03-02T11:51:34.473479Z","iopub.status.idle":"2024-03-02T11:51:34.480483Z","shell.execute_reply.started":"2024-03-02T11:51:34.473449Z","shell.execute_reply":"2024-03-02T11:51:34.478489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and preprocess images\ndef load_and_preprocess_images(image_paths):\n    images = []\n    for path in image_paths:\n        # Load image using OpenCV\n        img = cv2.imread(image_path + path + '.jpg')\n        # Convert image to RGB format\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        # Resize image to fit ResNet input size\n        img = cv2.resize(img, (224, 224))\n        # Normalize image\n        img = img / 255.0  # Normalize pixel values to [0, 1]\n        img = (img - [0.485, 0.456, 0.406]) / [0.229, 0.224, 0.225]  # Normalize using ImageNet mean and std\n        images.append(img)\n    return np.array(images)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:26.428608Z","iopub.execute_input":"2024-03-02T12:21:26.429113Z","iopub.status.idle":"2024-03-02T12:21:26.920867Z","shell.execute_reply.started":"2024-03-02T12:21:26.429076Z","shell.execute_reply":"2024-03-02T12:21:26.918216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:27.699418Z","iopub.execute_input":"2024-03-02T12:21:27.699883Z","iopub.status.idle":"2024-03-02T12:21:28.231003Z","shell.execute_reply.started":"2024-03-02T12:21:27.699848Z","shell.execute_reply":"2024-03-02T12:21:28.229912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_distribution = train_df['target'].value_counts()\nclass_distribution","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:29.253307Z","iopub.execute_input":"2024-03-02T12:21:29.254323Z","iopub.status.idle":"2024-03-02T12:21:29.753567Z","shell.execute_reply.started":"2024-03-02T12:21:29.254274Z","shell.execute_reply":"2024-03-02T12:21:29.751997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df['image_name'] #images\ny = train_df['target'] #target","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:30.57286Z","iopub.execute_input":"2024-03-02T12:21:30.573185Z","iopub.status.idle":"2024-03-02T12:21:31.027925Z","shell.execute_reply.started":"2024-03-02T12:21:30.57316Z","shell.execute_reply":"2024-03-02T12:21:31.027085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming X and y are your original data and labels\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:32.003187Z","iopub.execute_input":"2024-03-02T12:21:32.0035Z","iopub.status.idle":"2024-03-02T12:21:32.426171Z","shell.execute_reply.started":"2024-03-02T12:21:32.003476Z","shell.execute_reply":"2024-03-02T12:21:32.424656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate majority and minority classes in training data\nmajority_class = X_train[y_train == 0]\nminority_class = X_train[y_train == 1]\n\nprint(\"Length of Majority \",len(majority_class),\"Minority\",len(minority_class))","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:33.513694Z","iopub.execute_input":"2024-03-02T12:21:33.514041Z","iopub.status.idle":"2024-03-02T12:21:33.928649Z","shell.execute_reply.started":"2024-03-02T12:21:33.514013Z","shell.execute_reply":"2024-03-02T12:21:33.927793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"undersampled_majority_class = resample(majority_class,\n                                        replace=False,\n                                        n_samples=len(minority_class),\n                                        random_state=42)\n\t\t\t\t\t\t\t\t\t\t\nprint(\"Length of undersampled_majority \",len(undersampled_majority_class))\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:34.998215Z","iopub.execute_input":"2024-03-02T12:21:34.998853Z","iopub.status.idle":"2024-03-02T12:21:35.39029Z","shell.execute_reply.started":"2024-03-02T12:21:34.998824Z","shell.execute_reply":"2024-03-02T12:21:35.389106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and preprocess images for majority class\nundersampled_majority_images = load_and_preprocess_images(undersampled_majority_class)\n# Convert labels to float\nundersampled_majority_labels = np.zeros(len(undersampled_majority_images))\n\n# Load and preprocess images for minority class\nminority_images = load_and_preprocess_images(minority_class)\n# Convert labels to float\nminority_labels = np.ones(len(minority_images))","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:36.352994Z","iopub.execute_input":"2024-03-02T12:21:36.353301Z","iopub.status.idle":"2024-03-02T12:21:43.607532Z","shell.execute_reply.started":"2024-03-02T12:21:36.353277Z","shell.execute_reply":"2024-03-02T12:21:43.606435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Combine minority class with undersampled majority class\nundersampled_X_train = np.concatenate([undersampled_majority_images, minority_images])\nundersampled_y_train = np.concatenate([undersampled_majority_labels, minority_labels])\n\nundersampled_X_train_tensor = torch.tensor(undersampled_X_train, dtype=torch.float32)\nundersampled_y_train_tensor = torch.tensor(undersampled_y_train, dtype=torch.float32)\n\n# Shuffle the data\nshuffled_indices = np.random.permutation(len(undersampled_y_train))\nundersampled_X_train = undersampled_X_train_tensor[shuffled_indices]\nundersampled_y_train = undersampled_y_train_tensor[shuffled_indices]\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:45.313251Z","iopub.execute_input":"2024-03-02T12:21:45.313577Z","iopub.status.idle":"2024-03-02T12:21:46.538571Z","shell.execute_reply.started":"2024-03-02T12:21:45.313553Z","shell.execute_reply":"2024-03-02T12:21:46.537158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define transformations\ntransform = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))\n])","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:48.332859Z","iopub.execute_input":"2024-03-02T12:21:48.333507Z","iopub.status.idle":"2024-03-02T12:21:48.811661Z","shell.execute_reply.started":"2024-03-02T12:21:48.333468Z","shell.execute_reply":"2024-03-02T12:21:48.810448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a custom dataset\nundersampled_dataset = TensorDataset(undersampled_X_train_tensor, undersampled_y_train_tensor.long())\n","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:49.952595Z","iopub.execute_input":"2024-03-02T12:21:49.952917Z","iopub.status.idle":"2024-03-02T12:21:50.410071Z","shell.execute_reply.started":"2024-03-02T12:21:49.952893Z","shell.execute_reply":"2024-03-02T12:21:50.408774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define data loader\nbatch_size = 32\nundersampled_dataset.transform =  transform\nundersampled_dataloader = DataLoader(undersampled_dataset, batch_size=batch_size, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:51.398718Z","iopub.execute_input":"2024-03-02T12:21:51.399084Z","iopub.status.idle":"2024-03-02T12:21:51.860919Z","shell.execute_reply.started":"2024-03-02T12:21:51.399054Z","shell.execute_reply":"2024-03-02T12:21:51.86003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wandb\nwandb.init(project='Skin Melanoma Project ', save_code=True,name=\"Resnet50\")","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:21:52.928006Z","iopub.execute_input":"2024-03-02T12:21:52.928383Z","iopub.status.idle":"2024-03-02T12:22:29.494722Z","shell.execute_reply.started":"2024-03-02T12:21:52.928355Z","shell.execute_reply":"2024-03-02T12:22:29.493584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision.models as models","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:22:31.29322Z","iopub.execute_input":"2024-03-02T12:22:31.293828Z","iopub.status.idle":"2024-03-02T12:22:31.298862Z","shell.execute_reply.started":"2024-03-02T12:22:31.293797Z","shell.execute_reply":"2024-03-02T12:22:31.297654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load pre-trained ResNet-50 model\nresnet_model = models.resnet50(pretrained=True)\nnum_features = resnet_model.fc.in_features\nresnet_model.fc = nn.Linear(num_features, 2)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:22:50.354119Z","iopub.execute_input":"2024-03-02T12:22:50.354454Z","iopub.status.idle":"2024-03-02T12:22:50.793991Z","shell.execute_reply.started":"2024-03-02T12:22:50.354429Z","shell.execute_reply":"2024-03-02T12:22:50.79241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define loss function and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(resnet_model.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:22:52.468591Z","iopub.execute_input":"2024-03-02T12:22:52.468943Z","iopub.status.idle":"2024-03-02T12:22:52.475648Z","shell.execute_reply.started":"2024-03-02T12:22:52.468916Z","shell.execute_reply":"2024-03-02T12:22:52.474463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training loop\nnum_epochs = 10\nfor epoch in range(num_epochs):\n    resnet_model.train()\n    for inputs, labels in undersampled_dataloader:\n        optimizer.zero_grad()\n        outputs = resnet_model(inputs)\n        loss = criterion(outputs, labels.long())\n        loss.backward()\n        optimizer.step()\n        \n        # Log loss and epoch number to WandB\n        wandb.log({\"loss\": loss.item(), \"epoch\": epoch})\n\n    print(f'Epoch [{epoch+1}/{num_epochs}], Loss: {loss.item():.4f}')","metadata":{"execution":{"iopub.status.busy":"2024-03-02T12:22:53.706487Z","iopub.execute_input":"2024-03-02T12:22:53.70697Z","iopub.status.idle":"2024-03-02T12:22:53.851122Z","shell.execute_reply.started":"2024-03-02T12:22:53.70693Z","shell.execute_reply":"2024-03-02T12:22:53.849651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}