{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":1225697,"sourceType":"datasetVersion","datasetId":701123}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Overview SIIM-ISIC dataset (test)**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load the CSV file\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')\n\n# Count the number of rows\nnum_rows_test = len(test_df)\nnum_rows_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:39:30.945453Z","iopub.execute_input":"2024-11-25T13:39:30.946163Z","iopub.status.idle":"2024-11-25T13:39:31.280715Z","shell.execute_reply.started":"2024-11-25T13:39:30.946126Z","shell.execute_reply":"2024-11-25T13:39:31.279862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract only 2000 rows from test_df\nnew_test_df = test_df.head(2000)\n\n# Save the extracted rows into a new CSV file under /kaggle/working/\nnew_test_df.to_csv('/kaggle/working/test.csv', index=False)\n\nprint(\"2000 rows have been saved to /kaggle/working/test.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:39:33.805196Z","iopub.execute_input":"2024-11-25T13:39:33.805877Z","iopub.status.idle":"2024-11-25T13:39:33.819104Z","shell.execute_reply.started":"2024-11-25T13:39:33.805844Z","shell.execute_reply":"2024-11-25T13:39:33.81827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make a copy of new_test_df to avoid SettingWithCopyWarning\nnew_test_df = new_test_df.copy()\n\n# Rename columns\nnew_test_df.rename(columns={'age_approx': 'age', 'anatom_site_general_challenge': 'anatomy'}, inplace=True)\n\n# Display the updated DataFrame\nprint(new_test_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:39:38.138651Z","iopub.execute_input":"2024-11-25T13:39:38.139339Z","iopub.status.idle":"2024-11-25T13:39:38.153452Z","shell.execute_reply.started":"2024-11-25T13:39:38.139306Z","shell.execute_reply":"2024-11-25T13:39:38.152588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\n\n# Directory paths\nsource_dir = '/kaggle/input/siim-isic-melanoma-classification/jpeg/test'\ndestination_dir = '/kaggle/working/test'\n\n# Create the destination directory if it doesn't exist\nos.makedirs(destination_dir, exist_ok=True)\n\n# Loop through the image names in new_test_df and copy matching images to the new folder\nfor image_name in new_test_df['image_name']:\n    # Construct the image filename with .jpg\n    image_filename = image_name + '.jpg'\n    \n    # Construct full paths for source and destination\n    source_path = os.path.join(source_dir, image_filename)\n    destination_path = os.path.join(destination_dir, image_filename)\n    \n    # Check if the file exists in the source directory\n    if os.path.exists(source_path):\n        # Copy the file to the destination directory\n        shutil.copy2(source_path, destination_path)\n        \n\nprint(f\"All found images have been copied to: {destination_dir}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:39:41.546559Z","iopub.execute_input":"2024-11-25T13:39:41.546852Z","iopub.status.idle":"2024-11-25T13:40:11.515815Z","shell.execute_reply.started":"2024-11-25T13:39:41.546828Z","shell.execute_reply":"2024-11-25T13:40:11.514698Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Overview External Data (Train)**","metadata":{}},{"cell_type":"code","source":"# File paths\ntrain_path_external = '/kaggle/input/melanoma-external-malignant-256/train_concat.csv'\n\n# Read the CSV files\ntrain_df_external = pd.read_csv(train_path_external)\n\n# Display the first few rows of each DataFrame\nprint(\"Train CSV Head:\")\nprint(train_df_external.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:40:25.235423Z","iopub.execute_input":"2024-11-25T13:40:25.235805Z","iopub.status.idle":"2024-11-25T13:40:25.30267Z","shell.execute_reply.started":"2024-11-25T13:40:25.235759Z","shell.execute_reply":"2024-11-25T13:40:25.301876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count the number of rows in each DataFrame\ntrain_row_count = len(train_df_external)\nprint(train_row_count)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:40:30.184467Z","iopub.execute_input":"2024-11-25T13:40:30.184826Z","iopub.status.idle":"2024-11-25T13:40:30.189526Z","shell.execute_reply.started":"2024-11-25T13:40:30.184796Z","shell.execute_reply":"2024-11-25T13:40:30.188625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count the number of rows in each DataFrame\ntrain_row_count = len(train_df_external)\n\n# Count occurrences of target values in the training data\ntarget_counts = train_df_external['target'].value_counts()\n\n# Display the results\nprint(f\"Number of rows in train_df_external: {train_row_count}\")\nprint(f\"\\nTarget counts in train_df_external:\")\nprint(f\"Target = 0: {target_counts.get(0, 0)}\")\nprint(f\"Target = 1: {target_counts.get(1, 0)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:40:33.806317Z","iopub.execute_input":"2024-11-25T13:40:33.80696Z","iopub.status.idle":"2024-11-25T13:40:33.817408Z","shell.execute_reply.started":"2024-11-25T13:40:33.806926Z","shell.execute_reply":"2024-11-25T13:40:33.81638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# File paths\ntrain_path_external = '/kaggle/input/melanoma-external-malignant-256/train_concat.csv'\n\n# Read the CSV files\ntrain_df_external = pd.read_csv(train_path_external)# Extract rows where target == 0 and target == 1\n\ntarget_0 = train_df_external[train_df_external['target'] == 0].sample(n=5000, random_state=42)\ntarget_1 = train_df_external[train_df_external['target'] == 1].sample(n=5000, random_state=42)\n\n# Combine the two subsets\nsampled_df = pd.concat([target_0, target_1])\n\n# Shuffle the rows (optional, if you want random order)\nsampled_df = sampled_df.sample(frac=1, random_state=42).reset_index(drop=True)\n\n# Save the resulting DataFrame to a new CSV file\nsampled_df.to_csv('/kaggle/working/train.csv', index=False)\n\nprint(\"CSV file with 5000 rows for target 0 and 5000 rows for target 1 has been created.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:40:37.29526Z","iopub.execute_input":"2024-11-25T13:40:37.295597Z","iopub.status.idle":"2024-11-25T13:40:37.386104Z","shell.execute_reply.started":"2024-11-25T13:40:37.295569Z","shell.execute_reply":"2024-11-25T13:40:37.385278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\n\n# Define the source and destination directories\nsource_directory = '/kaggle/input/melanoma-external-malignant-256/train/train/'\ndestination_directory = '/kaggle/working/train/'\n\n# Create the destination directory if it doesn't exist\nos.makedirs(destination_directory, exist_ok=True)\n\n# Iterate over the image names in sampled_df\nfor image_name in sampled_df['image_name']:\n    # Append '.jpg' to each image_name\n    image_path = os.path.join(source_directory, image_name + '.jpg')\n    \n    # Check if the image exists in the source directory\n    if os.path.exists(image_path):\n        # Define the new destination path\n        destination_path = os.path.join(destination_directory, image_name + '.jpg')\n        \n        # Copy the image to the new directory\n        shutil.copy2(image_path, destination_path)  # copy2 to preserve metadata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:40:42.826565Z","iopub.execute_input":"2024-11-25T13:40:42.826908Z","iopub.status.idle":"2024-11-25T13:41:51.207107Z","shell.execute_reply.started":"2024-11-25T13:40:42.826878Z","shell.execute_reply":"2024-11-25T13:41:51.206404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Add image path in train.csv\n\n# Directory where images are stored\nimage_directory = '/kaggle/working/train/'\n\n# Add a new column 'image_path' with the full path including the .jpg extension\nsampled_df['image_path'] = image_directory + sampled_df['image_name'] + '.jpg'\n\n# Display the updated dataframe\nprint(sampled_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:42:31.498609Z","iopub.execute_input":"2024-11-25T13:42:31.499205Z","iopub.status.idle":"2024-11-25T13:42:31.509924Z","shell.execute_reply.started":"2024-11-25T13:42:31.499171Z","shell.execute_reply":"2024-11-25T13:42:31.509027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Rename columns in sampled_df\nsampled_df.rename(columns={'age_approx': 'age', 'anatom_site_general_challenge': 'anatomy'}, inplace=True)\n\n# Display the updated DataFrame\nprint(sampled_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:42:35.581435Z","iopub.execute_input":"2024-11-25T13:42:35.581784Z","iopub.status.idle":"2024-11-25T13:42:35.589534Z","shell.execute_reply.started":"2024-11-25T13:42:35.581755Z","shell.execute_reply":"2024-11-25T13:42:35.58864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get unique values in the 'anatomy' column\nunique_anatomy_values = sampled_df['anatomy'].unique()\n\n# Display the unique values\nprint(\"Unique values in the 'anatomy' column:\")\nprint(unique_anatomy_values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:42:39.435358Z","iopub.execute_input":"2024-11-25T13:42:39.435669Z","iopub.status.idle":"2024-11-25T13:42:39.442759Z","shell.execute_reply.started":"2024-11-25T13:42:39.435644Z","shell.execute_reply":"2024-11-25T13:42:39.441804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Missing Values Analysis\n# Display the count of missing values per column\nprint(\"\\nMissing Values in Train Dataset:\")\nmissing_values = sampled_df.isna().sum()\nprint(missing_values)\n\n# Visualize missing values\nplt.figure(figsize=(12, 6))\nsns.heatmap(sampled_df.isna(), cbar=False, cmap='viridis')\nplt.title(\"Missing Values Heatmap for Train Dataset\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:42:42.754291Z","iopub.execute_input":"2024-11-25T13:42:42.755115Z","iopub.status.idle":"2024-11-25T13:42:43.870178Z","shell.execute_reply.started":"2024-11-25T13:42:42.755084Z","shell.execute_reply":"2024-11-25T13:42:43.869295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(sampled_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:42:49.234969Z","iopub.execute_input":"2024-11-25T13:42:49.235773Z","iopub.status.idle":"2024-11-25T13:42:49.244172Z","shell.execute_reply.started":"2024-11-25T13:42:49.235738Z","shell.execute_reply":"2024-11-25T13:42:49.243167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Handling missing values in the Combined Train Dataset\n\n# 1. Fill missing 'sex' values with the most common value\nsampled_df['sex'] = sampled_df['sex'].fillna(sampled_df['sex'].mode()[0])\n\n# 2. Fill missing 'age' values with the median age\nsampled_df['age'] = sampled_df['age'].fillna(sampled_df['age'].median())\n\n# 3. Fill missing 'anatomy' with the most common value\nsampled_df['anatomy'] = sampled_df['anatomy'].fillna(sampled_df['anatomy'].mode()[0])\n\n# 4. Handle missing 'patient_id' by filling it with 'unknown'\nsampled_df['patient_id'] = sampled_df['patient_id'].fillna('unknown')\n\n# Summary statistics after missing values treatment\nprint(\"\\nSummary of the train dataset after handling missing values:\")\n\n# Save the cleaned dataset (overwrite the original train.csv)\nsampled_df.to_csv('/kaggle/working/train.csv', index=False)\nprint(f\"\\nCleaned combined training dataset saved back to '/kaggle/working/train/train.csv' for further processing.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:42:52.843655Z","iopub.execute_input":"2024-11-25T13:42:52.844008Z","iopub.status.idle":"2024-11-25T13:42:52.896985Z","shell.execute_reply.started":"2024-11-25T13:42:52.843978Z","shell.execute_reply":"2024-11-25T13:42:52.896188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Missing Values Analysis\n# Display the count of missing values per column\nprint(\"\\nMissing Values in Train Dataset:\")\nmissing_values = sampled_df.isna().sum()\nprint(missing_values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:42:56.814816Z","iopub.execute_input":"2024-11-25T13:42:56.815154Z","iopub.status.idle":"2024-11-25T13:42:56.825035Z","shell.execute_reply.started":"2024-11-25T13:42:56.815124Z","shell.execute_reply":"2024-11-25T13:42:56.823973Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Data Augumentation**\nApply augumentations to all images in data","metadata":{}},{"cell_type":"code","source":"import os\nimport shutil\nimport cv2\nimport pandas as pd\nimport numpy as np\nfrom albumentations import Compose, RandomResizedCrop, ShiftScaleRotate, HorizontalFlip, VerticalFlip, HueSaturationValue, RandomBrightnessContrast, Normalize\nfrom albumentations.pytorch import ToTensorV2\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:43:00.707299Z","iopub.execute_input":"2024-11-25T13:43:00.70784Z","iopub.status.idle":"2024-11-25T13:43:04.799211Z","shell.execute_reply.started":"2024-11-25T13:43:00.707804Z","shell.execute_reply":"2024-11-25T13:43:04.798291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Paths\ntrain_folder = '/kaggle/working/train'  # Folder where the original train images are stored\naugmented_folder = '/kaggle/working/augmented_train'  # Folder where augmented images will be saved\naugmented_csv_path = '/kaggle/working/augmented_train.csv'  # Path to save the new augmented CSV file\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:43:08.929875Z","iopub.execute_input":"2024-11-25T13:43:08.93092Z","iopub.status.idle":"2024-11-25T13:43:08.934595Z","shell.execute_reply.started":"2024-11-25T13:43:08.930883Z","shell.execute_reply":"2024-11-25T13:43:08.933761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# augmentations without hair removal ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:33:25.534754Z","iopub.execute_input":"2024-11-25T13:33:25.535019Z","iopub.status.idle":"2024-11-25T13:33:25.546077Z","shell.execute_reply.started":"2024-11-25T13:33:25.534994Z","shell.execute_reply":"2024-11-25T13:33:25.545183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nimport cv2\nimport pandas as pd\nfrom tqdm import tqdm\nfrom albumentations import (\n    Compose,\n    RandomResizedCrop,\n    ShiftScaleRotate,\n    HorizontalFlip,\n    VerticalFlip,\n    HueSaturationValue,\n    RandomBrightnessContrast,\n    Normalize,\n)\nfrom albumentations.pytorch import ToTensorV2\n\n# Create the augmented folder if it doesn't exist\nos.makedirs(augmented_folder, exist_ok=True)\n\n# Copy the original images to the augmented folder and log their data\naugmented_data = []\nfor index, row in tqdm(sampled_df.iterrows(), total=len(sampled_df)):\n    original_image_path = os.path.join(train_folder, row['image_path'])\n    original_image_name = os.path.basename(original_image_path)\n    augmented_image_path = os.path.join(augmented_folder, original_image_name)\n    \n    shutil.copy(original_image_path, augmented_image_path)\n    \n    # Log the original image metadata\n    augmented_data.append({\n        'image_path': augmented_image_path,\n        'image_name': original_image_name,\n        'patient_id': row['patient_id'],\n        'sex': row['sex'],\n        'age': row['age'],\n        'anatomy': row['anatomy'],\n        'target': row['target']\n    })\n\n# Create the augmentation transformation pipeline\ntransform = Compose([\n    RandomResizedCrop(height=224, width=224, scale=(0.4, 1.0)),\n    ShiftScaleRotate(rotate_limit=90, scale_limit=[0.8, 1.2]),\n    HorizontalFlip(p=0.5),\n    VerticalFlip(p=0.5),\n    HueSaturationValue(sat_shift_limit=[0.7, 1.3], hue_shift_limit=[-0.1, 0.1]),\n    RandomBrightnessContrast(brightness_limit=[0.7, 1.3], contrast_limit=[0.7, 1.3]),\n    Normalize(),\n    ToTensorV2()\n])\n\n# Apply augmentations and save new images\nfor index, row in tqdm(sampled_df.iterrows(), total=len(sampled_df)):\n    image_path = os.path.join(train_folder, row['image_path'])\n    image = cv2.imread(image_path)\n    \n    # Apply transformations (augmentations)\n    augmented_image = transform(image=image)['image']\n    \n    # Save the augmented image\n    augmented_image_name = f\"aug_{os.path.basename(row['image_path'])}\"\n    augmented_image_path = os.path.join(augmented_folder, augmented_image_name)\n    cv2.imwrite(augmented_image_path, augmented_image.permute(1, 2, 0).cpu().numpy() * 255)\n    \n    # Append metadata for the augmented image\n    augmented_data.append({\n        'image_path': augmented_image_path,\n        'image_name': augmented_image_name,\n        'patient_id': row['patient_id'],\n        'sex': row['sex'],\n        'age': row['age'],\n        'anatomy': row['anatomy'],\n        'target': row['target']\n    })\n\n# Convert the augmented data list to a DataFrame\naugmented_df = pd.DataFrame(augmented_data)\n\n# Save the augmented DataFrame to a CSV file\naugmented_df.to_csv(augmented_csv_path, index=False)\n\nprint(\"Augmentation, image saving, and CSV update complete.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:43:13.415738Z","iopub.execute_input":"2024-11-25T13:43:13.416166Z","iopub.status.idle":"2024-11-25T13:43:54.871598Z","shell.execute_reply.started":"2024-11-25T13:43:13.416122Z","shell.execute_reply":"2024-11-25T13:43:54.870547Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Model EVA**","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nimport timm\n\n# Define MedicalDataset\nclass MedicalDataset(Dataset):\n    def __init__(self, csv_file, image_dir, metadata_columns, transforms=None):\n        self.data = pd.read_csv(csv_file)\n        self.image_dir = image_dir\n        self.metadata_columns = metadata_columns\n        self.transforms = transforms\n\n        # Handle categorical encoding\n        self.data['sex'] = self.data['sex'].map({'male': 0, 'female': 1}).fillna(-1)\n        self.data['anatomy'] = self.data['anatomy'].astype('category').cat.codes\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        row = self.data.iloc[idx]\n\n        # Load and transform image\n        img_path = os.path.join(self.image_dir, row['image_name'])\n        image = Image.open(img_path).convert(\"RGB\")\n        if self.transforms:\n            image = self.transforms(image)\n\n        # Prepare metadata\n        metadata = row[self.metadata_columns].values.astype(np.float32)\n\n        # Get target\n        target = row['target']\n        return image, torch.tensor(metadata), torch.tensor(target, dtype=torch.float32)\n\n# Define EVAWithMetadata model\nclass EVAWithMetadata(nn.Module):\n    def __init__(self, output_size, metadata_size):\n        super().__init__()\n        # Load the EVA model from timm\n        self.eva = timm.create_model('eva02_small_patch14_336.mim_in22k_ft_in1k', pretrained=True, num_classes=0)\n        image_feature_dim = self.eva.embed_dim\n\n        # Metadata processing\n        self.metadata_fc = nn.Sequential(\n            nn.Linear(metadata_size, 128),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n        )\n\n        # Classification layer\n        self.classification = nn.Sequential(\n            nn.Linear(image_feature_dim + 128, 256),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(256, output_size),\n        )\n\n    def forward(self, image, metadata):\n        image_features = self.eva(image)\n        metadata_features = self.metadata_fc(metadata)\n        combined_features = torch.cat((image_features, metadata_features), dim=1)\n        output = self.classification(combined_features)\n        return output\n\n# Define transforms\nimage_transforms = transforms.Compose([\n    transforms.Resize((336, 336)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5]),\n])\n\n# Load dataset\nmetadata_columns = ['age', 'sex', 'anatomy']\ndataset = MedicalDataset(csv_file='/kaggle/working/augmented_train.csv', \n                         image_dir='/kaggle/working/augmented_train', \n                         metadata_columns=metadata_columns, \n                         transforms=image_transforms)\ndataloader = DataLoader(dataset, batch_size=16, shuffle=True)\n\n# Initialize model\nmetadata_size = len(metadata_columns)\nmodel_EVA = EVAWithMetadata(output_size=1, metadata_size=metadata_size)\nmodel_EVA = model_EVA.to('cuda')\n\n# Define optimizer and loss\noptimizer = optim.Adam(model_EVA.parameters(), lr=1e-4)\ncriterion = nn.BCEWithLogitsLoss()\n\n# Training loop\nfor epoch in range(10):\n    model_EVA.train()\n    epoch_loss = 0.0\n    correct_predictions = 0\n    total_predictions = 0\n\n    for images, metadata, labels in dataloader:\n        images, metadata, labels = images.to('cuda'), metadata.to('cuda'), labels.to('cuda')\n\n        optimizer.zero_grad()\n        outputs = model_EVA(images, metadata).squeeze()\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        epoch_loss += loss.item()\n\n        # Calculate accuracy\n        predicted = torch.sigmoid(outputs)\n        predicted_class = (predicted > 0.5).float()\n        correct_predictions += (predicted_class == labels).sum().item()\n        total_predictions += labels.size(0)\n\n    epoch_accuracy = correct_predictions / total_predictions * 100\n    print(f\"Epoch {epoch+1}, Loss: {epoch_loss / len(dataloader):.4f}, Accuracy: {epoch_accuracy:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T15:31:25.873109Z","iopub.execute_input":"2024-11-25T15:31:25.873474Z","iopub.status.idle":"2024-11-25T17:06:41.355528Z","shell.execute_reply.started":"2024-11-25T15:31:25.873443Z","shell.execute_reply":"2024-11-25T17:06:41.354568Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Efficient net7**","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models, transforms\nimport pandas as pd\nfrom PIL import Image\nimport numpy as np\n\n# Define Dataset\nclass MoleDataset(Dataset):\n    def __init__(self, csv_file, image_dir, metadata_columns, transforms=None):\n        self.data = pd.read_csv(csv_file)\n        self.image_dir = image_dir\n        self.metadata_columns = metadata_columns\n        self.transforms = transforms\n\n        # Convert categorical columns to numerical\n        self.data['sex'] = self.data['sex'].map({'male': 0, 'female': 1}).fillna(-1)\n        self.data['anatomy'] = self.data['anatomy'].astype('category').cat.codes\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        row = self.data.iloc[idx]\n\n        # Load and transform image\n        img_path = f\"{self.image_dir}/{row['image_name']}\"\n        image = Image.open(img_path).convert(\"RGB\")\n        if self.transforms:\n            image = self.transforms(image)\n\n        # Prepare metadata\n        metadata = row[self.metadata_columns].values.astype(np.float32)\n\n        # Get target\n        target = row['target']\n        return image, torch.tensor(metadata), torch.tensor(target, dtype=torch.float32)\n\n# Define Model\nclass EfficientNetWithMetadata(nn.Module):\n    def __init__(self, num_metadata_features, output_size):\n        super(EfficientNetWithMetadata, self).__init__()\n        # Load pre-trained EfficientNet-B7\n        self.backbone = models.efficientnet_b7(pretrained=True)\n\n        # Extract the number of input features from the original classifier\n        image_features_size = self.backbone.classifier[1].in_features\n\n        # Replace the classifier with Identity\n        self.backbone.classifier = nn.Identity()\n\n        # Metadata processing\n        self.metadata_fc = nn.Linear(num_metadata_features, 128)\n\n        # Combined classifier\n        self.classifier = nn.Sequential(\n            nn.Linear(image_features_size + 128, 256),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(256, output_size)\n        )\n\n    def forward(self, image, metadata):\n        # Extract image features\n        image_features = self.backbone(image)\n\n        # Process metadata\n        metadata_features = self.metadata_fc(metadata)\n\n        # Combine image and metadata features\n        combined_features = torch.cat((image_features, metadata_features), dim=1)\n\n        # Pass through the final classifier\n        return self.classifier(combined_features)\n\n# Data Transforms\nimage_transforms = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\n# Load Dataset\nmetadata_columns = ['age', 'sex', 'anatomy']\ndataset = MoleDataset(\n    csv_file='/kaggle/working/augmented_train.csv',\n    image_dir='/kaggle/working/augmented_train',\n    metadata_columns=metadata_columns,\n    transforms=image_transforms\n)\ndataloader = DataLoader(dataset, batch_size=16, shuffle=True)\n\n# Model, Loss, and Optimizer\nnum_metadata_features = len(metadata_columns)\nmodel_EffNet = EfficientNetWithMetadata(num_metadata_features=num_metadata_features, output_size=1).to('cuda')\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(model_EffNet.parameters(), lr=1e-4)\n\n# Training Loop\nfor epoch in range(10):\n    model_EffNet.train()\n    epoch_loss = 0.0\n    correct_predictions = 0\n    total_predictions = 0\n    \n    for images, metadata, targets in dataloader:\n        images, metadata, targets = images.to('cuda'), metadata.to('cuda'), targets.to('cuda')\n\n        optimizer.zero_grad()\n        outputs = model_EffNet(images, metadata).squeeze()  # Squeeze for single-dimension output\n        \n        # Calculate loss\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n\n        # Calculate accuracy\n        predicted = torch.sigmoid(outputs)  # Apply sigmoid to get probability values\n        predicted_class = (predicted > 0.5).float()  # Convert probabilities to binary class predictions\n        correct_predictions += (predicted_class == targets).sum().item()\n        total_predictions += targets.size(0)\n\n        # Accumulate epoch loss\n        epoch_loss += loss.item()\n\n    # Calculate average loss and accuracy for the epoch\n    epoch_accuracy = correct_predictions / total_predictions * 100\n    print(f\"Epoch {epoch+1}, Loss: {epoch_loss / len(dataloader):.4f}, Accuracy: {epoch_accuracy:.2f}%\")\n\n# Save Model\ntorch.save(model_EffNet.state_dict(), \"efficientnet_with_metadata.pth\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T13:44:19.557759Z","iopub.execute_input":"2024-11-25T13:44:19.558145Z","iopub.status.idle":"2024-11-25T15:17:43.547599Z","shell.execute_reply.started":"2024-11-25T13:44:19.558114Z","shell.execute_reply":"2024-11-25T15:17:43.546797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models, transforms\nimport pandas as pd\nfrom PIL import Image\nimport numpy as np\n\n# Define Dataset\nclass MoleDataset(Dataset):\n    def __init__(self, csv_file, image_dir, metadata_columns, transforms=None):\n        self.data = pd.read_csv(csv_file)\n        self.image_dir = image_dir\n        self.metadata_columns = metadata_columns\n        self.transforms = transforms\n\n        # Convert categorical columns to numerical\n        self.data['sex'] = self.data['sex'].map({'male': 0, 'female': 1}).fillna(-1)\n        self.data['anatomy'] = self.data['anatomy'].astype('category').cat.codes\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        row = self.data.iloc[idx]\n\n        # Load and transform image\n        img_path = f\"{self.image_dir}/{row['image_name']}\"\n        image = Image.open(img_path).convert(\"RGB\")\n        if self.transforms:\n            image = self.transforms(image)\n\n        # Prepare metadata\n        metadata = row[self.metadata_columns].values.astype(np.float32)\n\n        # Get target\n        target = row['target']\n        return image, torch.tensor(metadata), torch.tensor(target, dtype=torch.float32)\n\n# Define Model\nclass EfficientNetWithMetadata(nn.Module):\n    def __init__(self, num_metadata_features, output_size):\n        super(EfficientNetWithMetadata, self).__init__()\n        # Load pre-trained EfficientNet-B7\n        self.backbone = models.efficientnet_b7(pretrained=True)\n\n        # Extract the number of input features from the original classifier\n        image_features_size = self.backbone.classifier[1].in_features\n\n        # Replace the classifier with Identity\n        self.backbone.classifier = nn.Identity()\n\n        # Metadata processing\n        self.metadata_fc = nn.Linear(num_metadata_features, 128)\n\n        # Combined classifier\n        self.classifier = nn.Sequential(\n            nn.Linear(image_features_size + 128, 256),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(256, output_size)\n        )\n\n    def forward(self, image, metadata):\n        # Extract image features\n        image_features = self.backbone(image)\n\n        # Process metadata\n        metadata_features = self.metadata_fc(metadata)\n\n        # Combine image and metadata features\n        combined_features = torch.cat((image_features, metadata_features), dim=1)\n\n        # Pass through the final classifier\n        return self.classifier(combined_features)\n\n# Data Transforms\nimage_transforms = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\n# Load Dataset\nmetadata_columns = ['age', 'sex', 'anatomy']\ndataset = MoleDataset(\n    csv_file='/kaggle/working/augmented_train.csv',\n    image_dir='/kaggle/working/augmented_train',\n    metadata_columns=metadata_columns,\n    transforms=image_transforms\n)\ndataloader = DataLoader(dataset, batch_size=16, shuffle=True)\n\n# Model, Loss, and Optimizer\nnum_metadata_features = len(metadata_columns)\nmodel_EffNet = EfficientNetWithMetadata(num_metadata_features=num_metadata_features, output_size=1).to('cuda')\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(model_EffNet.parameters(), lr=1e-4)\n\n# Training Loop\nfor epoch in range(20):\n    model_EffNet.train()\n    epoch_loss = 0.0\n    correct_predictions = 0\n    total_predictions = 0\n    \n    for images, metadata, targets in dataloader:\n        images, metadata, targets = images.to('cuda'), metadata.to('cuda'), targets.to('cuda')\n\n        optimizer.zero_grad()\n        outputs = model_EffNet(images, metadata).squeeze()  # Squeeze for single-dimension output\n        \n        # Calculate loss\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n\n        # Calculate accuracy\n        predicted = torch.sigmoid(outputs)  # Apply sigmoid to get probability values\n        predicted_class = (predicted > 0.5).float()  # Convert probabilities to binary class predictions\n        correct_predictions += (predicted_class == targets).sum().item()\n        total_predictions += targets.size(0)\n\n        # Accumulate epoch loss\n        epoch_loss += loss.item()\n\n    # Calculate average loss and accuracy for the epoch\n    epoch_accuracy = correct_predictions / total_predictions * 100\n    print(f\"Epoch {epoch+1}, Loss: {epoch_loss / len(dataloader):.4f}, Accuracy: {epoch_accuracy:.2f}%\")\n\n# Save Model\ntorch.save(model_EffNet.state_dict(), \"efficientnet_with_metadata.pth\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T17:13:55.259639Z","iopub.execute_input":"2024-11-25T17:13:55.26043Z","iopub.status.idle":"2024-11-25T20:21:01.483643Z","shell.execute_reply.started":"2024-11-25T17:13:55.260393Z","shell.execute_reply":"2024-11-25T20:21:01.482672Z"}},"outputs":[],"execution_count":null}]}