{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":1225697,"sourceType":"datasetVersion","datasetId":701123}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Overview SIIM-ISIC dataset (test)**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load the CSV file\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')\n\n# Count the number of rows\nnum_rows_test = len(test_df)\nnum_rows_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T23:11:37.165290Z","iopub.execute_input":"2025-11-26T23:11:37.165957Z","iopub.status.idle":"2025-11-26T23:11:37.186308Z","shell.execute_reply.started":"2025-11-26T23:11:37.165922Z","shell.execute_reply":"2025-11-26T23:11:37.185328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract only 2000 rows from test_df\nnew_test_df = test_df.head(2000)\n\n# Save the extracted rows into a new CSV file under /kaggle/working/\nnew_test_df.to_csv('/kaggle/working/test.csv', index=False)\n\nprint(\"2000 rows have been saved to /kaggle/working/test.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:05:38.239099Z","iopub.execute_input":"2025-11-26T22:05:38.240100Z","iopub.status.idle":"2025-11-26T22:05:38.254313Z","shell.execute_reply.started":"2025-11-26T22:05:38.240060Z","shell.execute_reply":"2025-11-26T22:05:38.253305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make a copy of new_test_df to avoid SettingWithCopyWarning\nnew_test_df = new_test_df.copy()\n\n# Rename columns\nnew_test_df.rename(columns={'age_approx': 'age', 'anatom_site_general_challenge': 'anatomy'}, inplace=True)\n\n# Display the updated DataFrame\nprint(new_test_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:05:41.927097Z","iopub.execute_input":"2025-11-26T22:05:41.927950Z","iopub.status.idle":"2025-11-26T22:05:41.944549Z","shell.execute_reply.started":"2025-11-26T22:05:41.927896Z","shell.execute_reply":"2025-11-26T22:05:41.943457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\n\n# Directory paths\nsource_dir = '/kaggle/input/siim-isic-melanoma-classification/jpeg/test'\ndestination_dir = '/kaggle/working/test'\n\n# Create the destination directory if it doesn't exist\nos.makedirs(destination_dir, exist_ok=True)\n\n# Loop through the image names in new_test_df and copy matching images to the new folder\nfor image_name in new_test_df['image_name']:\n    # Construct the image filename with .jpg\n    image_filename = image_name + '.jpg'\n    \n    # Construct full paths for source and destination\n    source_path = os.path.join(source_dir, image_filename)\n    destination_path = os.path.join(destination_dir, image_filename)\n    \n    # Check if the file exists in the source directory\n    if os.path.exists(source_path):\n        # Copy the file to the destination directory\n        shutil.copy2(source_path, destination_path)\n        \n\nprint(f\"All found images have been copied to: {destination_dir}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:05:46.193226Z","iopub.execute_input":"2025-11-26T22:05:46.193971Z","iopub.status.idle":"2025-11-26T22:06:25.210447Z","shell.execute_reply.started":"2025-11-26T22:05:46.193933Z","shell.execute_reply":"2025-11-26T22:06:25.209598Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Overview External Data (Train)**","metadata":{}},{"cell_type":"code","source":"# File paths\ntrain_path_external = '/kaggle/input/melanoma-external-malignant-256/train_concat.csv'\n\n# Read the CSV files\ntrain_df_external = pd.read_csv(train_path_external)\n\n# Display the first few rows of each DataFrame\nprint(\"Train CSV Head:\")\nprint(train_df_external.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:06:51.829956Z","iopub.execute_input":"2025-11-26T22:06:51.830320Z","iopub.status.idle":"2025-11-26T22:06:51.886435Z","shell.execute_reply.started":"2025-11-26T22:06:51.830288Z","shell.execute_reply":"2025-11-26T22:06:51.885561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count the number of rows in each DataFrame\ntrain_row_count = len(train_df_external)\nprint(train_row_count)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:06:55.104031Z","iopub.execute_input":"2025-11-26T22:06:55.104372Z","iopub.status.idle":"2025-11-26T22:06:55.108744Z","shell.execute_reply.started":"2025-11-26T22:06:55.104342Z","shell.execute_reply":"2025-11-26T22:06:55.107831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count the number of rows in each DataFrame\ntrain_row_count = len(train_df_external)\n\n# Count occurrences of target values in the training data\ntarget_counts = train_df_external['target'].value_counts()\n\n# Display the results\nprint(f\"Number of rows in train_df_external: {train_row_count}\")\nprint(f\"\\nTarget counts in train_df_external:\")\nprint(f\"Target = 0: {target_counts.get(0, 0)}\")\nprint(f\"Target = 1: {target_counts.get(1, 0)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:06:57.845832Z","iopub.execute_input":"2025-11-26T22:06:57.849118Z","iopub.status.idle":"2025-11-26T22:06:57.871860Z","shell.execute_reply.started":"2025-11-26T22:06:57.849043Z","shell.execute_reply":"2025-11-26T22:06:57.870479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Count occurrences of target values\ntarget_counts = train_df_external['target'].value_counts()\n\n# Plot\nplt.figure(figsize=(6,4))\nplt.bar(target_counts.index.astype(str), target_counts.values)\nplt.title(\"Target Distribution Before Augmentation\")\nplt.xlabel(\"Target Value\")\nplt.ylabel(\"Count\")\nplt.grid(axis='y', linestyle='--', alpha=0.5)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T23:12:20.532756Z","iopub.execute_input":"2025-11-26T23:12:20.533187Z","iopub.status.idle":"2025-11-26T23:12:20.751123Z","shell.execute_reply.started":"2025-11-26T23:12:20.533153Z","shell.execute_reply":"2025-11-26T23:12:20.750182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Count occurrences of target values (ensures both classes appear)\ntarget_counts = sampled_df['target'].value_counts().reindex([0, 1], fill_value=0)\n\n# Plot\nplt.figure(figsize=(6,4))\nbars = plt.bar(target_counts.index.astype(str), target_counts.values)\n\n# Add value labels on top of bars\nfor bar in bars:\n    height = bar.get_height()\n    plt.text(bar.get_x() + bar.get_width()/2, height, str(height),\n             ha='center', va='bottom', fontsize=10)\n\nplt.title(\"Target Distribution Before Augmentation\")\nplt.xlabel(\"Target Value\")\nplt.ylabel(\"Count\")\nplt.grid(axis='y', linestyle='--', alpha=0.5)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T23:23:15.916166Z","iopub.execute_input":"2025-11-26T23:23:15.916971Z","iopub.status.idle":"2025-11-26T23:23:16.171413Z","shell.execute_reply.started":"2025-11-26T23:23:15.916935Z","shell.execute_reply":"2025-11-26T23:23:16.170507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# File paths\ntrain_path_external = '/kaggle/input/melanoma-external-malignant-256/train_concat.csv'\n\n# Read the CSV files\ntrain_df_external = pd.read_csv(train_path_external)# Extract rows where target == 0 and target == 1\n\ntarget_0 = train_df_external[train_df_external['target'] == 0].sample(n=5000, random_state=42)\ntarget_1 = train_df_external[train_df_external['target'] == 1].sample(n=5000, random_state=42)\n\n# Combine the two subsets\nsampled_df = pd.concat([target_0, target_1])\n\n# Shuffle the rows (optional, if you want random order)\nsampled_df = sampled_df.sample(frac=1, random_state=42).reset_index(drop=True)\n\n# Save the resulting DataFrame to a new CSV file\nsampled_df.to_csv('/kaggle/working/train.csv', index=False)\n\nprint(\"CSV file with 5000 rows for target 0 and 5000 rows for target 1 has been created.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:07:11.671240Z","iopub.execute_input":"2025-11-26T22:07:11.672098Z","iopub.status.idle":"2025-11-26T22:07:11.762025Z","shell.execute_reply.started":"2025-11-26T22:07:11.672062Z","shell.execute_reply":"2025-11-26T22:07:11.761117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\n\n# Define the source and destination directories\nsource_directory = '/kaggle/input/melanoma-external-malignant-256/train/train/'\ndestination_directory = '/kaggle/working/train/'\n\n# Create the destination directory if it doesn't exist\nos.makedirs(destination_directory, exist_ok=True)\n\n# Iterate over the image names in sampled_df\nfor image_name in sampled_df['image_name']:\n    # Append '.jpg' to each image_name\n    image_path = os.path.join(source_directory, image_name + '.jpg')\n    \n    # Check if the image exists in the source directory\n    if os.path.exists(image_path):\n        # Define the new destination path\n        destination_path = os.path.join(destination_directory, image_name + '.jpg')\n        \n        # Copy the image to the new directory\n        shutil.copy2(image_path, destination_path)  # copy2 to preserve metadata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:07:15.556379Z","iopub.execute_input":"2025-11-26T22:07:15.557153Z","iopub.status.idle":"2025-11-26T22:07:54.642510Z","shell.execute_reply.started":"2025-11-26T22:07:15.557111Z","shell.execute_reply":"2025-11-26T22:07:54.641521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Add image path in train.csv\n\n# Directory where images are stored\nimage_directory = '/kaggle/working/train/'\n\n# Add a new column 'image_path' with the full path including the .jpg extension\nsampled_df['image_path'] = image_directory + sampled_df['image_name'] + '.jpg'\n\n# Display the updated dataframe\nprint(sampled_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:13:48.222798Z","iopub.execute_input":"2025-11-26T22:13:48.223175Z","iopub.status.idle":"2025-11-26T22:13:48.234772Z","shell.execute_reply.started":"2025-11-26T22:13:48.223139Z","shell.execute_reply":"2025-11-26T22:13:48.233952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Rename columns in sampled_df\nsampled_df.rename(columns={'age_approx': 'age', 'anatom_site_general_challenge': 'anatomy'}, inplace=True)\n\n# Display the updated DataFrame\nprint(sampled_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T23:19:10.309484Z","iopub.execute_input":"2025-11-26T23:19:10.309885Z","iopub.status.idle":"2025-11-26T23:19:10.318711Z","shell.execute_reply.started":"2025-11-26T23:19:10.309850Z","shell.execute_reply":"2025-11-26T23:19:10.317701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get unique values in the 'anatomy' column\nunique_anatomy_values = sampled_df['anatomy'].unique()\n\n# Display the unique values\nprint(\"Unique values in the 'anatomy' column:\")\nprint(unique_anatomy_values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:14:17.712185Z","iopub.execute_input":"2025-11-26T22:14:17.712944Z","iopub.status.idle":"2025-11-26T22:14:17.718256Z","shell.execute_reply.started":"2025-11-26T22:14:17.712906Z","shell.execute_reply":"2025-11-26T22:14:17.717339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Missing Values Analysis\n# Display the count of missing values per column\nprint(\"\\nMissing Values in Train Dataset:\")\nmissing_values = sampled_df.isna().sum()\nprint(missing_values)\n\n# Visualize missing values\nplt.figure(figsize=(12, 6))\nsns.heatmap(sampled_df.isna(), cbar=False, cmap='viridis')\nplt.title(\"Missing Values Heatmap for Train Dataset\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:14:21.527330Z","iopub.execute_input":"2025-11-26T22:14:21.527933Z","iopub.status.idle":"2025-11-26T22:14:22.742004Z","shell.execute_reply.started":"2025-11-26T22:14:21.527897Z","shell.execute_reply":"2025-11-26T22:14:22.740959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(sampled_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T18:57:59.650176Z","iopub.execute_input":"2025-11-25T18:57:59.650801Z","iopub.status.idle":"2025-11-25T18:57:59.658762Z","shell.execute_reply.started":"2025-11-25T18:57:59.650759Z","shell.execute_reply":"2025-11-25T18:57:59.657954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Handling missing values in the Combined Train Dataset\n\n# 1. Fill missing 'sex' values with the most common value\nsampled_df['sex'] = sampled_df['sex'].fillna(sampled_df['sex'].mode()[0])\n\n# 2. Fill missing 'age' values with the median age\nsampled_df['age'] = sampled_df['age'].fillna(sampled_df['age'].median())\n\n# 3. Fill missing 'anatomy' with the most common value\nsampled_df['anatomy'] = sampled_df['anatomy'].fillna(sampled_df['anatomy'].mode()[0])\n\n# 4. Handle missing 'patient_id' by filling it with 'unknown'\nsampled_df['patient_id'] = sampled_df['patient_id'].fillna('unknown')\n\n# Summary statistics after missing values treatment\nprint(\"\\nSummary of the train dataset after handling missing values:\")\n\n# Save the cleaned dataset (overwrite the original train.csv)\nsampled_df.to_csv('/kaggle/working/train.csv', index=False)\nprint(f\"\\nCleaned combined training dataset saved back to '/kaggle/working/train/train.csv' for further processing.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:14:29.147903Z","iopub.execute_input":"2025-11-26T22:14:29.148375Z","iopub.status.idle":"2025-11-26T22:14:29.207611Z","shell.execute_reply.started":"2025-11-26T22:14:29.148344Z","shell.execute_reply":"2025-11-26T22:14:29.206646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Missing Values Analysis\n# Display the count of missing values per column\nprint(\"\\nMissing Values in Train Dataset:\")\nmissing_values = sampled_df.isna().sum()\nprint(missing_values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:14:33.428252Z","iopub.execute_input":"2025-11-26T22:14:33.428556Z","iopub.status.idle":"2025-11-26T22:14:33.437820Z","shell.execute_reply.started":"2025-11-26T22:14:33.428530Z","shell.execute_reply":"2025-11-26T22:14:33.436838Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Data Augumentation**\nApply augumentations to all images in data","metadata":{}},{"cell_type":"code","source":"import os\nimport shutil\nimport cv2\nimport pandas as pd\nimport numpy as np\nfrom albumentations import Compose, RandomResizedCrop, ShiftScaleRotate, HorizontalFlip, VerticalFlip, HueSaturationValue, RandomBrightnessContrast, Normalize\nfrom albumentations.pytorch import ToTensorV2\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:15:00.318235Z","iopub.execute_input":"2025-11-26T22:15:00.319435Z","iopub.status.idle":"2025-11-26T22:15:04.661313Z","shell.execute_reply.started":"2025-11-26T22:15:00.319393Z","shell.execute_reply":"2025-11-26T22:15:04.660367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install -U albumentations","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:14:38.973923Z","iopub.execute_input":"2025-11-26T22:14:38.974673Z","iopub.status.idle":"2025-11-26T22:14:51.163348Z","shell.execute_reply.started":"2025-11-26T22:14:38.974638Z","shell.execute_reply":"2025-11-26T22:14:51.162279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Paths\ntrain_folder = '/kaggle/working/train'  # Folder where the original train images are stored\naugmented_folder = '/kaggle/working/augmented_train'  # Folder where augmented images will be saved\naugmented_csv_path = '/kaggle/working/augmented_train.csv'  # Path to save the new augmented CSV file\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T23:17:17.590721Z","iopub.execute_input":"2025-11-26T23:17:17.591155Z","iopub.status.idle":"2025-11-26T23:17:17.595726Z","shell.execute_reply.started":"2025-11-26T23:17:17.591121Z","shell.execute_reply":"2025-11-26T23:17:17.594804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# augmentations without hair removal ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T18:58:07.770699Z","iopub.execute_input":"2025-11-25T18:58:07.771303Z","iopub.status.idle":"2025-11-25T18:58:07.780423Z","shell.execute_reply.started":"2025-11-25T18:58:07.771263Z","shell.execute_reply":"2025-11-25T18:58:07.779674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nimport cv2\nimport pandas as pd\nfrom tqdm import tqdm\nfrom albumentations import (\n    Compose,\n    RandomResizedCrop,\n    ShiftScaleRotate,\n    HorizontalFlip,\n    VerticalFlip,\n    HueSaturationValue,\n    RandomBrightnessContrast,\n    Normalize,\n)\nfrom albumentations.pytorch import ToTensorV2\n\n# Create the augmented folder if it doesn't exist\nos.makedirs(augmented_folder, exist_ok=True)\n\n# Copy the original images to the augmented folder and log their data\naugmented_data = []\nfor index, row in tqdm(sampled_df.iterrows(), total=len(sampled_df)):\n    original_image_path = os.path.join(train_folder, row['image_path'])\n    original_image_name = os.path.basename(original_image_path)\n    augmented_image_path = os.path.join(augmented_folder, original_image_name)\n    \n    shutil.copy(original_image_path, augmented_image_path)\n    \n    # Log the original image metadata\n    augmented_data.append({\n        'image_path': augmented_image_path,\n        'image_name': original_image_name,\n        'patient_id': row['patient_id'],\n        'sex': row['sex'],\n        'age': row['age'],\n        'anatomy': row['anatomy'],\n        'target': row['target']\n    })\n\n# Create the augmentation transformation pipeline\ntransform = Compose([\n    RandomResizedCrop(size=(224, 224), scale=(0.4, 1.0)),\n    ShiftScaleRotate(rotate_limit=90, scale_limit=[0.8, 1.2]),\n    HorizontalFlip(p=0.5),\n    VerticalFlip(p=0.5),\n    HueSaturationValue(sat_shift_limit=[0.7, 1.3], hue_shift_limit=[-0.1, 0.1]),\n    RandomBrightnessContrast(brightness_limit=[0.7, 1.3], contrast_limit=[0.7, 1.3]),\n    Normalize(),\n    ToTensorV2()\n])\n\n# Apply augmentations and save new images\nfor index, row in tqdm(sampled_df.iterrows(), total=len(sampled_df)):\n    image_path = os.path.join(train_folder, row['image_path'])\n    image = cv2.imread(image_path)\n    \n    # Apply transformations (augmentations)\n    augmented_image = transform(image=image)['image']\n    \n    # Save the augmented image\n    augmented_image_name = f\"aug_{os.path.basename(row['image_path'])}\"\n    augmented_image_path = os.path.join(augmented_folder, augmented_image_name)\n    cv2.imwrite(augmented_image_path, augmented_image.permute(1, 2, 0).cpu().numpy() * 255)\n    \n    # Append metadata for the augmented image\n    augmented_data.append({\n        'image_path': augmented_image_path,\n        'image_name': augmented_image_name,\n        'patient_id': row['patient_id'],\n        'sex': row['sex'],\n        'age': row['age'],\n        'anatomy': row['anatomy'],\n        'target': row['target']\n    })\n\n# Convert the augmented data list to a DataFrame\naugmented_df = pd.DataFrame(augmented_data)\n\n# Save the augmented DataFrame to a CSV file\naugmented_df.to_csv(augmented_csv_path, index=False)\n\nprint(\"Augmentation, image saving, and CSV update complete.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T23:20:32.878660Z","iopub.execute_input":"2025-11-26T23:20:32.879526Z","iopub.status.idle":"2025-11-26T23:21:02.216612Z","shell.execute_reply.started":"2025-11-26T23:20:32.879489Z","shell.execute_reply":"2025-11-26T23:21:02.215667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#augmentation with hair removal","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T18:59:00.038425Z","iopub.execute_input":"2025-11-25T18:59:00.039012Z","iopub.status.idle":"2025-11-25T18:59:00.045731Z","shell.execute_reply.started":"2025-11-25T18:59:00.038969Z","shell.execute_reply":"2025-11-25T18:59:00.044676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ================================\n#   IMPORTS\n# ================================\nimport cv2\nimport os\nimport pandas as pd\nfrom tqdm import tqdm\nimport numpy as np\n\n# ================================\n#   HAIR REMOVAL FUNCTION\n# ================================\ndef remove_hair(image):\n    \"\"\"\n    Removes hair from the input image using blackhat morphology and inpainting.\n\n    Args:\n        image (numpy.ndarray): BGR image.\n\n    Returns:\n        numpy.ndarray: Hair-removed image (BGR).\n    \"\"\"\n    # Convert to grayscale\n    gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n    \n    # Apply blackhat morphology to detect hair\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (17, 17))\n    blackhat = cv2.morphologyEx(gray, cv2.MORPH_BLACKHAT, kernel)\n    \n    # Create mask where hair is detected\n    _, mask = cv2.threshold(blackhat, 10, 255, cv2.THRESH_BINARY)\n    \n    # Inpaint to remove hair\n    inpainted = cv2.inpaint(image, mask, inpaintRadius=1, flags=cv2.INPAINT_TELEA)\n    \n    return inpainted\n\n# ================================\n#   FOLDERS AND CSV\n# ================================\n# Folder containing original images\noriginal_folder = '/kaggle/working/augmented_train'  # <-- corrected path (augmented images folder)\n\n# Folder to save hair-removed images\noutput_folder = '/kaggle/working/hairless_train'\nos.makedirs(output_folder, exist_ok=True)\n\n# CSV file with image metadata\ncsv_file = '/kaggle/working/augmented_train.csv'\ndf = pd.read_csv(csv_file)\n\n# ================================\n#   PROCESS IMAGES\n# ================================\nprint(f\"Processing {len(df)} images for hair removal...\")\n\nfor idx, row in tqdm(df.iterrows(), total=len(df)):\n    image_name = row['image_name']\n    input_path = os.path.join(original_folder, image_name)\n    \n    # Load image\n    image = cv2.imread(input_path)\n    if image is None:\n        print(f\"Failed to load: {input_path}\")\n        continue\n\n    # Remove hair\n    hairless_image = remove_hair(image)\n\n    # Save processed image to output folder\n    output_path = os.path.join(output_folder, image_name)\n    cv2.imwrite(output_path, hairless_image)\n\nprint(\"Hair removal complete. Processed images saved to:\", output_folder)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T23:02:40.397465Z","iopub.execute_input":"2025-11-26T23:02:40.398362Z","iopub.status.idle":"2025-11-26T23:04:15.038950Z","shell.execute_reply.started":"2025-11-26T23:02:40.398324Z","shell.execute_reply":"2025-11-26T23:04:15.038064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport os\nimport matplotlib.pyplot as plt\nimport random\n\n# ================================\n#   FOLDERS\n# ================================\noriginal_folder = '/kaggle/working/train'             # original images\nprocessed_folder = '/kaggle/working/augmented_train'  # hair-removed images\n\n# Get list of images present in both folders\nimage_files = [f for f in os.listdir(original_folder) if os.path.exists(os.path.join(processed_folder, f))]\nsample_files = random.sample(image_files, min(5, len(image_files)))  # visualize 5 random samples\n\n# ================================\n#   VISUALIZATION\n# ================================\nfor image_name in sample_files:\n    orig_path = os.path.join(original_folder, image_name)\n    proc_path = os.path.join(processed_folder, image_name)\n    \n    # Load images\n    orig_image = cv2.imread(orig_path)\n    proc_image = cv2.imread(proc_path)\n    \n    if orig_image is None or proc_image is None:\n        continue\n    \n    # Convert BGR -> RGB for plotting\n    orig_image = cv2.cvtColor(orig_image, cv2.COLOR_BGR2RGB)\n    proc_image = cv2.cvtColor(proc_image, cv2.COLOR_BGR2RGB)\n    \n    # Plot side by side\n    plt.figure(figsize=(10,5))\n    plt.subplot(1,2,1)\n    plt.imshow(orig_image)\n    plt.title(\"Original Image\")\n    plt.axis('off')\n    \n    plt.subplot(1,2,2)\n    plt.imshow(proc_image)\n    plt.title(\"Hair Removed Image\")\n    plt.axis('off')\n    \n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:51:59.071546Z","iopub.execute_input":"2025-11-26T22:51:59.072437Z","iopub.status.idle":"2025-11-26T22:52:01.069741Z","shell.execute_reply.started":"2025-11-26T22:51:59.072395Z","shell.execute_reply":"2025-11-26T22:52:01.068541Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Model EVA**","metadata":{}},{"cell_type":"code","source":"!pip install ipywidgets==8.1.1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T22:20:34.764481Z","iopub.execute_input":"2025-11-26T22:20:34.765311Z","iopub.status.idle":"2025-11-26T22:20:43.883857Z","shell.execute_reply.started":"2025-11-26T22:20:34.765275Z","shell.execute_reply":"2025-11-26T22:20:43.882838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"import os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nimport timm\n\n# Define MedicalDataset\nclass MedicalDataset(Dataset):\n    def __init__(self, csv_file, image_dir, metadata_columns, transforms=None):\n        self.data = pd.read_csv(csv_file)\n        self.image_dir = image_dir\n        self.metadata_columns = metadata_columns\n        self.transforms = transforms\n\n        # Handle categorical encoding\n        self.data['sex'] = self.data['sex'].map({'male': 0, 'female': 1}).fillna(-1)\n        self.data['anatomy'] = self.data['anatomy'].astype('category').cat.codes\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        row = self.data.iloc[idx]\n\n        # Load and transform image\n        img_path = os.path.join(self.image_dir, row['image_name'])\n        image = Image.open(img_path).convert(\"RGB\")\n        if self.transforms:\n            image = self.transforms(image)\n\n        # Prepare metadata\n        metadata = row[self.metadata_columns].values.astype(np.float32)\n\n        # Get target\n        target = row['target']\n        return image, torch.tensor(metadata), torch.tensor(target, dtype=torch.float32)\n\n# Define EVAWithMetadata model\nclass EVAWithMetadata(nn.Module):\n    def __init__(self, output_size, metadata_size):\n        super().__init__()\n        # Load the EVA model from timm\n        self.eva = timm.create_model('eva02_small_patch14_336.mim_in22k_ft_in1k', pretrained=True, num_classes=0)\n        image_feature_dim = self.eva.embed_dim\n\n        # Metadata processing\n        self.metadata_fc = nn.Sequential(\n            nn.Linear(metadata_size, 128),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n        )\n\n        # Classification layer\n        self.classification = nn.Sequential(\n            nn.Linear(image_feature_dim + 128, 256),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(256, output_size),\n        )\n\n    def forward(self, image, metadata):\n        image_features = self.eva(image)\n        metadata_features = self.metadata_fc(metadata)\n        combined_features = torch.cat((image_features, metadata_features), dim=1)\n        output = self.classification(combined_features)\n        return output\n\n# Define transforms\nimage_transforms = transforms.Compose([\n    transforms.Resize((336, 336)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5]),\n])\n\n# Load dataset\nmetadata_columns = ['age', 'sex', 'anatomy']\ndataset = MedicalDataset(csv_file='/kaggle/working/augmented_train.csv', \n                         image_dir='/kaggle/working/augmented_train', \n                         metadata_columns=metadata_columns, \n                         transforms=image_transforms)\ndataloader = DataLoader(dataset, batch_size=16, shuffle=True)\n\n# Initialize model\nmetadata_size = len(metadata_columns)\nmodel_EVA = EVAWithMetadata(output_size=1, metadata_size=metadata_size)\nmodel_EVA = model_EVA.to('cuda')\n\n# Define optimizer and loss\noptimizer = optim.Adam(model_EVA.parameters(), lr=1e-4)\ncriterion = nn.BCEWithLogitsLoss()\n\n# Training loop\nfor epoch in range(10):\n    model_EVA.train()\n    epoch_loss = 0.0\n    correct_predictions = 0\n    total_predictions = 0\n\n    #loop through batches\n    for images, metadata, labels in dataloader:\n        images, metadata, labels = images.to('cuda'), metadata.to('cuda'), labels.to('cuda')\n\n        optimizer.zero_grad()\n        outputs = model_EVA(images, metadata).squeeze()\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        epoch_loss += loss.item()\n\n        # Calculate accuracy\n        predicted = torch.sigmoid(outputs)\n        predicted_class = (predicted > 0.5).float()\n        correct_predictions += (predicted_class == labels).sum().item()\n        total_predictions += labels.size(0)\n\n    epoch_accuracy = correct_predictions / total_predictions * 100\n    print(f\"Epoch {epoch+1}, Loss: {epoch_loss / len(dataloader):.4f}, Accuracy: {epoch_accuracy:.2f}%\")\n\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T18:59:08.952311Z","iopub.execute_input":"2025-11-25T18:59:08.952703Z","iopub.status.idle":"2025-11-25T19:33:23.857460Z","shell.execute_reply.started":"2025-11-25T18:59:08.952660Z","shell.execute_reply":"2025-11-25T19:33:23.854999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport os\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader, random_split\nfrom torchvision import transforms\nimport timm\nimport matplotlib.pyplot as plt\n\n# ================================\n#   DATASET CLASS\n# ================================\nclass MedicalDataset(Dataset):\n    def __init__(self, csv_file, image_dir, metadata_columns, transforms=None):\n        self.data = pd.read_csv(csv_file)\n        self.image_dir = image_dir\n        self.metadata_columns = metadata_columns\n        self.transforms = transforms\n\n        # Encode categorical metadata\n        self.data['sex'] = self.data['sex'].map({'male': 0, 'female': 1}).fillna(-1)\n        self.data['anatomy'] = self.data['anatomy'].astype('category').cat.codes\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        row = self.data.iloc[idx]\n        img_path = os.path.join(self.image_dir, row['image_name'])\n\n        # Load preprocessed hair-removed image\n        image = Image.open(img_path).convert(\"RGB\")\n        if self.transforms:\n            image = self.transforms(image)\n\n        metadata = row[self.metadata_columns].values.astype(np.float32)\n        target = row['target']\n\n        return image, torch.tensor(metadata), torch.tensor(target, dtype=torch.float32)\n\n# ================================\n#   EVA MODEL WITH METADATA\n# ================================\nclass EVAWithMetadata(nn.Module):\n    def __init__(self, output_size, metadata_size):\n        super().__init__()\n        # Load pretrained EVA model from timm\n        self.eva = timm.create_model(\n            'eva02_small_patch14_336.mim_in22k_ft_in1k',\n            pretrained=True,\n            num_classes=0  # feature extraction\n        )\n        image_feature_dim = self.eva.embed_dim\n\n        # Metadata fully-connected layers\n        self.metadata_fc = nn.Sequential(\n            nn.Linear(metadata_size, 128),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n        )\n\n        # Classification head\n        self.classification = nn.Sequential(\n            nn.Linear(image_feature_dim + 128, 256),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(256, output_size),\n        )\n\n    def forward(self, image, metadata):\n        image_features = self.eva(image)\n        metadata_features = self.metadata_fc(metadata)\n        combined = torch.cat((image_features, metadata_features), dim=1)\n        return self.classification(combined)\n\n# ================================\n#   TRANSFORMS\n# ================================\nimage_transforms = transforms.Compose([\n    transforms.Resize((336, 336)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.5,0.5,0.5], std=[0.5,0.5,0.5])\n])\n\n# ================================\n#   LOAD DATA\n# ================================\nmetadata_columns = ['age', 'sex', 'anatomy']\n\ndataset = MedicalDataset(\n    csv_file='/kaggle/working/augmented_train.csv',\n    image_dir='/kaggle/working/augmented_train',\n    metadata_columns=metadata_columns,\n    transforms=image_transforms\n)\n\n# Split train/validation\ntrain_size = int(0.8 * len(dataset))\nval_size = len(dataset) - train_size\ntrain_dataset, val_dataset = random_split(dataset, [train_size, val_size])\n\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False)\n\n# ================================\n#   MODEL, LOSS, OPTIMIZER\n# ================================\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'\nmodel = EVAWithMetadata(output_size=1, metadata_size=len(metadata_columns)).to(device)\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters(), lr=1e-4)\n\n# ================================\n#   TRAINING LOOP\n# ================================\nEPOCHS = 10\ntrain_losses, val_losses = [], []\n\nfor epoch in range(EPOCHS):\n    model.train()\n    running_train_loss = 0.0\n\n    for images, metadata, labels in train_loader:\n        images, metadata, labels = images.to(device), metadata.to(device), labels.to(device)\n\n        optimizer.zero_grad()\n        outputs = model(images, metadata).squeeze()\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n\n        running_train_loss += loss.item()\n\n    avg_train_loss = running_train_loss / len(train_loader)\n    train_losses.append(avg_train_loss)\n\n    # Validation\n    model.eval()\n    running_val_loss = 0.0\n    with torch.no_grad():\n        for images, metadata, labels in val_loader:\n            images, metadata, labels = images.to(device), metadata.to(device), labels.to(device)\n            outputs = model(images, metadata).squeeze()\n            loss = criterion(outputs, labels)\n            running_val_loss += loss.item()\n\n    avg_val_loss = running_val_loss / len(val_loader)\n    val_losses.append(avg_val_loss)\n\n    print(f\"Epoch {epoch+1}/{EPOCHS} | Train Loss: {avg_train_loss:.4f} | Val Loss: {avg_val_loss:.4f}\")\n\n# ================================\n#   PLOT LOSS CURVES\n# ================================\nplt.figure(figsize=(8,5))\nplt.plot(train_losses, label='Training Loss')\nplt.plot(val_losses, label='Validation Loss')\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.title(\"Training & Validation Loss\")\nplt.grid(True)\nplt.legend()\nplt.show()\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:09:31.343877Z","iopub.execute_input":"2025-11-27T03:09:31.344245Z","iopub.status.idle":"2025-11-27T03:09:31.968957Z","shell.execute_reply.started":"2025-11-27T03:09:31.344216Z","shell.execute_reply":"2025-11-27T03:09:31.967176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\n\n# Accuracy\nplt.subplot(1,2,1)\nplt.plot(train_acc, label='Train Acc', marker='o')\nplt.plot(val_acc, label='Val Acc', marker='o')\nplt.plot(test_acc, label='Test Acc', marker='o')\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Accuracy\")\nplt.title(\"Accuracy Overview\")\nplt.grid(True)\nplt.legend()\n\n# F1 Score\nplt.subplot(1,2,2)\nplt.plot(train_f1, label='Train F1', marker='s')\nplt.plot(val_f1, label='Val F1', marker='s')\nplt.plot(test_f1, label='Test F1', marker='s')\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"F1 Score\")\nplt.title(\"F1 Score Overview\")\nplt.grid(True)\nplt.legend()\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T02:48:43.869930Z","iopub.execute_input":"2025-11-27T02:48:43.870794Z","iopub.status.idle":"2025-11-27T02:48:44.110263Z","shell.execute_reply.started":"2025-11-27T02:48:43.870739Z","shell.execute_reply":"2025-11-27T02:48:44.109058Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Efficient net7**","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader, random_split\nfrom torchvision import models, transforms\nimport pandas as pd\nfrom PIL import Image\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# ================================\n# Dataset (Hair already removed)\n# ================================\nclass MoleDataset(Dataset):\n    def __init__(self, csv_file, image_dir, metadata_columns, transforms=None):\n        self.data = pd.read_csv(csv_file)\n        self.image_dir = image_dir\n        self.metadata_columns = metadata_columns\n        self.transforms = transforms\n\n        # Convert categorical columns to numerical\n        self.data['sex'] = self.data['sex'].map({'male': 0, 'female': 1}).fillna(-1)\n        self.data['anatomy'] = self.data['anatomy'].astype('category').cat.codes\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        row = self.data.iloc[idx]\n\n        img_path = f\"{self.image_dir}/{row['image_name']}\"\n        image = Image.open(img_path).convert(\"RGB\")\n        if self.transforms:\n            image = self.transforms(image)\n\n        metadata = row[self.metadata_columns].values.astype(np.float32)\n        target = row['target']\n\n        return image, torch.tensor(metadata), torch.tensor(target, dtype=torch.float32)\n\n# ================================\n# Model\n# ================================\nclass EfficientNetWithMetadata(nn.Module):\n    def __init__(self, num_metadata_features, output_size):\n        super(EfficientNetWithMetadata, self).__init__()\n\n        self.backbone = models.efficientnet_b7(pretrained=True)\n        image_features_size = self.backbone.classifier[1].in_features\n        self.backbone.classifier = nn.Identity()\n\n        self.metadata_fc = nn.Linear(num_metadata_features, 128)\n\n        self.classifier = nn.Sequential(\n            nn.Linear(image_features_size + 128, 256),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(256, output_size)\n        )\n\n    def forward(self, image, metadata):\n        image_features = self.backbone(image)\n        metadata_features = self.metadata_fc(metadata)\n        combined = torch.cat((image_features, metadata_features), dim=1)\n        return self.classifier(combined)\n\n# ================================\n# Transforms\n# ================================\nimage_transforms = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std=[0.229, 0.224, 0.225]),\n])\n\n# ================================\n# Dataset & Splits\n# ================================\nmetadata_columns = ['age', 'sex', 'anatomy']\ndataset = MoleDataset(\n    csv_file='/kaggle/working/augmented_train.csv',\n    image_dir='/kaggle/working/augmented_train',\n    metadata_columns=metadata_columns,\n    transforms=image_transforms\n)\n\ntrain_size = int(0.9 * len(dataset))\nval_size = len(dataset) - train_size\n\ntrain_dataset, val_dataset = random_split(dataset, [train_size, val_size])\n\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False)\n\n# ================================\n# Model, Loss, Optimizer\n# ================================\nnum_metadata_features = len(metadata_columns)\nmodel = EfficientNetWithMetadata(num_metadata_features, output_size=1).to(\"cuda\")\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-4)\n\n# Lists to store losses\ntrain_losses = []\nval_losses = []\n\n# ================================\n# Training Loop (with validation)\n# ================================\nnum_epochs = 10\nfor epoch in range(num_epochs):\n    # ---- Training ----\n    model.train()\n    running_loss = 0\n\n    for images, metadata, targets in train_loader:\n        images, metadata, targets = images.cuda(), metadata.cuda(), targets.cuda()\n\n        optimizer.zero_grad()\n        outputs = model(images, metadata).squeeze()\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item()\n\n    train_loss = running_loss / len(train_loader)\n    train_losses.append(train_loss)\n\n    # ---- Validation ----\n    model.eval()\n    val_running_loss = 0\n\n    with torch.no_grad():\n        for images, metadata, targets in val_loader:\n            images, metadata, targets = images.cuda(), metadata.cuda(), targets.cuda()\n            outputs = model(images, metadata).squeeze()\n            loss = criterion(outputs, targets)\n            val_running_loss += loss.item()\n\n    val_loss = val_running_loss / len(val_loader)\n    val_losses.append(val_loss)\n\n    print(f\"Epoch {epoch+1}/{num_epochs} | Train Loss: {train_loss:.4f} | Val Loss: {val_loss:.4f}\")\n\n# ================================\n# Plot Train vs Validation Loss\n# ================================\nplt.figure(figsize=(8,5))\nplt.plot(range(1, num_epochs+1), train_losses, marker='o', label='Training Loss')\nplt.plot(range(1, num_epochs+1), val_losses, marker='o', label='Validation Loss')\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.title(\"Training vs Validation Loss (EfficientNet)\")\nplt.grid(True)\nplt.legend()\nplt.show()\n\n# ================================\n# Save Model\n# ================================\ntorch.save(model.state_dict(), \"efficientnet_with_metadata.pth\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T03:04:45.810618Z","iopub.execute_input":"2025-11-27T03:04:45.811002Z","iopub.status.idle":"2025-11-27T03:04:47.549475Z","shell.execute_reply.started":"2025-11-27T03:04:45.810968Z","shell.execute_reply":"2025-11-27T03:04:47.547757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import models, transforms\nimport pandas as pd\nfrom PIL import Image\nimport numpy as np\n\n# ================================\n# Dataset (Hair already removed)\n# ================================\nclass MoleDataset(Dataset):\n    def __init__(self, csv_file, image_dir, metadata_columns, transforms=None):\n        self.data = pd.read_csv(csv_file)\n        self.image_dir = image_dir\n        self.metadata_columns = metadata_columns\n        self.transforms = transforms\n\n        # Convert categorical columns to numerical\n        self.data['sex'] = self.data['sex'].map({'male': 0, 'female': 1}).fillna(-1)\n        self.data['anatomy'] = self.data['anatomy'].astype('category').cat.codes\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        row = self.data.iloc[idx]\n\n        # Load preprocessed image (hair already removed)\n        img_path = f\"{self.image_dir}/{row['image_name']}\"\n        image = Image.open(img_path).convert(\"RGB\")\n        if self.transforms:\n            image = self.transforms(image)\n\n        # Prepare metadata\n        metadata = row[self.metadata_columns].values.astype(np.float32)\n\n        # Get target\n        target = row['target']\n        return image, torch.tensor(metadata), torch.tensor(target, dtype=torch.float32)\n\n# ================================\n# Model\n# ================================\nclass EfficientNetWithMetadata(nn.Module):\n    def __init__(self, num_metadata_features, output_size):\n        super(EfficientNetWithMetadata, self).__init__()\n        # Load pre-trained EfficientNet-B7\n        self.backbone = models.efficientnet_b7(pretrained=True)\n\n        # Extract image features size\n        image_features_size = self.backbone.classifier[1].in_features\n        self.backbone.classifier = nn.Identity()  # Remove original classifier\n\n        # Metadata processing\n        self.metadata_fc = nn.Linear(num_metadata_features, 128)\n\n        # Combined classifier\n        self.classifier = nn.Sequential(\n            nn.Linear(image_features_size + 128, 256),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(256, output_size)\n        )\n\n    def forward(self, image, metadata):\n        image_features = self.backbone(image)\n        metadata_features = self.metadata_fc(metadata)\n        combined_features = torch.cat((image_features, metadata_features), dim=1)\n        return self.classifier(combined_features)\n\n# ================================\n# Transforms\n# ================================\nimage_transforms = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std=[0.229, 0.224, 0.225]),\n])\n\n# ================================\n# Dataset & DataLoader\n# ================================\nmetadata_columns = ['age', 'sex', 'anatomy']\ndataset = MoleDataset(\n    csv_file='/kaggle/working/augmented_train.csv',\n    image_dir='/kaggle/working/augmented_train',\n    metadata_columns=metadata_columns,\n    transforms=image_transforms\n)\ndataloader = DataLoader(dataset, batch_size=16, shuffle=True)\n\n# ================================\n# Model, Loss, Optimizer\n# ================================\nnum_metadata_features = len(metadata_columns)\nmodel_EffNet = EfficientNetWithMetadata(num_metadata_features=num_metadata_features, output_size=1).to('cuda')\n\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = torch.optim.Adam(model_EffNet.parameters(), lr=1e-4)\n\n# ================================\n# Training Loop\n# ================================\nfor epoch in range(10):\n    model_EffNet.train()\n    epoch_loss = 0.0\n    correct_predictions = 0\n    total_predictions = 0\n    \n    for images, metadata, targets in dataloader:\n        images, metadata, targets = images.to('cuda'), metadata.to('cuda'), targets.to('cuda')\n\n        optimizer.zero_grad()\n        outputs = model_EffNet(images, metadata).squeeze()\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n\n        predicted = torch.sigmoid(outputs)\n        predicted_class = (predicted > 0.5).float()\n        correct_predictions += (predicted_class == targets).sum().item()\n        total_predictions += targets.size(0)\n        epoch_loss += loss.item()\n\n    epoch_accuracy = correct_predictions / total_predictions * 100\n    print(f\"Epoch {epoch+1}, Loss: {epoch_loss / len(dataloader):.4f}, Accuracy: {epoch_accuracy:.2f}%\")\n\n# ================================\n# Save Model\n# ================================\ntorch.save(model_EffNet.state_dict(), \"efficientnet_with_metadata.pth\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T01:03:06.651910Z","iopub.execute_input":"2025-11-27T01:03:06.652296Z","iopub.status.idle":"2025-11-27T02:37:47.687841Z","shell.execute_reply.started":"2025-11-27T01:03:06.652267Z","shell.execute_reply":"2025-11-27T02:37:47.686841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Loss and Accuracy lists\nepoch_losses = [0.3785, 0.3033, 0.2508, 0.2154, 0.1910, 0.1734, 0.1671, 0.1627, 0.1592, 0.1568]\nepoch_accuracies = [82.4, 86.34, 88.69, 90.09, 91.34, 91.88, 91.88, 92.09, 92.38, 92.33]\n\nepochs = range(1, len(epoch_losses) + 1)\n\n# ---- Plot LOSS ----\nplt.figure(figsize=(8,5))\nplt.plot(epochs, epoch_losses, marker='o')\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Loss\")\nplt.title(\"Training Loss Curve (EfficientNet)\")\nplt.grid(True)\nplt.show()\n\n# ---- Plot ACCURACY ----\nplt.figure(figsize=(8,5))\nplt.plot(epochs, epoch_accuracies, marker='o')\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Accuracy (%)\")\nplt.title(\"Training Accuracy Curve (EfficientNet)\")\nplt.grid(True)\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T02:57:38.415472Z","iopub.execute_input":"2025-11-27T02:57:38.415877Z","iopub.status.idle":"2025-11-27T02:57:38.909004Z","shell.execute_reply.started":"2025-11-27T02:57:38.415832Z","shell.execute_reply":"2025-11-27T02:57:38.908098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\n\n# Assuming y_true and y_pred are tensors or arrays for the test set\ny_true = test_labels.cpu().numpy()   # shape=(num_samples,)\ny_pred_probs = torch.sigmoid(test_outputs).cpu().numpy()\ny_pred = (y_pred_probs > 0.5).astype(int)  # binary classification\n\n# Classification report per class\nreport = classification_report(y_true, y_pred, target_names=['Class 0', 'Class 1'])\nprint(\"Per-Class Metrics (Test Set):\")\nprint(report)\n\n# Optional: confusion matrix\ncm = confusion_matrix(y_true, y_pred)\nplt.figure(figsize=(4,4))\nplt.imshow(cm, cmap='Blues')\nplt.title(\"Confusion Matrix\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"Actual\")\nfor i in range(cm.shape[0]):\n    for j in range(cm.shape[1]):\n        plt.text(j, i, cm[i,j], ha='center', va='center', color='red', fontsize=12)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T02:48:54.046629Z","iopub.execute_input":"2025-11-27T02:48:54.047593Z","iopub.status.idle":"2025-11-27T02:48:54.229490Z","shell.execute_reply.started":"2025-11-27T02:48:54.047532Z","shell.execute_reply":"2025-11-27T02:48:54.228264Z"}},"outputs":[],"execution_count":null}]}