{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":1225697,"sourceType":"datasetVersion","datasetId":701123}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Overview SIIM-ISIC dataset (test)**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load the CSV file\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')\n\n# Count the number of rows\nnum_rows_test = len(test_df)\nnum_rows_test","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:54:44.63827Z","iopub.execute_input":"2024-11-20T08:54:44.638538Z","iopub.status.idle":"2024-11-20T08:54:45.61964Z","shell.execute_reply.started":"2024-11-20T08:54:44.638512Z","shell.execute_reply":"2024-11-20T08:54:45.61875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract only 2000 rows from test_df\nnew_test_df = test_df.head(2000)\n\n# Save the extracted rows into a new CSV file under /kaggle/working/\nnew_test_df.to_csv('/kaggle/working/test.csv', index=False)\n\nprint(\"2000 rows have been saved to /kaggle/working/test.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:54:45.621016Z","iopub.execute_input":"2024-11-20T08:54:45.621276Z","iopub.status.idle":"2024-11-20T08:54:45.636253Z","shell.execute_reply.started":"2024-11-20T08:54:45.621249Z","shell.execute_reply":"2024-11-20T08:54:45.635226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a copy of new_test_df to avoid SettingWithCopyWarning\nnew_test_df = new_test_df.copy()\n\n# Rename columns\nnew_test_df.rename(columns={'age_approx': 'age', 'anatom_site_general_challenge': 'anatomy'}, inplace=True)\n\n# Display the updated DataFrame\nprint(new_test_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:54:45.637468Z","iopub.execute_input":"2024-11-20T08:54:45.637854Z","iopub.status.idle":"2024-11-20T08:54:45.653256Z","shell.execute_reply.started":"2024-11-20T08:54:45.63781Z","shell.execute_reply":"2024-11-20T08:54:45.652345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dictionary mapping anatomy values to corresponding numbers\nanatomy_mapping = {\n    'anterior torso': 1,\n    'torso': 2,\n    'lower extremity': 3,\n    'posterior torso': 4,\n    'upper extremity': 5,\n    'head/neck': 6,\n    'lateral torso': 7,\n    'palms/soles': 8,\n    'oral/genital': 9\n}\n\n# Replace values in 'anatomy' column using the mapping\nnew_test_df['anatomy'] = new_test_df['anatomy'].replace(anatomy_mapping)\n\n# Display the updated DataFrame\nprint(new_test_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:54:45.655639Z","iopub.execute_input":"2024-11-20T08:54:45.655961Z","iopub.status.idle":"2024-11-20T08:54:45.670812Z","shell.execute_reply.started":"2024-11-20T08:54:45.655923Z","shell.execute_reply":"2024-11-20T08:54:45.669816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dictionary mapping anatomy values to corresponding numbers\nsex_mapping = {\n    'male': 0,\n    'female': 1\n}\n\n# Replace values in 'age' column using the mapping\nnew_test_df['sex'] = new_test_df['sex'].replace(sex_mapping)\n\n# Display the updated DataFrame\nprint(new_test_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:54:45.672008Z","iopub.execute_input":"2024-11-20T08:54:45.672322Z","iopub.status.idle":"2024-11-20T08:54:45.681675Z","shell.execute_reply.started":"2024-11-20T08:54:45.672286Z","shell.execute_reply":"2024-11-20T08:54:45.680552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\n\n# Directory paths\nsource_dir = '/kaggle/input/siim-isic-melanoma-classification/jpeg/test'\ndestination_dir = '/kaggle/working/test'\n\n# Create the destination directory if it doesn't exist\nos.makedirs(destination_dir, exist_ok=True)\n\n# Loop through the image names in new_test_df and copy matching images to the new folder\nfor image_name in new_test_df['image_name']:\n    # Construct the image filename with .jpg\n    image_filename = image_name + '.jpg'\n    \n    # Construct full paths for source and destination\n    source_path = os.path.join(source_dir, image_filename)\n    destination_path = os.path.join(destination_dir, image_filename)\n    \n    # Check if the file exists in the source directory\n    if os.path.exists(source_path):\n        # Copy the file to the destination directory\n        shutil.copy2(source_path, destination_path)\n        print(f\"Copied {image_filename} to {destination_dir}\")\n    else:\n        print(f\"Image {image_filename} not found in {source_dir}\")\n\nprint(f\"All found images have been copied to: {destination_dir}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:54:45.682884Z","iopub.execute_input":"2024-11-20T08:54:45.683203Z","iopub.status.idle":"2024-11-20T08:55:25.837696Z","shell.execute_reply.started":"2024-11-20T08:54:45.683166Z","shell.execute_reply":"2024-11-20T08:55:25.836743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Overview External Data (Train)**","metadata":{}},{"cell_type":"code","source":"# File paths\ntrain_path_external = '/kaggle/input/melanoma-external-malignant-256/train_concat.csv'\n\n# Read the CSV files\ntrain_df_external = pd.read_csv(train_path_external)\n\n# Display the first few rows of each DataFrame\nprint(\"Train CSV Head:\")\nprint(train_df_external.head())","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:55:25.838977Z","iopub.execute_input":"2024-11-20T08:55:25.839237Z","iopub.status.idle":"2024-11-20T08:55:25.914338Z","shell.execute_reply.started":"2024-11-20T08:55:25.839211Z","shell.execute_reply":"2024-11-20T08:55:25.913501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add a new column 'diagnosis' with the value 'melanoma' for all rows\ntrain_df_external['diagnosis'] = 'melanoma'\n\n# Display the head of the DataFrame to verify\nprint(train_df_external.head())","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:55:25.915391Z","iopub.execute_input":"2024-11-20T08:55:25.915761Z","iopub.status.idle":"2024-11-20T08:55:25.923606Z","shell.execute_reply.started":"2024-11-20T08:55:25.915721Z","shell.execute_reply":"2024-11-20T08:55:25.92275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count the number of rows in each DataFrame\ntrain_row_count = len(train_df_external)\nprint(train_row_count)","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:55:25.924678Z","iopub.execute_input":"2024-11-20T08:55:25.924931Z","iopub.status.idle":"2024-11-20T08:55:25.934163Z","shell.execute_reply.started":"2024-11-20T08:55:25.924905Z","shell.execute_reply":"2024-11-20T08:55:25.933423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count the number of rows in each DataFrame\ntrain_row_count = len(train_df_external)\n\n# Count occurrences of target values in the training data\ntarget_counts = train_df_external['target'].value_counts()\n\n# Display the results\nprint(f\"Number of rows in train_df_external: {train_row_count}\")\nprint(f\"\\nTarget counts in train_df_external:\")\nprint(f\"Target = 0: {target_counts.get(0, 0)}\")\nprint(f\"Target = 1: {target_counts.get(1, 0)}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:55:25.937312Z","iopub.execute_input":"2024-11-20T08:55:25.937656Z","iopub.status.idle":"2024-11-20T08:55:25.95243Z","shell.execute_reply.started":"2024-11-20T08:55:25.93763Z","shell.execute_reply":"2024-11-20T08:55:25.951492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# File paths\ntrain_path_external = '/kaggle/input/melanoma-external-malignant-256/train_concat.csv'\n\n# Read the CSV files\ntrain_df_external = pd.read_csv(train_path_external)# Extract rows where target == 0 and target == 1\n\ntarget_0 = train_df_external[train_df_external['target'] == 0].sample(n=5000, random_state=42)\ntarget_1 = train_df_external[train_df_external['target'] == 1].sample(n=5000, random_state=42)\n\n# Combine the two subsets\nsampled_df = pd.concat([target_0, target_1])\n\n# Shuffle the rows (optional, if you want random order)\nsampled_df = sampled_df.sample(frac=1, random_state=42).reset_index(drop=True)\n\n# Save the resulting DataFrame to a new CSV file\nsampled_df.to_csv('/kaggle/working/train.csv', index=False)\n\nprint(\"CSV file with 5000 rows for target 0 and 5000 rows for target 1 has been created.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:55:25.95335Z","iopub.execute_input":"2024-11-20T08:55:25.95372Z","iopub.status.idle":"2024-11-20T08:55:26.050778Z","shell.execute_reply.started":"2024-11-20T08:55:25.953683Z","shell.execute_reply":"2024-11-20T08:55:26.049753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\n\n# Define the source and destination directories\nsource_directory = '/kaggle/input/melanoma-external-malignant-256/train/train/'\ndestination_directory = '/kaggle/working/train/'\n\n# Create the destination directory if it doesn't exist\nos.makedirs(destination_directory, exist_ok=True)\n\n# Iterate over the image names in sampled_df\nfor image_name in sampled_df['image_name']:\n    # Append '.jpg' to each image_name\n    image_path = os.path.join(source_directory, image_name + '.jpg')\n    \n    # Check if the image exists in the source directory\n    if os.path.exists(image_path):\n        # Define the new destination path\n        destination_path = os.path.join(destination_directory, image_name + '.jpg')\n        \n        # Copy the image to the new directory\n        shutil.copy2(image_path, destination_path)  # copy2 to preserve metadata\n        print(f\"Copied {image_name}.jpg to {destination_directory}\")\n    else:\n        print(f\"Image {image_name}.jpg does not exist in the source directory.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:55:26.05196Z","iopub.execute_input":"2024-11-20T08:55:26.052248Z","iopub.status.idle":"2024-11-20T08:57:05.24048Z","shell.execute_reply.started":"2024-11-20T08:55:26.05222Z","shell.execute_reply":"2024-11-20T08:57:05.239651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Add image path in train.csv\n\n# Directory where images are stored\nimage_directory = '/kaggle/working/train/'\n\n# Add a new column 'image_path' with the full path including the .jpg extension\nsampled_df['image_path'] = image_directory + sampled_df['image_name'] + '.jpg'\n\n# Display the updated dataframe\nprint(sampled_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:57:05.241704Z","iopub.execute_input":"2024-11-20T08:57:05.242077Z","iopub.status.idle":"2024-11-20T08:57:05.253586Z","shell.execute_reply.started":"2024-11-20T08:57:05.242035Z","shell.execute_reply":"2024-11-20T08:57:05.252732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Rename columns in sampled_df\nsampled_df.rename(columns={'age_approx': 'age', 'anatom_site_general_challenge': 'anatomy'}, inplace=True)\n\n# Display the updated DataFrame\nprint(sampled_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:57:05.254632Z","iopub.execute_input":"2024-11-20T08:57:05.254881Z","iopub.status.idle":"2024-11-20T08:57:05.271194Z","shell.execute_reply.started":"2024-11-20T08:57:05.254857Z","shell.execute_reply":"2024-11-20T08:57:05.270432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get unique values in the 'anatomy' column\nunique_anatomy_values = sampled_df['anatomy'].unique()\n\n# Display the unique values\nprint(\"Unique values in the 'anatomy' column:\")\nprint(unique_anatomy_values)","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:57:05.272332Z","iopub.execute_input":"2024-11-20T08:57:05.272633Z","iopub.status.idle":"2024-11-20T08:57:05.284934Z","shell.execute_reply.started":"2024-11-20T08:57:05.272607Z","shell.execute_reply":"2024-11-20T08:57:05.284234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dictionary mapping anatomy values to corresponding numbers\nanatomy_mapping = {\n    'anterior torso': 1,\n    'torso': 2,\n    'lower extremity': 3,\n    'posterior torso': 4,\n    'upper extremity': 5,\n    'head/neck': 6,\n    'lateral torso': 7,\n    'palms/soles': 8,\n    'oral/genital': 9\n}\n\n# Replace values in 'anatomy' column using the mapping\nsampled_df['anatomy'] = sampled_df['anatomy'].replace(anatomy_mapping)\n\n# Display the updated DataFrame\nprint(sampled_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:57:05.285879Z","iopub.execute_input":"2024-11-20T08:57:05.28612Z","iopub.status.idle":"2024-11-20T08:57:05.305768Z","shell.execute_reply.started":"2024-11-20T08:57:05.286097Z","shell.execute_reply":"2024-11-20T08:57:05.304999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dictionary mapping anatomy values to corresponding numbers\nsex_mapping = {\n    'male': 0,\n    'female': 1\n}\n\n# Replace values in 'age' column using the mapping\nsampled_df['sex'] = sampled_df['sex'].replace(sex_mapping)\n\n# Display the updated DataFrame\nprint(sampled_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:57:05.306827Z","iopub.execute_input":"2024-11-20T08:57:05.307562Z","iopub.status.idle":"2024-11-20T08:57:05.320342Z","shell.execute_reply.started":"2024-11-20T08:57:05.307516Z","shell.execute_reply":"2024-11-20T08:57:05.319598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Load the concatenated datasets\ntrain_df = pd.read_csv('/kaggle/working/train.csv')\ntest_df = pd.read_csv('/kaggle/working/test.csv')\n\n# Cell 1: Basic Data Information - Combined Train Dataset\ndef data_summary(df, dataset_name):\n    print(f\"\\nDataset Summary: {dataset_name}\\n\")\n    print(\"Shape:\", df.shape)\n    print(\"\\nColumns:\", df.columns.tolist())\n    print(\"\\nFirst 5 Rows:\\n\", df.head())\n    print(\"\\nSummary Statistics:\\n\", df.describe())\n    print(\"\\nMissing Values:\")\n    print(df.isna().sum())\n\ndata_summary(train_df, \"Combined Train Dataset\")\ndata_summary(test_df, \"Test Dataset\")\n\n# Cell 2: Analyzing the Target Column - Combined Train Dataset\ntrain_target_counts = train_df['target'].value_counts()\nprint(\"\\nCombined Train Dataset Target Counts:\\n\", train_target_counts)\n\n# Cell 3: Plotting Class Distribution for the Combined Train Dataset\ndef plot_class_distribution(counts, title):\n    sns.barplot(x=counts.index, y=counts.values)\n    plt.xlabel('Target Class')\n    plt.ylabel('Count')\n    plt.title(title)\n    plt.show()\n\nplot_class_distribution(train_target_counts, \"Combined Train Dataset Target Class Distribution\")\n\n# Cell 4: Exploring Features - Distribution by Sex, Age, and Anatomical Site in Combined Train Dataset\nfeatures = ['sex', 'age_approx', 'anatom_site_general_challenge']\nfor feature in features:\n    plt.figure(figsize=(10, 5))\n    sns.countplot(data=train_df, x=feature, hue='target')\n    plt.title(f\"{feature.capitalize()} Distribution by Target (Combined Train Dataset)\")\n    plt.show()\n\n# Cell 5: Check for Duplicated Records - Combined Train Dataset\nduplicated_records = train_df.duplicated().sum()\nprint(f\"\\nNumber of duplicated records in combined training data: {duplicated_records}\")\n\n# Cell 6: Correlation Matrix for Numerical Features - Combined Train Dataset\nplt.figure(figsize=(10, 8))\nnumeric_features = train_df.select_dtypes(include=['int64', 'float64'])\nsns.heatmap(numeric_features.corr(), annot=True, cmap='coolwarm', linewidths=0.5)\nplt.title(\"Correlation Matrix of Combined Train Dataset\")\nplt.show()\n\n# Cell 7: Missing Values Analysis\n# Display the count of missing values per column\nprint(\"\\nMissing Values in Combined Train Dataset:\")\nmissing_values = train_df.isna().sum()\nprint(missing_values)\n\n# Visualize missing values\nplt.figure(figsize=(12, 6))\nsns.heatmap(train_df.isna(), cbar=False, cmap='viridis')\nplt.title(\"Missing Values Heatmap for Combined Train Dataset\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:57:05.321478Z","iopub.execute_input":"2024-11-20T08:57:05.32208Z","iopub.status.idle":"2024-11-20T08:57:07.478309Z","shell.execute_reply.started":"2024-11-20T08:57:05.322042Z","shell.execute_reply":"2024-11-20T08:57:07.477419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Load the concatenated datasets\ntrain_df = pd.read_csv('/kaggle/working/train.csv')\ntest_df = pd.read_csv('/kaggle/working/test.csv')\n\n# Cell 1: Basic Data Information - Combined Train Dataset\ndef data_summary(df, dataset_name):\n    print(f\"\\nDataset Summary: {dataset_name}\\n\")\n    print(\"Shape:\", df.shape)\n    print(\"\\nColumns:\", df.columns.tolist())\n    print(\"\\nFirst 5 Rows:\\n\", df.head())\n    print(\"\\nSummary Statistics:\\n\", df.describe())\n    print(\"\\nMissing Values:\")\n    print(df.isna().sum())\n\ndata_summary(train_df, \"Combined Train Dataset\")\ndata_summary(test_df, \"Test Dataset\")\n\n# Cell 2: Analyzing the Target Column - Combined Train Dataset\ntrain_target_counts = train_df['target'].value_counts()\nprint(\"\\nCombined Train Dataset Target Counts:\\n\", train_target_counts)\n\n# Cell 3: Plotting Class Distribution for the Combined Train Dataset\ndef plot_class_distribution(counts, title):\n    sns.barplot(x=counts.index, y=counts.values)\n    plt.xlabel('Target Class')\n    plt.ylabel('Count')\n    plt.title(title)\n    plt.show()\n\nplot_class_distribution(train_target_counts, \"Combined Train Dataset Target Class Distribution\")\n\n# Cell 4: Exploring Features - Distribution by Sex, Age, and Anatomical Site in Combined Train Dataset\nfeatures = ['sex', 'age_approx', 'anatom_site_general_challenge']\nfor feature in features:\n    plt.figure(figsize=(10, 5))\n    sns.countplot(data=train_df, x=feature, hue='target')\n    plt.title(f\"{feature.capitalize()} Distribution by Target (Combined Train Dataset)\")\n    plt.show()\n\n# Cell 5: Check for Duplicated Records - Combined Train Dataset\nduplicated_records = train_df.duplicated().sum()\nprint(f\"\\nNumber of duplicated records in combined training data: {duplicated_records}\")\n\n# Cell 6: Correlation Matrix for Numerical Features - Combined Train Dataset\nplt.figure(figsize=(10, 8))\nnumeric_features = train_df.select_dtypes(include=['int64', 'float64'])\nsns.heatmap(numeric_features.corr(), annot=True, cmap='coolwarm', linewidths=0.5)\nplt.title(\"Correlation Matrix of Combined Train Dataset\")\nplt.show()\n\n# Cell 7: Missing Values Analysis\n# Handling missing values in the Combined Train Dataset\n\n# 1. Fill missing 'sex' values with the most common value\ntrain_df['sex'] = train_df['sex'].fillna(train_df['sex'].mode()[0])\n\n# 2. Fill missing 'age_approx' values with the median age\ntrain_df['age_approx'] = train_df['age_approx'].fillna(train_df['age_approx'].median())\n\n# 3. Fill missing 'anatom_site_general_challenge' with the most common value\ntrain_df['anatom_site_general_challenge'] = train_df['anatom_site_general_challenge'].fillna(train_df['anatom_site_general_challenge'].mode()[0])\n\n# 4. Handle missing 'patient_id' by filling it with 'unknown'\ntrain_df['patient_id'] = train_df['patient_id'].fillna('unknown')\n\n# Summary statistics after missing values treatment\nprint(\"\\nSummary of the combined dataset after handling missing values:\")\ndata_summary(train_df, \"Combined Train Dataset After Handling Missing Values\")\n\n# Save the cleaned dataset (overwrite the original train.csv)\ntrain_df.to_csv('/kaggle/working/train.csv', index=False)\nprint(f\"\\nCleaned combined training dataset saved back to '/kaggle/working/train/train.csv' for further processing.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T08:57:07.479444Z","iopub.execute_input":"2024-11-20T08:57:07.479724Z","iopub.status.idle":"2024-11-20T08:57:08.94569Z","shell.execute_reply.started":"2024-11-20T08:57:07.479696Z","shell.execute_reply":"2024-11-20T08:57:08.944811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport albumentations as A\nimport matplotlib.pyplot as plt\nimport random\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nfrom PIL import Image\n\n# Suppress Albumentations update check warning\nos.environ[\"NO_ALBUMENTATIONS_UPDATE\"] = \"1\"\n\n# Resize parameters for consistent model input size\nTARGET_SIZE = (224, 224)\n\n# Define the directory containing training images\ntrain_images_dir = '/kaggle/working/train'  # Update this to match the correct path to your images\n\n# Load the training dataframe\ntrain_df_path = '/kaggle/working/train.csv'  # Update this to match the correct CSV path\ntrain_df = pd.read_csv(train_df_path)\n\n# Function to resize and augment an image\ndef process_image(image_path, augmentation_type=None):\n    image = cv2.imread(image_path)\n    if image is None:\n        return None, None\n\n    # Convert BGR to RGB for consistent color channels\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    # Resize the image while maintaining the aspect ratio\n    height, width = image.shape[:2]\n    scale = min(TARGET_SIZE[0] / height, TARGET_SIZE[1] / width)\n    resized_dim = (int(width * scale), int(height * scale))\n    resized_image = cv2.resize(image, resized_dim, interpolation=cv2.INTER_AREA)\n    \n    # Pad the resized image to match TARGET_SIZE\n    delta_w = TARGET_SIZE[1] - resized_dim[0]\n    delta_h = TARGET_SIZE[0] - resized_dim[1]\n    top, bottom = delta_h // 2, delta_h - (delta_h // 2)\n    left, right = delta_w // 2, delta_w - (delta_w // 2)\n    color = [0, 0, 0]  # Black padding\n    padded_image = cv2.copyMakeBorder(resized_image, top, bottom, left, right, cv2.BORDER_CONSTANT, value=color)\n    \n    # Apply augmentation based on the specified type\n    if augmentation_type == \"blur\":\n        augmented_image = cv2.addWeighted(padded_image, 4, cv2.GaussianBlur(padded_image, (0, 0), 256 / 10), -4, 128)\n    elif augmentation_type == \"flip_rotate\":\n        augmentor = A.Compose([\n            A.HorizontalFlip(p=0.5),\n            A.VerticalFlip(p=0.5),\n            A.Rotate(limit=45, p=0.5),\n        ])\n        augmented = augmentor(image=padded_image)\n        augmented_image = augmented['image']\n    elif augmentation_type == \"elastic\":\n        augmentor = A.ElasticTransform(p=1.0)\n        augmented = augmentor(image=padded_image)\n        augmented_image = augmented['image']\n    else:\n        augmented_image = padded_image\n    \n    return padded_image, augmented_image\n\n# Data augmentation on the training data and saving the new images\nnew_rows = []\naugmentation_types = [\"blur\", \"flip_rotate\", \"elastic\"]\n\n# Iterating through the training dataset\nfor _, row in train_df.iterrows():\n    image_name = row['image_name']\n    image_path = os.path.join(train_images_dir, image_name + '.jpg')\n    if not os.path.exists(image_path):\n        print(f\"Warning: Image {image_path} not found. Skipping.\")\n        continue\n\n    # Apply each augmentation type to the image\n    for aug_type in augmentation_types:\n        _, augmented_image = process_image(image_path, augmentation_type=aug_type)\n        \n        # Skip if the image could not be processed\n        if augmented_image is None:\n            print(f\"Warning: Could not process image '{image_name}' with augmentation '{aug_type}'. Skipping.\")\n            continue\n        \n        # Create a new image name for the augmented image\n        new_image_name = f\"{image_name}_{aug_type}\"\n        new_image_path = os.path.join('/kaggle/working/', new_image_name + '.jpg')\n        \n        # Save the augmented image in the output directory\n        cv2.imwrite(new_image_path, cv2.cvtColor(augmented_image, cv2.COLOR_RGB2BGR))\n        \n        # Create a new row for the augmented image with the same attributes as the original image\n        new_row = row.copy()\n        new_row['image_name'] = new_image_name\n        new_rows.append(new_row)\n\n# Create a DataFrame for the augmented data\naugmented_df = pd.DataFrame(new_rows)\n\n# Combine the original training DataFrame with the augmented data\ncombined_train_df = pd.concat([train_df, augmented_df], ignore_index=True)\n\n# Save the updated combined training dataset with augmented data\ncombined_train_df.to_csv('/kaggle/working/train_augmented.csv', index=False)\nprint(\"Augmented training dataset saved to '/kaggle/working/train_augmented.csv'\")\n\n# Visualization: Compare Original and Augmented Images\n# Select a random image from the original dataset for visualization\noriginal_image_name = random.choice(train_df['image_name'].tolist())\noriginal_image_path = os.path.join(train_images_dir, original_image_name + '.jpg')\n\n# Display the original and its augmentations\nfig, axes = plt.subplots(1, 4, figsize=(20, 5))\nplt.suptitle(f\"Original and Augmented Versions of Image: {original_image_name}\", fontsize=16)\n\n# Display the original image\noriginal_image = cv2.imread(original_image_path)\n\n# Check if the original image was loaded successfully\nif original_image is None:\n    print(f\"Warning: Could not load the image '{original_image_path}'. Please check the file path.\")\n    axes[0].set_title(\"Original Image Not Found\")\n    axes[0].axis('off')\nelse:\n    original_image = cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB)\n    axes[0].imshow(original_image)\n    axes[0].set_title(\"Original\")\n    axes[0].axis('off')\n\n# Display the augmented images\nfor idx, aug_type in enumerate(augmentation_types):\n    augmented_image_name = f\"{original_image_name}_{aug_type}\"\n    augmented_image_path = os.path.join('/kaggle/working/', augmented_image_name + '.jpg')\n    augmented_image = cv2.imread(augmented_image_path)\n\n    # Check if the augmented image was loaded successfully\n    if augmented_image is None:\n        print(f\"Warning: Could not load the augmented image '{augmented_image_path}'. Please check the file path.\")\n        axes[idx + 1].set_title(f\"{aug_type} Not Found\")\n        axes[idx + 1].axis('off')\n    else:\n        augmented_image = cv2.cvtColor(augmented_image, cv2.COLOR_BGR2RGB)\n        axes[idx + 1].imshow(augmented_image)\n        axes[idx + 1].set_title(aug_type)\n        axes[idx + 1].axis('off')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:04:45.526173Z","iopub.execute_input":"2024-11-20T09:04:45.526514Z","iopub.status.idle":"2024-11-20T09:15:40.680919Z","shell.execute_reply.started":"2024-11-20T09:04:45.526487Z","shell.execute_reply":"2024-11-20T09:15:40.680071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport albumentations as A\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nimport skimage\nimport random\nimport torchvision.transforms as transforms\nfrom torch.utils.data import DataLoader, Dataset\nimport torch\nimport torchvision\nfrom PIL import Image\nfrom scipy.stats import entropy\nfrom skimage.filters import gabor\nfrom skimage.feature import local_binary_pattern\nfrom skimage.measure import shannon_entropy\n\n# Step 2: Enhanced Color, Texture, and Size Feature Extraction\ncolor_features = []\ntexture_features = []\nshape_features = []\n\nfor _, row in combined_df.iterrows():\n    image_path = os.path.join(train_images_dir, row['image_name'] + '.jpg')\n    if not os.path.exists(image_path):\n        continue\n    \n    # Process image (resize, pad, and no augmentation for feature extraction)\n    image = cv2.imread(image_path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    \n    # Color Feature Extraction\n    reshaped_img = image.reshape((-1, 3))\n    rgb_mean = np.mean(reshaped_img, axis=0)\n    rgb_std = np.std(reshaped_img, axis=0)\n    color_entropy = shannon_entropy(image)\n    color_features.append([rgb_mean[0], rgb_mean[1], rgb_mean[2], rgb_std[0], rgb_std[1], rgb_std[2], color_entropy])\n    \n    # Texture Feature Extraction using Local Binary Pattern (LBP)\n    gray_image = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n    # Using LBP for texture feature extraction\n    lbp = local_binary_pattern(gray_image, P=8, R=1, method='uniform')\n    lbp_hist, _ = np.histogram(lbp.ravel(), bins=np.arange(0, lbp.max() + 1), density=True)\n    texture_features.append(lbp_hist.tolist())\n\n    # Shape Feature Extraction\n    gray_blurred = cv2.GaussianBlur(gray_image, (5, 5), 0)\n    _, thresh = cv2.threshold(gray_blurred, 60, 255, cv2.THRESH_BINARY)\n    contours, _ = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    if contours:\n        c = max(contours, key=cv2.contourArea)\n        perimeter = cv2.arcLength(c, True)\n        area = cv2.contourArea(c)\n        aspect_ratio = float(image.shape[1]) / image.shape[0]\n        circularity = 4 * np.pi * (area / (perimeter * perimeter)) if perimeter != 0 else 0\n        shape_features.append([aspect_ratio, circularity, area, perimeter])\n    else:\n        shape_features.append([0, 0, 0, 0])\n\n# Convert extracted features to DataFrame\ncolor_columns = ['color_r_mean', 'color_g_mean', 'color_b_mean', 'color_r_std', 'color_g_std', 'color_b_std', 'color_entropy']\ntexture_columns = [f'texture_lbp_{i}' for i in range(len(texture_features[0]))]\nshape_columns = ['shape_aspect_ratio', 'shape_circularity', 'shape_area', 'shape_perimeter']\n\ncolor_df = pd.DataFrame(color_features, columns=color_columns)\ntexture_df = pd.DataFrame(texture_features, columns=texture_columns)\nshape_df = pd.DataFrame(shape_features, columns=shape_columns)\n\n# Combine extracted features with original tabular data\ncombined_features_df = pd.concat([combined_df.reset_index(drop=True), color_df, texture_df, shape_df], axis=1)\n\n# Step 3: Preprocessing\n# Fill missing values only for numerical columns\nnumeric_columns = combined_features_df.select_dtypes(include=['number']).columns\ncombined_features_df[numeric_columns] = combined_features_df[numeric_columns].fillna(combined_features_df[numeric_columns].median())\n\n# Standardize the newly added features (only numeric columns)\nscaler = StandardScaler()\ncombined_features_df[numeric_columns] = scaler.fit_transform(combined_features_df[numeric_columns])\n\n# Save the updated dataset with augmented data and features\ncombined_features_df.to_csv('/kaggle/working/train.csv', index=False)\nprint(\"Augmented and enhanced dataset with extracted features saved to '/kaggle/working/train.csv'\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-20T09:29:56.364518Z","iopub.execute_input":"2024-11-20T09:29:56.36537Z","iopub.status.idle":"2024-11-20T09:33:48.983203Z","shell.execute_reply.started":"2024-11-20T09:29:56.365336Z","shell.execute_reply":"2024-11-20T09:33:48.982297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}