{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nimport cv2\nfrom PIL import Image\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:54:24.425187Z","iopub.execute_input":"2025-08-28T07:54:24.425507Z","iopub.status.idle":"2025-08-28T07:54:24.430723Z","shell.execute_reply.started":"2025-08-28T07:54:24.425486Z","shell.execute_reply":"2025-08-28T07:54:24.429737Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Handling & Preprocessing (Look Wey Shen)","metadata":{}},{"cell_type":"code","source":"# (1) Explore Dataset\n# Dataset path\ndata_path = \"/kaggle/input/siim-isic-melanoma-classification\"\n\n# List files\nprint(os.listdir(data_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:19:52.043067Z","iopub.execute_input":"2025-08-28T07:19:52.043744Z","iopub.status.idle":"2025-08-28T07:19:52.048786Z","shell.execute_reply.started":"2025-08-28T07:19:52.043715Z","shell.execute_reply":"2025-08-28T07:19:52.048139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load Metadata (CSV File)\ntrain_df = pd.read_csv(f\"{data_path}/train.csv\")\ntest_df = pd.read_csv(f\"{data_path}/test.csv\")\n\nprint(f\"Training data shape: {train_df.shape}\")\nprint(f\"Test data shape: {test_df.shape}\")\nprint(\"\\nTraining data columns:\", train_df.columns.tolist())\nprint(\"\\nFirst 5 rows:\")\nprint(train_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:19:53.391882Z","iopub.execute_input":"2025-08-28T07:19:53.392377Z","iopub.status.idle":"2025-08-28T07:19:53.643599Z","shell.execute_reply.started":"2025-08-28T07:19:53.392351Z","shell.execute_reply":"2025-08-28T07:19:53.642816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# (2) Metadata Cleaning and Preprocessing\ndef clean_metadata(df, is_train=True):\n    \"\"\"Clean and preprocess metadata\"\"\"\n    df_clean = df.copy()\n    \n    # Handle missing values in age\n    if 'age_approx' in df_clean.columns:\n        # Fill missing age with median\n        median_age = df_clean['age_approx'].median()\n        df_clean['age_approx'].fillna(median_age, inplace=True)\n        \n        # Normalize age (0-1 scale)\n        df_clean['age_normalized'] = df_clean['age_approx'] / 100.0\n        print(f\"Age missing values filled with median: {median_age}\")\n    \n    # Handle missing values in sex\n    if 'sex' in df_clean.columns:\n        # Fill missing sex with mode\n        mode_sex = df_clean['sex'].mode()[0] if not df_clean['sex'].mode().empty else 'male'\n        df_clean['sex'].fillna(mode_sex, inplace=True)\n        \n        # One-hot encode sex\n        sex_dummies = pd.get_dummies(df_clean['sex'], prefix='sex')\n        df_clean = pd.concat([df_clean, sex_dummies], axis=1)\n        print(f\"Sex missing values filled with mode  : {mode_sex}\")\n    \n    # Handle missing values in anatomical site\n    if 'anatom_site_general_challenge' in df_clean.columns:\n        # Fill missing site with 'unknown'\n        df_clean['anatom_site_general_challenge'].fillna('unknown', inplace=True)\n        \n        # One-hot encode anatomical site\n        site_dummies = pd.get_dummies(df_clean['anatom_site_general_challenge'], prefix='site')\n        df_clean = pd.concat([df_clean, site_dummies], axis=1)\n        print(\"Anatomical site missing values filled with 'unknown'\")\n    \n    # Create additional features\n    if is_train and 'target' in df_clean.columns:\n        # Calculate class weights for imbalanced dataset\n        target_counts = df_clean['target'].value_counts()\n        print(f\"\\nClass distribution:\")\n        print(f\"Benign (0)   : {target_counts[0]} ({target_counts[0]/len(df_clean)*100:.2f}%)\")\n        print(f\"Malignant (1): {target_counts[1]} ({target_counts[1]/len(df_clean)*100:.2f}%)\")\n    \n    return df_clean","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:19:58.003368Z","iopub.execute_input":"2025-08-28T07:19:58.003642Z","iopub.status.idle":"2025-08-28T07:19:58.011660Z","shell.execute_reply.started":"2025-08-28T07:19:58.003622Z","shell.execute_reply":"2025-08-28T07:19:58.010928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clean training and test metadata\ntrain_clean = clean_metadata(train_df, is_train=True)\ntest_clean = clean_metadata(test_df, is_train=False)\n\nprint(f\"\\nCleaned training data shape: {train_clean.shape}\")\nprint(f\"Cleaned test data shape    : {test_clean.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:20:00.640431Z","iopub.execute_input":"2025-08-28T07:20:00.640760Z","iopub.status.idle":"2025-08-28T07:20:00.688065Z","shell.execute_reply.started":"2025-08-28T07:20:00.640734Z","shell.execute_reply":"2025-08-28T07:20:00.687226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# (3) Patient-based Train/Validation Split\ndef create_patient_split(df, test_size=0.2, random_state=42):\n    \"\"\"Create train/validation split by patient ID to avoid data leakage\"\"\"\n    \n    # Get unique patients\n    unique_patients = df['patient_id'].unique()\n    print(f\"Total unique patients: {len(unique_patients)}\")\n    \n    # Split patients (not individual images)\n    train_patients, val_patients = train_test_split(\n        unique_patients, \n        test_size=test_size, \n        random_state=random_state,\n        stratify=None  # Can't stratify by patient easily, would need more complex logic\n    )\n    \n    # Create train/validation dataframes\n    train_split = df[df['patient_id'].isin(train_patients)].copy()\n    val_split = df[df['patient_id'].isin(val_patients)].copy()\n    \n    print(f\"Training patients    : {len(train_patients)}\")\n    print(f\"Validation patients  : {len(val_patients)}\")\n    print(f\"Training images      : {len(train_split)}\")\n    print(f\"Validation images    : {len(val_split)}\")\n    \n    # Check target distribution in splits\n    if 'target' in df.columns:\n        print(f\"\\nTarget distribution in training split:\")\n        print(train_split['target'].value_counts(normalize=True))\n        print(f\"\\nTarget distribution in validation split:\")\n        print(val_split['target'].value_counts(normalize=True))\n    \n    return train_split, val_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:20:03.836907Z","iopub.execute_input":"2025-08-28T07:20:03.837240Z","iopub.status.idle":"2025-08-28T07:20:03.844892Z","shell.execute_reply.started":"2025-08-28T07:20:03.837219Z","shell.execute_reply":"2025-08-28T07:20:03.843869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create patient-based split\ntrain_split, val_split = create_patient_split(train_clean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:20:06.535015Z","iopub.execute_input":"2025-08-28T07:20:06.535320Z","iopub.status.idle":"2025-08-28T07:20:06.558672Z","shell.execute_reply.started":"2025-08-28T07:20:06.535296Z","shell.execute_reply":"2025-08-28T07:20:06.557998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# (4) Image Preprocessing Functions\nclass ImagePreprocessor:\n    def __init__(self, target_size=(224, 224), normalize=True):\n        self.target_size = target_size\n        self.normalize = normalize\n        \n    def load_and_preprocess_image(self, image_path, augment=False):\n        \"\"\"Load and preprocess a single image\"\"\"\n        try:\n            # Load image\n            image = cv2.imread(image_path)\n            if image is None:\n                print(f\"Warning: Could not load image {image_path}\")\n                return None\n                \n            # Convert BGR to RGB\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            \n            # Resize image\n            image = cv2.resize(image, self.target_size)\n            \n            # Normalize pixel values to [0, 1]\n            if self.normalize:\n                image = image.astype(np.float32) / 255.0\n            \n            # Apply augmentation if specified\n            if augment:\n                image = self.apply_augmentation(image)\n                \n            return image\n            \n        except Exception as e:\n            print(f\"Error processing image {image_path}: {str(e)}\")\n            return None\n    \n    def apply_augmentation(self, image):\n        \"\"\"Apply basic data augmentation\"\"\"\n        # Random horizontal flip\n        if np.random.random() > 0.5:\n            image = cv2.flip(image, 1)\n        \n        # Random rotation (small angle)\n        if np.random.random() > 0.5:\n            angle = np.random.uniform(-15, 15)\n            rows, cols = image.shape[:2]\n            M = cv2.getRotationMatrix2D((cols/2, rows/2), angle, 1)\n            image = cv2.warpAffine(image, M, (cols, rows))\n        \n        # Random brightness adjustment\n        if np.random.random() > 0.5:\n            brightness = np.random.uniform(0.8, 1.2)\n            image = np.clip(image * brightness, 0, 1)\n            \n        return image\n    \n    def preprocess_batch(self, image_paths, augment=False, batch_size=32):\n        \"\"\"Preprocess a batch of images\"\"\"\n        images = []\n        valid_paths = []\n        \n        for path in image_paths:\n            img = self.load_and_preprocess_image(path, augment=augment)\n            if img is not None:\n                images.append(img)\n                valid_paths.append(path)\n                \n        return np.array(images), valid_paths","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:20:10.677928Z","iopub.execute_input":"2025-08-28T07:20:10.678503Z","iopub.status.idle":"2025-08-28T07:20:10.687809Z","shell.execute_reply.started":"2025-08-28T07:20:10.678478Z","shell.execute_reply":"2025-08-28T07:20:10.686763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize preprocessor\npreprocessor = ImagePreprocessor(target_size=(224, 224), normalize=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:20:11.405306Z","iopub.execute_input":"2025-08-28T07:20:11.405591Z","iopub.status.idle":"2025-08-28T07:20:11.409899Z","shell.execute_reply.started":"2025-08-28T07:20:11.405569Z","shell.execute_reply":"2025-08-28T07:20:11.409055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# (5) Create Data Loading Functions\ndef create_image_paths(df, image_dir):\n    \"\"\"Create full image paths from dataframe\"\"\"\n    return [os.path.join(image_dir, f\"{img_id}.jpg\") for img_id in df['image_name']]\n\ndef save_processed_data(train_df, val_df, test_df, output_dir='processed_data'):\n    \"\"\"Save processed dataframes\"\"\"\n    os.makedirs(output_dir, exist_ok=True)\n    \n    # Save cleaned metadata\n    train_df.to_csv(os.path.join(output_dir, 'train_processed.csv'), index=False)\n    val_df.to_csv(os.path.join(output_dir, 'val_processed.csv'), index=False)\n    test_df.to_csv(os.path.join(output_dir, 'test_processed.csv'), index=False)\n    \n    print(f\"Processed data saved to {output_dir}/\")\n    \n    # Save preprocessing summary\n    with open(os.path.join(output_dir, 'preprocessing_summary.txt'), 'w') as f:\n        f.write(\"SIIM-ISIC Data Preprocessing Summary\\n\")\n        f.write(\"=\"*40 + \"\\n\\n\")\n        f.write(f\"Training samples: {len(train_df)}\\n\")\n        f.write(f\"Validation samples: {len(val_df)}\\n\")\n        f.write(f\"Test samples: {len(test_df)}\\n\")\n        f.write(f\"Image target size: {preprocessor.target_size}\\n\")\n        f.write(f\"Normalization applied: {preprocessor.normalize}\\n\")\n        \n        # Feature columns\n        feature_cols = [col for col in train_df.columns if col not in ['image_name', 'patient_id', 'target']]\n        f.write(f\"\\nFeature columns ({len(feature_cols)}):\\n\")\n        for col in feature_cols:\n            f.write(f\"  - {col}\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:20:13.268386Z","iopub.execute_input":"2025-08-28T07:20:13.268892Z","iopub.status.idle":"2025-08-28T07:20:13.276315Z","shell.execute_reply.started":"2025-08-28T07:20:13.268869Z","shell.execute_reply":"2025-08-28T07:20:13.275262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# (6) Execute Preprocessing Pipeline\n# Create image paths\ntrain_paths = create_image_paths(train_split, f\"{data_path}/jpeg/train/\")\nval_paths = create_image_paths(val_split, f\"{data_path}/jpeg/train/\")\ntest_paths = create_image_paths(test_clean, f\"{data_path}/jpeg/test/\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:20:13.850094Z","iopub.execute_input":"2025-08-28T07:20:13.850850Z","iopub.status.idle":"2025-08-28T07:20:13.899215Z","shell.execute_reply.started":"2025-08-28T07:20:13.850822Z","shell.execute_reply":"2025-08-28T07:20:13.898363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Validate that images exist\ndef validate_image_paths(paths, df, split_name):\n    \"\"\"Validate that image files exist\"\"\"\n    existing_paths = []\n    valid_indices = []\n    \n    for i, path in enumerate(paths):\n        if os.path.exists(path):\n            existing_paths.append(path)\n            valid_indices.append(i)\n    \n    print(f\"{split_name}: {len(existing_paths)}/{len(paths)} images found\")\n    return df.iloc[valid_indices].reset_index(drop=True), existing_paths","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:20:23.363021Z","iopub.execute_input":"2025-08-28T07:20:23.363764Z","iopub.status.idle":"2025-08-28T07:20:23.369901Z","shell.execute_reply.started":"2025-08-28T07:20:23.363731Z","shell.execute_reply":"2025-08-28T07:20:23.369059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Validate all splits\ntrain_final, train_paths_final = validate_image_paths(train_paths, train_split, \"Training\")\nval_final, val_paths_final = validate_image_paths(val_paths, val_split, \"Validation\")\ntest_final, test_paths_final = validate_image_paths(test_paths, test_clean, \"Test\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:20:24.482664Z","iopub.execute_input":"2025-08-28T07:20:24.482984Z","iopub.status.idle":"2025-08-28T07:22:16.540527Z","shell.execute_reply.started":"2025-08-28T07:20:24.482962Z","shell.execute_reply":"2025-08-28T07:22:16.539876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save processed data\nsave_processed_data(train_final, val_final, test_final)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-28T07:22:18.855184Z","iopub.execute_input":"2025-08-28T07:22:18.855483Z","iopub.status.idle":"2025-08-28T07:22:19.149907Z","shell.execute_reply.started":"2025-08-28T07:22:18.855460Z","shell.execute_reply":"2025-08-28T07:22:19.149210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Training samples   : {len(train_final)}\")\nprint(f\"Validation samples : {len(val_final)}\")\nprint(f\"Test samples       : {len(test_final)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T12:47:41.884305Z","iopub.execute_input":"2025-08-27T12:47:41.884595Z","iopub.status.idle":"2025-08-27T12:47:41.892409Z","shell.execute_reply.started":"2025-08-27T12:47:41.884579Z","shell.execute_reply":"2025-08-27T12:47:41.891458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and display a sample image\nif len(train_paths_final) > 0:\n    # Load a sample image\n    sample_image = preprocessor.load_and_preprocess_image(train_paths_final[0])\n    if sample_image is not None:\n        print(f\"Sample image shape: {sample_image.shape}\")\n        print(f\"Sample image data type: {sample_image.dtype}\")\n        print(f\"Sample image value range: [{sample_image.min():.3f}, {sample_image.max():.3f}]\")\n        \n        plt.figure(figsize=(8, 6))\n        plt.imshow(sample_image)\n        plt.title(\"Sample Preprocessed Image\")\n        plt.axis('off')\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-27T12:53:00.212017Z","iopub.execute_input":"2025-08-27T12:53:00.212293Z","iopub.status.idle":"2025-08-27T12:53:00.859366Z","shell.execute_reply.started":"2025-08-27T12:53:00.212275Z","shell.execute_reply":"2025-08-27T12:53:00.858409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Image Model (CNN) (Tan Jian Hua)","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Metadata + Fusion Model (Seng Zi Jun)","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluation, Analysis & Reporting (Wong Kang Yi)","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}