{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedGroupKFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:10.204041Z","iopub.execute_input":"2026-06-26T20:22:10.204410Z","iopub.status.idle":"2026-06-26T20:22:10.210171Z","shell.execute_reply.started":"2026-06-26T20:22:10.204379Z","shell.execute_reply":"2026-06-26T20:22:10.208887Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset Paths","metadata":{}},{"cell_type":"code","source":"ISIC2019_IMG = (\n    \"/kaggle/input/datasets/cdeotte/jpeg-isic2019-512x512/train\"\n)\n\nISIC2020_IMG = (\n    \"/kaggle/input/competitions/siim-isic-melanoma-classification/jpeg/train\"\n)\n\nISIC2020_META = (\n    \"/kaggle/input/competitions/siim-isic-melanoma-classification/train.csv\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:10.211987Z","iopub.execute_input":"2026-06-26T20:22:10.212380Z","iopub.status.idle":"2026-06-26T20:22:10.236100Z","shell.execute_reply.started":"2026-06-26T20:22:10.212339Z","shell.execute_reply":"2026-06-26T20:22:10.234994Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load ISIC 2020","metadata":{}},{"cell_type":"code","source":"df2020 = pd.read_csv(ISIC2020_META)\n\ndf2020[\"image_path\"] = df2020[\"image_name\"].apply(\n    lambda x: os.path.join(\n        ISIC2020_IMG,\n        x + \".jpg\"\n    )\n)\n\ndf2020[\"dataset\"] = \"2020\"\n\n# patient-wise grouping\ndf2020[\"group_id\"] = (\n    \"2020_\" +\n    df2020[\"patient_id\"].astype(str)\n)\n\ndf2020 = df2020[\n    [\n        \"image_name\",\n        \"group_id\",\n        \"sex\",\n        \"age_approx\",\n        \"anatom_site_general_challenge\",\n        \"target\",\n        \"image_path\",\n        \"dataset\"\n    ]\n]\n\nprint(df2020.shape)\ndf2020.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:10.237467Z","iopub.execute_input":"2026-06-26T20:22:10.237868Z","iopub.status.idle":"2026-06-26T20:22:10.387279Z","shell.execute_reply.started":"2026-06-26T20:22:10.237828Z","shell.execute_reply":"2026-06-26T20:22:10.386307Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load ISIC 2019","metadata":{}},{"cell_type":"code","source":"df2019 = pd.read_csv(\n    \"/kaggle/input/datasets/cdeotte/jpeg-isic2019-512x512/train.csv\"\n)\n\ndf2019[\"image_path\"] = df2019[\"image_name\"].apply(\n    lambda x: os.path.join(\n        ISIC2019_IMG,\n        x + \".jpg\"\n    )\n)\n\ndf2019[\"dataset\"] = \"2019\"\n\n# image-wise grouping\ndf2019[\"group_id\"] = (\n    \"2019_\" +\n    df2019[\"image_name\"].astype(str)\n)\n\ndf2019 = df2019[\n    [\n        \"image_name\",\n        \"group_id\",\n        \"sex\",\n        \"age_approx\",\n        \"anatom_site_general_challenge\",\n        \"target\",\n        \"image_path\",\n        \"dataset\"\n    ]\n]\n\nprint(df2019.shape)\ndf2019.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:10.389420Z","iopub.execute_input":"2026-06-26T20:22:10.389757Z","iopub.status.idle":"2026-06-26T20:22:10.505369Z","shell.execute_reply.started":"2026-06-26T20:22:10.389728Z","shell.execute_reply":"2026-06-26T20:22:10.504486Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Merge Dataset","metadata":{}},{"cell_type":"code","source":"df = pd.concat(\n    [df2019, df2020],\n    ignore_index=True\n)\n\nprint(\"ISIC2019:\", len(df2019))\nprint(\"ISIC2020:\", len(df2020))\nprint(\"Merged   :\", len(df))\n\nprint(\"\\nClass Distribution\")\nprint(df[\"target\"].value_counts())\n\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:10.506604Z","iopub.execute_input":"2026-06-26T20:22:10.506881Z","iopub.status.idle":"2026-06-26T20:22:10.531755Z","shell.execute_reply.started":"2026-06-26T20:22:10.506854Z","shell.execute_reply":"2026-06-26T20:22:10.530857Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create Missingness Indicators and Fill Missing Values","metadata":{}},{"cell_type":"code","source":"# MISSINGNESS INDICATORS\n\ndf[\"age_missing\"] = (\n    df[\"age_approx\"].isna().astype(int)\n)\n\ndf[\"sex_missing\"] = (\n    df[\"sex\"].isna().astype(int)\n)\n\ndf[\"site_missing\"] = (\n    df[\"anatom_site_general_challenge\"]\n    .isna()\n    .astype(int)\n)\n\n# IMPUTATION\n\ndf[\"age_approx\"] = df[\"age_approx\"].fillna(\n    df[\"age_approx\"].median()\n)\n\ndf[\"sex\"] = df[\"sex\"].map({\n    \"male\": 1,\n    \"female\": 0\n})\n\ndf[\"sex\"] = df[\"sex\"].fillna(-1)\n\ndf[\"anatom_site_general_challenge\"] = (\n    df[\"anatom_site_general_challenge\"]\n    .fillna(\"unknown\")\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:10.532875Z","iopub.execute_input":"2026-06-26T20:22:10.533180Z","iopub.status.idle":"2026-06-26T20:22:10.563395Z","shell.execute_reply.started":"2026-06-26T20:22:10.533150Z","shell.execute_reply":"2026-06-26T20:22:10.562512Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encode Metadata","metadata":{}},{"cell_type":"code","source":"le = LabelEncoder()\n\ndf[\"site_encoded\"] = le.fit_transform(\n    df[\"anatom_site_general_challenge\"]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:10.564598Z","iopub.execute_input":"2026-06-26T20:22:10.564877Z","iopub.status.idle":"2026-06-26T20:22:10.580893Z","shell.execute_reply.started":"2026-06-26T20:22:10.564852Z","shell.execute_reply":"2026-06-26T20:22:10.579632Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create Missingness Vector","metadata":{}},{"cell_type":"code","source":"df[\"missing_vector\"] = df.apply(\n\n    lambda row: [\n\n        row[\"age_missing\"],\n\n        row[\"sex_missing\"],\n\n        row[\"site_missing\"]\n\n    ],\n\n    axis=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:10.582589Z","iopub.execute_input":"2026-06-26T20:22:10.583039Z","iopub.status.idle":"2026-06-26T20:22:11.248530Z","shell.execute_reply.started":"2026-06-26T20:22:10.582996Z","shell.execute_reply":"2026-06-26T20:22:11.247416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:11.250981Z","iopub.execute_input":"2026-06-26T20:22:11.251321Z","iopub.status.idle":"2026-06-26T20:22:11.266871Z","shell.execute_reply.started":"2026-06-26T20:22:11.251293Z","shell.execute_reply":"2026-06-26T20:22:11.266133Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Create Fixed Five Folds\nstrategy:\n* ISIC 2020 → patient-wise grouping using patient_id\n* ISIC 2019 → image-wise grouping using image_name\n* Merge the datasets\n* Create one fixed 5-fold split\n* Save the missingness information for the gating model","metadata":{}},{"cell_type":"code","source":"df[\"fold\"] = -1\n\nsgkf = StratifiedGroupKFold(\n\n    n_splits=5,\n\n    shuffle=True,\n\n    random_state=42\n)\n\nfor fold, (train_idx, val_idx) in enumerate(\n\n    sgkf.split(\n\n        df,\n\n        y=df[\"target\"],\n\n        groups=df[\"group_id\"]\n\n    )\n):\n\n    df.loc[val_idx, \"fold\"] = fold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:11.268102Z","iopub.execute_input":"2026-06-26T20:22:11.269105Z","iopub.status.idle":"2026-06-26T20:22:20.457074Z","shell.execute_reply.started":"2026-06-26T20:22:11.269074Z","shell.execute_reply":"2026-06-26T20:22:20.455926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Verify Fold Distribution","metadata":{}},{"cell_type":"code","source":"print(df.fold.value_counts())\n\nprint()\n\nprint(\n\n    df.groupby(\"fold\")[\"target\"]\n\n    .value_counts()\n\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:22:20.458180Z","iopub.execute_input":"2026-06-26T20:22:20.458494Z","iopub.status.idle":"2026-06-26T20:22:20.472392Z","shell.execute_reply.started":"2026-06-26T20:22:20.458467Z","shell.execute_reply":"2026-06-26T20:22:20.471274Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Save Dataset","metadata":{}},{"cell_type":"code","source":"df.to_csv(\n\n    \"isic_missingness_fold_assignments.csv\",\n\n    index=False\n)\n\nprint(\n\n    \"Saved successfully.\"\n\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-26T20:23:11.944267Z","iopub.execute_input":"2026-06-26T20:23:11.944960Z","iopub.status.idle":"2026-06-26T20:23:12.405047Z","shell.execute_reply.started":"2026-06-26T20:23:11.944918Z","shell.execute_reply":"2026-06-26T20:23:12.404100Z"}},"outputs":[],"execution_count":null}]}