{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":20270,"databundleVersionId":1222630},{"sourceType":"datasetVersion","sourceId":13229715,"datasetId":8385883,"databundleVersionId":13925133},{"sourceType":"datasetVersion","sourceId":13229732,"datasetId":8385893,"databundleVersionId":13925151},{"sourceType":"kernelVersion","sourceId":38667125}],"dockerImageVersionId":31286,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n        print(dirname)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2026-03-02T01:56:27.372169Z","iopub.execute_input":"2026-03-02T01:56:27.372542Z","iopub.status.idle":"2026-03-02T01:57:17.862006Z","shell.execute_reply.started":"2026-03-02T01:56:27.372511Z","shell.execute_reply":"2026-03-02T01:57:17.860461Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip -q install plotly","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-02T01:57:34.854124Z","iopub.execute_input":"2026-03-02T01:57:34.854463Z","iopub.status.idle":"2026-03-02T01:57:37.550796Z","shell.execute_reply.started":"2026-03-02T01:57:34.854433Z","shell.execute_reply":"2026-03-02T01:57:37.549868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport cv2\nfrom tqdm import tqdm_notebook as tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly\nimport plotly.graph_objects as go\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2026-03-02T01:57:40.793366Z","iopub.execute_input":"2026-03-02T01:57:40.793652Z","iopub.status.idle":"2026-03-02T01:57:40.800604Z","shell.execute_reply.started":"2026-03-02T01:57:40.793625Z","shell.execute_reply":"2026-03-02T01:57:40.799362Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_image_names(dataframe) : \n    image_names = dataframe[\"image_name\"].values\n    image_names = image_names + \".jpg\"\n    return image_names","metadata":{"execution":{"iopub.status.busy":"2026-03-02T01:57:46.424523Z","iopub.execute_input":"2026-03-02T01:57:46.424774Z","iopub.status.idle":"2026-03-02T01:57:46.429056Z","shell.execute_reply.started":"2026-03-02T01:57:46.424755Z","shell.execute_reply":"2026-03-02T01:57:46.42829Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_info(image_names) : \n    image_names = np.array(image_names)\n    \n    print(\"Length = \", len(image_names))\n    print(\"Type = \", type(image_names))\n    print(\"Shape = \", image_names.shape)\n    \n    return image_names","metadata":{"execution":{"iopub.status.busy":"2026-03-02T01:57:48.899901Z","iopub.execute_input":"2026-03-02T01:57:48.900279Z","iopub.status.idle":"2026-03-02T01:57:48.90615Z","shell.execute_reply.started":"2026-03-02T01:57:48.900247Z","shell.execute_reply":"2026-03-02T01:57:48.905007Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew\n\ndef extract_information(image_names, directory) : \n    image_statistics = pd.DataFrame(index = np.arange(len(image_names)),\n                                    columns = [\"image_name\", \"path\", \"rows\", \"columns\", \"channels\", \n                                              \"image_mean\", \"image_standard_deviation\", \"image_skewness\",\n                                              \"mean_red_value\", \"mean_green_value\", \"mean_blue_value\"])\n    i = 0 \n    for name in tqdm(image_names) : \n        path = os.path.join(directory, name)\n        image = cv2.imread(path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        \n        image_statistics.iloc[i][\"image_name\"] = name\n        image_statistics.iloc[i][\"path\"] = path\n        image_statistics.iloc[i][\"rows\"] = image.shape[0]\n        image_statistics.iloc[i][\"columns\"] = image.shape[1]\n        image_statistics.iloc[i][\"channels\"] = image.shape[2]\n        image_statistics.iloc[i][\"image_mean\"] = np.mean(image.flatten())\n        image_statistics.iloc[i][\"image_standard_deviation\"] = np.std(image.flatten())\n        image_statistics.iloc[i][\"image_skewness\"] = skew(image.flatten())\n        image_statistics.iloc[i][\"mean_red_value\"] = np.mean(image[:,:,0])\n        image_statistics.iloc[i][\"mean_green_value\"] = np.mean(image[:,:,1])\n        image_statistics.iloc[i][\"mean_blue_value\"] = np.mean(image[:,:,2])\n        \n        i = i + 1\n        del image\n        \n    return image_statistics","metadata":{"execution":{"iopub.status.busy":"2026-03-02T01:57:51.237597Z","iopub.execute_input":"2026-03-02T01:57:51.237944Z","iopub.status.idle":"2026-03-02T01:57:51.24926Z","shell.execute_reply.started":"2026-03-02T01:57:51.237919Z","shell.execute_reply":"2026-03-02T01:57:51.248287Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nROOT = \"/kaggle/input/competitions/siim-isic-melanoma-classification\"\ntrain_dir = f\"{ROOT}/jpeg/train\"\ntrain = pd.read_csv(f\"{ROOT}/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-02T02:01:23.105117Z","iopub.execute_input":"2026-03-02T02:01:23.105357Z","iopub.status.idle":"2026-03-02T02:01:23.231383Z","shell.execute_reply.started":"2026-03-02T02:01:23.105339Z","shell.execute_reply":"2026-03-02T02:01:23.230706Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_names = get_image_names(train)\nimage_names = get_info(image_names)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.610698Z","iopub.status.idle":"2026-03-01T20:02:09.611093Z","shell.execute_reply.started":"2026-03-01T20:02:09.610852Z","shell.execute_reply":"2026-03-01T20:02:09.610868Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dir = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/test/\"\ntest = pd.DataFrame(pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/test.csv\"))\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.612359Z","iopub.status.idle":"2026-03-01T20:02:09.613684Z","shell.execute_reply.started":"2026-03-01T20:02:09.613467Z","shell.execute_reply":"2026-03-01T20:02:09.613499Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_names = get_image_names(test)\nimage_names = get_info(image_names)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.61552Z","iopub.status.idle":"2026-03-01T20:02:09.616325Z","shell.execute_reply.started":"2026-03-01T20:02:09.616072Z","shell.execute_reply":"2026-03-01T20:02:09.616092Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.DataFrame(pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\"))\ntest = pd.DataFrame(pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/test.csv\"))","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.618332Z","iopub.status.idle":"2026-03-01T20:02:09.618645Z","shell.execute_reply.started":"2026-03-01T20:02:09.618475Z","shell.execute_reply":"2026-03-01T20:02:09.61849Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.619323Z","iopub.status.idle":"2026-03-01T20:02:09.619695Z","shell.execute_reply.started":"2026-03-01T20:02:09.619537Z","shell.execute_reply":"2026-03-01T20:02:09.619559Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.621024Z","iopub.status.idle":"2026-03-01T20:02:09.621451Z","shell.execute_reply.started":"2026-03-01T20:02:09.621241Z","shell.execute_reply":"2026-03-01T20:02:09.62126Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.622993Z","iopub.status.idle":"2026-03-01T20:02:09.623422Z","shell.execute_reply.started":"2026-03-01T20:02:09.623175Z","shell.execute_reply":"2026-03-01T20:02:09.623191Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.625154Z","iopub.status.idle":"2026-03-01T20:02:09.625559Z","shell.execute_reply.started":"2026-03-01T20:02:09.625355Z","shell.execute_reply":"2026-03-01T20:02:09.625382Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.62674Z","iopub.status.idle":"2026-03-01T20:02:09.627015Z","shell.execute_reply.started":"2026-03-01T20:02:09.626877Z","shell.execute_reply":"2026-03-01T20:02:09.626892Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train[\"patient_id\"].unique()), len(test[\"patient_id\"].unique())\n","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.62809Z","iopub.status.idle":"2026-03-01T20:02:09.628719Z","shell.execute_reply.started":"2026-03-01T20:02:09.628578Z","shell.execute_reply":"2026-03-01T20:02:09.628596Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train[\"target\"].value_counts())","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.62967Z","iopub.status.idle":"2026-03-01T20:02:09.629912Z","shell.execute_reply.started":"2026-03-01T20:02:09.629793Z","shell.execute_reply":"2026-03-01T20:02:09.629807Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"malignant = len(train[train[\"target\"] == 1])\nbenign = len(train[train[\"target\"] == 0])\n\nlabels = [\"Malignant\", \"Benign\"] \nsize = [malignant, benign]\n\nplt.figure(figsize = (8, 8))\nplt.pie(size, labels = labels, shadow = True, startangle = 90, colors = [\"r\", \"g\"])\nplt.title(\"Malignant VS Benign Cases\")\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.630593Z","iopub.status.idle":"2026-03-01T20:02:09.63082Z","shell.execute_reply.started":"2026-03-01T20:02:09.630705Z","shell.execute_reply":"2026-03-01T20:02:09.630719Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_males = len(train[train[\"sex\"] == \"male\"])\ntrain_females  = len(train[train[\"sex\"] == \"female\"])\n\ntest_males = len(test[test[\"sex\"] == \"male\"])\ntest_females  = len(test[test[\"sex\"] == \"female\"])\n\nlabels = [\"Males\", \"Female\"] \n\nsize = [train_males, train_females]\nexplode = [0.1, 0.0]\n\nplt.figure(figsize = (16, 16))\nplt.subplot(1,2,1)\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90, colors = [\"b\", \"g\"])\nplt.title(\"Male VS Female Training Set Count\", fontsize = 18)\nplt.legend()\n\nprint(\"Number of males in training set = \", train_males)\nprint(\"Number of females in training set= \", train_females)\n\nsize = [test_males, test_females]\n\nplt.subplot(1,2,2)\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90, colors = [\"b\", \"g\"])\nplt.title(\"Male VS Female Test Set Count\", fontsize = 18)\nplt.legend()\n\nprint(\"Number of males in testing set = \", test_males)\nprint(\"Number of females in testing set= \", test_females)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.631475Z","iopub.status.idle":"2026-03-01T20:02:09.631718Z","shell.execute_reply.started":"2026-03-01T20:02:09.631602Z","shell.execute_reply":"2026-03-01T20:02:09.631615Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_malignant  = train[train[\"target\"] == 1]\ntrain_malignant_males = len(train_malignant[train_malignant[\"sex\"] == \"male\"])\ntrain_malignant_females  = len(train_malignant[train_malignant[\"sex\"] == \"female\"])\n\nlabels = [\"Malignant Male Cases\", \"Malignant Female Cases\"] \nsize = [train_malignant_males, train_malignant_females]\nexplode = [0.1, 0.0]\n\nplt.figure(figsize = (10, 10))\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90, colors = [\"r\", \"c\"])\nplt.title(\"Malignant Male VS Female Cases\", fontsize = 18)\nplt.legend()\nprint(\"Malignant Male Cases = \", train_malignant_males)\nprint(\"Malignant Female Cases = \", train_malignant_females)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.633529Z","iopub.status.idle":"2026-03-01T20:02:09.63393Z","shell.execute_reply.started":"2026-03-01T20:02:09.633727Z","shell.execute_reply":"2026-03-01T20:02:09.63375Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_benign  = train[train[\"target\"] == 0]\n\ntrain_benign_males = len(train_benign[train_benign[\"sex\"] == \"male\"])\ntrain_benign_females  = len(train_benign[train_benign[\"sex\"] == \"female\"]) \n\nlabels = [\"Benign Male Cases\", \"Benign Female Cases\"] \nsize = [train_benign_males, train_benign_females]\nexplode = [0.1, 0.0]\n\nplt.figure(figsize = (10, 10))\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90, colors = [\"g\", \"y\"])\nplt.title(\"Benign Male VS Benign Female Cases\", fontsize = 18)\nplt.legend()\nprint(\"Benign Male Cases = \", train_benign_males)\nprint(\"Benign Female Cases = \", train_benign_females)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.635193Z","iopub.status.idle":"2026-03-01T20:02:09.635564Z","shell.execute_reply.started":"2026-03-01T20:02:09.635389Z","shell.execute_reply":"2026-03-01T20:02:09.635406Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cancer_versus_sex = train.groupby([\"benign_malignant\", \"sex\"]).size()\nprint(cancer_versus_sex)\nprint(type(cancer_versus_sex))","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.637321Z","iopub.status.idle":"2026-03-01T20:02:09.637601Z","shell.execute_reply.started":"2026-03-01T20:02:09.637477Z","shell.execute_reply":"2026-03-01T20:02:09.637493Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cancer_versus_sex = cancer_versus_sex.unstack(level = 1) / len(train) * 100\nprint(cancer_versus_sex)\nprint(type(cancer_versus_sex))","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.638614Z","iopub.status.idle":"2026-03-01T20:02:09.63887Z","shell.execute_reply.started":"2026-03-01T20:02:09.638742Z","shell.execute_reply":"2026-03-01T20:02:09.638756Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set(style='whitegrid')\nsns.set_context(\"paper\", rc={\"font.size\":12,\"axes.titlesize\":20,\"axes.labelsize\":18})   \n\nplt.figure(figsize = (10, 6))\nsns.heatmap(cancer_versus_sex, annot=True, cmap=\"icefire\", cbar=True)\nplt.title(\"Cancer VS Sex Heatmap Analysis Normalized\", fontsize = 18)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.640374Z","iopub.status.idle":"2026-03-01T20:02:09.640762Z","shell.execute_reply.started":"2026-03-01T20:02:09.640567Z","shell.execute_reply":"2026-03-01T20:02:09.640588Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train\ntrain_torso = len(train[train[\"anatom_site_general_challenge\"] == \"torso\"])\ntrain_lower_extremity = len(train[train[\"anatom_site_general_challenge\"] == \"lower extremity\"])\ntrain_upper_extremity = len(train[train[\"anatom_site_general_challenge\"] == \"upper extremity\"])\ntrain_head_neck = len(train[train[\"anatom_site_general_challenge\"] == \"head/neck\"])\ntrain_palms_soles = len(train[train[\"anatom_site_general_challenge\"] == \"palms/soles\"])\ntrain_oral_genital = len(train[train[\"anatom_site_general_challenge\"] == \"oral/genital\"])\n\n# test\ntest_torso = len(test[test[\"anatom_site_general_challenge\"] == \"torso\"])\ntest_lower_extremity = len(test[test[\"anatom_site_general_challenge\"] == \"lower extremity\"])\ntest_upper_extremity = len(test[test[\"anatom_site_general_challenge\"] == \"upper extremity\"])\ntest_head_neck = len(test[test[\"anatom_site_general_challenge\"] == \"head/neck\"])\ntest_palms_soles = len(test[test[\"anatom_site_general_challenge\"] == \"palms/soles\"])\ntest_oral_genital = len(test[test[\"anatom_site_general_challenge\"] == \"oral/genital\"])\n\nlabels = [\"Torso\", \"Lower Extremity\", \"Upper Extremity\", \"Head/Neck\", \"Palms/Soles\", \"Oral/Genital\"] \n\nplt.figure(figsize = (16, 16))\n\nplt.subplot(1,2,1)\nsize = [train_torso, train_lower_extremity, train_upper_extremity, train_head_neck, train_palms_soles, train_oral_genital]\nexplode = [0.05, 0.05, 0.05, 0.05, 0.05, 0.1]\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90)\nplt.title(\"Anatomy Sites In Training Set\", fontsize = 18)\nplt.legend()\n\nplt.subplot(1,2,2)\nsize = [test_torso, test_lower_extremity, test_upper_extremity, test_head_neck, test_palms_soles, test_oral_genital]\nexplode = [0.05, 0.05, 0.05, 0.05, 0.05, 0.1]\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90)\nplt.title(\"Anatomy Sites In Testing Set\", fontsize = 18)\nplt.legend()\n\n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.642149Z","iopub.status.idle":"2026-03-01T20:02:09.642655Z","shell.execute_reply.started":"2026-03-01T20:02:09.642431Z","shell.execute_reply":"2026-03-01T20:02:09.642454Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ages_benign = train.loc[train[\"target\"] == 0, \"age_approx\"]\ntrain_ages_malignant = train.loc[train[\"target\"] == 1 , \"age_approx\"]\n\nplt.figure(figsize = (10, 8))\nsns.kdeplot(train_ages_benign, label = \"Benign\", shade = True, legend = True, cbar = True)\nsns.kdeplot(train_ages_malignant, label = \"Malignant\", shade = True, legend = True, cbar = True)\nplt.grid(True)\nplt.xlabel(\"Age Of The Patients\", fontsize = 18)\nplt.ylabel(\"Probability Density\", fontsize = 18)\nplt.grid(which = \"minor\", axis = \"both\")\nplt.title(\"Probabilistic Age Distribution In Training Set\", fontsize = 18)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.644518Z","iopub.status.idle":"2026-03-01T20:02:09.644832Z","shell.execute_reply.started":"2026-03-01T20:02:09.644683Z","shell.execute_reply":"2026-03-01T20:02:09.644699Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_stats_01 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_01\"))\ntrain_image_stats_02 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_02\"))\ntrain_image_stats_03 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_03\"))\ntrain_image_stats_04 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_04\"))\ntrain_image_stats_05 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_05\"))\ntrain_image_stats_06 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_06\"))\n\nprint(train_image_stats_01.shape)\nprint(train_image_stats_02.shape)\nprint(train_image_stats_03.shape)\nprint(train_image_stats_04.shape)\nprint(train_image_stats_05.shape)\nprint(train_image_stats_06.shape)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.645941Z","iopub.status.idle":"2026-03-01T20:02:09.646342Z","shell.execute_reply.started":"2026-03-01T20:02:09.646161Z","shell.execute_reply":"2026-03-01T20:02:09.646178Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics = pd.concat([train_image_stats_01, train_image_stats_02, train_image_stats_03,\n                                   train_image_stats_04, train_image_stats_05, train_image_stats_06],\n                                  ignore_index = True)\ntrain_image_statistics.shape","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.647448Z","iopub.status.idle":"2026-03-01T20:02:09.647733Z","shell.execute_reply.started":"2026-03-01T20:02:09.647607Z","shell.execute_reply":"2026-03-01T20:02:09.647623Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics.info()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.650879Z","iopub.status.idle":"2026-03-01T20:02:09.651269Z","shell.execute_reply.started":"2026-03-01T20:02:09.651063Z","shell.execute_reply":"2026-03-01T20:02:09.651086Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image_stats_01 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled-test/melanoma_image_statistics_compiled_test_01\"))\ntest_image_stats_02 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled-test/melanoma_image_statistics_compiled_test_02\"))\n\nprint(test_image_stats_01.shape)\nprint(test_image_stats_02.shape)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.652885Z","iopub.status.idle":"2026-03-01T20:02:09.65331Z","shell.execute_reply.started":"2026-03-01T20:02:09.65307Z","shell.execute_reply":"2026-03-01T20:02:09.653094Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image_statistics = pd.concat([test_image_stats_01, test_image_stats_02], ignore_index = True)\n\ntest_image_statistics.shape","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.654826Z","iopub.status.idle":"2026-03-01T20:02:09.655205Z","shell.execute_reply.started":"2026-03-01T20:02:09.655009Z","shell.execute_reply":"2026-03-01T20:02:09.655025Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image_statistics.info()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.656624Z","iopub.status.idle":"2026-03-01T20:02:09.656877Z","shell.execute_reply.started":"2026-03-01T20:02:09.656745Z","shell.execute_reply":"2026-03-01T20:02:09.656758Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.657549Z","iopub.status.idle":"2026-03-01T20:02:09.65779Z","shell.execute_reply.started":"2026-03-01T20:02:09.65767Z","shell.execute_reply":"2026-03-01T20:02:09.657684Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image_statistics.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.660626Z","iopub.status.idle":"2026-03-01T20:02:09.661013Z","shell.execute_reply.started":"2026-03-01T20:02:09.660818Z","shell.execute_reply":"2026-03-01T20:02:09.66084Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_names = train_image_statistics[\"image_name\"].values\nrandom_images = [np.random.choice(image_names) for i in range(4)] # Generates a random sample from a given 1-D array\nrandom_images ","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.662193Z","iopub.status.idle":"2026-03-01T20:02:09.662591Z","shell.execute_reply.started":"2026-03-01T20:02:09.662426Z","shell.execute_reply":"2026-03-01T20:02:09.662457Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\"","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.664434Z","iopub.status.idle":"2026-03-01T20:02:09.664787Z","shell.execute_reply.started":"2026-03-01T20:02:09.664605Z","shell.execute_reply":"2026-03-01T20:02:09.66463Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nfor i in range(4) : \n    plt.subplot(2, 2, i + 1) \n    image = cv2.imread(os.path.join(train_dir, random_images[i]))\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image, cmap = \"gray\")\n    plt.grid(True)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.666164Z","iopub.status.idle":"2026-03-01T20:02:09.666481Z","shell.execute_reply.started":"2026-03-01T20:02:09.666341Z","shell.execute_reply":"2026-03-01T20:02:09.666362Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"benign_mean_red_value = []\nbenign_mean_green_value = []\nbenign_mean_blue_value = []\n\nmalignant_mean_red_value = []\nmalignant_mean_green_value = []\nmalignant_mean_blue_value = []\n\nfor image_name in tqdm(train_image_statistics[\"image_name\"]) : \n    name = image_name[0:len(image_name)-4] \n    extracted_section = train[train[\"image_name\"] == name]\n    r = int(train_image_statistics[train_image_statistics[\"image_name\"] == image_name][\"mean_red_value\"])\n    g = int(train_image_statistics[train_image_statistics[\"image_name\"] == image_name][\"mean_green_value\"])\n    b = int(train_image_statistics[train_image_statistics[\"image_name\"] == image_name][\"mean_blue_value\"])\n    if int(extracted_section[\"target\"]) == 0 : # benign\n        benign_mean_red_value.append(r)\n        benign_mean_green_value.append(g)\n        benign_mean_blue_value.append(b)\n    else:\n        malignant_mean_red_value.append(r)\n        malignant_mean_green_value.append(g)\n        malignant_mean_blue_value.append(b)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.667045Z","iopub.status.idle":"2026-03-01T20:02:09.667389Z","shell.execute_reply.started":"2026-03-01T20:02:09.66721Z","shell.execute_reply":"2026-03-01T20:02:09.667225Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"range_of_spread = max(benign_mean_red_value) - min(benign_mean_red_value)\n\nplt.figure(figsize = (12, 8))\nplt.rc(\"font\", weight = \"bold\")\nsns.set_style(\"whitegrid\")\nfig = sns.distplot(benign_mean_red_value, hist = True, kde = True, label = \"Mean Red Channel Intensities\", color = \"r\")\nfig.set(xlabel = \"Mean red channel intensities observed in each image\",\n        ylabel = \"Probability Density\")\nplt.title(\"Spread Of Red Channel In Benign Cases\", fontsize = 18)\nplt.legend()\nprint(\"The range of spread = {:.2f}\".format(range_of_spread))","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.667945Z","iopub.status.idle":"2026-03-01T20:02:09.668549Z","shell.execute_reply.started":"2026-03-01T20:02:09.668247Z","shell.execute_reply":"2026-03-01T20:02:09.668268Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"range_of_spread = max(benign_mean_green_value) - min(benign_mean_green_value)\n\nplt.figure(figsize = (12, 8))\nplt.rc(\"font\", weight = \"bold\")\nsns.set_style(\"whitegrid\")\nfig = sns.distplot(benign_mean_green_value, hist = True, kde = True, label = \"Mean Green Channel Intensities\", color = \"g\")\nfig.set(xlabel = \"Mean green channel intensities observed in each image\",\n        ylabel = \"Probability Density\") \nplt.title(\"Spread Of Green Channel In Benign Cases\", fontsize = 18)\nplt.legend()\nprint(\"The range of spread = {:.2f}\".format(range_of_spread))","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.671562Z","iopub.status.idle":"2026-03-01T20:02:09.671981Z","shell.execute_reply.started":"2026-03-01T20:02:09.671811Z","shell.execute_reply":"2026-03-01T20:02:09.671827Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"range_of_spread = max(benign_mean_blue_value) - min(benign_mean_blue_value)\n\nplt.figure(figsize = (12, 8))\nplt.rc(\"font\", weight = \"bold\")\nsns.set_style(\"whitegrid\")\nfig = sns.distplot(benign_mean_blue_value, hist = True, kde = True, label = \"Mean Blue Channel Intensities\", color = \"b\")\nfig.set(xlabel = \"Mean blue channel intensities observed in each image\",\n        ylabel = \"Probability Density\") \nplt.title(\"Spread Of Blue Channel In Benign Cases\", fontsize = 18)\nplt.legend()\nprint(\"The range of spread = {:.2f}\".format(range_of_spread))","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.673204Z","iopub.status.idle":"2026-03-01T20:02:09.67364Z","shell.execute_reply.started":"2026-03-01T20:02:09.673498Z","shell.execute_reply":"2026-03-01T20:02:09.673521Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nplt.rc(\"font\", weight = \"bold\")\nsns.set_style(\"whitegrid\")\nfig = sns.distplot(benign_mean_blue_value, hist = False, kde = True, label = \"Mean Blue Channel Intensities\", color = \"b\")\nfig = sns.distplot(benign_mean_red_value, hist = False, kde = True, label = \"Mean Red Channel Intensities\", color = \"r\")\nfig = sns.distplot(benign_mean_green_value, hist = False, kde = True, label = \"Mean Green Channel Intensities\", color = \"g\")\n\nfig.set(xlabel = \"Mean channel intensities observed in each image\",\n        ylabel = \"Probability Density\") \nplt.title(\"Spread Of Channels In Benign Cases\", fontsize = 18)\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.674878Z","iopub.status.idle":"2026-03-01T20:02:09.675246Z","shell.execute_reply.started":"2026-03-01T20:02:09.675038Z","shell.execute_reply":"2026-03-01T20:02:09.675078Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del benign_mean_red_value\ndel benign_mean_green_value\ndel benign_mean_blue_value","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.677462Z","iopub.status.idle":"2026-03-01T20:02:09.677864Z","shell.execute_reply.started":"2026-03-01T20:02:09.677664Z","shell.execute_reply":"2026-03-01T20:02:09.677688Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.678604Z","iopub.status.idle":"2026-03-01T20:02:09.678979Z","shell.execute_reply.started":"2026-03-01T20:02:09.678777Z","shell.execute_reply":"2026-03-01T20:02:09.678801Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nplt.rc(\"font\", weight = \"bold\")\nsns.set_style(\"whitegrid\")\nfig = sns.distplot(malignant_mean_blue_value, hist = False, kde = True, label = \"Mean Blue Channel Intensities\", color = \"b\")\nfig = sns.distplot(malignant_mean_red_value, hist = False, kde = True, label = \"Mean Red Channel Intensities\", color = \"r\")\nfig = sns.distplot(malignant_mean_green_value, hist = False, kde = True, label = \"Mean Green Channel Intensities\", color = \"g\")\n\nfig.set(xlabel = \"Mean channel intensities observed in each image\",\n        ylabel = \"Probability Density\") \nplt.title(\"Spread Of Channels In Malignant Cases\", fontsize = 18)\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.679917Z","iopub.status.idle":"2026-03-01T20:02:09.680348Z","shell.execute_reply.started":"2026-03-01T20:02:09.680092Z","shell.execute_reply":"2026-03-01T20:02:09.680114Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.682122Z","iopub.status.idle":"2026-03-01T20:02:09.682435Z","shell.execute_reply.started":"2026-03-01T20:02:09.682259Z","shell.execute_reply":"2026-03-01T20:02:09.682311Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.683746Z","iopub.status.idle":"2026-03-01T20:02:09.684131Z","shell.execute_reply.started":"2026-03-01T20:02:09.683883Z","shell.execute_reply":"2026-03-01T20:02:09.68392Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing = len(train[train[\"sex\"].isna() == True])\navailable = len(train[train[\"sex\"].isna() == False])\n\nx = [\"Availabe data\", \"Unavailable data\"]\ny = [np.log(available), np.log(missing)]\n\nprint(\"Count of missing data = \", missing)\nprint(\"Count of available data = \", available)\n\nplt.figure(figsize = (12, 8))\nplt.subplot(1,1,1)\nplt.barh(x, y, color = \"m\")\nplt.grid(True)\nplt.title(\"Data On Patient's Sex\")","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.685328Z","iopub.status.idle":"2026-03-01T20:02:09.685583Z","shell.execute_reply.started":"2026-03-01T20:02:09.685468Z","shell.execute_reply":"2026-03-01T20:02:09.685482Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['sex'].fillna('male', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.686973Z","iopub.status.idle":"2026-03-01T20:02:09.687445Z","shell.execute_reply.started":"2026-03-01T20:02:09.687161Z","shell.execute_reply":"2026-03-01T20:02:09.687183Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing =  len(train[train[\"age_approx\"].isna() == True]) \navailable = len(train[train[\"age_approx\"].isna() == False]) \n\nprint(\"Missing age values = \", missing)\nprint(\"Available age data = \", available)\n\nx = [\"Availabe data\", \"Unavailable data\"]\ny = [np.log(available), np.log(missing)] \n\nplt.figure(figsize = (12, 8))\nplt.subplot(1,1,1)\nplt.barh(x, y, color = \"y\")\nplt.grid(True)\nplt.title(\"Data On Patient's Age\")","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.689481Z","iopub.status.idle":"2026-03-01T20:02:09.68988Z","shell.execute_reply.started":"2026-03-01T20:02:09.689678Z","shell.execute_reply":"2026-03-01T20:02:09.689703Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train\nanatomy_sites = [\"torso\", \"upper extremity\", \"lower extremity\"]\n\nrelevant_dataframe_part = train[(train[\"sex\"] == \"male\") &\n                     (train[\"anatom_site_general_challenge\"].isin(anatomy_sites)) &\n                     (train[\"target\"] == 0)]\n\nmedian_value = relevant_dataframe_part[\"age_approx\"].median()\n\nprint(\"Median value = \", median_value)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.691558Z","iopub.status.idle":"2026-03-01T20:02:09.69212Z","shell.execute_reply.started":"2026-03-01T20:02:09.691977Z","shell.execute_reply":"2026-03-01T20:02:09.691996Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[\"age_approx\"].fillna(median_value, inplace = True)\n","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.693466Z","iopub.status.idle":"2026-03-01T20:02:09.693862Z","shell.execute_reply.started":"2026-03-01T20:02:09.693633Z","shell.execute_reply":"2026-03-01T20:02:09.693659Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[\"anatom_site_general_challenge\"].fillna(\"torso\", inplace = True)\ntest[\"anatom_site_general_challenge\"].fillna(\"torso\", inplace = True)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.695618Z","iopub.status.idle":"2026-03-01T20:02:09.696042Z","shell.execute_reply.started":"2026-03-01T20:02:09.695782Z","shell.execute_reply":"2026-03-01T20:02:09.695818Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.697421Z","iopub.status.idle":"2026-03-01T20:02:09.69781Z","shell.execute_reply.started":"2026-03-01T20:02:09.697649Z","shell.execute_reply":"2026-03-01T20:02:09.697665Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()\n","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.699253Z","iopub.status.idle":"2026-03-01T20:02:09.699597Z","shell.execute_reply.started":"2026-03-01T20:02:09.699467Z","shell.execute_reply":"2026-03-01T20:02:09.699483Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_stats_01 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_01\"))\ntrain_image_stats_02 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_02\"))\ntrain_image_stats_03 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_03\"))\ntrain_image_stats_04 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_04\"))\ntrain_image_stats_05 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_05\"))\ntrain_image_stats_06 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_06\"))\n\nprint(train_image_stats_01.shape)\nprint(train_image_stats_02.shape)\nprint(train_image_stats_03.shape)\nprint(train_image_stats_04.shape)\nprint(train_image_stats_05.shape)\nprint(train_image_stats_06.shape)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.701151Z","iopub.status.idle":"2026-03-01T20:02:09.701555Z","shell.execute_reply.started":"2026-03-01T20:02:09.701362Z","shell.execute_reply":"2026-03-01T20:02:09.701389Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics = pd.concat([train_image_stats_01, train_image_stats_02, train_image_stats_03,\n                                   train_image_stats_04, train_image_stats_05, train_image_stats_06],\n                                  ignore_index = True)\ntrain_image_statistics.shape","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.703002Z","iopub.status.idle":"2026-03-01T20:02:09.703474Z","shell.execute_reply.started":"2026-03-01T20:02:09.703187Z","shell.execute_reply":"2026-03-01T20:02:09.703247Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.705097Z","iopub.status.idle":"2026-03-01T20:02:09.705415Z","shell.execute_reply.started":"2026-03-01T20:02:09.705229Z","shell.execute_reply":"2026-03-01T20:02:09.705243Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics.info()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.706335Z","iopub.status.idle":"2026-03-01T20:02:09.706649Z","shell.execute_reply.started":"2026-03-01T20:02:09.706486Z","shell.execute_reply":"2026-03-01T20:02:09.706501Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\"","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.708174Z","iopub.status.idle":"2026-03-01T20:02:09.708486Z","shell.execute_reply.started":"2026-03-01T20:02:09.708356Z","shell.execute_reply":"2026-03-01T20:02:09.708372Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_names = train_image_statistics[\"image_name\"].values\nrandom_images = [np.random.choice(image_names) for i in range(4)] # Generates a random sample from a given 1-D array\nrandom_images ","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.709214Z","iopub.status.idle":"2026-03-01T20:02:09.709574Z","shell.execute_reply.started":"2026-03-01T20:02:09.70945Z","shell.execute_reply":"2026-03-01T20:02:09.709468Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nfor i in range(4) : \n    plt.subplot(2, 2, i + 1) \n    image = cv2.imread(os.path.join(train_dir, random_images[i]))\n    # cv2 reads images in BGR format. Hence we convert it to RGB\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image, cmap = \"gray\")\n    plt.grid(True)\n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.711195Z","iopub.status.idle":"2026-03-01T20:02:09.71181Z","shell.execute_reply.started":"2026-03-01T20:02:09.711658Z","shell.execute_reply":"2026-03-01T20:02:09.711678Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def non_local_means_denoising(image) : \n    denoised_image = cv2.fastNlMeansDenoisingColored(image, None, 10, 10, 7, 21)\n    return denoised_image","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.712782Z","iopub.status.idle":"2026-03-01T20:02:09.713069Z","shell.execute_reply.started":"2026-03-01T20:02:09.71294Z","shell.execute_reply":"2026-03-01T20:02:09.712955Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_image = cv2.imread(os.path.join(train_dir, random_images[0]))\n# cv2 reads images in BGR format. Hence we convert it to RGB\nsample_image = cv2.cvtColor(sample_image, cv2.COLOR_BGR2RGB)\ndenoised_image = non_local_means_denoising(sample_image)\n\n\nplt.figure(figsize = (12, 8))\nplt.subplot(1,2,1)\nplt.imshow(sample_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Normal Image\")\n\nplt.subplot(1,2,2)  \nplt.imshow(denoised_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Denoised image\")    \n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout() ","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.714013Z","iopub.status.idle":"2026-03-01T20:02:09.714324Z","shell.execute_reply.started":"2026-03-01T20:02:09.714149Z","shell.execute_reply":"2026-03-01T20:02:09.714164Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def hair_removal(image):\n    gray = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (17,17))\n    blackhat = cv2.morphologyEx(gray, cv2.MORPH_BLACKHAT, kernel)\n\n    _, thresh = cv2.threshold(blackhat, 10, 255, cv2.THRESH_BINARY)\n    thresh = cv2.dilate(thresh, None, iterations=2)\n    thresh = cv2.erode(thresh, None, iterations=2)\n\n    dst = cv2.inpaint(image, thresh, 3, cv2.INPAINT_TELEA)\n\n    return dst, thresh","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.715521Z","iopub.status.idle":"2026-03-01T20:02:09.715817Z","shell.execute_reply.started":"2026-03-01T20:02:09.715671Z","shell.execute_reply":"2026-03-01T20:02:09.715704Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hair_removed, hair_mask = hair_removal(sample_image)\n\nplt.figure(figsize=(15,5))\n\nplt.subplot(1,3,1)\nplt.imshow(sample_image)\nplt.title(\"Original Image\")\nplt.axis(\"off\")\n\nplt.subplot(1,3,2)\nplt.imshow(hair_mask, cmap=\"gray\")\nplt.title(\"Detected Hair Mask\")\nplt.axis(\"off\")\n\nplt.subplot(1,3,3)\nplt.imshow(hair_removed)\nplt.title(\"Hair Removed Image\")\nplt.axis(\"off\")\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.717556Z","iopub.status.idle":"2026-03-01T20:02:09.717846Z","shell.execute_reply.started":"2026-03-01T20:02:09.717718Z","shell.execute_reply":"2026-03-01T20:02:09.717734Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def histogram_equalization(image) : \n    image_ycrcb = cv2.cvtColor(image, cv2.COLOR_RGB2YCR_CB)\n    y_channel = image_ycrcb[:,:,0] # apply local histogram processing on this channel\n    cr_channel = image_ycrcb[:,:,1]\n    cb_channel = image_ycrcb[:,:,2]\n    \n    # Local histogram equalization\n    clahe = cv2.createCLAHE(clipLimit = 2.0, tileGridSize=(8,8))\n    equalized = clahe.apply(y_channel)\n    equalized_image = cv2.merge([equalized, cr_channel, cb_channel])\n    equalized_image = cv2.cvtColor(equalized_image, cv2.COLOR_YCR_CB2RGB)\n    return equalized_image","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.718749Z","iopub.status.idle":"2026-03-01T20:02:09.719022Z","shell.execute_reply.started":"2026-03-01T20:02:09.718864Z","shell.execute_reply":"2026-03-01T20:02:09.718878Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Denoise\ndenoised_image = non_local_means_denoising(sample_image)\n\n# 2. Hair Removal\nhair_removed, hair_mask = hair_removal(denoised_image)\n\n# 3. Histogram Equalization (dùng ảnh đã xóa lông)\nequalized_image = histogram_equalization(hair_removed)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.720007Z","iopub.status.idle":"2026-03-01T20:02:09.72039Z","shell.execute_reply.started":"2026-03-01T20:02:09.72018Z","shell.execute_reply":"2026-03-01T20:02:09.720204Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nplt.subplot(1,4,1)\nplt.imshow(sample_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Normal img\", fontsize = 14)\n\nplt.subplot(1,4,2)  \nplt.imshow(denoised_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"denoised img\", fontsize = 14)\n\nplt.subplot(1,4,3)  \nplt.imshow(hair_removed, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"img after hair removal\", fontsize = 14)\n\nplt.subplot(1,4,4)  \nplt.imshow(equalized_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Histogram equalized img\", fontsize = 14)\n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.723568Z","iopub.status.idle":"2026-03-01T20:02:09.723924Z","shell.execute_reply.started":"2026-03-01T20:02:09.723794Z","shell.execute_reply":"2026-03-01T20:02:09.723809Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def segmentation(image, k, attempts) : \n    vectorized = np.float32(image.reshape((-1, 3)))\n    criteria = (cv2.TERM_CRITERIA_EPS + cv2.TERM_CRITERIA_MAX_ITER, 20, 1.0)\n    res , label , center = cv2.kmeans(vectorized, k, None, criteria, attempts, cv2.KMEANS_PP_CENTERS)\n    center = np.uint8(center)\n    res = center[label.flatten()]\n    segmented_image = res.reshape((image.shape))\n    return segmented_image","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.725511Z","iopub.status.idle":"2026-03-01T20:02:09.725833Z","shell.execute_reply.started":"2026-03-01T20:02:09.725666Z","shell.execute_reply":"2026-03-01T20:02:09.72569Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nplt.subplot(1,1,1)\nplt.imshow(hair_removed, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"hair remove Image\")","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.726718Z","iopub.status.idle":"2026-03-01T20:02:09.72694Z","shell.execute_reply.started":"2026-03-01T20:02:09.726828Z","shell.execute_reply":"2026-03-01T20:02:09.726841Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nsegmented_image = segmentation(hair_removed, 3, 10) # k = 3, attempt = 10\nplt.subplot(1,3,1)\nplt.imshow(segmented_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Segmented Img k = 3\")\n\nsegmented_image = segmentation(hair_removed, 4, 10) # k = 4, attempt = 10\nplt.subplot(1,3,2)\nplt.imshow(segmented_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Segmented Img k = 4\")\n\nsegmented_image = segmentation(hair_removed, 5, 10) # k = 5, attempt = 10\nplt.subplot(1,3,3)\nplt.imshow(segmented_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Segmented Img k = 5\")","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.727549Z","iopub.status.idle":"2026-03-01T20:02:09.727773Z","shell.execute_reply.started":"2026-03-01T20:02:09.727661Z","shell.execute_reply":"2026-03-01T20:02:09.727674Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SHAPE = (224, 224, 3)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.728536Z","iopub.status.idle":"2026-03-01T20:02:09.729357Z","shell.execute_reply.started":"2026-03-01T20:02:09.72915Z","shell.execute_reply":"2026-03-01T20:02:09.729168Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def resize(image, shape) : \n    image = cv2.resize(image, (shape[0], shape[1]))\n    return image   ","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.731263Z","iopub.status.idle":"2026-03-01T20:02:09.73178Z","shell.execute_reply.started":"2026-03-01T20:02:09.731583Z","shell.execute_reply":"2026-03-01T20:02:09.731607Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -----------------------------\n# Train/Validation split (patient-level) + generators (match paper)\n# -----------------------------\n\n# 1) Labels\n# Use 'target' as binary label: 0 = benign, 1 = malignant\ntrain[\"target\"] = train[\"target\"].astype(int)\n\n# 2) Metadata cleaning (match paper)\n# sex: fill missing with mode; age: fill missing with median; anatom_site: fill missing with 'torso'\nif \"sex\" in train.columns:\n    sex_mode = train[\"sex\"].mode(dropna=True)[0]\n    train[\"sex\"] = train[\"sex\"].fillna(sex_mode)\n\nif \"age_approx\" in train.columns:\n    age_median = float(train[\"age_approx\"].median())\n    train[\"age_approx\"] = train[\"age_approx\"].fillna(age_median)\n\nif \"anatom_site_general_challenge\" in train.columns:\n    train[\"anatom_site_general_challenge\"] = train[\"anatom_site_general_challenge\"].fillna(\"torso\")\n\n# 3) File names\ntrain[\"image_name\"] = train[\"image_name\"].astype(str) + \".jpg\"\n\n# 4) Patient-level split (external, reproducible)\nfrom sklearn.model_selection import StratifiedGroupKFold, GroupShuffleSplit\n\ndef make_patient_split(df, label_col=\"target\", group_col=\"patient_id\", val_ratio=0.2, seed=42):\n    y = df[label_col].values\n    groups = df[group_col].values if group_col in df.columns else None\n\n    # Prefer StratifiedGroupKFold if available\n    try:\n        sgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=seed)\n        train_idx, val_idx = next(sgkf.split(df, y=y, groups=groups))\n        return df.iloc[train_idx].reset_index(drop=True), df.iloc[val_idx].reset_index(drop=True)\n    except Exception:\n        # Fallback: GroupShuffleSplit (may not perfectly stratify)\n        gss = GroupShuffleSplit(n_splits=1, test_size=val_ratio, random_state=seed)\n        train_idx, val_idx = next(gss.split(df, y=y, groups=groups))\n        return df.iloc[train_idx].reset_index(drop=True), df.iloc[val_idx].reset_index(drop=True)\n\ntrain_df, val_df = make_patient_split(train, label_col=\"target\", group_col=\"patient_id\", val_ratio=0.2, seed=42)\n\nprint(\"Train size:\", len(train_df), \" Val size:\", len(val_df))\nprint(\"Train positives:\", train_df[\"target\"].sum(), \" Val positives:\", val_df[\"target\"].sum())\nif \"patient_id\" in train.columns:\n    print(\"Unique patients - train:\", train_df[\"patient_id\"].nunique(), \" val:\", val_df[\"patient_id\"].nunique())\n\n# 5) Class imbalance handling (oversample malignant WITH REPLACEMENT on TRAIN only)\nbenign_df = train_df[train_df[\"target\"] == 0]\nmalignant_df = train_df[train_df[\"target\"] == 1]\n\nmalignant_oversampled = malignant_df.sample(len(benign_df), replace=True, random_state=42)\ntrain_balanced = pd.concat([benign_df, malignant_oversampled], axis=0).sample(frac=1, random_state=42).reset_index(drop=True)\n\nprint(\"Balanced train distribution:\")\nprint(train_balanced[\"target\"].value_counts())\n\n# 6) Artifact-aware preprocessing as generator preprocessing_function (match paper)\nimport cv2\nimport numpy as np\n\ndef preprocessing_function_cv2(img):\n    # img is a float array (usually 0..255). Convert to uint8 for OpenCV.\n    x = img.astype(np.uint8)\n\n    # Non-local means denoising (OpenCV)\n    x = cv2.fastNlMeansDenoisingColored(x, None, 10, 10, 7, 21)\n\n    # Hair removal (black-hat + inpainting)\n    gray = cv2.cvtColor(x, cv2.COLOR_RGB2GRAY)\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (17, 17))\n    blackhat = cv2.morphologyEx(gray, cv2.MORPH_BLACKHAT, kernel)\n    _, thresh = cv2.threshold(blackhat, 10, 255, cv2.THRESH_BINARY)\n    thresh = cv2.dilate(thresh, None, iterations=2)\n    thresh = cv2.erode(thresh, None, iterations=2)\n    x = cv2.inpaint(x, thresh, 3, cv2.INPAINT_TELEA)\n\n    # CLAHE on luminance channel (Y in YCrCb)\n    ycrcb = cv2.cvtColor(x, cv2.COLOR_RGB2YCrCb)\n    y = ycrcb[:, :, 0]\n    cr = ycrcb[:, :, 1]\n    cb = ycrcb[:, :, 2]\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    y_eq = clahe.apply(y)\n    x = cv2.cvtColor(cv2.merge([y_eq, cr, cb]), cv2.COLOR_YCrCb2RGB)\n\n    return x.astype(np.float32)\n\n# 7) Generators (NO internal validation_split)\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\naugment_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=20,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    vertical_flip=True,\n    preprocessing_function=preprocessing_function_cv2\n)\n\ntrain_generator = augment_datagen.flow_from_dataframe(\n    dataframe=train_balanced,\n    directory=train_dir,\n    x_col=\"image_name\",\n    y_col=\"target\",\n    target_size=(224, 224),\n    class_mode=\"raw\",\n    batch_size=32,\n    shuffle=True,\n    seed=42\n)\n\nval_datagen = ImageDataGenerator(\n    rescale=1./255,\n    preprocessing_function=preprocessing_function_cv2\n)\n\nval_generator = val_datagen.flow_from_dataframe(\n    dataframe=val_df,\n    directory=train_dir,\n    x_col=\"image_name\",\n    y_col=\"target\",\n    target_size=(224, 224),\n    class_mode=\"raw\",\n    batch_size=32,\n    shuffle=False\n)","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.732642Z","iopub.status.idle":"2026-03-01T20:02:09.733072Z","shell.execute_reply.started":"2026-03-01T20:02:09.732858Z","shell.execute_reply":"2026-03-01T20:02:09.732885Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sanity-check: class distribution before/after oversampling (train split only)\nimport matplotlib.pyplot as plt\n\nbefore = train_df[\"target\"].value_counts().sort_index()\nafter = train_balanced[\"target\"].value_counts().sort_index()\n\nplt.figure(figsize=(6,4))\nplt.bar([\"benign (0)\", \"malignant (1)\"], before.values)\nplt.title(\"Train split class distribution (before)\")\nplt.show()\n\nplt.figure(figsize=(6,4))\nplt.bar([\"benign (0)\", \"malignant (1)\"], after.values)\nplt.title(\"Train split class distribution (after oversampling)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2026-03-01T20:02:09.735086Z","iopub.status.idle":"2026-03-01T20:02:09.735477Z","shell.execute_reply.started":"2026-03-01T20:02:09.735263Z","shell.execute_reply":"2026-03-01T20:02:09.735312Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model training (MobileNetV2) and evaluation (matches paper)","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\n\ntf.random.set_seed(42)\n\n# Build MobileNetV2 with logits head (for calibration)\nbase = MobileNetV2(include_top=False, weights=\"imagenet\", input_shape=(224,224,3))\nbase.trainable = False  # warm-up stage\n\ninputs = layers.Input(shape=(224,224,3))\nx = base(inputs, training=False)\nx = layers.GlobalAveragePooling2D()(x)\nx = layers.Dropout(0.2)(x)\nlogits = layers.Dense(1, activation=None, name=\"logits\")(x)\nprobs = layers.Activation(\"sigmoid\", name=\"prob\")(logits)\n\nmodel = Model(inputs, probs)\nlogits_model = Model(inputs, logits)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n    loss=tf.keras.losses.BinaryCrossentropy(),\n    metrics=[\n        tf.keras.metrics.AUC(name=\"roc_auc\"),\n        tf.keras.metrics.BinaryAccuracy(name=\"accuracy\"),\n        tf.keras.metrics.Precision(name=\"precision\"),\n        tf.keras.metrics.Recall(name=\"recall\")\n    ]\n)\n\ncallbacks = [\n    EarlyStopping(monitor=\"val_roc_auc\", mode=\"max\", patience=3, restore_best_weights=True),\n    ReduceLROnPlateau(monitor=\"val_roc_auc\", mode=\"max\", factor=0.5, patience=2, min_lr=1e-6),\n    ModelCheckpoint(\"mobilenetv2_best.h5\", monitor=\"val_roc_auc\", mode=\"max\", save_best_only=True, save_weights_only=False)\n]\n\nhistory = model.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=10,\n    callbacks=callbacks,\n    verbose=1\n)\n\n# Optional fine-tuning: unfreeze top layers of backbone\nbase.trainable = True\nfor layer in base.layers[:-30]:\n    layer.trainable = False\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5),\n    loss=tf.keras.losses.BinaryCrossentropy(),\n    metrics=[\n        tf.keras.metrics.AUC(name=\"roc_auc\"),\n        tf.keras.metrics.BinaryAccuracy(name=\"accuracy\"),\n        tf.keras.metrics.Precision(name=\"precision\"),\n        tf.keras.metrics.Recall(name=\"recall\")\n    ]\n)\n\nhistory_ft = model.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=5,\n    callbacks=callbacks,\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-01T20:02:09.736761Z","iopub.status.idle":"2026-03-01T20:02:09.737105Z","shell.execute_reply.started":"2026-03-01T20:02:09.736965Z","shell.execute_reply":"2026-03-01T20:02:09.736989Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Discrimination metrics + confusion matrix (validation)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import roc_auc_score, average_precision_score, confusion_matrix, classification_report\n\n# Collect validation predictions\nval_probs = model.predict(val_generator, verbose=0).reshape(-1)\nval_y = val_df[\"target\"].values.astype(int)\n\nroc_auc = roc_auc_score(val_y, val_probs)\npr_auc = average_precision_score(val_y, val_probs)\n\nprint(\"Validation ROC-AUC:\", roc_auc)\nprint(\"Validation PR-AUC:\", pr_auc)\n\n# Threshold at 0.5 (reporting)\nval_pred = (val_probs >= 0.5).astype(int)\nprint(\"Confusion matrix (thr=0.5):\")\nprint(confusion_matrix(val_y, val_pred))\nprint(\"\\nClassification report:\")\nprint(classification_report(val_y, val_pred, digits=4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-01T20:02:09.738383Z","iopub.status.idle":"2026-03-01T20:02:09.738712Z","shell.execute_reply.started":"2026-03-01T20:02:09.73854Z","shell.execute_reply":"2026-03-01T20:02:09.738564Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Calibration: Temperature scaling + ECE/Brier","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\ndef fit_temperature(logits, labels, lr=0.05, steps=300):\n    # logits: shape (N,1) or (N,)\n    logits = tf.convert_to_tensor(logits.reshape(-1,1), dtype=tf.float32)\n    labels = tf.convert_to_tensor(labels.reshape(-1,1), dtype=tf.float32)\n\n    T = tf.Variable(1.0, dtype=tf.float32)\n\n    opt = tf.keras.optimizers.Adam(lr)\n    for _ in range(steps):\n        with tf.GradientTape() as tape:\n            scaled_logits = logits / tf.maximum(T, 1e-6)\n            loss = tf.reduce_mean(tf.nn.sigmoid_cross_entropy_with_logits(labels=labels, logits=scaled_logits))\n        grads = tape.gradient(loss, [T])\n        opt.apply_gradients(zip(grads, [T]))\n\n    return float(T.numpy())\n\ndef brier_score(y_true, p):\n    y_true = y_true.astype(np.float32)\n    p = p.astype(np.float32)\n    return float(np.mean((p - y_true) ** 2))\n\ndef expected_calibration_error(y_true, p, n_bins=15):\n    y_true = y_true.astype(np.int32)\n    p = p.astype(np.float32)\n    bins = np.linspace(0.0, 1.0, n_bins + 1)\n    ece = 0.0\n    for i in range(n_bins):\n        lo, hi = bins[i], bins[i+1]\n        mask = (p > lo) & (p <= hi) if i > 0 else (p >= lo) & (p <= hi)\n        if np.any(mask):\n            conf = np.mean(p[mask])\n            acc = np.mean(y_true[mask])\n            ece += np.abs(acc - conf) * (np.sum(mask) / len(p))\n    return float(ece)\n\n# Get validation logits for calibration\nval_logits = logits_model.predict(val_generator, verbose=0).reshape(-1)\nT = fit_temperature(val_logits, val_y, lr=0.05, steps=300)\nprint(\"Fitted temperature T:\", T)\n\nval_probs_cal = 1.0 / (1.0 + np.exp(-(val_logits / T)))\n\nprint(\"Brier (uncalibrated):\", brier_score(val_y, val_probs))\nprint(\"Brier (calibrated):  \", brier_score(val_y, val_probs_cal))\n\nprint(\"ECE (uncalibrated):\", expected_calibration_error(val_y, val_probs, n_bins=15))\nprint(\"ECE (calibrated):  \", expected_calibration_error(val_y, val_probs_cal, n_bins=15))\n\n# Optional: reliability diagram\nimport matplotlib.pyplot as plt\n\ndef reliability_diagram(y_true, p, n_bins=10, title=\"Reliability diagram\"):\n    bins = np.linspace(0.0, 1.0, n_bins + 1)\n    bin_centers = []\n    accs = []\n    confs = []\n    for i in range(n_bins):\n        lo, hi = bins[i], bins[i+1]\n        mask = (p > lo) & (p <= hi) if i > 0 else (p >= lo) & (p <= hi)\n        if np.any(mask):\n            bin_centers.append((lo + hi) / 2)\n            confs.append(np.mean(p[mask]))\n            accs.append(np.mean(y_true[mask]))\n    plt.figure(figsize=(5,5))\n    plt.plot([0,1],[0,1], linestyle=\"--\")\n    plt.plot(confs, accs, marker=\"o\")\n    plt.title(title)\n    plt.xlabel(\"Confidence\")\n    plt.ylabel(\"Accuracy\")\n    plt.xlim(0,1); plt.ylim(0,1)\n    plt.grid(True)\n    plt.show()\n\nreliability_diagram(val_y, val_probs, n_bins=10, title=\"Uncalibrated\")\nreliability_diagram(val_y, val_probs_cal, n_bins=10, title=\"Temperature scaled\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-01T20:02:09.741066Z","iopub.status.idle":"2026-03-01T20:02:09.741696Z","shell.execute_reply.started":"2026-03-01T20:02:09.741549Z","shell.execute_reply":"2026-03-01T20:02:09.741573Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Uncertainty (MC Dropout) + selective prediction (risk-coverage)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\n\n# Build a feature extractor that always runs backbone in inference mode\nfeature_extractor = tf.keras.Model(model.input, model.get_layer(\"global_average_pooling2d\").output)\n\n# Recreate the head (dropout + dense) using existing weights\ndrop_layer = model.get_layer(index=[i for i,l in enumerate(model.layers) if isinstance(l, tf.keras.layers.Dropout)][0])\ndense_layer = model.get_layer(\"logits\")\n\ndef mc_predict_probs(dataset, n_passes=10):\n    all_mean = []\n    all_entropy = []\n    all_var = []\n\n    for batch_x, _ in dataset:\n        feats = feature_extractor(batch_x, training=False)\n        probs_passes = []\n        for _ in range(n_passes):\n            # activate dropout only\n            x = drop_layer(feats, training=True)\n            logits = dense_layer(x)\n            probs = tf.sigmoid(logits)\n            probs_passes.append(probs.numpy().reshape(-1))\n        probs_passes = np.stack(probs_passes, axis=0)  # (P, B)\n        mean_p = probs_passes.mean(axis=0)\n        var_p = probs_passes.var(axis=0)\n        # predictive entropy for binary\n        eps = 1e-7\n        ent = -(mean_p*np.log(mean_p+eps) + (1-mean_p)*np.log(1-mean_p+eps))\n\n        all_mean.append(mean_p)\n        all_var.append(var_p)\n        all_entropy.append(ent)\n\n    return np.concatenate(all_mean), np.concatenate(all_var), np.concatenate(all_entropy)\n\n# Create a tf.data dataset from val_generator for MC prediction\nval_tf = tf.data.Dataset.from_generator(\n    lambda: val_generator,\n    output_signature=(\n        tf.TensorSpec(shape=(None,224,224,3), dtype=tf.float32),\n        tf.TensorSpec(shape=(None,), dtype=tf.float32)\n    )\n)\n\nmc_mean_p, mc_var, mc_entropy = mc_predict_probs(val_tf, n_passes=10)\n\n# Risk-coverage curve (using entropy as uncertainty score)\ny_true = val_y\npred = (mc_mean_p >= 0.5).astype(int)\nerrors = (pred != y_true).astype(float)\n\norder = np.argsort(mc_entropy)  # low entropy = more confident\nerrors_sorted = errors[order]\ncoverage = np.arange(1, len(errors_sorted)+1) / len(errors_sorted)\nrisk = np.cumsum(errors_sorted) / np.arange(1, len(errors_sorted)+1)\n\nimport matplotlib.pyplot as plt\nplt.figure(figsize=(6,4))\nplt.plot(coverage, risk)\nplt.xlabel(\"Coverage (fraction accepted)\")\nplt.ylabel(\"Risk (error on accepted)\")\nplt.title(\"Risk-Coverage (selective prediction)\")\nplt.grid(True)\nplt.show()\n\nprint(\"Example: risk at 80% coverage =\", float(risk[int(0.8*len(risk))-1]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-01T20:02:09.742832Z","iopub.status.idle":"2026-03-01T20:02:09.74357Z","shell.execute_reply.started":"2026-03-01T20:02:09.743328Z","shell.execute_reply":"2026-03-01T20:02:09.743355Z"}},"outputs":[],"execution_count":null}]}