{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":13229715,"sourceType":"datasetVersion","datasetId":8385883},{"sourceId":13229732,"sourceType":"datasetVersion","datasetId":8385893},{"sourceId":38667125,"sourceType":"kernelVersion"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n        print(dirname)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:35:24.693342Z","iopub.execute_input":"2025-10-02T06:35:24.69357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport cv2\nfrom tqdm import tqdm_notebook as tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly\nimport plotly.graph_objects as go\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:32.981385Z","iopub.execute_input":"2025-10-02T06:38:32.982063Z","iopub.status.idle":"2025-10-02T06:38:34.648922Z","shell.execute_reply.started":"2025-10-02T06:38:32.982028Z","shell.execute_reply":"2025-10-02T06:38:34.648248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_image_names(dataframe) : \n    image_names = dataframe[\"image_name\"].values\n    image_names = image_names + \".jpg\"\n    return image_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:34.649971Z","iopub.execute_input":"2025-10-02T06:38:34.650306Z","iopub.status.idle":"2025-10-02T06:38:34.653718Z","shell.execute_reply.started":"2025-10-02T06:38:34.650287Z","shell.execute_reply":"2025-10-02T06:38:34.653222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_info(image_names) : \n    image_names = np.array(image_names)\n    \n    print(\"Length = \", len(image_names))\n    print(\"Type = \", type(image_names))\n    print(\"Shape = \", image_names.shape)\n    \n    return image_names","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:36.082978Z","iopub.execute_input":"2025-10-02T06:38:36.083249Z","iopub.status.idle":"2025-10-02T06:38:36.087474Z","shell.execute_reply.started":"2025-10-02T06:38:36.083227Z","shell.execute_reply":"2025-10-02T06:38:36.086918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy.stats import skew\n\ndef extract_information(image_names, directory) : \n    image_statistics = pd.DataFrame(index = np.arange(len(image_names)),\n                                    columns = [\"image_name\", \"path\", \"rows\", \"columns\", \"channels\", \n                                              \"image_mean\", \"image_standard_deviation\", \"image_skewness\",\n                                              \"mean_red_value\", \"mean_green_value\", \"mean_blue_value\"])\n    i = 0 \n    for name in tqdm(image_names) : \n        path = os.path.join(directory, name)\n        image = cv2.imread(path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        \n        image_statistics.iloc[i][\"image_name\"] = name\n        image_statistics.iloc[i][\"path\"] = path\n        image_statistics.iloc[i][\"rows\"] = image.shape[0]\n        image_statistics.iloc[i][\"columns\"] = image.shape[1]\n        image_statistics.iloc[i][\"channels\"] = image.shape[2]\n        image_statistics.iloc[i][\"image_mean\"] = np.mean(image.flatten())\n        image_statistics.iloc[i][\"image_standard_deviation\"] = np.std(image.flatten())\n        image_statistics.iloc[i][\"image_skewness\"] = skew(image.flatten())\n        image_statistics.iloc[i][\"mean_red_value\"] = np.mean(image[:,:,0])\n        image_statistics.iloc[i][\"mean_green_value\"] = np.mean(image[:,:,1])\n        image_statistics.iloc[i][\"mean_blue_value\"] = np.mean(image[:,:,2])\n        \n        i = i + 1\n        del image\n        \n    return image_statistics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:37.47539Z","iopub.execute_input":"2025-10-02T06:38:37.476308Z","iopub.status.idle":"2025-10-02T06:38:37.483665Z","shell.execute_reply.started":"2025-10-02T06:38:37.47627Z","shell.execute_reply":"2025-10-02T06:38:37.482847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\"\ntrain = pd.DataFrame(pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\"))\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:39.295534Z","iopub.execute_input":"2025-10-02T06:38:39.296206Z","iopub.status.idle":"2025-10-02T06:38:39.411443Z","shell.execute_reply.started":"2025-10-02T06:38:39.29618Z","shell.execute_reply":"2025-10-02T06:38:39.410862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_names = get_image_names(train)\nimage_names = get_info(image_names)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:41.917677Z","iopub.execute_input":"2025-10-02T06:38:41.918251Z","iopub.status.idle":"2025-10-02T06:38:41.928034Z","shell.execute_reply.started":"2025-10-02T06:38:41.918224Z","shell.execute_reply":"2025-10-02T06:38:41.927464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dir = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/test/\"\ntest = pd.DataFrame(pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/test.csv\"))\ntest.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:43.344659Z","iopub.execute_input":"2025-10-02T06:38:43.344887Z","iopub.status.idle":"2025-10-02T06:38:43.38099Z","shell.execute_reply.started":"2025-10-02T06:38:43.34487Z","shell.execute_reply":"2025-10-02T06:38:43.380443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_names = get_image_names(test)\nimage_names = get_info(image_names)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:44.928818Z","iopub.execute_input":"2025-10-02T06:38:44.929332Z","iopub.status.idle":"2025-10-02T06:38:44.935465Z","shell.execute_reply.started":"2025-10-02T06:38:44.929306Z","shell.execute_reply":"2025-10-02T06:38:44.934915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.DataFrame(pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\"))\ntest = pd.DataFrame(pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/test.csv\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:46.121669Z","iopub.execute_input":"2025-10-02T06:38:46.122392Z","iopub.status.idle":"2025-10-02T06:38:46.185377Z","shell.execute_reply.started":"2025-10-02T06:38:46.122366Z","shell.execute_reply":"2025-10-02T06:38:46.184832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape, test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:47.328307Z","iopub.execute_input":"2025-10-02T06:38:47.328761Z","iopub.status.idle":"2025-10-02T06:38:47.333449Z","shell.execute_reply.started":"2025-10-02T06:38:47.328736Z","shell.execute_reply":"2025-10-02T06:38:47.332825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:47.677159Z","iopub.execute_input":"2025-10-02T06:38:47.677365Z","iopub.status.idle":"2025-10-02T06:38:47.687016Z","shell.execute_reply.started":"2025-10-02T06:38:47.677349Z","shell.execute_reply":"2025-10-02T06:38:47.686223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:48.040437Z","iopub.execute_input":"2025-10-02T06:38:48.040614Z","iopub.status.idle":"2025-10-02T06:38:48.048741Z","shell.execute_reply.started":"2025-10-02T06:38:48.0406Z","shell.execute_reply":"2025-10-02T06:38:48.048222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:48.484646Z","iopub.execute_input":"2025-10-02T06:38:48.484817Z","iopub.status.idle":"2025-10-02T06:38:48.530228Z","shell.execute_reply.started":"2025-10-02T06:38:48.484803Z","shell.execute_reply":"2025-10-02T06:38:48.529464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:49.78877Z","iopub.execute_input":"2025-10-02T06:38:49.789082Z","iopub.status.idle":"2025-10-02T06:38:49.800066Z","shell.execute_reply.started":"2025-10-02T06:38:49.78906Z","shell.execute_reply":"2025-10-02T06:38:49.799484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train[\"patient_id\"].unique()), len(test[\"patient_id\"].unique())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:50.007678Z","iopub.execute_input":"2025-10-02T06:38:50.007876Z","iopub.status.idle":"2025-10-02T06:38:50.014547Z","shell.execute_reply.started":"2025-10-02T06:38:50.007859Z","shell.execute_reply":"2025-10-02T06:38:50.013808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train[\"target\"].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:50.26805Z","iopub.execute_input":"2025-10-02T06:38:50.268675Z","iopub.status.idle":"2025-10-02T06:38:50.275563Z","shell.execute_reply.started":"2025-10-02T06:38:50.268654Z","shell.execute_reply":"2025-10-02T06:38:50.274952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"malignant = len(train[train[\"target\"] == 1])\nbenign = len(train[train[\"target\"] == 0])\n\nlabels = [\"Malignant\", \"Benign\"] \nsize = [malignant, benign]\n\nplt.figure(figsize = (8, 8))\nplt.pie(size, labels = labels, shadow = True, startangle = 90, colors = [\"r\", \"g\"])\nplt.title(\"Malignant VS Benign Cases\")\nplt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:38:51.488826Z","iopub.execute_input":"2025-10-02T06:38:51.489552Z","iopub.status.idle":"2025-10-02T06:38:51.835503Z","shell.execute_reply.started":"2025-10-02T06:38:51.489525Z","shell.execute_reply":"2025-10-02T06:38:51.834846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_males = len(train[train[\"sex\"] == \"male\"])\ntrain_females  = len(train[train[\"sex\"] == \"female\"])\n\ntest_males = len(test[test[\"sex\"] == \"male\"])\ntest_females  = len(test[test[\"sex\"] == \"female\"])\n\nlabels = [\"Males\", \"Female\"] \n\nsize = [train_males, train_females]\nexplode = [0.1, 0.0]\n\nplt.figure(figsize = (16, 16))\nplt.subplot(1,2,1)\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90, colors = [\"b\", \"g\"])\nplt.title(\"Male VS Female Training Set Count\", fontsize = 18)\nplt.legend()\n\nprint(\"Number of males in training set = \", train_males)\nprint(\"Number of females in training set= \", train_females)\n\nsize = [test_males, test_females]\n\nplt.subplot(1,2,2)\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90, colors = [\"b\", \"g\"])\nplt.title(\"Male VS Female Test Set Count\", fontsize = 18)\nplt.legend()\n\nprint(\"Number of males in testing set = \", test_males)\nprint(\"Number of females in testing set= \", test_females)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:06.673267Z","iopub.execute_input":"2025-10-02T06:39:06.673834Z","iopub.status.idle":"2025-10-02T06:39:06.962913Z","shell.execute_reply.started":"2025-10-02T06:39:06.673813Z","shell.execute_reply":"2025-10-02T06:39:06.962276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_malignant  = train[train[\"target\"] == 1]\ntrain_malignant_males = len(train_malignant[train_malignant[\"sex\"] == \"male\"])\ntrain_malignant_females  = len(train_malignant[train_malignant[\"sex\"] == \"female\"])\n\nlabels = [\"Malignant Male Cases\", \"Malignant Female Cases\"] \nsize = [train_malignant_males, train_malignant_females]\nexplode = [0.1, 0.0]\n\nplt.figure(figsize = (10, 10))\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90, colors = [\"r\", \"c\"])\nplt.title(\"Malignant Male VS Female Cases\", fontsize = 18)\nplt.legend()\nprint(\"Malignant Male Cases = \", train_malignant_males)\nprint(\"Malignant Female Cases = \", train_malignant_females)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:09.638583Z","iopub.execute_input":"2025-10-02T06:39:09.639343Z","iopub.status.idle":"2025-10-02T06:39:09.819648Z","shell.execute_reply.started":"2025-10-02T06:39:09.639304Z","shell.execute_reply":"2025-10-02T06:39:09.818823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_benign  = train[train[\"target\"] == 0]\n\ntrain_benign_males = len(train_benign[train_benign[\"sex\"] == \"male\"])\ntrain_benign_females  = len(train_benign[train_benign[\"sex\"] == \"female\"]) \n\nlabels = [\"Benign Male Cases\", \"Benign Female Cases\"] \nsize = [train_benign_males, train_benign_females]\nexplode = [0.1, 0.0]\n\nplt.figure(figsize = (10, 10))\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90, colors = [\"g\", \"y\"])\nplt.title(\"Benign Male VS Benign Female Cases\", fontsize = 18)\nplt.legend()\nprint(\"Benign Male Cases = \", train_benign_males)\nprint(\"Benign Female Cases = \", train_benign_females)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:12.013494Z","iopub.execute_input":"2025-10-02T06:39:12.014033Z","iopub.status.idle":"2025-10-02T06:39:12.253827Z","shell.execute_reply.started":"2025-10-02T06:39:12.014008Z","shell.execute_reply":"2025-10-02T06:39:12.253178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cancer_versus_sex = train.groupby([\"benign_malignant\", \"sex\"]).size()\nprint(cancer_versus_sex)\nprint(type(cancer_versus_sex))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:12.3322Z","iopub.execute_input":"2025-10-02T06:39:12.332722Z","iopub.status.idle":"2025-10-02T06:39:12.345265Z","shell.execute_reply.started":"2025-10-02T06:39:12.332703Z","shell.execute_reply":"2025-10-02T06:39:12.344675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cancer_versus_sex = cancer_versus_sex.unstack(level = 1) / len(train) * 100\nprint(cancer_versus_sex)\nprint(type(cancer_versus_sex))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:12.67434Z","iopub.execute_input":"2025-10-02T06:39:12.674528Z","iopub.status.idle":"2025-10-02T06:39:12.690163Z","shell.execute_reply.started":"2025-10-02T06:39:12.674513Z","shell.execute_reply":"2025-10-02T06:39:12.68934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set(style='whitegrid')\nsns.set_context(\"paper\", rc={\"font.size\":12,\"axes.titlesize\":20,\"axes.labelsize\":18})   \n\nplt.figure(figsize = (10, 6))\nsns.heatmap(cancer_versus_sex, annot=True, cmap=\"icefire\", cbar=True)\nplt.title(\"Cancer VS Sex Heatmap Analysis Normalized\", fontsize = 18)\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:15.729284Z","iopub.execute_input":"2025-10-02T06:39:15.729575Z","iopub.status.idle":"2025-10-02T06:39:15.972363Z","shell.execute_reply.started":"2025-10-02T06:39:15.729554Z","shell.execute_reply":"2025-10-02T06:39:15.97174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train\ntrain_torso = len(train[train[\"anatom_site_general_challenge\"] == \"torso\"])\ntrain_lower_extremity = len(train[train[\"anatom_site_general_challenge\"] == \"lower extremity\"])\ntrain_upper_extremity = len(train[train[\"anatom_site_general_challenge\"] == \"upper extremity\"])\ntrain_head_neck = len(train[train[\"anatom_site_general_challenge\"] == \"head/neck\"])\ntrain_palms_soles = len(train[train[\"anatom_site_general_challenge\"] == \"palms/soles\"])\ntrain_oral_genital = len(train[train[\"anatom_site_general_challenge\"] == \"oral/genital\"])\n\n# test\ntest_torso = len(test[test[\"anatom_site_general_challenge\"] == \"torso\"])\ntest_lower_extremity = len(test[test[\"anatom_site_general_challenge\"] == \"lower extremity\"])\ntest_upper_extremity = len(test[test[\"anatom_site_general_challenge\"] == \"upper extremity\"])\ntest_head_neck = len(test[test[\"anatom_site_general_challenge\"] == \"head/neck\"])\ntest_palms_soles = len(test[test[\"anatom_site_general_challenge\"] == \"palms/soles\"])\ntest_oral_genital = len(test[test[\"anatom_site_general_challenge\"] == \"oral/genital\"])\n\nlabels = [\"Torso\", \"Lower Extremity\", \"Upper Extremity\", \"Head/Neck\", \"Palms/Soles\", \"Oral/Genital\"] \n\nplt.figure(figsize = (16, 16))\n\nplt.subplot(1,2,1)\nsize = [train_torso, train_lower_extremity, train_upper_extremity, train_head_neck, train_palms_soles, train_oral_genital]\nexplode = [0.05, 0.05, 0.05, 0.05, 0.05, 0.1]\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90)\nplt.title(\"Anatomy Sites In Training Set\", fontsize = 18)\nplt.legend()\n\nplt.subplot(1,2,2)\nsize = [test_torso, test_lower_extremity, test_upper_extremity, test_head_neck, test_palms_soles, test_oral_genital]\nexplode = [0.05, 0.05, 0.05, 0.05, 0.05, 0.1]\nplt.pie(size, labels = labels, explode = explode, shadow = True, startangle = 90)\nplt.title(\"Anatomy Sites In Testing Set\", fontsize = 18)\nplt.legend()\n\n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:17.82056Z","iopub.execute_input":"2025-10-02T06:39:17.820836Z","iopub.status.idle":"2025-10-02T06:39:18.349296Z","shell.execute_reply.started":"2025-10-02T06:39:17.820816Z","shell.execute_reply":"2025-10-02T06:39:18.348627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ages_benign = train.loc[train[\"target\"] == 0, \"age_approx\"]\ntrain_ages_malignant = train.loc[train[\"target\"] == 1 , \"age_approx\"]\n\nplt.figure(figsize = (10, 8))\nsns.kdeplot(train_ages_benign, label = \"Benign\", shade = True, legend = True, cbar = True)\nsns.kdeplot(train_ages_malignant, label = \"Malignant\", shade = True, legend = True, cbar = True)\nplt.grid(True)\nplt.xlabel(\"Age Of The Patients\", fontsize = 18)\nplt.ylabel(\"Probability Density\", fontsize = 18)\nplt.grid(which = \"minor\", axis = \"both\")\nplt.legend()\nplt.title(\"Probabilistic Age Distribution In Training Set\", fontsize = 18)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:16:52.550127Z","iopub.execute_input":"2025-10-02T07:16:52.550412Z","iopub.status.idle":"2025-10-02T07:16:53.071624Z","shell.execute_reply.started":"2025-10-02T07:16:52.550391Z","shell.execute_reply":"2025-10-02T07:16:53.070883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_stats_01 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_01\"))\ntrain_image_stats_02 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_02\"))\ntrain_image_stats_03 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_03\"))\ntrain_image_stats_04 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_04\"))\ntrain_image_stats_05 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_05\"))\ntrain_image_stats_06 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_06\"))\n\nprint(train_image_stats_01.shape)\nprint(train_image_stats_02.shape)\nprint(train_image_stats_03.shape)\nprint(train_image_stats_04.shape)\nprint(train_image_stats_05.shape)\nprint(train_image_stats_06.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:27.775337Z","iopub.execute_input":"2025-10-02T06:39:27.775602Z","iopub.status.idle":"2025-10-02T06:39:28.016632Z","shell.execute_reply.started":"2025-10-02T06:39:27.775582Z","shell.execute_reply":"2025-10-02T06:39:28.016012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics = pd.concat([train_image_stats_01, train_image_stats_02, train_image_stats_03,\n                                   train_image_stats_04, train_image_stats_05, train_image_stats_06],\n                                  ignore_index = True)\ntrain_image_statistics.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:28.017593Z","iopub.execute_input":"2025-10-02T06:39:28.017891Z","iopub.status.idle":"2025-10-02T06:39:28.024564Z","shell.execute_reply.started":"2025-10-02T06:39:28.017873Z","shell.execute_reply":"2025-10-02T06:39:28.023899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:30.591023Z","iopub.execute_input":"2025-10-02T06:39:30.591287Z","iopub.status.idle":"2025-10-02T06:39:30.603817Z","shell.execute_reply.started":"2025-10-02T06:39:30.59127Z","shell.execute_reply":"2025-10-02T06:39:30.603208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image_stats_01 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled-test/melanoma_image_statistics_compiled_test_01\"))\ntest_image_stats_02 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled-test/melanoma_image_statistics_compiled_test_02\"))\n\nprint(test_image_stats_01.shape)\nprint(test_image_stats_02.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:31.82803Z","iopub.execute_input":"2025-10-02T06:39:31.828291Z","iopub.status.idle":"2025-10-02T06:39:31.908065Z","shell.execute_reply.started":"2025-10-02T06:39:31.828276Z","shell.execute_reply":"2025-10-02T06:39:31.907456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image_statistics = pd.concat([test_image_stats_01, test_image_stats_02], ignore_index = True)\n\ntest_image_statistics.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:33.196285Z","iopub.execute_input":"2025-10-02T06:39:33.196511Z","iopub.status.idle":"2025-10-02T06:39:33.202833Z","shell.execute_reply.started":"2025-10-02T06:39:33.196495Z","shell.execute_reply":"2025-10-02T06:39:33.20228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image_statistics.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:33.444781Z","iopub.execute_input":"2025-10-02T06:39:33.445391Z","iopub.status.idle":"2025-10-02T06:39:33.455847Z","shell.execute_reply.started":"2025-10-02T06:39:33.445373Z","shell.execute_reply":"2025-10-02T06:39:33.455091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:33.728922Z","iopub.execute_input":"2025-10-02T06:39:33.729176Z","iopub.status.idle":"2025-10-02T06:39:33.740047Z","shell.execute_reply.started":"2025-10-02T06:39:33.729158Z","shell.execute_reply":"2025-10-02T06:39:33.739291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image_statistics.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:34.181238Z","iopub.execute_input":"2025-10-02T06:39:34.181998Z","iopub.status.idle":"2025-10-02T06:39:34.193978Z","shell.execute_reply.started":"2025-10-02T06:39:34.181965Z","shell.execute_reply":"2025-10-02T06:39:34.193229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_names = train_image_statistics[\"image_name\"].values\nrandom_images = [np.random.choice(image_names) for i in range(4)] # Generates a random sample from a given 1-D array\nrandom_images ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:35.65905Z","iopub.execute_input":"2025-10-02T06:39:35.659339Z","iopub.status.idle":"2025-10-02T06:39:35.665204Z","shell.execute_reply.started":"2025-10-02T06:39:35.659317Z","shell.execute_reply":"2025-10-02T06:39:35.664499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:35.93164Z","iopub.execute_input":"2025-10-02T06:39:35.932063Z","iopub.status.idle":"2025-10-02T06:39:35.935216Z","shell.execute_reply.started":"2025-10-02T06:39:35.932042Z","shell.execute_reply":"2025-10-02T06:39:35.934589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nfor i in range(4) : \n    plt.subplot(2, 2, i + 1) \n    image = cv2.imread(os.path.join(train_dir, random_images[i]))\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image, cmap = \"gray\")\n    plt.grid(True)\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:36.68796Z","iopub.execute_input":"2025-10-02T06:39:36.688206Z","iopub.status.idle":"2025-10-02T06:39:44.840498Z","shell.execute_reply.started":"2025-10-02T06:39:36.688187Z","shell.execute_reply":"2025-10-02T06:39:44.839781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"benign_mean_red_value = []\nbenign_mean_green_value = []\nbenign_mean_blue_value = []\n\nmalignant_mean_red_value = []\nmalignant_mean_green_value = []\nmalignant_mean_blue_value = []\n\nfor image_name in tqdm(train_image_statistics[\"image_name\"]) : \n    name = image_name[0:len(image_name)-4] \n    extracted_section = train[train[\"image_name\"] == name]\n    r = int(train_image_statistics[train_image_statistics[\"image_name\"] == image_name][\"mean_red_value\"])\n    g = int(train_image_statistics[train_image_statistics[\"image_name\"] == image_name][\"mean_green_value\"])\n    b = int(train_image_statistics[train_image_statistics[\"image_name\"] == image_name][\"mean_blue_value\"])\n    if int(extracted_section[\"target\"]) == 0 : # benign\n        benign_mean_red_value.append(r)\n        benign_mean_green_value.append(g)\n        benign_mean_blue_value.append(b)\n    else:\n        malignant_mean_red_value.append(r)\n        malignant_mean_green_value.append(g)\n        malignant_mean_blue_value.append(b)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:39:53.167891Z","iopub.execute_input":"2025-10-02T06:39:53.168192Z","iopub.status.idle":"2025-10-02T06:45:45.97843Z","shell.execute_reply.started":"2025-10-02T06:39:53.168171Z","shell.execute_reply":"2025-10-02T06:45:45.977712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"range_of_spread = max(benign_mean_red_value) - min(benign_mean_red_value)\n\nplt.figure(figsize = (12, 8))\nplt.rc(\"font\", weight = \"bold\")\nsns.set_style(\"whitegrid\")\nfig = sns.distplot(benign_mean_red_value, hist = True, kde = True, label = \"Mean Red Channel Intensities\", color = \"r\")\nfig.set(xlabel = \"Mean red channel intensities observed in each image\",\n        ylabel = \"Probability Density\")\nplt.title(\"Spread Of Red Channel In Benign Cases\", fontsize = 18)\nplt.legend()\nprint(\"The range of spread = {:.2f}\".format(range_of_spread))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:11.82523Z","iopub.execute_input":"2025-10-02T06:47:11.825989Z","iopub.status.idle":"2025-10-02T06:47:12.403042Z","shell.execute_reply.started":"2025-10-02T06:47:11.825954Z","shell.execute_reply":"2025-10-02T06:47:12.402358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"range_of_spread = max(benign_mean_green_value) - min(benign_mean_green_value)\n\nplt.figure(figsize = (12, 8))\nplt.rc(\"font\", weight = \"bold\")\nsns.set_style(\"whitegrid\")\nfig = sns.distplot(benign_mean_green_value, hist = True, kde = True, label = \"Mean Green Channel Intensities\", color = \"g\")\nfig.set(xlabel = \"Mean green channel intensities observed in each image\",\n        ylabel = \"Probability Density\") \nplt.title(\"Spread Of Green Channel In Benign Cases\", fontsize = 18)\nplt.legend()\nprint(\"The range of spread = {:.2f}\".format(range_of_spread))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:14.864318Z","iopub.execute_input":"2025-10-02T06:47:14.864589Z","iopub.status.idle":"2025-10-02T06:47:15.45311Z","shell.execute_reply.started":"2025-10-02T06:47:14.864568Z","shell.execute_reply":"2025-10-02T06:47:15.45242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"range_of_spread = max(benign_mean_blue_value) - min(benign_mean_blue_value)\n\nplt.figure(figsize = (12, 8))\nplt.rc(\"font\", weight = \"bold\")\nsns.set_style(\"whitegrid\")\nfig = sns.distplot(benign_mean_blue_value, hist = True, kde = True, label = \"Mean Blue Channel Intensities\", color = \"b\")\nfig.set(xlabel = \"Mean blue channel intensities observed in each image\",\n        ylabel = \"Probability Density\") \nplt.title(\"Spread Of Blue Channel In Benign Cases\", fontsize = 18)\nplt.legend()\nprint(\"The range of spread = {:.2f}\".format(range_of_spread))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:17.302816Z","iopub.execute_input":"2025-10-02T06:47:17.303549Z","iopub.status.idle":"2025-10-02T06:47:17.848402Z","shell.execute_reply.started":"2025-10-02T06:47:17.303523Z","shell.execute_reply":"2025-10-02T06:47:17.847698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nplt.rc(\"font\", weight = \"bold\")\nsns.set_style(\"whitegrid\")\nfig = sns.distplot(benign_mean_blue_value, hist = False, kde = True, label = \"Mean Blue Channel Intensities\", color = \"b\")\nfig = sns.distplot(benign_mean_red_value, hist = False, kde = True, label = \"Mean Red Channel Intensities\", color = \"r\")\nfig = sns.distplot(benign_mean_green_value, hist = False, kde = True, label = \"Mean Green Channel Intensities\", color = \"g\")\n\nfig.set(xlabel = \"Mean channel intensities observed in each image\",\n        ylabel = \"Probability Density\") \nplt.title(\"Spread Of Channels In Benign Cases\", fontsize = 18)\nplt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:19.983977Z","iopub.execute_input":"2025-10-02T06:47:19.984274Z","iopub.status.idle":"2025-10-02T06:47:20.812026Z","shell.execute_reply.started":"2025-10-02T06:47:19.984254Z","shell.execute_reply":"2025-10-02T06:47:20.811319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del benign_mean_red_value\ndel benign_mean_green_value\ndel benign_mean_blue_value","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:23.413821Z","iopub.execute_input":"2025-10-02T06:47:23.414455Z","iopub.status.idle":"2025-10-02T06:47:23.417881Z","shell.execute_reply.started":"2025-10-02T06:47:23.414428Z","shell.execute_reply":"2025-10-02T06:47:23.417311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:24.907304Z","iopub.execute_input":"2025-10-02T06:47:24.907562Z","iopub.status.idle":"2025-10-02T06:47:25.006152Z","shell.execute_reply.started":"2025-10-02T06:47:24.907543Z","shell.execute_reply":"2025-10-02T06:47:25.005515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nplt.rc(\"font\", weight = \"bold\")\nsns.set_style(\"whitegrid\")\nfig = sns.distplot(malignant_mean_blue_value, hist = False, kde = True, label = \"Mean Blue Channel Intensities\", color = \"b\")\nfig = sns.distplot(malignant_mean_red_value, hist = False, kde = True, label = \"Mean Red Channel Intensities\", color = \"r\")\nfig = sns.distplot(malignant_mean_green_value, hist = False, kde = True, label = \"Mean Green Channel Intensities\", color = \"g\")\n\nfig.set(xlabel = \"Mean channel intensities observed in each image\",\n        ylabel = \"Probability Density\") \nplt.title(\"Spread Of Channels In Malignant Cases\", fontsize = 18)\nplt.legend()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:26.29558Z","iopub.execute_input":"2025-10-02T06:47:26.295841Z","iopub.status.idle":"2025-10-02T06:47:26.693711Z","shell.execute_reply.started":"2025-10-02T06:47:26.29582Z","shell.execute_reply":"2025-10-02T06:47:26.692972Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:29.051629Z","iopub.execute_input":"2025-10-02T06:47:29.051899Z","iopub.status.idle":"2025-10-02T06:47:29.129276Z","shell.execute_reply.started":"2025-10-02T06:47:29.051875Z","shell.execute_reply":"2025-10-02T06:47:29.128719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:29.941646Z","iopub.execute_input":"2025-10-02T06:47:29.942163Z","iopub.status.idle":"2025-10-02T06:47:29.951553Z","shell.execute_reply.started":"2025-10-02T06:47:29.94214Z","shell.execute_reply":"2025-10-02T06:47:29.950989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing = len(train[train[\"sex\"].isna() == True])\navailable = len(train[train[\"sex\"].isna() == False])\n\nx = [\"Availabe data\", \"Unavailable data\"]\ny = [np.log(available), np.log(missing)]\n\nprint(\"Count of missing data = \", missing)\nprint(\"Count of available data = \", available)\n\nplt.figure(figsize = (12, 8))\nplt.subplot(1,1,1)\nplt.barh(x, y, color = \"m\")\nplt.grid(True)\nplt.title(\"Data On Patient's Sex\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:30.831402Z","iopub.execute_input":"2025-10-02T06:47:30.832086Z","iopub.status.idle":"2025-10-02T06:47:31.03189Z","shell.execute_reply.started":"2025-10-02T06:47:30.83206Z","shell.execute_reply":"2025-10-02T06:47:31.031204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['sex'] = train['sex'].fillna('male')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:33.125776Z","iopub.execute_input":"2025-10-02T06:47:33.126306Z","iopub.status.idle":"2025-10-02T06:47:33.13258Z","shell.execute_reply.started":"2025-10-02T06:47:33.12628Z","shell.execute_reply":"2025-10-02T06:47:33.131962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing =  len(train[train[\"age_approx\"].isna() == True]) \navailable = len(train[train[\"age_approx\"].isna() == False]) \n\nprint(\"Missing age values = \", missing)\nprint(\"Available age data = \", available)\n\nx = [\"Availabe data\", \"Unavailable data\"]\ny = [np.log(available), np.log(missing)] \n\nplt.figure(figsize = (12, 8))\nplt.subplot(1,1,1)\nplt.barh(x, y, color = \"y\")\nplt.grid(True)\nplt.title(\"Data On Patient's Age\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:34.208403Z","iopub.execute_input":"2025-10-02T06:47:34.209233Z","iopub.status.idle":"2025-10-02T06:47:34.402541Z","shell.execute_reply.started":"2025-10-02T06:47:34.209206Z","shell.execute_reply":"2025-10-02T06:47:34.401746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train\nanatomy_sites = [\"torso\", \"upper extremity\", \"lower extremity\"]\n\nrelevant_dataframe_part = train[(train[\"sex\"] == \"male\") &\n                     (train[\"anatom_site_general_challenge\"].isin(anatomy_sites)) &\n                     (train[\"target\"] == 0)]\n\nmedian_value = relevant_dataframe_part[\"age_approx\"].median()\n\nprint(\"Median value = \", median_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:36.571991Z","iopub.execute_input":"2025-10-02T06:47:36.572246Z","iopub.status.idle":"2025-10-02T06:47:36.584492Z","shell.execute_reply.started":"2025-10-02T06:47:36.572229Z","shell.execute_reply":"2025-10-02T06:47:36.583772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[\"age_approx\"]= train[\"age_approx\"].fillna(median_value)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:38.065903Z","iopub.execute_input":"2025-10-02T06:47:38.066183Z","iopub.status.idle":"2025-10-02T06:47:38.070852Z","shell.execute_reply.started":"2025-10-02T06:47:38.066164Z","shell.execute_reply":"2025-10-02T06:47:38.070152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[\"anatom_site_general_challenge\"] = train[\"anatom_site_general_challenge\"].fillna(\"torso\")\ntest[\"anatom_site_general_challenge\"] = test[\"anatom_site_general_challenge\"].fillna(\"torso\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:38.935106Z","iopub.execute_input":"2025-10-02T06:47:38.935359Z","iopub.status.idle":"2025-10-02T06:47:38.943566Z","shell.execute_reply.started":"2025-10-02T06:47:38.93534Z","shell.execute_reply":"2025-10-02T06:47:38.942815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:39.568362Z","iopub.execute_input":"2025-10-02T06:47:39.568689Z","iopub.status.idle":"2025-10-02T06:47:39.596353Z","shell.execute_reply.started":"2025-10-02T06:47:39.568668Z","shell.execute_reply":"2025-10-02T06:47:39.595433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:41.66988Z","iopub.execute_input":"2025-10-02T06:47:41.670199Z","iopub.status.idle":"2025-10-02T06:47:41.681266Z","shell.execute_reply.started":"2025-10-02T06:47:41.670177Z","shell.execute_reply":"2025-10-02T06:47:41.680497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_stats_01 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_01\"))\ntrain_image_stats_02 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_02\"))\ntrain_image_stats_03 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_03\"))\ntrain_image_stats_04 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_04\"))\ntrain_image_stats_05 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_05\"))\ntrain_image_stats_06 = pd.DataFrame(pd.read_csv(\"/kaggle/input/compiled/melanoma_image_statistics_compiled_06\"))\n\nprint(train_image_stats_01.shape)\nprint(train_image_stats_02.shape)\nprint(train_image_stats_03.shape)\nprint(train_image_stats_04.shape)\nprint(train_image_stats_05.shape)\nprint(train_image_stats_06.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:42.460591Z","iopub.execute_input":"2025-10-02T06:47:42.460815Z","iopub.status.idle":"2025-10-02T06:47:42.573974Z","shell.execute_reply.started":"2025-10-02T06:47:42.460798Z","shell.execute_reply":"2025-10-02T06:47:42.573365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics = pd.concat([train_image_stats_01, train_image_stats_02, train_image_stats_03,\n                                   train_image_stats_04, train_image_stats_05, train_image_stats_06],\n                                  ignore_index = True)\ntrain_image_statistics.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:44.073528Z","iopub.execute_input":"2025-10-02T06:47:44.073776Z","iopub.status.idle":"2025-10-02T06:47:44.081486Z","shell.execute_reply.started":"2025-10-02T06:47:44.073759Z","shell.execute_reply":"2025-10-02T06:47:44.08081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:47.705995Z","iopub.execute_input":"2025-10-02T06:47:47.706291Z","iopub.status.idle":"2025-10-02T06:47:47.717747Z","shell.execute_reply.started":"2025-10-02T06:47:47.70627Z","shell.execute_reply":"2025-10-02T06:47:47.716912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_image_statistics.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:47:49.011719Z","iopub.execute_input":"2025-10-02T06:47:49.012091Z","iopub.status.idle":"2025-10-02T06:47:49.025167Z","shell.execute_reply.started":"2025-10-02T06:47:49.012068Z","shell.execute_reply":"2025-10-02T06:47:49.024368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:59:42.330345Z","iopub.execute_input":"2025-10-02T06:59:42.330863Z","iopub.status.idle":"2025-10-02T06:59:42.33442Z","shell.execute_reply.started":"2025-10-02T06:59:42.330839Z","shell.execute_reply":"2025-10-02T06:59:42.333644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_names = train_image_statistics[\"image_name\"].values\nrandom_images = [np.random.choice(image_names) for i in range(4)] # Generates a random sample from a given 1-D array\nrandom_images ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:02:48.727917Z","iopub.execute_input":"2025-10-02T07:02:48.728211Z","iopub.status.idle":"2025-10-02T07:02:48.733869Z","shell.execute_reply.started":"2025-10-02T07:02:48.728192Z","shell.execute_reply":"2025-10-02T07:02:48.733217Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Viewing Our Training Images : \n\nLet's plot some of our training images : ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nfor i in range(4) : \n    plt.subplot(2, 2, i + 1) \n    image = cv2.imread(os.path.join(train_dir, random_images[i]))\n    # cv2 reads images in BGR format. Hence we convert it to RGB\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    plt.imshow(image, cmap = \"gray\")\n    plt.grid(True)\n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:02:51.397483Z","iopub.execute_input":"2025-10-02T07:02:51.397738Z","iopub.status.idle":"2025-10-02T07:03:00.356601Z","shell.execute_reply.started":"2025-10-02T07:02:51.397718Z","shell.execute_reply":"2025-10-02T07:03:00.355746Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Image Denoising :\n\nMany image smoothing techniques like Gaussian Blurring, Median Blurring etc were good to some extent in removing small quantities of noise. In those techniques, we took a small neighbourhood around a pixel and performed some operations like gaussian weighted average, median of the values etc to replace the central element. In short, noise removal at a pixel was local to its neighbourhood.\n\nThere is a property of noise. **Noise is generally considered to be a random variable with zero mean.**\n\nSuppose we hold a static camera to a certain location for a couple of seconds. This will give us plenty of frames, or a lot of images of the same scene. Then averaging all the frames, we compare the final result and first frame. Reduction in noise would be easily observed.\n\nSo idea is simple, we need a set of similar images to average out the noise. Considering a small window (say 5x5 window) in the image, chance is large that the same patch may be somewhere else in the image. Sometimes in a small neighbourhood around it. Hence, using these similar patches together averaging them can lead to an efficient denoised image.\n\nThis method is **Non-Local Means Denoising. It takes more time compared to blurring techniques, but the result are very satisfying.**\n\nDenoising illustration :\n![image.png](attachment:image.png) ","metadata":{}},{"cell_type":"markdown","source":"## OpenCV implementation of the aforementioned approach :\ncv2.fastNlMeansDenoisingColored() - Works on Colored images cv2.fastNlMeansDenoising() - Works on graysacle images\n\nCommon arguments are:\n\n* h : parameter deciding filter strength. Higher h value removes noise better, but removes details of image also. (10 is ok)\n* hForColorComponents : same as h, but for color images only. (normally same as h)\n* templateWindowSize : should be odd. (recommended 7)\n* searchWindowSize : should be odd. (recommended 21)","metadata":{}},{"cell_type":"code","source":"def non_local_means_denoising(image) : \n    denoised_image = cv2.fastNlMeansDenoisingColored(image, None, 10, 10, 7, 21)\n    return denoised_image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:03:08.505253Z","iopub.execute_input":"2025-10-02T07:03:08.505503Z","iopub.status.idle":"2025-10-02T07:03:08.509311Z","shell.execute_reply.started":"2025-10-02T07:03:08.505484Z","shell.execute_reply":"2025-10-02T07:03:08.508651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_image = cv2.imread(os.path.join(train_dir, random_images[0]))\n# cv2 reads images in BGR format. Hence we convert it to RGB\nsample_image = cv2.cvtColor(sample_image, cv2.COLOR_BGR2RGB)\ndenoised_image = non_local_means_denoising(sample_image)\n\n\nplt.figure(figsize = (12, 8))\nplt.subplot(1,2,1)\nplt.imshow(sample_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Normal Image\")\n\nplt.subplot(1,2,2)  \nplt.imshow(denoised_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Denoised image\")    \n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:03:12.309199Z","iopub.execute_input":"2025-10-02T07:03:12.309427Z","iopub.status.idle":"2025-10-02T07:03:25.559825Z","shell.execute_reply.started":"2025-10-02T07:03:12.309411Z","shell.execute_reply":"2025-10-02T07:03:25.55916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def hair_removal(image):\n    gray = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\n\n    kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (17,17))\n    blackhat = cv2.morphologyEx(gray, cv2.MORPH_BLACKHAT, kernel)\n\n    _, thresh = cv2.threshold(blackhat, 10, 255, cv2.THRESH_BINARY)\n    thresh = cv2.dilate(thresh, None, iterations=2)\n    thresh = cv2.erode(thresh, None, iterations=2)\n\n    dst = cv2.inpaint(image, thresh, 3, cv2.INPAINT_TELEA)\n\n    return dst, thresh","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:03:47.685049Z","iopub.execute_input":"2025-10-02T07:03:47.685333Z","iopub.status.idle":"2025-10-02T07:03:47.690301Z","shell.execute_reply.started":"2025-10-02T07:03:47.68531Z","shell.execute_reply":"2025-10-02T07:03:47.689685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hair_removed, hair_mask = hair_removal(sample_image)\n\nplt.figure(figsize=(15,5))\n\nplt.subplot(1,3,1)\nplt.imshow(sample_image)\nplt.title(\"Original Image\")\nplt.axis(\"off\")\n\nplt.subplot(1,3,2)\nplt.imshow(hair_mask, cmap=\"gray\")\nplt.title(\"Detected Hair Mask\")\nplt.axis(\"off\")\n\nplt.subplot(1,3,3)\nplt.imshow(hair_removed)\nplt.title(\"Hair Removed Image\")\nplt.axis(\"off\")\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:03:49.657253Z","iopub.execute_input":"2025-10-02T07:03:49.657977Z","iopub.status.idle":"2025-10-02T07:03:54.283431Z","shell.execute_reply.started":"2025-10-02T07:03:49.657953Z","shell.execute_reply":"2025-10-02T07:03:54.282743Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Local Histogram Pre-Processing\n\nFirst of all, why can't we apply histogram equalization directly to an RGB image?\nHistogram equalization is a non-linear process. Channel splitting and equalizing each channel separately is incorrect. *Equalization involves intensity values of the image, not the color components*. \n\nSo for a simple RGB color image, histogram equalization cannot be applied directly on the channels.*It needs to be applied in such a way that the intensity values are equalized without disturbing the color balance of the image. So, the first step is to convert the color space of the image from RGB into one of the color spaces that separates intensity values from color components. Some of the possible options are HSV/HLS, YUV, YCbCr, etc. YCbCr is preferred as it is designed for digital images. Perform histogram equalization on the intensity plane Y. Now convert the resultant YCbCr image back to RGB.*\n\n(Excerpt taken from :\n\nhttps://prateekvjoshi.com/2013/11/22/histogram-equalization-of-rgb-images/ )\n\nAn illustration of histogram equalization : **Observe the intensity difference**\n![image.png](attachment:image.png) \n\nHere the third one is actually local histogram equalization, where we equalize intensities inside a rolling window of certain dimension instead of the whole image at once.","metadata":{}},{"cell_type":"code","source":"def histogram_equalization(image) : \n    image_ycrcb = cv2.cvtColor(image, cv2.COLOR_RGB2YCR_CB)\n    y_channel = image_ycrcb[:,:,0] # apply local histogram processing on this channel\n    cr_channel = image_ycrcb[:,:,1]\n    cb_channel = image_ycrcb[:,:,2]\n    \n    # Local histogram equalization\n    clahe = cv2.createCLAHE(clipLimit = 2.0, tileGridSize=(8,8))\n    equalized = clahe.apply(y_channel)\n    equalized_image = cv2.merge([equalized, cr_channel, cb_channel])\n    equalized_image = cv2.cvtColor(equalized_image, cv2.COLOR_YCR_CB2RGB)\n    return equalized_image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:06:51.238784Z","iopub.execute_input":"2025-10-02T07:06:51.23951Z","iopub.status.idle":"2025-10-02T07:06:51.243885Z","shell.execute_reply.started":"2025-10-02T07:06:51.239482Z","shell.execute_reply":"2025-10-02T07:06:51.243257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Denoise\ndenoised_image = non_local_means_denoising(sample_image)\n\n# 2. Hair Removal\nhair_removed, hair_mask = hair_removal(denoised_image)\n\n# 3. Histogram Equalization (dùng ảnh đã xóa lông)\nequalized_image = histogram_equalization(hair_removed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:06:52.787191Z","iopub.execute_input":"2025-10-02T07:06:52.787456Z","iopub.status.idle":"2025-10-02T07:07:04.052069Z","shell.execute_reply.started":"2025-10-02T07:06:52.787437Z","shell.execute_reply":"2025-10-02T07:07:04.051245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nplt.subplot(1,4,1)\nplt.imshow(sample_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Normal img\", fontsize = 14)\n\nplt.subplot(1,4,2)  \nplt.imshow(denoised_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"denoised img\", fontsize = 14)\n\nplt.subplot(1,4,3)  \nplt.imshow(hair_removed, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"img after hair removal\", fontsize = 14)\n\nplt.subplot(1,4,4)  \nplt.imshow(equalized_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Histogram equalized img\", fontsize = 14)\n# Automatically adjust subplot parameters to give specified padding.\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:07:04.053237Z","iopub.execute_input":"2025-10-02T07:07:04.053503Z","iopub.status.idle":"2025-10-02T07:07:09.528065Z","shell.execute_reply.started":"2025-10-02T07:07:04.053485Z","shell.execute_reply":"2025-10-02T07:07:09.527318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Image Segmentation :\n\nIs the technique of dividing or partitioning an image into parts, called segments. It is mostly useful for applications like image compression or object recognition, because for these types of applications, it is inefficient to process the whole image.\n\nWe will use **K-means clustering algorithm** to segment the images.\n![image.png](attachment:image.png)       \nK-Means Segmentation Approach Using OpenCV\n\n* `samples` : It should be of np.float32 data type, and each feature should be put in a single column. Here we have 3 channels, so every channel features have to be in one column. So, total columns we have are 3, while we don't care about the number of rows, hence -1. So, shape : (-1, 3).\n\n* `nclusters(K)` : Number of clusters required at end.\n\n* `criteria` : It is the iteration termination criteria. When this criteria is satisfied, algorithm iteration stops. Actually, it should be a tuple of 3 parameters. They are `( type, max_iter, epsilon )`:\n\nType of termination criteria. It has 3 flags as below:\n\n1. `cv.TERM_CRITERIA_EPS` - stop the algorithm iteration if specified accuracy, epsilon, is reached.\n2. `cv.TERM_CRITERIA_MAX_ITER` - stop the algorithm after the specified number of iterations, max_iter.\n3. `cv.TERM_CRITERIA_EPS + cv.TERM_CRITERIA_MAX_ITER` - stop the iteration when any of the above condition is met.\n\n*max_iter - An integer specifying maximum number of iterations. epsilon - Required accuracy*\n\n* `attempts` : Flag to specify the number of times the algorithm is executed using different initial labellings. The algorithm returns the labels that yield the best compactness. This compactness is returned as output.\n\n* `flags` : This flag is used to specify how initial centers are taken. Normally two flags are used for this : cv.KMEANS_PP_CENTERS and cv.KMEANS_RANDOM_CENTERS.\n\n****\n\nOutput parameters :\n\n* `compactness` : It is the sum of squared distance from each point to their corresponding centers.\n* `labels` : This is the label array (same as 'code' in previous article) where each element marked '0','1'.....\n* `centers` : This is array of centers of clusters.","metadata":{}},{"cell_type":"code","source":"def segmentation(image, k, attempts) : \n    vectorized = np.float32(image.reshape((-1, 3)))\n    criteria = (cv2.TERM_CRITERIA_EPS + cv2.TERM_CRITERIA_MAX_ITER, 20, 1.0)\n    res , label , center = cv2.kmeans(vectorized, k, None, criteria, attempts, cv2.KMEANS_PP_CENTERS)\n    center = np.uint8(center)\n    res = center[label.flatten()]\n    segmented_image = res.reshape((image.shape))\n    return segmented_image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:51:20.323883Z","iopub.execute_input":"2025-10-02T06:51:20.324216Z","iopub.status.idle":"2025-10-02T06:51:20.329357Z","shell.execute_reply.started":"2025-10-02T06:51:20.324194Z","shell.execute_reply":"2025-10-02T06:51:20.328442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nplt.subplot(1,1,1)\nplt.imshow(hair_removed, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"hair remove Image\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:51:22.795207Z","iopub.execute_input":"2025-10-02T06:51:22.795833Z","iopub.status.idle":"2025-10-02T06:51:26.88474Z","shell.execute_reply.started":"2025-10-02T06:51:22.795813Z","shell.execute_reply":"2025-10-02T06:51:26.884036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (12, 8))\nsegmented_image = segmentation(hair_removed, 3, 10) # k = 3, attempt = 10\nplt.subplot(1,3,1)\nplt.imshow(segmented_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Segmented Img k = 3\")\n\nsegmented_image = segmentation(hair_removed, 4, 10) # k = 4, attempt = 10\nplt.subplot(1,3,2)\nplt.imshow(segmented_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Segmented Img k = 4\")\n\nsegmented_image = segmentation(hair_removed, 5, 10) # k = 5, attempt = 10\nplt.subplot(1,3,3)\nplt.imshow(segmented_image, cmap = \"gray\")\nplt.grid(False)\nplt.title(\"Segmented Img k = 5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T07:47:02.839578Z","iopub.execute_input":"2025-10-02T07:47:02.840302Z","iopub.status.idle":"2025-10-02T07:47:48.547652Z","shell.execute_reply.started":"2025-10-02T07:47:02.840276Z","shell.execute_reply":"2025-10-02T07:47:48.547004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SHAPE = (224, 224, 3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:55:08.332537Z","iopub.execute_input":"2025-10-02T06:55:08.332841Z","iopub.status.idle":"2025-10-02T06:55:08.336489Z","shell.execute_reply.started":"2025-10-02T06:55:08.332819Z","shell.execute_reply":"2025-10-02T06:55:08.335764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def resize(image, shape) : \n    image = cv2.resize(image, (shape[0], shape[1]))\n    return image   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:55:11.768605Z","iopub.execute_input":"2025-10-02T06:55:11.76933Z","iopub.status.idle":"2025-10-02T06:55:11.772987Z","shell.execute_reply.started":"2025-10-02T06:55:11.769302Z","shell.execute_reply":"2025-10-02T06:55:11.772247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dùng cột benign_malignant làm nhãn\ntrain[\"label\"] = train[\"benign_malignant\"].astype(str)\n\nprint(train[\"label\"].value_counts())  # kiểm tra số lượng\n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\naugment_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=20,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    vertical_flip=True,\n    validation_split=0.2\n)\n\n# Thêm đuôi .jpg vào image_name\ntrain[\"image_name\"] = train[\"image_name\"].astype(str) + \".jpg\"\n\nprint(train[\"image_name\"].head())  # kiểm tra lại tên file\n\n\n# ✅ Train generator\ntrain_generator = augment_datagen.flow_from_dataframe(\n    dataframe=train,\n    directory=train_dir,\n    x_col=\"image_name\",\n    y_col=\"label\",          # <--- dùng label mới\n    target_size=(224,224),\n    class_mode=\"binary\",\n    subset=\"training\",\n    batch_size=32,\n    shuffle=True\n)\n\n# ✅ Validation generator\nval_generator = augment_datagen.flow_from_dataframe(\n    dataframe=train,\n    directory=train_dir,\n    x_col=\"image_name\",\n    y_col=\"label\",          # <--- dùng label mới\n    target_size=(224,224),\n    class_mode=\"binary\",\n    subset=\"validation\",\n    batch_size=32,\n    shuffle=False\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:55:14.03001Z","iopub.execute_input":"2025-10-02T06:55:14.030659Z","iopub.status.idle":"2025-10-02T06:57:44.81278Z","shell.execute_reply.started":"2025-10-02T06:55:14.030636Z","shell.execute_reply":"2025-10-02T06:57:44.812165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Đếm trước khi cân bằng\ncounts_before = train[\"label\"].value_counts()\n\n# ✅ Oversampling malignant cho cân bằng\nbenign_df = train[train[\"label\"] == \"benign\"]\nmalignant_df = train[train[\"label\"] == \"malignant\"]\n\n# Nhân malignant lên cho gần bằng benign\nmalignant_oversampled = malignant_df.sample(len(benign_df), replace=True, random_state=42)\n\n# Ghép lại\nbalanced_train = pd.concat([benign_df, malignant_oversampled], axis=0).reset_index(drop=True)\n\n# Đếm sau khi cân bằng\ncounts_after = balanced_train[\"label\"].value_counts()\n\n# ✅ Vẽ biểu đồ so sánh\nfig, axes = plt.subplots(1, 2, figsize=(12, 5))\n\n# Trước khi cân bằng\naxes[0].bar(counts_before.index, counts_before.values, color=[\"skyblue\", \"salmon\"])\naxes[0].set_title(\"Before balanced\")\naxes[0].set_ylabel(\"Number of the img\")\n\n# Sau khi cân bằng\naxes[1].bar(counts_after.index, counts_after.values, color=[\"skyblue\", \"salmon\"])\naxes[1].set_title(\"After balanced\")\n\nplt.suptitle(\"Comparing to Benign vs Malignant\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T06:57:44.81406Z","iopub.execute_input":"2025-10-02T06:57:44.814676Z","iopub.status.idle":"2025-10-02T06:57:45.084733Z","shell.execute_reply.started":"2025-10-02T06:57:44.814656Z","shell.execute_reply":"2025-10-02T06:57:45.083998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}