{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":31154,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:48:49.492925Z","iopub.execute_input":"2025-10-10T06:48:49.493549Z","iopub.status.idle":"2025-10-10T06:49:17.407211Z","shell.execute_reply.started":"2025-10-10T06:48:49.493526Z","shell.execute_reply":"2025-10-10T06:49:17.406425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\nfrom os import listdir\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n#plotly\n!pip install chart_studio\nimport plotly.express as px\nimport chart_studio.plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nimport cufflinks\ncufflinks.go_offline()\ncufflinks.set_config_file(world_readable=True, theme='pearl')\n\nimport seaborn as sns\nsns.set(style=\"whitegrid\")\n\n\n#pydicom\nimport pydicom\n\n# Suppress warnings \nimport warnings\nwarnings.filterwarnings('ignore')\n\n\n# Settings for pretty nice plots\nplt.style.use('fivethirtyeight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:49:34.150622Z","iopub.execute_input":"2025-10-10T06:49:34.151243Z","iopub.status.idle":"2025-10-10T06:49:42.799573Z","shell.execute_reply.started":"2025-10-10T06:49:34.151222Z","shell.execute_reply":"2025-10-10T06:49:42.798905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.listdir(\"../input/siim-isic-melanoma-classification\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:49:54.204562Z","iopub.execute_input":"2025-10-10T06:49:54.205270Z","iopub.status.idle":"2025-10-10T06:49:54.209346Z","shell.execute_reply.started":"2025-10-10T06:49:54.205249Z","shell.execute_reply":"2025-10-10T06:49:54.208695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMAGE_PATH = \"../input/siim-isic-melanoma-classification/\"\n\ntrain_df = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\n\n\n#Training data\nprint('Training data shape: ', train_df.shape)\ntrain_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:50:03.701102Z","iopub.execute_input":"2025-10-10T06:50:03.701724Z","iopub.status.idle":"2025-10-10T06:50:03.823269Z","shell.execute_reply.started":"2025-10-10T06:50:03.701701Z","shell.execute_reply":"2025-10-10T06:50:03.822513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.groupby(['benign_malignant']).count()['sex'].to_frame()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:50:14.889216Z","iopub.execute_input":"2025-10-10T06:50:14.889902Z","iopub.status.idle":"2025-10-10T06:50:14.916340Z","shell.execute_reply.started":"2025-10-10T06:50:14.889878Z","shell.execute_reply":"2025-10-10T06:50:14.915743Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Data Exploration","metadata":{}},{"cell_type":"markdown","source":"### Missing Values","metadata":{}},{"cell_type":"code","source":"print('Train Set')\nprint(train_df.info())\nprint('-------------')\nprint('Test Set')\nprint(test_df.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:50:24.439257Z","iopub.execute_input":"2025-10-10T06:50:24.439672Z","iopub.status.idle":"2025-10-10T06:50:24.469156Z","shell.execute_reply.started":"2025-10-10T06:50:24.439647Z","shell.execute_reply":"2025-10-10T06:50:24.468418Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Total Number of images","metadata":{}},{"cell_type":"code","source":"print(\"Total images in Train set: \",train_df['image_name'].count())\nprint(\"Total images in Test set: \",test_df['image_name'].count())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:50:38.309625Z","iopub.execute_input":"2025-10-10T06:50:38.309901Z","iopub.status.idle":"2025-10-10T06:50:38.316821Z","shell.execute_reply.started":"2025-10-10T06:50:38.309881Z","shell.execute_reply":"2025-10-10T06:50:38.316072Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Unique Ids","metadata":{}},{"cell_type":"code","source":"print(f\"The total patient ids are {train_df['patient_id'].count()}, from those the unique ids are {train_df['patient_id'].value_counts().shape[0]} \")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:52:08.425348Z","iopub.execute_input":"2025-10-10T06:52:08.426091Z","iopub.status.idle":"2025-10-10T06:52:08.435235Z","shell.execute_reply.started":"2025-10-10T06:52:08.426061Z","shell.execute_reply":"2025-10-10T06:52:08.434450Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns = train_df.keys()\ncolumns = list(columns)\nprint(columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:52:15.274560Z","iopub.execute_input":"2025-10-10T06:52:15.275055Z","iopub.status.idle":"2025-10-10T06:52:15.278936Z","shell.execute_reply.started":"2025-10-10T06:52:15.275035Z","shell.execute_reply":"2025-10-10T06:52:15.278079Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Exploring the Target Column","metadata":{}},{"cell_type":"code","source":"train_df['target'].value_counts()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:52:34.329449Z","iopub.execute_input":"2025-10-10T06:52:34.330155Z","iopub.status.idle":"2025-10-10T06:52:34.337338Z","shell.execute_reply.started":"2025-10-10T06:52:34.330135Z","shell.execute_reply":"2025-10-10T06:52:34.336514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['target'].value_counts(normalize=True).iplot(kind='bar',\n                                                      yTitle='Percentage', \n                                                      linecolor='black', \n                                                      opacity=0.7,\n                                                      color='red',\n                                                      theme='pearl',\n                                                      bargap=0.8,\n                                                      gridcolor='white',\n                                                     \n                                                      title='Distribution of the Target column in the training set')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:52:45.428980Z","iopub.execute_input":"2025-10-10T06:52:45.429557Z","iopub.status.idle":"2025-10-10T06:52:46.291164Z","shell.execute_reply.started":"2025-10-10T06:52:45.429534Z","shell.execute_reply":"2025-10-10T06:52:46.290513Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Gender wise distribution","metadata":{}},{"cell_type":"code","source":"train_df['sex'].value_counts(normalize=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:53:14.514400Z","iopub.execute_input":"2025-10-10T06:53:14.514658Z","iopub.status.idle":"2025-10-10T06:53:14.523525Z","shell.execute_reply.started":"2025-10-10T06:53:14.514641Z","shell.execute_reply":"2025-10-10T06:53:14.522763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['sex'].value_counts(normalize=True).iplot(kind='bar',\n                                                      yTitle='Percentage', \n                                                      linecolor='black', \n                                                      opacity=0.7,\n                                                      color='green',\n                                                      theme='pearl',\n                                                      bargap=0.8,\n                                                      gridcolor='white',\n                                                     \n                                                      title='Distribution of the Sex column in the training set')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:53:28.459681Z","iopub.execute_input":"2025-10-10T06:53:28.460356Z","iopub.status.idle":"2025-10-10T06:53:28.496636Z","shell.execute_reply.started":"2025-10-10T06:53:28.460332Z","shell.execute_reply":"2025-10-10T06:53:28.495903Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Gender V/s Target","metadata":{}},{"cell_type":"code","source":"z=train_df.groupby(['target','sex'])['benign_malignant'].count().to_frame().reset_index()\nz.style.background_gradient(cmap='Reds')  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:54:33.979486Z","iopub.execute_input":"2025-10-10T06:54:33.980304Z","iopub.status.idle":"2025-10-10T06:54:34.074803Z","shell.execute_reply.started":"2025-10-10T06:54:33.980274Z","shell.execute_reply":"2025-10-10T06:54:34.074166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.catplot(x='target',y='benign_malignant', hue='sex',data=z,kind='bar')\nplt.ylabel('Count')\nplt.xlabel('benign:0 vs malignant:1')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:54:42.240577Z","iopub.execute_input":"2025-10-10T06:54:42.241436Z","iopub.status.idle":"2025-10-10T06:54:42.682053Z","shell.execute_reply.started":"2025-10-10T06:54:42.241415Z","shell.execute_reply":"2025-10-10T06:54:42.681365Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Localisation of image site","metadata":{}},{"cell_type":"code","source":"train_df['anatom_site_general_challenge'].value_counts(normalize=True).sort_values()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:55:08.610385Z","iopub.execute_input":"2025-10-10T06:55:08.611027Z","iopub.status.idle":"2025-10-10T06:55:08.619397Z","shell.execute_reply.started":"2025-10-10T06:55:08.611006Z","shell.execute_reply":"2025-10-10T06:55:08.618816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['anatom_site_general_challenge'].value_counts(normalize=True).sort_values().iplot(kind='barh',\n                                                      xTitle='Percentage', \n                                                      linecolor='black', \n                                                      opacity=0.7,\n                                                      color='#FB8072',\n                                                      theme='pearl',\n                                                      bargap=0.2,\n                                                      gridcolor='white',\n                                                      title='Distribution of the imaged site in the training set')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:55:18.385894Z","iopub.execute_input":"2025-10-10T06:55:18.386156Z","iopub.status.idle":"2025-10-10T06:55:18.423367Z","shell.execute_reply.started":"2025-10-10T06:55:18.386137Z","shell.execute_reply":"2025-10-10T06:55:18.422750Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Location of imaged site w.r.t gender","metadata":{}},{"cell_type":"code","source":"z1=train_df.groupby(['sex','anatom_site_general_challenge'])['benign_malignant'].count().to_frame().reset_index()\nz1.style.background_gradient(cmap='Reds')\nsns.catplot(x='anatom_site_general_challenge',y='benign_malignant', hue='sex',data=z1,kind='bar')\nplt.gcf().set_size_inches(10,8)\nplt.xlabel('location of imaged site')\nplt.xticks(rotation=45,fontsize='10', horizontalalignment='right')\nplt.ylabel('count of melanoma cases')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:55:54.246894Z","iopub.execute_input":"2025-10-10T06:55:54.247161Z","iopub.status.idle":"2025-10-10T06:55:54.670930Z","shell.execute_reply.started":"2025-10-10T06:55:54.247141Z","shell.execute_reply":"2025-10-10T06:55:54.670107Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Age Distribution of patients","metadata":{}},{"cell_type":"code","source":"train_df['age_approx'].iplot(kind='hist',bins=30,color='orange',xTitle='Age distribution',yTitle='Count')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:56:15.940205Z","iopub.execute_input":"2025-10-10T06:56:15.940729Z","iopub.status.idle":"2025-10-10T06:56:16.125052Z","shell.execute_reply.started":"2025-10-10T06:56:15.940706Z","shell.execute_reply":"2025-10-10T06:56:16.124316Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualising Age KDEs","metadata":{}},{"cell_type":"markdown","source":"### Distribution of Ages w.r.t Target¶","metadata":{}},{"cell_type":"code","source":"# KDE plot of age that were diagnosed as benign\nsns.kdeplot(train_df.loc[train_df['target'] == 0, 'age_approx'], label = 'Benign',shade=True)\n\n# KDE plot of age that were diagnosed as malignant\nsns.kdeplot(train_df.loc[train_df['target'] == 1, 'age_approx'], label = 'Malignant',shade=True)\n\n# Labeling of plot\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages');","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:57:09.763860Z","iopub.execute_input":"2025-10-10T06:57:09.764587Z","iopub.status.idle":"2025-10-10T06:57:10.251295Z","shell.execute_reply.started":"2025-10-10T06:57:09.764562Z","shell.execute_reply":"2025-10-10T06:57:10.250592Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Distribution of Ages w.r.t gender","metadata":{}},{"cell_type":"code","source":"# KDE plot of age that were diagnosed as benign\nsns.kdeplot(train_df.loc[train_df['sex'] == 'male', 'age_approx'], label = 'Male',shade=True)\n\n# KDE plot of age that were diagnosed as malignant\nsns.kdeplot(train_df.loc[train_df['sex'] == 'female', 'age_approx'], label = 'Female',shade=True)\n\n# Labeling of plot\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages');","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:57:32.983801Z","iopub.execute_input":"2025-10-10T06:57:32.984344Z","iopub.status.idle":"2025-10-10T06:57:33.453207Z","shell.execute_reply.started":"2025-10-10T06:57:32.984324Z","shell.execute_reply":"2025-10-10T06:57:33.452329Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Distribution of Diagnosis","metadata":{}},{"cell_type":"code","source":"train_df['diagnosis'].value_counts()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:57:51.017987Z","iopub.execute_input":"2025-10-10T06:57:51.018675Z","iopub.status.idle":"2025-10-10T06:57:51.026190Z","shell.execute_reply.started":"2025-10-10T06:57:51.018651Z","shell.execute_reply":"2025-10-10T06:57:51.025333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['diagnosis'].value_counts(normalize=True).sort_values().iplot(kind='barh',\n                                                      xTitle='Percentage', \n                                                      linecolor='black', \n                                                      opacity=0.7,\n                                                      color='blue',\n                                                      theme='pearl',\n                                                      bargap=0.2,\n                                                      gridcolor='white',\n                                                      title='Distribution in the training set')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:58:00.487531Z","iopub.execute_input":"2025-10-10T06:58:00.488103Z","iopub.status.idle":"2025-10-10T06:58:00.523093Z","shell.execute_reply.started":"2025-10-10T06:58:00.488081Z","shell.execute_reply":"2025-10-10T06:58:00.522317Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Patient Overlap","metadata":{}},{"cell_type":"code","source":"# Extract patient id's for the training set\nids_train = train_df.patient_id.values\n# Extract patient id's for the validation set\nids_test = test_df.patient_id.values\n\n# Create a \"set\" datastructure of the training set id's to identify unique id's\nids_train_set = set(ids_train)\nprint(f'There are {len(ids_train_set)} unique Patient IDs in the training set')\n# Create a \"set\" datastructure of the validation set id's to identify unique id's\nids_test_set = set(ids_test)\nprint(f'There are {len(ids_test_set)} unique Patient IDs in the training set')\n\n# Identify patient overlap by looking at the intersection between the sets\npatient_overlap = list(ids_train_set.intersection(ids_test_set))\nn_overlap = len(patient_overlap)\nprint(f'There are {n_overlap} Patient IDs in both the training and test sets')\nprint('')\nprint(f'These patients are in both the training and test datasets:')\nprint(f'{patient_overlap}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T06:59:42.140951Z","iopub.execute_input":"2025-10-10T06:59:42.141555Z","iopub.status.idle":"2025-10-10T06:59:42.148313Z","shell.execute_reply.started":"2025-10-10T06:59:42.141528Z","shell.execute_reply":"2025-10-10T06:59:42.147505Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Visualising Images : JPEG","metadata":{}},{"cell_type":"markdown","source":"### Visualizing a random selection of images","metadata":{}},{"cell_type":"code","source":"images = train_df['image_name'].values\n\n# Extract 9 random images from it\nrandom_images = [np.random.choice(images+'.jpg') for i in range(9)]\n\n# Location of the image dir\nimg_dir = IMAGE_PATH+'/jpeg/train'\n\nprint('Display Random Images')\n\n# Adjust the size of your images\nplt.figure(figsize=(10,8))\n\n# Iterate and plot random images\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img = plt.imread(os.path.join(img_dir, random_images[i]))\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    \n# Adjust subplot parameters to give specified padding\nplt.tight_layout() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:01:10.439327Z","iopub.execute_input":"2025-10-10T07:01:10.439834Z","iopub.status.idle":"2025-10-10T07:01:27.340446Z","shell.execute_reply.started":"2025-10-10T07:01:10.439812Z","shell.execute_reply":"2025-10-10T07:01:27.339629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Visualizing Images with benign lesions","metadata":{}},{"cell_type":"code","source":"benign = train_df[train_df['benign_malignant']=='benign']\nmalignant = train_df[train_df['benign_malignant']=='malignant']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:01:57.711036Z","iopub.execute_input":"2025-10-10T07:01:57.711555Z","iopub.status.idle":"2025-10-10T07:01:57.724291Z","shell.execute_reply.started":"2025-10-10T07:01:57.711528Z","shell.execute_reply":"2025-10-10T07:01:57.723762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images = benign['image_name'].values\n\n# Extract 9 random images from it\nrandom_images = [np.random.choice(images+'.jpg') for i in range(9)]\n\n# Location of the image dir\nimg_dir = IMAGE_PATH+'/jpeg/train'\n\nprint('Display benign Images')\n\n# Adjust the size of your images\nplt.figure(figsize=(10,8))\n\n# Iterate and plot random images\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img = plt.imread(os.path.join(img_dir, random_images[i]))\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    \n# Adjust subplot parameters to give specified padding\nplt.tight_layout()   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:02:08.037298Z","iopub.execute_input":"2025-10-10T07:02:08.037872Z","iopub.status.idle":"2025-10-10T07:02:23.718822Z","shell.execute_reply.started":"2025-10-10T07:02:08.037848Z","shell.execute_reply":"2025-10-10T07:02:23.716562Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Visualizing Images with Malignant lesions","metadata":{}},{"cell_type":"code","source":"images = malignant['image_name'].values\n\n# Extract 9 random images from it\nrandom_images = [np.random.choice(images+'.jpg') for i in range(9)]\n\n# Location of the image dir\nimg_dir = IMAGE_PATH+'/jpeg/train'\n\nprint('Display malignant Images')\n\n# Adjust the size of your images\nplt.figure(figsize=(10,8))\n\n# Iterate and plot random images\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img = plt.imread(os.path.join(img_dir, random_images[i]))\n    plt.imshow(img, cmap='gray')\n    plt.axis('off')\n    \n# Adjust subplot parameters to give specified padding\nplt.tight_layout()   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:02:33.414242Z","iopub.execute_input":"2025-10-10T07:02:33.414532Z","iopub.status.idle":"2025-10-10T07:02:43.178430Z","shell.execute_reply.started":"2025-10-10T07:02:33.414505Z","shell.execute_reply":"2025-10-10T07:02:43.177800Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Histograms","metadata":{}},{"cell_type":"markdown","source":"#### Benign category","metadata":{}},{"cell_type":"code","source":"f = plt.figure(figsize=(16,8))\nf.add_subplot(1,2, 1)\n\nsample_img = benign['image_name'][0]+'.jpg'\nraw_image = plt.imread(os.path.join(img_dir, sample_img))\nplt.imshow(raw_image, cmap='gray')\nplt.colorbar()\nplt.title('Benign Image')\nprint(f\"Image dimensions:  {raw_image.shape[0],raw_image.shape[1]}\")\nprint(f\"Maximum pixel value : {raw_image.max():.1f} ; Minimum pixel value:{raw_image.min():.1f}\")\nprint(f\"Mean value of the pixels : {raw_image.mean():.1f} ; Standard deviation : {raw_image.std():.1f}\")\n\nf.add_subplot(1,2, 2)\n\n#_ = plt.hist(raw_image.ravel(),bins = 256, color = 'orange',)\n_ = plt.hist(raw_image[:, :, 0].ravel(), bins = 256, color = 'red', alpha = 0.5)\n_ = plt.hist(raw_image[:, :, 1].ravel(), bins = 256, color = 'Green', alpha = 0.5)\n_ = plt.hist(raw_image[:, :, 2].ravel(), bins = 256, color = 'Blue', alpha = 0.5)\n_ = plt.xlabel('Intensity Value')\n_ = plt.ylabel('Count')\n_ = plt.legend(['Red_Channel', 'Green_Channel', 'Blue_Channel'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:03:32.974500Z","iopub.execute_input":"2025-10-10T07:03:32.975078Z","iopub.status.idle":"2025-10-10T07:03:38.844781Z","shell.execute_reply.started":"2025-10-10T07:03:32.975059Z","shell.execute_reply":"2025-10-10T07:03:38.843955Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Malignant category","metadata":{}},{"cell_type":"code","source":"f = plt.figure(figsize=(16,8))\nf.add_subplot(1,2, 1)\n\nsample_img = malignant['image_name'][235]+'.jpg'\nraw_image = plt.imread(os.path.join(img_dir, sample_img))\nplt.imshow(raw_image, cmap='gray')\nplt.colorbar()\nplt.title('Malignant Image')\nprint(f\"Image dimensions:  {raw_image.shape[0],raw_image.shape[1]}\")\nprint(f\"Maximum pixel value : {raw_image.max():.1f} ; Minimum pixel value:{raw_image.min():.1f}\")\nprint(f\"Mean value of the pixels : {raw_image.mean():.1f} ; Standard deviation : {raw_image.std():.1f}\")\n\nf.add_subplot(1,2, 2)\n\n#_ = plt.hist(raw_image.ravel(),bins = 256, color = 'orange',)\n_ = plt.hist(raw_image[:, :, 0].ravel(), bins = 256, color = 'red', alpha = 0.5)\n_ = plt.hist(raw_image[:, :, 1].ravel(), bins = 256, color = 'Green', alpha = 0.5)\n_ = plt.hist(raw_image[:, :, 2].ravel(), bins = 256, color = 'Blue', alpha = 0.5)\n_ = plt.xlabel('Intensity Value')\n_ = plt.ylabel('Count')\n_ = plt.legend(['Red_Channel', 'Green_Channel', 'Blue_Channel'])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:04:04.688195Z","iopub.execute_input":"2025-10-10T07:04:04.688500Z","iopub.status.idle":"2025-10-10T07:04:07.491087Z","shell.execute_reply.started":"2025-10-10T07:04:04.688479Z","shell.execute_reply":"2025-10-10T07:04:07.490371Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5 Preprocessing DIOCOM files","metadata":{}},{"cell_type":"code","source":"print (pydicom.__version__)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:04:27.974969Z","iopub.execute_input":"2025-10-10T07:04:27.975588Z","iopub.status.idle":"2025-10-10T07:04:27.980037Z","shell.execute_reply.started":"2025-10-10T07:04:27.975566Z","shell.execute_reply":"2025-10-10T07:04:27.979137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# https://www.kaggle.com/schlerp/getting-to-know-dicom-and-the-data\ndef show_dcm_info(dataset):\n    print(\"Filename.........:\", file_path)\n    print(\"Storage type.....:\", dataset.SOPClassUID)\n    print()\n\n    pat_name = dataset.PatientName\n    display_name = pat_name.family_name + \", \" + pat_name.given_name\n    print(\"Patient's name......:\", display_name)\n    print(\"Patient id..........:\", dataset.PatientID)\n    print(\"Patient's Age.......:\", dataset.PatientAge)\n    print(\"Patient's Sex.......:\", dataset.PatientSex)\n    print(\"Modality............:\", dataset.Modality)\n    print(\"Body Part Examined..:\", dataset.BodyPartExamined)\n   \n    \n    \n    if 'PixelData' in dataset:\n        rows = int(dataset.Rows)\n        cols = int(dataset.Columns)\n        print(\"Image size.......: {rows:d} x {cols:d}, {size:d} bytes\".format(\n            rows=rows, cols=cols, size=len(dataset.PixelData)))\n        if 'PixelSpacing' in dataset:\n            print(\"Pixel spacing....:\", dataset.PixelSpacing)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:04:42.837664Z","iopub.execute_input":"2025-10-10T07:04:42.837930Z","iopub.status.idle":"2025-10-10T07:04:42.843305Z","shell.execute_reply.started":"2025-10-10T07:04:42.837911Z","shell.execute_reply":"2025-10-10T07:04:42.842509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_pixel_array(dataset, figsize=(5,5)):\n    plt.figure(figsize=figsize)\n    plt.grid(False)\n    plt.imshow(dataset.pixel_array)\n    plt.show()\n    \ni = 1\nnum_to_plot = 5\nfor file_name in os.listdir('../input/siim-isic-melanoma-classification/train/'):\n        file_path = os.path.join('../input/siim-isic-melanoma-classification/train/',file_name)\n        dataset = pydicom.dcmread(file_path)\n        show_dcm_info(dataset)\n        plot_pixel_array(dataset)\n    \n        if i >= num_to_plot:\n            break\n    \n        i += 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:04:55.629222Z","iopub.execute_input":"2025-10-10T07:04:55.629916Z","iopub.status.idle":"2025-10-10T07:05:04.872589Z","shell.execute_reply.started":"2025-10-10T07:04:55.629892Z","shell.execute_reply":"2025-10-10T07:05:04.871668Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Extracting DIOCOM files information in a dataframe","metadata":{}},{"cell_type":"code","source":"# source: https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/154658\nfolder='train'\nPATH='../input/siim-isic-melanoma-classification/'\n\ndef extract_DICOM_attributes(folder):\n    images = list(os.listdir(os.path.join(PATH, folder)))\n    df = pd.DataFrame()\n    for image in tqdm(images):\n        image_name = image.split(\".\")[0]\n        dicom_file_path = os.path.join(PATH,folder,image)\n        dicom_file_dataset = pydicom.dcmread(dicom_file_path)\n        study_date = dicom_file_dataset.StudyDate\n        modality = dicom_file_dataset.Modality\n        age = dicom_file_dataset.PatientAge\n        sex = dicom_file_dataset.PatientSex\n        body_part_examined = dicom_file_dataset.BodyPartExamined\n        patient_orientation = dicom_file_dataset.PatientOrientation\n        photometric_interpretation = dicom_file_dataset.PhotometricInterpretation\n        rows = dicom_file_dataset.Rows\n        columns = dicom_file_dataset.Columns\n\n        df = pd.concat([\n            df,pd.DataFrame(\n                {\n            'image_name': image_name, \n            'dcm_modality': modality,\n            'dcm_study_date':study_date, \n            'dcm_age': age, \n            'dcm_sex': sex,            \n            'dcm_body_part_examined': body_part_examined,\n            'dcm_patient_orientation': patient_orientation,            \n            'dcm_photometric_interpretation': photometric_interpretation,           \n            'dcm_rows': rows, \n            'dcm_columns': columns\n        }, \n                                    \n                index=[0],\n                )\n                       ])\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:05:36.676053Z","iopub.execute_input":"2025-10-10T07:05:36.676330Z","iopub.status.idle":"2025-10-10T07:05:36.682722Z","shell.execute_reply.started":"2025-10-10T07:05:36.676311Z","shell.execute_reply":"2025-10-10T07:05:36.681881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"extract_DICOM_attributes('train')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-10T07:05:46.317701Z","iopub.execute_input":"2025-10-10T07:05:46.317962Z","iopub.status.idle":"2025-10-10T07:20:32.066740Z","shell.execute_reply.started":"2025-10-10T07:05:46.317944Z","shell.execute_reply":"2025-10-10T07:20:32.066079Z"}},"outputs":[],"execution_count":null}]}