{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":29926,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#NumPy와 Pandas 라이브러리 불러오기\n\nimport numpy as np \nimport pandas as pd \n\n#/kaggle/input' 디렉토리 내의 파일과 하위 디렉토리를 불러오기(이 코드 작성자 기준에서의 디렉토리 내 파일 경로)\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename)) #디렉토리 경로와 파일 이름을 결합하여 전체 파일 경로를 생성","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-12-04T01:15:35.872718Z","iopub.execute_input":"2023-12-04T01:15:35.873140Z","iopub.status.idle":"2023-12-04T01:16:39.268910Z","shell.execute_reply.started":"2023-12-04T01:15:35.873105Z","shell.execute_reply":"2023-12-04T01:16:39.267901Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<<과제 2>>\n\n                   232146 의예과 정선우\nTitle : Melanoma Classification : EDA starter\n\nhttps://www.kaggle.com/code/parulpandey/melanoma-classification-eda-starter\n\n<요약 및 결과>\n\n이미지의 병변이 악성인지, 양성인지 예측하는 모델을 만들기 위한 피부 병변 이미지를 가지고 흑색종을 이미지의 특성을 분하는 과정을 담고 있습니다.\n\n먼저, 데이터를 불러오고 각각의 데이터의 형식, 종류 등을 탐색하는 과정이 이루어진다. 이와 함께 Plotly를 사용하여 여러 데이터 간의 분포나 상관관계를 파악하는 과정도 동반된다.\n\n밀도 플롯을 사용하여 데이터의 중심, 밀도 분포 등을 파악 커널 밀도 추정 플롯은 단일 변수의 분포를 보여주며 이 과정에서 확률 밀도 함수를 추정하는 미모수적 방법인 커널 밀도 추정을 사용하기도 한다.\n\n다음으로는 DICOM파일을 다루는데, 필요한 라이브러리를 불러와 이미지의 픽셀 값을 분석한다. (malignant와 benign 각각의 이미지에서의 픽셀 값 정보를 추출) 마지막으로 DICOM을 사용하여 파일에 대한 정보를 가져와 각각의 속성을 추출하여 각각의 Malignant, benign이미지에 대한 특성을 이해한다.","metadata":{}},{"cell_type":"markdown","source":"\n# 1. Importing the necessary libraries\n\nIncase you fork the notebook, make sure to keep the Internet in `ON` mode.","metadata":{}},{"cell_type":"code","source":"#필요한 함수, 라이브러리, pyplot 모듈 불러오기\nfrom os import listdir  # listdir 함수는 지정된 디렉토리의 파일 및 디렉토리 목록을 반환\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n#plotly\n#chart_studio 설치, 필요한 모듈 가져오기 및 여러 기타 설정\n!pip install chart_studio\nimport plotly.express as px\nimport chart_studio.plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nimport cufflinks\ncufflinks.go_offline()\ncufflinks.set_config_file(world_readable=True, theme='pearl')    #기타 설정\n\nimport seaborn as sns\nsns.set(style=\"whitegrid\")   #시각화 스타일 설정\n\n\n#pydicom 가져오기\nimport pydicom\n\n# Suppress warnings (경고 메세지 무시 설정)\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\n# Settings for pretty nice plots (그림 스타일 설정)\nplt.style.use('fivethirtyeight')\nplt.show()\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:39.271746Z","iopub.execute_input":"2023-12-04T01:16:39.272086Z","iopub.status.idle":"2023-12-04T01:16:47.443807Z","shell.execute_reply.started":"2023-12-04T01:16:39.272053Z","shell.execute_reply":"2023-12-04T01:16:47.442692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Reading the Image datasets","metadata":{}},{"cell_type":"code","source":"# List files available (디렉토리에 포함된 파일과 하위 목록을 출력)\nprint(os.listdir(\"../input/siim-isic-melanoma-classification\"))","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:47.445787Z","iopub.execute_input":"2023-12-04T01:16:47.446145Z","iopub.status.idle":"2023-12-04T01:16:47.453152Z","shell.execute_reply.started":"2023-12-04T01:16:47.446109Z","shell.execute_reply":"2023-12-04T01:16:47.452109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 파일을 읽어와 Pandas의 DataFrame 형태로 저장\nIMAGE_PATH = \"../input/siim-isic-melanoma-classification/\"\n\ntrain_df = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\n\n\n#데이터의 구조와 특성을 간략히 파악하기 위한 선제 작업\nprint('Training data shape: ', train_df.shape)\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:47.454858Z","iopub.execute_input":"2023-12-04T01:16:47.455200Z","iopub.status.idle":"2023-12-04T01:16:47.599938Z","shell.execute_reply.started":"2023-12-04T01:16:47.455161Z","shell.execute_reply":"2023-12-04T01:16:47.598804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#양성과 악성으로 나뉘는 것을 이용한 그룹화, 각각의 개수 나타내기\ntrain_df.groupby(['benign_malignant']).count()['sex'].to_frame()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:47.603995Z","iopub.execute_input":"2023-12-04T01:16:47.604493Z","iopub.status.idle":"2023-12-04T01:16:47.645459Z","shell.execute_reply.started":"2023-12-04T01:16:47.604457Z","shell.execute_reply":"2023-12-04T01:16:47.644397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Data Exploration\n\n## Missing Values","metadata":{}},{"cell_type":"code","source":"# 정보 유형을 확인 Null values and Data types\nprint('Train Set')\nprint(train_df.info())\nprint('-------------')\nprint('Test Set')\nprint(test_df.info())","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:47.649098Z","iopub.execute_input":"2023-12-04T01:16:47.649504Z","iopub.status.idle":"2023-12-04T01:16:47.699089Z","shell.execute_reply.started":"2023-12-04T01:16:47.649454Z","shell.execute_reply":"2023-12-04T01:16:47.697942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are some missing values in some of the columns. We shall deal with them later.","metadata":{}},{"cell_type":"markdown","source":"## Total Number of images","metadata":{}},{"cell_type":"code","source":"# train(학습) data set, test(훈련 평가, 검증) data set에 대한 각각의 이미지 개수\nprint(\"Total images in Train set: \",train_df['image_name'].count())\nprint(\"Total images in Test set: \",test_df['image_name'].count())","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:47.700988Z","iopub.execute_input":"2023-12-04T01:16:47.701526Z","iopub.status.idle":"2023-12-04T01:16:47.715699Z","shell.execute_reply.started":"2023-12-04T01:16:47.701472Z","shell.execute_reply":"2023-12-04T01:16:47.714100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Unique IDs ","metadata":{}},{"cell_type":"code","source":"# 'patient_id' , unique ids 의 총 데이터 수\nprint(f\"The total patient ids are {train_df['patient_id'].count()}, from those the unique ids are {train_df['patient_id'].value_counts().shape[0]} \")","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:47.717587Z","iopub.execute_input":"2023-12-04T01:16:47.718099Z","iopub.status.idle":"2023-12-04T01:16:47.742523Z","shell.execute_reply.started":"2023-12-04T01:16:47.718049Z","shell.execute_reply":"2023-12-04T01:16:47.741385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The number of unique patients is less than the total number of patients. This means that, patients have multiple records.","metadata":{}},{"cell_type":"code","source":"# 데이터프레임의 열 레이블을 가져와 리스트 변환 및 출력\ncolumns = train_df.keys()\ncolumns = list(columns)\nprint(columns)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:47.744137Z","iopub.execute_input":"2023-12-04T01:16:47.744519Z","iopub.status.idle":"2023-12-04T01:16:47.750577Z","shell.execute_reply.started":"2023-12-04T01:16:47.744477Z","shell.execute_reply":"2023-12-04T01:16:47.749558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Exploring the Target column","metadata":{}},{"cell_type":"code","source":"# target 열의 값들의 개수\ntrain_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:47.752280Z","iopub.execute_input":"2023-12-04T01:16:47.752944Z","iopub.status.idle":"2023-12-04T01:16:47.765039Z","shell.execute_reply.started":"2023-12-04T01:16:47.752903Z","shell.execute_reply":"2023-12-04T01:16:47.763684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotly를 사용하여 'target' 분포를 시각화\ntrain_df['target'].value_counts(normalize=True).iplot(kind='bar',\n                                                      yTitle='Percentage',\n                                                      linecolor='black',\n                                                      opacity=0.7,\n                                                      color='red',\n                                                      theme='pearl',\n                                                      bargap=0.8,\n                                                      gridcolor='white',\n\n                                                      title='Distribution of the Target column in the training set')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:47.767095Z","iopub.execute_input":"2023-12-04T01:16:47.767503Z","iopub.status.idle":"2023-12-04T01:16:47.844236Z","shell.execute_reply.started":"2023-12-04T01:16:47.767448Z","shell.execute_reply":"2023-12-04T01:16:47.843290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Gender wise distribution\n","metadata":{}},{"cell_type":"code","source":"# sex 열의 값의 비율 출력\ntrain_df['sex'].value_counts(normalize=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:47.845683Z","iopub.execute_input":"2023-12-04T01:16:47.846227Z","iopub.status.idle":"2023-12-04T01:16:47.865815Z","shell.execute_reply.started":"2023-12-04T01:16:47.846183Z","shell.execute_reply":"2023-12-04T01:16:47.864704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotly를 사용하여 'sex' 분포를 시각화\ntrain_df['sex'].value_counts(normalize=True).iplot(kind='bar',\n                                                      yTitle='Percentage',\n                                                      linecolor='black',\n                                                      opacity=0.7,\n                                                      color='green',\n                                                      theme='pearl',\n                                                      bargap=0.8,\n                                                      gridcolor='white',\n\n                                                      title='Distribution of the Sex column in the training set')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:47.867518Z","iopub.execute_input":"2023-12-04T01:16:47.867893Z","iopub.status.idle":"2023-12-04T01:16:47.944662Z","shell.execute_reply.started":"2023-12-04T01:16:47.867858Z","shell.execute_reply":"2023-12-04T01:16:47.943472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Gender vs Target","metadata":{}},{"cell_type":"code","source":"#'target', 'sex' 열을 기준으로 그룹화\n#'benign_malignant' 개수를 데이터프레임화\n# 히트맵 스타일로 시각화\nz=train_df.groupby(['target','sex'])['benign_malignant'].count().to_frame().reset_index()\nz.style.background_gradient(cmap='Reds')","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-12-04T01:16:47.946611Z","iopub.execute_input":"2023-12-04T01:16:47.947341Z","iopub.status.idle":"2023-12-04T01:16:47.994689Z","shell.execute_reply.started":"2023-12-04T01:16:47.947289Z","shell.execute_reply":"2023-12-04T01:16:47.993304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##target, sex에 따른 benign_malignant 분포를 바 형식으로 시각화\nsns.catplot(x='target',y='benign_malignant', hue='sex',data=z,kind='bar')\n#'target'을 x축으로, 'benign_malignant'을 y축으로 하고, 'sex'에 따라 색상을 구분\nplt.ylabel('Count')\nplt.xlabel('benign:0 vs malignant:1')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:47.996948Z","iopub.execute_input":"2023-12-04T01:16:47.997357Z","iopub.status.idle":"2023-12-04T01:16:48.395009Z","shell.execute_reply.started":"2023-12-04T01:16:47.997314Z","shell.execute_reply":"2023-12-04T01:16:48.394147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Location of imaged site","metadata":{}},{"cell_type":"code","source":"# anatom_site_general_challenge 열의 각각의 값들의 상대적인 비율을 출력, 정렬\ntrain_df['anatom_site_general_challenge'].value_counts(normalize=True).sort_values()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:48.396530Z","iopub.execute_input":"2023-12-04T01:16:48.397077Z","iopub.status.idle":"2023-12-04T01:16:48.416107Z","shell.execute_reply.started":"2023-12-04T01:16:48.397040Z","shell.execute_reply":"2023-12-04T01:16:48.414968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotly를 사용하여 'anatom_site_general_challenge' 분포를 시각화\ntrain_df['anatom_site_general_challenge'].value_counts(normalize=True).sort_values().iplot(kind='barh',\n                                                      xTitle='Percentage',\n                                                      linecolor='black',\n                                                      opacity=0.7,\n                                                      color='#FB8072',\n                                                      theme='pearl',\n                                                      bargap=0.2,\n                                                      gridcolor='white',\n                                                      title='Distribution of the imaged site in the training set')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:48.417840Z","iopub.execute_input":"2023-12-04T01:16:48.418225Z","iopub.status.idle":"2023-12-04T01:16:48.497045Z","shell.execute_reply.started":"2023-12-04T01:16:48.418189Z","shell.execute_reply":"2023-12-04T01:16:48.495945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Location of imaged site w.r.t gender","metadata":{}},{"cell_type":"code","source":"# sex와 anatom_site_general_challenge에 따른 benign_malignant 열의 분포를 바 차트로 시각화\nz1=train_df.groupby(['sex','anatom_site_general_challenge'])['benign_malignant'].count().to_frame().reset_index()\nz1.style.background_gradient(cmap='Reds')\nsns.catplot(x='anatom_site_general_challenge',y='benign_malignant', hue='sex',data=z1,kind='bar')\nplt.gcf().set_size_inches(10,8)\nplt.xlabel('location of imaged site')\nplt.xticks(rotation=45,fontsize='10', horizontalalignment='right')\nplt.ylabel('count of melanoma cases')\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:48.498589Z","iopub.execute_input":"2023-12-04T01:16:48.498927Z","iopub.status.idle":"2023-12-04T01:16:49.017858Z","shell.execute_reply.started":"2023-12-04T01:16:48.498894Z","shell.execute_reply":"2023-12-04T01:16:49.016393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Age Distribution of patients","metadata":{}},{"cell_type":"code","source":"#age_approx 열에 대한 히스토그램\ntrain_df['age_approx'].iplot(kind='hist',bins=30,color='orange',xTitle='Age distribution',yTitle='Count')  #막대수 30, 색상 : orange, x축 y축 제목 설정","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:49.019975Z","iopub.execute_input":"2023-12-04T01:16:49.020459Z","iopub.status.idle":"2023-12-04T01:16:49.950239Z","shell.execute_reply.started":"2023-12-04T01:16:49.020391Z","shell.execute_reply":"2023-12-04T01:16:49.949248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualising Age KDEs\nSummarizing the data with Density plots to see where the mass of the data is located. [A kernel density estimate plot](https://chemicalstatistician.wordpress.com/2013/06/09/exploratory-data-analysis-kernel-density-estimation-in-r-on-ozone-pollution-data-in-new-york-and-ozonopolis/) shows the distribution of a single variable and can be thought of as a smoothed histogram (it is created by computing a kernel, usually a Gaussian, at each data point and then averaging all the individual kernels to develop a single smooth curve). We will use the seaborn kdeplot for this graph.\n\n### Distribution of Ages w.r.t Target","metadata":{}},{"cell_type":"code","source":"# KDE plot of age that were diagnosed as benign\n#target== 0 (Benign인 경우)의 age_approx 열에 대한 커널 밀도 추정 플롯을 생성\nsns.kdeplot(train_df.loc[train_df['target'] == 0, 'age_approx'], label = 'Benign',shade=True)\n\n# KDE plot of age that were diagnosed as malignant\n#target== 1 (Malignant인 경우)의 age_approx 열에 대한 커널 밀도 추정 플롯을 생성\nsns.kdeplot(train_df.loc[train_df['target'] == 1, 'age_approx'], label = 'Malignant',shade=True)\n\n# plot의 제목, x축, y축 제목 설정\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages');","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:49.951688Z","iopub.execute_input":"2023-12-04T01:16:49.952475Z","iopub.status.idle":"2023-12-04T01:16:50.317121Z","shell.execute_reply.started":"2023-12-04T01:16:49.952411Z","shell.execute_reply":"2023-12-04T01:16:50.316203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n### Distribution of Ages w.r.t gender","metadata":{}},{"cell_type":"code","source":"# KDE plot of age that were diagnosed as benign\n#sex가 male인 경우, age_approx에 대한  커널 밀도 추정 플롯을 생성\nsns.kdeplot(train_df.loc[train_df['sex'] == 'male', 'age_approx'], label = 'Male',shade=True)\n\n# KDE plot of age that were diagnosed as malignant\n#sex가 female인 경우, age_approx에 대한  커널 밀도 추정 플롯을 생성\nsns.kdeplot(train_df.loc[train_df['sex'] == 'female', 'age_approx'], label = 'Female',shade=True)\n\n# plot의 제목, x축, y축 제목 설정\nplt.xlabel('Age (years)'); plt.ylabel('Density'); plt.title('Distribution of Ages');\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:50.318804Z","iopub.execute_input":"2023-12-04T01:16:50.319516Z","iopub.status.idle":"2023-12-04T01:16:50.684274Z","shell.execute_reply.started":"2023-12-04T01:16:50.319461Z","shell.execute_reply":"2023-12-04T01:16:50.682913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Distribution of Diagnosis","metadata":{}},{"cell_type":"code","source":"#diagnosis 각각의 값들의 빈도수\ntrain_df['diagnosis'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:16:50.686196Z","iopub.execute_input":"2023-12-04T01:16:50.686697Z","iopub.status.idle":"2023-12-04T01:16:50.705955Z","shell.execute_reply.started":"2023-12-04T01:16:50.686648Z","shell.execute_reply":"2023-12-04T01:16:50.704387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#diagnosis 값들의 상대적인 빈도를 수평 막대 그래프로 시각화\ntrain_df['diagnosis'].value_counts(normalize=True).sort_values().iplot(kind='barh',\n                                                      xTitle='Percentage',\n                                                      linecolor='black',\n                                                      opacity=0.7,\n                                                      color='blue',\n                                                      theme='pearl',\n                                                      bargap=0.2,\n                                                      gridcolor='white',\n                                                      title='Distribution in the training set')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:50.707357Z","iopub.execute_input":"2023-12-04T01:16:50.707688Z","iopub.status.idle":"2023-12-04T01:16:50.786801Z","shell.execute_reply.started":"2023-12-04T01:16:50.707656Z","shell.execute_reply":"2023-12-04T01:16:50.785692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Patient Overlap \nWe need to check that the the same patient lesion images shouldn't appear in both training and test set.","metadata":{}},{"cell_type":"code","source":"# Extract patient id's for the training set\n#train_df의 'patient_id' 열에 해당하는 값들이 넘파이 배열로 저장\nids_train = train_df.patient_id.values\n# Extract patient id's for the validation set\n#test_df의 'patient_id' 열에 해당하는 값들이 넘파이 배열로 저장\nids_test = test_df.patient_id.values\n\n# Create a \"set\" datastructure of the training set id's to identify unique id's\n#train_df에서의 고유한 환자 ID의 수(중복 제거) 및 출력\nids_train_set = set(ids_train)\nprint(f'There are {len(ids_train_set)} unique Patient IDs in the training set')\n\n# Create a \"set\" datastructure of the validation set id's to identify unique id's\n#test_df에서의 고유한 환자 ID의 수(중복 제거)\nids_test_set = set(ids_test)\nprint(f'There are {len(ids_test_set)} unique Patient IDs in the training set')\n\n\n#ids_train_set, ids_test_set 에서 겹치는 id 찾기\npatient_overlap = list(ids_train_set.intersection(ids_test_set))\nn_overlap = len(patient_overlap)\nprint(f'There are {n_overlap} Patient IDs in both the training and test sets')\nprint('')\nprint(f'These patients are in both the training and test datasets:')\nprint(f'{patient_overlap}')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:50.788486Z","iopub.execute_input":"2023-12-04T01:16:50.789117Z","iopub.status.idle":"2023-12-04T01:16:50.805940Z","shell.execute_reply.started":"2023-12-04T01:16:50.789067Z","shell.execute_reply":"2023-12-04T01:16:50.804695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Visualising Images : JPEG\n\n## Visualizing a random selection of images","metadata":{"trusted":true}},{"cell_type":"code","source":"# train dataframe에서의 image_name 값들을 images라는 변수에 할당\nimages = train_df['image_name'].values\n\n# 이미지 파일 중 9개를 무작위로 선택\nrandom_images = [np.random.choice(images+'.jpg') for i in range(9)]\n\n# random_images의 무작위로 선택된 이미지 파일을 나타내는 코드\nimg_dir = IMAGE_PATH+'/jpeg/train'\n\nprint('Display Random Images')\n\nplt.figure(figsize=(10,8))\n\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img = plt.imread(os.path.join(img_dir, random_images[i]))\n    plt.imshow(img, cmap='gray')                                   #흑백\n    plt.axis('off')                                                #축 제거\n\nplt.tight_layout()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:16:50.808193Z","iopub.execute_input":"2023-12-04T01:16:50.808679Z","iopub.status.idle":"2023-12-04T01:17:04.238257Z","shell.execute_reply.started":"2023-12-04T01:16:50.808632Z","shell.execute_reply":"2023-12-04T01:17:04.237285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We do see that the JPEG format images vary in sizes","metadata":{}},{"cell_type":"markdown","source":"## Visualizing Images with benign lesions","metadata":{"trusted":true}},{"cell_type":"code","source":"#train data frame에서 benign과 maligant를 각각 새로운 데이터 프레임에 저장\n\nbenign = train_df[train_df['benign_malignant']=='benign']\nmalignant = train_df[train_df['benign_malignant']=='malignant']","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:17:04.239625Z","iopub.execute_input":"2023-12-04T01:17:04.240138Z","iopub.status.idle":"2023-12-04T01:17:04.269781Z","shell.execute_reply.started":"2023-12-04T01:17:04.240090Z","shell.execute_reply":"2023-12-04T01:17:04.268671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# benign 이미지를 추출하여 images라는 변수에 저장\nimages = benign['image_name'].values\n\n# iimages에서 9개 무작위 선택 이미지 파일 할당\nrandom_images = [np.random.choice(images+'.jpg') for i in range(9)]\n\n# benign 무작위로 선택된 9개 이미지 출력\nimg_dir = IMAGE_PATH+'/jpeg/train'\n\nprint('Display benign Images')\n\nplt.figure(figsize=(10,8))\n\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img = plt.imread(os.path.join(img_dir, random_images[i]))\n    plt.imshow(img, cmap='gray')   #흑백\n    plt.axis('off')                #축 제거\n\nplt.tight_layout()                 # 간격 조절","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:17:04.271226Z","iopub.execute_input":"2023-12-04T01:17:04.271597Z","iopub.status.idle":"2023-12-04T01:17:20.913090Z","shell.execute_reply.started":"2023-12-04T01:17:04.271561Z","shell.execute_reply":"2023-12-04T01:17:20.908725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizing Images with Malignant lesions","metadata":{}},{"cell_type":"code","source":"# malignant 이미지를 추출하여 images라는 변수에 저장\nimages = malignant['image_name'].values\n\n# # iimages에서 9개 무작위 선택 이미지 파일 할당\nrandom_images = [np.random.choice(images+'.jpg') for i in range(9)]\n\n# benign 무작위로 선택된 9개 이미지 출력\nimg_dir = IMAGE_PATH+'/jpeg/train'\n\nprint('Display malignant Images')\n\nplt.figure(figsize=(10,8))\n\nfor i in range(9):\n    plt.subplot(3, 3, i + 1)\n    img = plt.imread(os.path.join(img_dir, random_images[i]))\n    plt.imshow(img, cmap='gray')   #흑백\n    plt.axis('off')                #축 제거\n\nplt.tight_layout()                 # 간격 조절","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:17:20.914979Z","iopub.execute_input":"2023-12-04T01:17:20.915803Z","iopub.status.idle":"2023-12-04T01:17:29.999789Z","shell.execute_reply.started":"2023-12-04T01:17:20.915743Z","shell.execute_reply":"2023-12-04T01:17:29.998729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Histograms\n\nHistograms are a graphical representation showing how frequently various color values occur in the image i.e frequency of pixels intensity values. In a RGB color space, pixel values range from 0 to 255 where 0 stands for black and 255 stands for white. Analysis of a histogram can help us understand thee brightness, contrast and intensity distribution of an image. Now let's look at the histogram of a random selected sample from each category.\n\n### Benign category","metadata":{}},{"cell_type":"code","source":"# benign 이미지 특성, 픽셀 값 정보를 표시\nf = plt.figure(figsize=(16,8))\nf.add_subplot(1,2, 1)\n\nsample_img = benign['image_name'][0]+'.jpg'\nraw_image = plt.imread(os.path.join(img_dir, sample_img))\nplt.imshow(raw_image, cmap='gray')\nplt.colorbar()\nplt.title('Benign Image')\nprint(f\"Image dimensions:  {raw_image.shape[0],raw_image.shape[1]}\")\nprint(f\"Maximum pixel value : {raw_image.max():.1f} ; Minimum pixel value:{raw_image.min():.1f}\")\nprint(f\"Mean value of the pixels : {raw_image.mean():.1f} ; Standard deviation : {raw_image.std():.1f}\")\n\nf.add_subplot(1,2, 2)\n\n# RGB에 따른 각 색상에 대한 픽셀 강도 분포를 히스토그램으로 시각화\n#ravel() : 다차원 배열을 1차원 배열화// bins:histogram 막대개수 // alpha 투명도\n_ = plt.hist(raw_image[:, :, 0].ravel(), bins = 256, color = 'red', alpha = 0.5)\n_ = plt.hist(raw_image[:, :, 1].ravel(), bins = 256, color = 'Green', alpha = 0.5)\n_ = plt.hist(raw_image[:, :, 2].ravel(), bins = 256, color = 'Blue', alpha = 0.5)\n_ = plt.xlabel('Intensity Value')   #x축 제목\n_ = plt.ylabel('Count')             #y축 제목\n_ = plt.legend(['Red_Channel', 'Green_Channel', 'Blue_Channel'])  #범례\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:17:30.001598Z","iopub.execute_input":"2023-12-04T01:17:30.002126Z","iopub.status.idle":"2023-12-04T01:17:38.175621Z","shell.execute_reply.started":"2023-12-04T01:17:30.002084Z","shell.execute_reply":"2023-12-04T01:17:38.174449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Malignant category","metadata":{}},{"cell_type":"code","source":"# malignant 이미지 특성, 픽셀 값 정보를 표시\nf = plt.figure(figsize=(16,8))\nf.add_subplot(1,2, 1)\n\nsample_img = malignant['image_name'][235]+'.jpg'\nraw_image = plt.imread(os.path.join(img_dir, sample_img))\nplt.imshow(raw_image, cmap='gray')\nplt.colorbar()\nplt.title('Malignant Image')\nprint(f\"Image dimensions:  {raw_image.shape[0],raw_image.shape[1]}\")\nprint(f\"Maximum pixel value : {raw_image.max():.1f} ; Minimum pixel value:{raw_image.min():.1f}\")\nprint(f\"Mean value of the pixels : {raw_image.mean():.1f} ; Standard deviation : {raw_image.std():.1f}\")\n\nf.add_subplot(1,2, 2)\n\n# RGB에 따른 각 색상에 대한 픽셀 강도 분포를 히스토그램으로 시각화\n#ravel() : 다차원 배열을 1차원 배열화// bins:histogram 막대개수 // alpha 투명도\n_ = plt.hist(raw_image[:, :, 0].ravel(), bins = 256, color = 'red', alpha = 0.5)\n_ = plt.hist(raw_image[:, :, 1].ravel(), bins = 256, color = 'Green', alpha = 0.5)\n_ = plt.hist(raw_image[:, :, 2].ravel(), bins = 256, color = 'Blue', alpha = 0.5)\n_ = plt.xlabel('Intensity Value')   #x축 제목\n_ = plt.ylabel('Count')             #y축 제목\n_ = plt.legend(['Red_Channel', 'Green_Channel', 'Blue_Channel'])  #범례\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:17:38.177641Z","iopub.execute_input":"2023-12-04T01:17:38.178091Z","iopub.status.idle":"2023-12-04T01:17:42.883496Z","shell.execute_reply.started":"2023-12-04T01:17:38.178049Z","shell.execute_reply":"2023-12-04T01:17:42.882226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5 Preprocessing DIOCOM files \n[Digital Imaging and Communications in Medicine (DICOM)](https://en.wikipedia.org/wiki/DICOM) is the standard for the communication and management of medical imaging information and related data.DICOM is most commonly used for storing and transmitting medical images enabling the integration of medical imaging devices such as scanners, servers, workstations, printers, network hardware, and picture archiving and communication systems (PACS) from multiple manufacturers\n\nDICOM images have the extension dcm. A DICOM file has two parts: the header and the dataset. The header contains information on the encapsulated dataset. It consists of a File Preamble, a DICOM prefix, and the File Meta Elements.\nFortunately we have a library in Python called Pydicom which can be used to read the DIOCOM files.pydicom makes it easy to read these complex files into natural pythonic structures for easy manipulation. Modified datasets can be written again to DICOM format files.\n\nThere is very nice [kernel](https://www.kaggle.com/schlerp/getting-to-know-dicom-and-the-data) from a competition couple of years ago which serves as a great introduction to DIOCOM image files.I have borrowed the below mentioned code from there.\nKernel: https://www.kaggle.com/schlerp/getting-to-know-dicom-and-the-data\n","metadata":{"trusted":true}},{"cell_type":"code","source":"#pydicom 버전 정보\nprint (pydicom.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:17:42.885055Z","iopub.execute_input":"2023-12-04T01:17:42.885532Z","iopub.status.idle":"2023-12-04T01:17:42.891421Z","shell.execute_reply.started":"2023-12-04T01:17:42.885484Z","shell.execute_reply":"2023-12-04T01:17:42.890285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DICOM에서 필요한 정보 추출하여 출력\ndef show_dcm_info(dataset):\n    print(\"Filename.........:\", file_path)\n    print(\"Storage type.....:\", dataset.SOPClassUID)\n    print()\n#환자 정보\n    pat_name = dataset.PatientName\n    display_name = pat_name.family_name + \", \" + pat_name.given_name\n    print(\"Patient's name......:\", display_name)\n    print(\"Patient id..........:\", dataset.PatientID)\n    print(\"Patient's Age.......:\", dataset.PatientAge)\n    print(\"Patient's Sex.......:\", dataset.PatientSex)\n#검사, 영상 정보\n    print(\"Modality............:\", dataset.Modality)\n    print(\"Body Part Examined..:\", dataset.BodyPartExamined)\n\n\n# pixel data의 이미지 크기 및 픽셀 데이터 크기 출력\n    if 'PixelData' in dataset:\n        rows = int(dataset.Rows)\n        cols = int(dataset.Columns)\n        print(\"Image size.......: {rows:d} x {cols:d}, {size:d} bytes\".format(\n            rows=rows, cols=cols, size=len(dataset.PixelData)))\n#pixelspacing이 있는 경우 해당 정보 출력\n        if 'PixelSpacing' in dataset:\n            print(\"Pixel spacing....:\", dataset.PixelSpacing)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:17:42.897797Z","iopub.execute_input":"2023-12-04T01:17:42.898293Z","iopub.status.idle":"2023-12-04T01:17:42.911932Z","shell.execute_reply.started":"2023-12-04T01:17:42.898243Z","shell.execute_reply":"2023-12-04T01:17:42.910725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#픽셀 배열을 가져와 시각적으로 출력\n#픽셀 배열을 플로팅\ndef plot_pixel_array(dataset, figsize=(5,5)):\n    plt.figure(figsize=figsize)\n    plt.grid(False)\n    plt.imshow(dataset.pixel_array)\n    plt.show()\n\ni = 1\nnum_to_plot = 5\n#DICOM 파일 경로 생성, 읽기, 출력\nfor file_name in os.listdir('../input/siim-isic-melanoma-classification/train/'):\n        file_path = os.path.join('../input/siim-isic-melanoma-classification/train/',file_name)\n        dataset = pydicom.dcmread(file_path)\n        show_dcm_info(dataset)\n        plot_pixel_array(dataset)\n\n        if i >= num_to_plot:\n            break\n\n        i += 1","metadata":{"execution":{"iopub.status.busy":"2023-12-04T01:17:42.913751Z","iopub.execute_input":"2023-12-04T01:17:42.914108Z","iopub.status.idle":"2023-12-04T01:17:51.682723Z","shell.execute_reply.started":"2023-12-04T01:17:42.914073Z","shell.execute_reply":"2023-12-04T01:17:51.681784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Extracting DIOCOM files information in a dataframe\n\n[Gabriel Preda](https://www.kaggle.com/gpreda) has shared the following code in the [discussion forum](https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/154658) which let's you easily extract the relevant information from the diocom files and store it in a dataframe.","metadata":{}},{"cell_type":"code","source":"folder='train'\nPATH='../input/siim-isic-melanoma-classification/'\n\n# DICOM 파일 가져오기\ndef extract_DICOM_attributes(folder):\n    images = list(os.listdir(os.path.join(PATH, folder)))\n    df = pd.DataFrame()      #결과를 저장할 Dataframe\n    for image in images:     #이미지 특성 추출, dataframe에 추가\n        image_name = image.split(\".\")[0]      #확장자 제거\n        dicom_file_path = os.path.join(PATH,folder,image)\n        dicom_file_dataset = pydicom.read_file(dicom_file_path)  #DICOM 파일 읽기\n        study_date = dicom_file_dataset.StudyDate      #DICOM 속성 추출\n        modality = dicom_file_dataset.Modality\n        age = dicom_file_dataset.PatientAge\n        sex = dicom_file_dataset.PatientSex\n        body_part_examined = dicom_file_dataset.BodyPartExamined\n        patient_orientation = dicom_file_dataset.PatientOrientation\n        photometric_interpretation = dicom_file_dataset.PhotometricInterpretation\n        rows = dicom_file_dataset.Rows\n        columns = dicom_file_dataset.Columns\n#dataframe의 각각의 행 추가\n        df = df.append(pd.DataFrame({'image_name': image_name,\n                        'dcm_modality': modality,'dcm_study_date':study_date, 'dcm_age': age, 'dcm_sex': sex,\n                        'dcm_body_part_examined': body_part_examined,'dcm_patient_orientation': patient_orientation,\n                        'dcm_photometric_interpretation': photometric_interpretation,\n                        'dcm_rows': rows, 'dcm_columns': columns}, index=[0]))\n    return df    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T01:17:51.684094Z","iopub.execute_input":"2023-12-04T01:17:51.684697Z","iopub.status.idle":"2023-12-04T01:17:51.698158Z","shell.execute_reply.started":"2023-12-04T01:17:51.684650Z","shell.execute_reply":"2023-12-04T01:17:51.697048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#최종 결과 출력\nextract_DICOM_attributes('train')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}