{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# libraries\nimport seaborn as sns\n\n#color\nfrom colorama import Fore, Back, Style\n\n#plotly\n!pip install chart_studio\nimport plotly.express as px\nimport chart_studio.plotly as py\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nimport cufflinks\ncufflinks.go_offline()\ncufflinks.set_config_file(world_readable=True, theme='pearl')\n\n#read the .dcm file\nimport pydicom","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":false},"cell_type":"markdown","source":"# 1. Input the DataFrame","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# preview the train dataframe\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# preview the train dataframe\ntest_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Check the list of files or folders in the data source\nlist(os.listdir(\"../input/siim-isic-melanoma-classification/\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_submission = pd.read_csv('../input/siim-isic-melanoma-classification/sample_submission.csv')\nsample_submission","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Because the `benign` is naturally highly possibile. If there is no machine learning and only guess the patient to be `benigh`. How good the result will be ?","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"submission1 = sample_submission\nsubmission1.to_csv('submission1.csv',index = False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2. Check the NULL value","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# check if there is missing data in the dataframe\n# check the null part in the whole data set, red part is missing data, blue is non-null\nsns.heatmap(train_df.isnull(),yticklabels=False,cbar=False,cmap='coolwarm')\ntrain_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the Missing data distribution in train_df\nfig = px.scatter(train_df.isnull().sum())\n\nfig.update_layout(\n    title=\"Missing Data in train_df\",\n    xaxis_title=\"Columns\",\n    yaxis_title=\"Missing data count\",\n    showlegend=False,\n    font=dict(\n        family=\"Courier New, monospace\",\n        size=12,\n        color=\"RebeccaPurple\"\n    )\n)\n\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check if there is missing data in the dataframe\n# check the null part in the whole data set, red part is missing data, blue is non-null\nsns.heatmap(test_df.isnull(),yticklabels=False,cbar=False,cmap='coolwarm')\ntest_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check the Missing data distribution in test_df\nfig = px.scatter(test_df.isnull().sum())\n\nfig.update_layout(\n    title=\"Missing Data in test_df\",\n    xaxis_title=\"Columns\",\n    yaxis_title=\"Missing data count\",\n    showlegend=False,\n    font=dict(\n        family=\"Courier New, monospace\",\n        size=12,\n        color=\"RebeccaPurple\"\n    )\n)\n\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Shape of train and test dataframe\nprint(Fore.RED + 'Training data shape: ',Style.RESET_ALL,train_df.shape)\nprint(Fore.BLUE + 'Test data shape: ',Style.RESET_ALL,test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Show the list of columns\ncolumns = train_df.keys()\ncolumns = list(columns)\nprint(Fore.RED + \"List of columns in the train_df\",Fore.GREEN + \"\", columns)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Clean data\n### Because `sex` and 'age-approx' are two important features. Comparing to the number of rows in train_df (33126), the missing data of `sex` (65) and `age-approx` (68) is less than 0.1%. We can drop the null elements.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# This dataset has some missing values, which we set to the median of the column for the purpose of this tutorial. \ncleaned_train_df = train_df.dropna()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# check if there is missing data in the dataframe\n# check the null part in the whole data set, red part is missing data, blue is non-null\nsns.heatmap(cleaned_train_df.isnull(),yticklabels=False,cbar=False,cmap='coolwarm')\ncleaned_train_df.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"![WechatIMG14.jpeg](http://github.com/daiwofei/skin_cancer_classification/blob/master/WechatIMG14.jpeg)","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# 3. Exploraty Data Analysis","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# verify if the patient_id is unique for the train_df\n\nprint ('Rows in trains_df is', len(train_df))\nprint ('Number of unique patient id is', train_df['patient_id'].nunique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There are only 2056 unique `patient id`.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# verify if the image_name is unique for the train_df\n\nprint ('Rows in trains_df is', len(train_df))\nprint ('Number of unique patient id is', train_df['image_name'].nunique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"`image_name` is unique in train_df.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# The Histogram of sex\ntrain_df['sex'].value_counts().iplot(kind='bar',yTitle='Counts',xTitle = 'Sex',linecolor='black',opacity=0.7,color='green',theme='pearl',bargap=0.5,\n                                       gridcolor='white',title='Distribution of the Sex column in the Unique Patient Set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# The Histogram of benign_malignant\ntrain_df['benign_malignant'].value_counts().iplot(kind='bar',yTitle='Counts',xTitle = 'Sex',linecolor='black',opacity=0.7,color='blue',theme='pearl',bargap=0.5,\n                                       gridcolor='white',title='Distribution of the Sex column in the Unique Patient Set')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# the 'benign' corresponds to 0 in 'target'.\n# The Histogram of target\ntrain_df['target'].value_counts().iplot(kind='bar',yTitle='Counts',xTitle = 'Sex',linecolor='black',opacity=0.7,color='orange',theme='pearl',bargap=0.5,\n                                       gridcolor='white',title='Distribution of the Sex column in the Unique Patient Set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# The Histogram of tadiagnosisrget\ntrain_df['diagnosis'].value_counts().iplot(kind='bar',yTitle='Counts',xTitle = 'Sex',linecolor='black',opacity=0.7,color='red',theme='pearl',bargap=0.5,\n                                       gridcolor='white',title='Distribution of the Sex column in the Unique Patient Set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# The Histogram of anatom_site_general_challenge\ntrain_df['anatom_site_general_challenge'].value_counts().iplot(kind='bar',yTitle='Counts',xTitle = 'Sex',linecolor='black',opacity=0.7,color='purple',theme='pearl',bargap=0.5,\n                                       gridcolor='white',title='Distribution of the Position column in the Unique Patient Set')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 4. Analyze the .dcm image","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/aadhavvignesh/lung-segmentation-by-marker-controlled-watershed\ndef load_scan(path):\n    \"\"\"\n    Loads scans from a folder and into a list.\n    \n    Parameters: path (Folder path)\n    \n    Returns: slices (List of slices)\n    \"\"\"\n    slices = pydicom.dcmread(path)\n    #slices = [pydicom.read_file(path + '/' + s) for s in os.listdir(path)]\n    #slices.sort(key = lambda x: int(x.InstanceNumber))\n        \n    return slices","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/aadhavvignesh/lung-segmentation-by-marker-controlled-watershed\ndef get_pixels_hu(scans):\n    \"\"\"\n    Converts raw images to Hounsfield Units (HU).\n    \n    Parameters: scans (Raw images)\n    \n    Returns: image (NumPy array)\n    \"\"\"\n    \n    image = np.stack([s.pixel_array for s in scans])\n    image = image.astype(np.int16)\n\n    # Since the scanning equipment is cylindrical in nature and image output is square,\n    # we set the out-of-scan pixels to 0\n    image[image == -2000] = 0\n    \n    \n    # HU = m*P + b\n    intercept = scans[0].RescaleIntercept\n    slope = scans[0].RescaleSlope\n    \n    if slope != 1:\n        image = slope * image.astype(np.float64)\n        image = image.astype(np.int16)\n        \n    image += np.int16(intercept)\n    \n    return np.array(image, dtype=np.int16)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"INPUT_FOLDER = '/kaggle/input/siim-isic-melanoma-classification/train/'\n\npictures = os.listdir(INPUT_FOLDER)\npictures.sort()\npictures[0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The image name is called `PatientID` in the `.dcm` file","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"test_patient_scans = load_scan(INPUT_FOLDER + pictures[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_patient_scans.dir()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_patient_scans.PixelData","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_patient_images = get_pixels_hu(test_patient_scans)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(test_patient_scans.PixelData, cmap='gray')\nplt.title(\"Original Slice\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}