{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\npd.options.mode.chained_assignment = None\nfrom torch.utils.data import Dataset, DataLoader, Subset\nimport torch\nimport torchvision\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib.pylab import rcParams\nimport cv2\nimport gc\nimport plotly.express as ex\nimport plotly.graph_objects as go\nfrom plotly.offline import iplot\n#cufflinks to link pandas to plotly\nimport cufflinks as cf\ncf.go_offline()\ncf.set_config_file(offline=False, world_readable=True)\n%matplotlib inline\nsns.set(style='whitegrid', palette='muted')\nrcParams['figure.figsize'] = 14,8\nimport torchvision.transforms as transforms\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"root_dir = '../input/siim-isic-melanoma-classification/'\n\ntrain = pd.read_csv(root_dir + 'train.csv')\n\ntrain.head()\n\n\ntest = pd.read_csv(root_dir + 'test.csv')\n\ntest.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print('Total number of entries in train dataset: ', len(train))\nprint('Unique patients in train dataset: ',train['patient_id'].nunique())\n\nprint('Total number of entries in test dataset: ', len(test))\nprint('Unique patients in test dataset: ',test['patient_id'].nunique())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Meta Data Analysis","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**Age Analysis for both train and test**","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"#Missing values in train\nprint('Number of missing age values in train data is :',train['age_approx'].isnull().sum())\nprint('Number of missing age values in test data is :',test['age_approx'].isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def show_age_dist(series, bins = None):\n    fig = go.Figure()\n    hist = go.Histogram(x = series, nbinsx=50)\n    fig.add_trace(hist)\n    fig.update_layout(title_text = 'Age distribution')\n    fig.update_yaxes(title_text='Number of Patients')\n    fig.update_xaxes(title_text='Age')\n    fig.show()\n#     return hist.xbins\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"group_age_train = train.groupby('patient_id')['age_approx'].mean()\ngroup_age_test = test.groupby('patient_id')['age_approx'].mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"sns.distplot(group_age_train,bins=50, kde = True).set_title('Age distribution in train dataset')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"sns.distplot(group_age_test,bins=50, kde = True).set_title('Age distribution in test dataset')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"train_withAge = train[train['age_approx']>0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Lets check its correlation with \"Target\" variable","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_withAge['age_approx'].corr(train_withAge['target'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Since the correlation is very low between 'target' and 'age', we can deduce that age is not a significant factor for Melanoma","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"** Gender Distribution **","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"print('Number of records with missing gender information in training set: ', train['sex'].isnull().sum())\nprint('Number of records with missing gender information in test set: ', test['sex'].isnull().sum())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def show_gender_dist(series, title):\n    fig = go.Figure()\n    df_ = series.value_counts(normalize = True)\n    bar_graph = go.Bar(x = df_.index, y = df_.values, width=[0.1,0.1])\n    fig.add_trace(bar_graph)\n    fig.update_layout(title_text = title, bargap=0.1)\n    fig.update_yaxes(title_text='Gender count percentage')\n    fig.update_xaxes(title_text='Gender')\n    fig.show()\n#     return hist.xbins\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_gender_dist(train['sex'], 'Gender Distribution for Train set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_gender_dist(test['sex'], 'Gender Distribution for Test set')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def show_gender_target_dist(df, title):\n    gender_target_df = df.groupby(['sex', 'target'])['benign_malignant'].count().to_frame().reset_index()\n    gender_target_df.target = gender_target_df.target.replace({0:'Benign', 1:'Malignant'})\n    fig = go.Figure()\n    fig.add_trace(go.Bar(x = gender_target_df[gender_target_df['sex']=='female'].target, \n                         y = gender_target_df[gender_target_df['sex']=='female'].benign_malignant, \n                         name='Female',\n                         marker_color='indianred',\n                         text = gender_target_df[gender_target_df['sex']=='female'].benign_malignant, \n                         textposition = 'auto',\n                         width=[0.25,0.25]))\n\n    fig.add_trace(go.Bar(x = gender_target_df[gender_target_df['sex']=='male'].target, \n                         y = gender_target_df[gender_target_df['sex']=='male'].benign_malignant, \n                         name='Male',\n                         marker_color='lightsalmon',\n                         text = gender_target_df[gender_target_df['sex']=='male'].benign_malignant, \n                         textposition = 'auto',\n                         width=[0.25,0.25]))\n\n\n    fig.update_layout(barmode='group', title=title, bargap=0)\n    fig.update_yaxes(title_text='Count')\n    fig.update_xaxes(title_text='Benign vs Malignant')\n    fig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_gender_target_dist(train, 'Gender-Target distirbution of Train set')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Although the difference is not huge, but is noticeable enough that male patients have higher malignant cases compared to female patients","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"** Anatomy Site EDA **","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"lesion_loc = train['anatom_site_general_challenge'].value_counts(normalize=True).sort_values(ascending=False)\nlesion_loc.iplot(kind='bar', \n                 xTitle='Percentage', \n                 text = lesion_loc.values.tolist(),\n                 \n                 textposition='outside',\n                 title='Distribution of lesion location across train dataset')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"lesion_loc = test['anatom_site_general_challenge'].value_counts(normalize=True).sort_values(ascending=False)\nlesion_loc.iplot(kind='bar', \n                 xTitle='Percentage', \n                 text = lesion_loc.values.tolist(),\n                 \n                 textposition='outside',\n                 title='Distribution of lesion location across test dataset')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def show_location_target_dist(df, title):\n    df.dropna(subset = [\"anatom_site_general_challenge\"], inplace=True)\n    loc_target_df = df.groupby(['anatom_site_general_challenge', 'target'])['benign_malignant'].count().to_frame().reset_index()\n    loc_target_df.target = loc_target_df.target.replace({0:'Benign', 1:'Malignant'})\n    locations = df['anatom_site_general_challenge'].unique()\n    total_benign = loc_target_df[loc_target_df['target']=='Benign'].benign_malignant.sum()\n    total_malignant = loc_target_df[loc_target_df['target']=='Malignant'].benign_malignant.sum()\n    fig = go.Figure()\n    for l in locations:\n#         print(l)\n        percent_text = round((loc_target_df[loc_target_df['anatom_site_general_challenge']==l].benign_malignant/[total_benign,total_malignant ])*100, 2)\n        percent_text = [str(t)+\"%\" for t in percent_text]\n        fig.add_trace(go.Bar(x = loc_target_df[loc_target_df['anatom_site_general_challenge']==l].target, \n                             y = loc_target_df[loc_target_df['anatom_site_general_challenge']==l].benign_malignant, \n                             name=l,\n                             text = percent_text, \n                             textposition = 'outside'\n                            ))\n\n\n\n\n    fig.update_layout(barmode='group', title=title, bargap=0.1)\n    fig.update_yaxes(title_text='Count')\n    fig.update_xaxes(title_text='Benign vs Malignant')\n    fig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_location_target_dist(train, 'Location ditribution relative to Target ')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's take a deeper look at location distribution for Malignant lesions","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":false},"cell_type":"code","source":"def show_location_malignant_dist(df, title):\n    df.dropna(subset = [\"anatom_site_general_challenge\"], inplace=True)\n    loc_target_df = df.groupby(['anatom_site_general_challenge', 'target'])['benign_malignant'].count().to_frame().reset_index()\n    loc_target_df.target = loc_target_df.target.replace({0:'Benign', 1:'Malignant'})\n    loc_target_df = loc_target_df[loc_target_df['target']=='Malignant']\n    locations = df['anatom_site_general_challenge'].unique()    \n    total_malignant = loc_target_df[loc_target_df['target']=='Malignant'].benign_malignant.sum()\n    fig = go.Figure()\n    for l in locations:\n#         print(l)\n        percent_text = round((loc_target_df[loc_target_df['anatom_site_general_challenge']==l].benign_malignant/total_malignant)*100, 2)\n        percent_text = [str(t)+\"%\" for t in percent_text]\n        fig.add_trace(go.Bar(x = loc_target_df[loc_target_df['anatom_site_general_challenge']==l].target, \n                             y = loc_target_df[loc_target_df['anatom_site_general_challenge']==l].benign_malignant, \n                             name=l,\n                             text = percent_text, \n                             textposition = 'outside'\n                            ))\n\n\n\n\n    fig.update_layout(barmode='group', title=title, bargap=0.1)\n    fig.update_yaxes(title_text='Count')\n    fig.update_xaxes(title_text='Malignant')\n    fig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_location_malignant_dist(train, 'Location ditribution relative to Malignant lesions')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"So, Torso, which is the trunk of human body has more chances to have malignant lesions. For that matter, any lesion since even it has higher percentage share for benign lesions as well","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"** Diagnosis EDA **","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def show_diagnosis_dist(df):\n    diagnosis = df['diagnosis'].value_counts().sort_values(ascending=False)\n    \n    fig = go.Figure()\n    fig.add_trace(go.Bar( x = diagnosis.index,\n                         y = diagnosis.values,\n                         text = diagnosis.values.tolist(),\n                         textposition = 'outside'\n    \n    ))\n    fig.update_layout(title='Diagnosis distribution')\n    fig.update_yaxes(title_text='Count')\n    fig.update_xaxes(title_text='Diagnosis')\n    fig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"show_diagnosis_dist(train)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"So lot of records have unknown diagnosis. However, this information is only available for train dataset","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# EDA on images","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"## Data Leak?\nOne of interesting discussions on forum is possible data leak through image resolution and image mean color. Is it really true? Let's check it now","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"** Data Leak through Image resolution: ** Images with high resolution have more pixels, so lets use this intuition and calculate coorelation between image resolution and target. Record with high number of pixels implies that it has high resolution","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"class MelanomaDataset(Dataset):\n    def __init__(self, df: pd.DataFrame, im_folder: str, train: bool = True, transforms = None):\n        \"\"\"\n        Class initialization\n        Args:\n            df (pd.DataFrame): DataFrame with data description\n            im_folder (str): folder with images\n            train (bool): flag of whether a training dataset is being initialized or testing one\n            transforms: image transformation method to be applied\n            \n        \"\"\"\n        self.df = df\n        self.transforms = transforms\n        self.train = train\n        self.im_folder = im_folder\n        \n    def __getitem__(self, index):\n        im_path = os.path.join(self.im_folder, self.df.iloc[index]['image_name'] + '.jpg')\n        x = cv2.imread(im_path)\n        x = cv2.cvtColor(x, cv2.COLOR_BGR2RGB)\n        if self.transforms:\n            x = self.transforms(x)\n            \n        if self.train:\n            y = self.df.iloc[index]['target']\n            return x, y\n        else:\n            return x\n    \n    def __len__(self):\n        return len(self.df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"melanoma_dataset = MelanomaDataset(train, root_dir+'jpeg/train/', True)\n\nmelanoma_dataloader = DataLoader(dataset = melanoma_dataset, batch_size=1, shuffle=False, num_workers=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"%%time\nfrom tqdm import tqdm\npixels = []\nmean_color = []\ntargets_temp = []\nwidths = []\nheights = []\nfor i, (img, target_) in enumerate(tqdm(melanoma_dataloader)):\n    img = img.squeeze()\n    target_ = target_.squeeze()\n#     print(target_.shape)\n    h,w = img.shape[0], img.shape[1]\n    pixels.append(h*w)\n    widths.append(w)\n    heights.append(h)\n    mean_color.append(np.mean(img.numpy()))\n    targets_temp.append(int(target_.numpy()))\n    del img\n    gc.collect()\n    if(i==1500):\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# plt.scatter(widths, heights, c = targets_temp, cmap = plt.cm.autumn, alpha = 0.5)\nfig = go.Figure()\nfig.add_trace(go.Scatter(x=widths, y=heights, mode='markers', marker = dict(color=targets_temp)))\nfig.update_xaxes(title_text='Widhts')\nfig.update_yaxes(title_text='Heights')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"targets_temp = np.array(targets_temp)\nwidths = np.array(widths)\nheights = np.array(heights)\nmalignant_indices = np.where(targets_temp==1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"benign_indices = np.where(targets_temp==0)\n\npixels = np.array(pixels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.distplot(pixels[malignant_indices]).set_title('Resolution distribution for Malignant Lesions')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.distplot(pixels[benign_indices]).set_title('Resolution distribution for Benign Lesions')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"!pip install chart_studio","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import chart_studio.plotly as py\nimport plotly.figure_factory as ff","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def density_plot_resolution(indices, title):\n    fig = go.Figure()\n    fig.add_trace(go.Histogram2dContour(x=widths[indices], y=heights[indices], histfunc='count', colorscale='blues'))\n    fig.add_trace(go.Scatter(x=widths[indices], y=heights[indices], mode='markers'))\n    fig.add_trace(go.Histogram(\n            y = heights[indices],\n            xaxis = 'x2',\n    \n        ))\n    fig.add_trace(go.Histogram(\n            x = widths[indices],\n            yaxis = 'y2',\n   \n        ))\n\n    fig.update_layout(\n\n        xaxis = dict(\n            zeroline = False,\n            domain = [0,0.85],\n            showgrid = False\n        ),\n        yaxis = dict(\n            zeroline = False,\n            domain = [0,0.85],\n            showgrid = False\n        ),\n        xaxis2 = dict(\n            zeroline = False,\n            domain = [0.85,1],\n            showgrid = False\n        ),\n        yaxis2 = dict(\n            zeroline = False,\n            domain = [0.85,1],\n            showgrid = False\n        ),\n\n        bargap = 0,\n        hovermode = 'closest',\n        showlegend = False,\n        title = title\n    )\n    fig.update_xaxes(title_text='Widhts')\n    fig.update_yaxes(title_text='Heights')\n    fig.show()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"density_plot_resolution(benign_indices, 'Density plot of resolution for Benign Lesions')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"density_plot_resolution(malignant_indices, 'Density plot of resolution for Malignant Lesions')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mean_color = np.array(mean_color)\n# fig = go.Figure()\nfig=(ff.create_distplot([mean_color[benign_indices], mean_color[malignant_indices]], group_labels=['Mean color for Benign', 'Mean color for Malignant']))\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# Corelation calculation:\npixels_series = pd.Series(pixels)\nmean_color_series = pd.Series(mean_color)\ntargets_temp_series = pd.Series(targets_temp)\nprint('Correlation between Image resolution and targets is :', pixels_series.corr(targets_temp_series))\nprint('Correlation between Mean color of images and targets is :', mean_color_series.corr(targets_temp_series))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Though there is no direct correlation of resolution & mean-color with Target i.e. no linear relationship, the density plots indicate a possible non-linear relationship which can be captured by a neural network model as malignant and benign lesions have different resolution distributions","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"np.save('widths.npy', widths)\nnp.save('heights.npy', heights)\nnp.save('targets_temp.npy', targets_temp)\nnp.save('mean_color.npy', mean_color)\nnp.save('pixels.npy',pixels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"markdown","source":"# Image Visualization Analysis","execution_count":null},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"transform_basic = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.Resize((256,256)),\n    transforms.ToTensor()\n    ])\n\nmelanoma_dataset = MelanomaDataset(train, root_dir+'jpeg/train/', True, transform_basic)\n\nmelanoma_dataloader = DataLoader(dataset = melanoma_dataset, batch_size=10, shuffle=False)\n\ndef imshow(img):\n    img = img.squeeze()\n    plt.imshow(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"dataiter = iter(melanoma_dataloader)\n\nimages, targets = dataiter.next()\nimages = images.permute(0,2,3,1)\nimages = images.numpy()\ntargets = targets.numpy()\nfig = plt.figure(figsize=(25, 10))\n# display 10 images\nfor idx in np.arange(10):\n    ax = fig.add_subplot(2, 10/2, idx+1, xticks=[], yticks=[])\n    imshow(images[idx])\n    ax.set_title(targets[idx])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Remember the distribution on diagnosis? Let's analyse that","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"show_diagnosis_dist(train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"def display_images_diagnosis(diagnosis, title):\n    \n    train_temp = train[train['diagnosis']==diagnosis]\n    random_indices = np.random.randint(0, len(train_temp), 5)\n    \n    fig = plt.figure(figsize=(25, 10))\n    for idx in np.arange(5):\n        img = cv2.imread(root_dir+'jpeg/train/'+train_temp.iloc[random_indices[idx]].image_name+'.jpg')\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = cv2.resize(img,(256,256))\n        target = train_temp.iloc[random_indices[idx]].target\n        ax = fig.add_subplot(2, 5, idx+1, xticks=[], yticks=[])\n        imshow(img)\n        ax.set_title(target) \n    plt.suptitle(title)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"display_images_diagnosis('nevus', 'Lesions diagnosed as Nevus')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"display_images_diagnosis('melanoma', 'Lesions diagnosed as Melanoma')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"display_images_diagnosis('seborrheic keratosis', 'Lesions diagnosed as Seborrheic Keratosis')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"display_images_diagnosis('lentigo NOS', 'Lesions diagnosed as Lentigo NOS')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"display_images_diagnosis('lichenoid keratosis', 'Lesions diagnosed as Lichenoid keratosis')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"From the above images, it can be seen that Nevus lesions have some sort of pinkish shade all over ( No matter how many random indices were picked all nevus images somehow has same shade). Could model easily capture this? The rest of the types i.e. Melanoma, Seborrheic Kertosis, Lentigo NOS, Lichenoid Keratosis etc seem to display similar characteristics at the first glance. To distinguish between them, model has to capture more insighful features like the ones described in ABCDE, 7-point derma checklist etc.","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"**To be continued ...**","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}