{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport os\n\nsns.set_style('darkgrid')\nplt.style.use('seaborn-notebook')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-14T08:12:14.783583Z","iopub.execute_input":"2022-10-14T08:12:14.784062Z","iopub.status.idle":"2022-10-14T08:12:16.170729Z","shell.execute_reply.started":"2022-10-14T08:12:14.784023Z","shell.execute_reply":"2022-10-14T08:12:16.169059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#read in train and test csv files to calculate descriptive stats and characteristics\n\ntrain = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\n\ntest = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T08:12:17.997932Z","iopub.execute_input":"2022-10-14T08:12:17.998802Z","iopub.status.idle":"2022-10-14T08:12:18.140956Z","shell.execute_reply.started":"2022-10-14T08:12:17.998756Z","shell.execute_reply":"2022-10-14T08:12:18.139140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:05.976664Z","iopub.execute_input":"2022-10-14T07:28:05.977169Z","iopub.status.idle":"2022-10-14T07:28:06.000717Z","shell.execute_reply.started":"2022-10-14T07:28:05.977122Z","shell.execute_reply":"2022-10-14T07:28:05.998965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:06.003834Z","iopub.execute_input":"2022-10-14T07:28:06.004575Z","iopub.status.idle":"2022-10-14T07:28:06.013646Z","shell.execute_reply.started":"2022-10-14T07:28:06.004523Z","shell.execute_reply":"2022-10-14T07:28:06.012214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:06.016654Z","iopub.execute_input":"2022-10-14T07:28:06.017239Z","iopub.status.idle":"2022-10-14T07:28:06.048205Z","shell.execute_reply.started":"2022-10-14T07:28:06.017178Z","shell.execute_reply":"2022-10-14T07:28:06.046927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:06.049951Z","iopub.execute_input":"2022-10-14T07:28:06.050983Z","iopub.status.idle":"2022-10-14T07:28:06.084983Z","shell.execute_reply.started":"2022-10-14T07:28:06.050931Z","shell.execute_reply":"2022-10-14T07:28:06.083727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax1 = sns.countplot(data=train, x='anatom_site_general_challenge')\nax1.set_title('Distribution of Anatomical Sites Training Set')\nax1.tick_params(axis='x', labelrotation = 45, labelsize = 12)\n\n#note: largest category by far is the torso, oral/genital is the least common category\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:06.086804Z","iopub.execute_input":"2022-10-14T07:28:06.088131Z","iopub.status.idle":"2022-10-14T07:28:06.421287Z","shell.execute_reply.started":"2022-10-14T07:28:06.088079Z","shell.execute_reply":"2022-10-14T07:28:06.420322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax2 = sns.countplot(data=train, x='sex')\nax2.set_title('Distribution of Gender')\n\n#note: gender is pretty evenly distributed amongst the image dataset","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:06.423231Z","iopub.execute_input":"2022-10-14T07:28:06.424071Z","iopub.status.idle":"2022-10-14T07:28:06.729512Z","shell.execute_reply.started":"2022-10-14T07:28:06.424024Z","shell.execute_reply":"2022-10-14T07:28:06.728302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax3 = sns.countplot(data=train, x='diagnosis')\nax3.tick_params(axis='x', labelrotation = 45, labelsize = 12)\nax3.set_title('Distribution of Diagnosis')\n\n#note: the vast majority of the images have no associated diagnosis, some have the nevus diagnosis, and very few have the melanoma diagnosis. \n#Melanoma is the target we're actually looking for.","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:06.732750Z","iopub.execute_input":"2022-10-14T07:28:06.733282Z","iopub.status.idle":"2022-10-14T07:28:07.092472Z","shell.execute_reply.started":"2022-10-14T07:28:06.733234Z","shell.execute_reply":"2022-10-14T07:28:07.091265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax4 = sns.countplot(data=train, x='benign_malignant')\nax4.tick_params(axis='x', labelrotation = 45, labelsize = 12)\nax4.set_title('Distribution of Diagnosis')\n\n#note: malignant here is defined as a have a \"melanoma\" disease label","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:07.096308Z","iopub.execute_input":"2022-10-14T07:28:07.096693Z","iopub.status.idle":"2022-10-14T07:28:07.365281Z","shell.execute_reply.started":"2022-10-14T07:28:07.096660Z","shell.execute_reply":"2022-10-14T07:28:07.363938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax5 = sns.histplot(data=train, x='age_approx', bins=18)\nax5.tick_params(axis='x', labelrotation = 45, labelsize = 12)\nax5.set_title('Distribution of Age')\n\n#age follows a normal distribution","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:07.370973Z","iopub.execute_input":"2022-10-14T07:28:07.372056Z","iopub.status.idle":"2022-10-14T07:28:07.716168Z","shell.execute_reply.started":"2022-10-14T07:28:07.371998Z","shell.execute_reply":"2022-10-14T07:28:07.713300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('n rows where target != benign_malignant: {}'.format(len(train.loc[(train['target'] == 1) & (train['benign_malignant'] != 'malignant')])))\nprint('n rows where benign_malignant != melanoma diagnosis: {}'.format(len(train.loc[(train['diagnosis'] != 'melanoma') & (train['benign_malignant'] == 'malignant')])))\n\n#my assumptions hold--this dataset is labeled based off of the 'melanoma' diagnosis, \n#and a positive value in the target column indicates a malignant melanoma tumor\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:07.717954Z","iopub.execute_input":"2022-10-14T07:28:07.718383Z","iopub.status.idle":"2022-10-14T07:28:07.739271Z","shell.execute_reply.started":"2022-10-14T07:28:07.718349Z","shell.execute_reply":"2022-10-14T07:28:07.738081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print the number of missing values from each col in train\nfor col in train.columns:\n    print(col + ' missing values: ' + str(train[col].isna().sum()))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:07.740799Z","iopub.execute_input":"2022-10-14T07:28:07.741765Z","iopub.status.idle":"2022-10-14T07:28:07.764795Z","shell.execute_reply.started":"2022-10-14T07:28:07.741720Z","shell.execute_reply":"2022-10-14T07:28:07.763425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the training dataframe holding data for 33,126 images, the column with the most missing values is \"anatom_site_challenge\" at 527 missing (1.6% of the data). Sex is missing for 65 entries, and age is missing for 68 images. This represents such a small proportion of our dataset that it may be reasonable to simply drop these images from the dataset. Before I do, though, I want to check to make sure there's no apparent pattern in the entries with missing values.","metadata":{}},{"cell_type":"code","source":"#Pull out just the observations that are positive for the target condition\nmalignant = train[train['target'] == 1]\nprint(malignant.head())\nprint(malignant.shape)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:07.766285Z","iopub.execute_input":"2022-10-14T07:28:07.766960Z","iopub.status.idle":"2022-10-14T07:28:07.782983Z","shell.execute_reply.started":"2022-10-14T07:28:07.766913Z","shell.execute_reply":"2022-10-14T07:28:07.781574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in malignant.columns:\n    print(col + ' missing values: ' + str(malignant[col].isna().sum()))\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:07.784694Z","iopub.execute_input":"2022-10-14T07:28:07.785976Z","iopub.status.idle":"2022-10-14T07:28:07.796904Z","shell.execute_reply.started":"2022-10-14T07:28:07.785926Z","shell.execute_reply":"2022-10-14T07:28:07.795700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on our analysis, 584 out of 33126 observations are positive for the malignant class. We could drop the rows with missing age and sex values, but if we simply drop the rows with missing anatomical region values, we would lose 9 positive cases. I'm going to fill the NaN values for the anatomical site feature with a new label: \"other or unknown\"","metadata":{}},{"cell_type":"code","source":"#fill missing anatom site values with \"unknown or other\"\ntrain.anatom_site_general_challenge.fillna('other or unknown', inplace=True)\n\n#verify there are no more missing values in that column\nprint('anatom_site_general_challenge' + ' missing values: ' + str(train['anatom_site_general_challenge'].isna().sum()))\n\n#plot the new distribution\nax = sns.countplot(data=train, x='anatom_site_general_challenge')\n#train.anatom_site_general_challenge.hist()\nax.set_title('Distribution of Anatomical Sites Training Set')\nax.tick_params(axis='x', labelrotation = 45, labelsize = 12)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:07.798678Z","iopub.execute_input":"2022-10-14T07:28:07.799922Z","iopub.status.idle":"2022-10-14T07:28:08.133926Z","shell.execute_reply.started":"2022-10-14T07:28:07.799850Z","shell.execute_reply":"2022-10-14T07:28:08.132643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Test Set exploration**","metadata":{}},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:08.135892Z","iopub.execute_input":"2022-10-14T07:28:08.137051Z","iopub.status.idle":"2022-10-14T07:28:08.153163Z","shell.execute_reply.started":"2022-10-14T07:28:08.137002Z","shell.execute_reply":"2022-10-14T07:28:08.151563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:08.155345Z","iopub.execute_input":"2022-10-14T07:28:08.156194Z","iopub.status.idle":"2022-10-14T07:28:08.165861Z","shell.execute_reply.started":"2022-10-14T07:28:08.156149Z","shell.execute_reply":"2022-10-14T07:28:08.164379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print the number of missing values from each col in train\nfor col in test.columns:\n    print(col + ' missing values: ' + str(test[col].isna().sum()))\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:08.168174Z","iopub.execute_input":"2022-10-14T07:28:08.168965Z","iopub.status.idle":"2022-10-14T07:28:08.184725Z","shell.execute_reply.started":"2022-10-14T07:28:08.168915Z","shell.execute_reply":"2022-10-14T07:28:08.183801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Test Set Conclusions:**\nWe can see that the test set provided by kaggle is truly a test set meant for the competition submission: The set is missing the target variable column \"target\" and therefore is unlabeled. We will only use this test set for the kaggle competition submission to evaluate the final model performance against others'. Interestingly, some of these images do have missing anatomical site information as well. If we end up using the metadata in the deep learning model, we'll need to be sure to clean these values first.","metadata":{}},{"cell_type":"markdown","source":"**Grouping by Patient**\nThis dataset made clear that there are multiple images from the same patient included. I'd like to understand how many images are from the same patient, and generally how the dataset looks when we group by patient_id","metadata":{}},{"cell_type":"code","source":"#how many unique patients are there anyway?\nn_patients = len(pd.unique(train['patient_id']))\n\nprint('There are {} unique patients in the training set'.format(n_patients))\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:08.188882Z","iopub.execute_input":"2022-10-14T07:28:08.189285Z","iopub.status.idle":"2022-10-14T07:28:08.199484Z","shell.execute_reply.started":"2022-10-14T07:28:08.189250Z","shell.execute_reply":"2022-10-14T07:28:08.198234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#how many images are associated with each patient?\nindividuals_count = train.groupby('patient_id').count()\n\nindividuals_count.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:08.201429Z","iopub.execute_input":"2022-10-14T07:28:08.202485Z","iopub.status.idle":"2022-10-14T07:28:08.243060Z","shell.execute_reply.started":"2022-10-14T07:28:08.202436Z","shell.execute_reply":"2022-10-14T07:28:08.241673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print summary stats for the number of images per patient.\nindividuals_count.image_name.describe()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:08.244999Z","iopub.execute_input":"2022-10-14T07:28:08.245915Z","iopub.status.idle":"2022-10-14T07:28:08.258778Z","shell.execute_reply.started":"2022-10-14T07:28:08.245849Z","shell.execute_reply":"2022-10-14T07:28:08.257824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show the disribution of image counts\nfig, ax = plt.subplots(figsize=(25,10))\nax = sns.histplot(data=individuals_count, x='image_name')\nax.set_title('Images per Patient')\nax.set_xlabel('Unique Images per Patient')\nax.set_ylabel('Number of Patients')\nax.set_xticks(ticks=range(0,120,10))\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:08.260929Z","iopub.execute_input":"2022-10-14T07:28:08.261968Z","iopub.status.idle":"2022-10-14T07:28:08.705478Z","shell.execute_reply.started":"2022-10-14T07:28:08.261917Z","shell.execute_reply":"2022-10-14T07:28:08.704009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('the mean number of images per patient is: {}'.format(round(individuals_count.image_name.mean(),2)))\nprint('the median number of images per patient is: {}'.format(round(individuals_count.image_name.median(),2)))\nprint('the standard deviation of the number of images per patient is: {}'.format(round(individuals_count.image_name.std(),2)))","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:08.707420Z","iopub.execute_input":"2022-10-14T07:28:08.707917Z","iopub.status.idle":"2022-10-14T07:28:08.717469Z","shell.execute_reply.started":"2022-10-14T07:28:08.707852Z","shell.execute_reply":"2022-10-14T07:28:08.716118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The number of images per patient varied broadly, and we have a right-tailed distribution. Clearly, some of the patients involved in this study had many more moles to photograph! There's not necessarily any indication that we need to exclude any of these outlier patients, but I do want to understand more about how the quantity of images in the dataset relates to the number of positive targets. I can also potentially impute the missing values for gender and age if that data exists for another image from the same patient.","metadata":{}},{"cell_type":"code","source":"#check out the distribution of positive cases grouped by patient\nindividuals_sum = train.groupby('patient_id').sum()\n\nprint(individuals_sum.target.describe())\n\n#fig, ax = plt.subplots(figsize=(18,10))\nax = sns.histplot(data=individuals_sum, x='target', bins=10)\nax.set_title('Quantity of Malignant Tumors per Patient')\nax.set_xlabel('Sum of Positives')\nax.set_ylabel('Number of Patients')\n#ax.set_xticks(ticks=range(0,20))\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:08.719820Z","iopub.execute_input":"2022-10-14T07:28:08.720427Z","iopub.status.idle":"2022-10-14T07:28:09.080057Z","shell.execute_reply.started":"2022-10-14T07:28:08.720379Z","shell.execute_reply":"2022-10-14T07:28:09.078849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Another pass at missing sex and age values**","metadata":{}},{"cell_type":"code","source":"#let's pull out individuals from individuals_count who have fewer counts of sex or age_approx than image_name\nindividuals_count.loc[(individuals_count['sex'] < individuals_count['image_name']) | (individuals_count['age_approx'] < individuals_count['image_name'])]\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.081505Z","iopub.execute_input":"2022-10-14T07:28:09.081853Z","iopub.status.idle":"2022-10-14T07:28:09.099267Z","shell.execute_reply.started":"2022-10-14T07:28:09.081821Z","shell.execute_reply":"2022-10-14T07:28:09.097683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like just 2 patients are missing sex, and their 65 images account for all of the 65 images missing sex. These two patients are included in the set of three patients who are missing age information, the third patient having just 3 images.\n\nSo, we can't infer sex or age for the missing values from other entries of the same patient. However, if these patients don't have any malignant targets, I'm comfortable dropping them from the original dataframe. If they do, I'm comfortable filling in age and gender based on the median/mode of the overall distribution. Let's check.","metadata":{}},{"cell_type":"code","source":"pats = ['IP_0550106','IP_5205991', 'IP_9835712']\nindividuals_sum.reset_index(inplace=True)\n\n\nindividuals_sum.query('patient_id in @pats')","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.101624Z","iopub.execute_input":"2022-10-14T07:28:09.102391Z","iopub.status.idle":"2022-10-14T07:28:09.119253Z","shell.execute_reply.started":"2022-10-14T07:28:09.102355Z","shell.execute_reply":"2022-10-14T07:28:09.117857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('percentage of missing patient data= {}%'.format(round((3+48+17)/len(train)*100,2)))","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.120409Z","iopub.execute_input":"2022-10-14T07:28:09.120950Z","iopub.status.idle":"2022-10-14T07:28:09.130334Z","shell.execute_reply.started":"2022-10-14T07:28:09.120912Z","shell.execute_reply":"2022-10-14T07:28:09.129025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These particular patients don't contribute any data to the imbalanced target class. They also represent just 0.21% of the overall data. Time to say goodbye for good!","metadata":{}},{"cell_type":"code","source":"#drop all rows from the train dataframe that are associated with the 3 patients identified above.\ntrain.drop(train[(train['patient_id'] == 'IP_0550106') | (train['patient_id'] == 'IP_5205991') |(train['patient_id'] == 'IP_9835712')].index, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.137107Z","iopub.execute_input":"2022-10-14T07:28:09.137489Z","iopub.status.idle":"2022-10-14T07:28:09.158882Z","shell.execute_reply.started":"2022-10-14T07:28:09.137458Z","shell.execute_reply":"2022-10-14T07:28:09.157370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check for any more missing values:\nfor col in train.columns:\n    print(col + ' missing values: ' + str(train[col].isna().sum()))","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.160277Z","iopub.execute_input":"2022-10-14T07:28:09.160608Z","iopub.status.idle":"2022-10-14T07:28:09.180306Z","shell.execute_reply.started":"2022-10-14T07:28:09.160579Z","shell.execute_reply":"2022-10-14T07:28:09.178973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Relationships between positive cases and other variables.**","metadata":{}},{"cell_type":"code","source":"#make sure index is the patient id so we can join these aggregate dfs\n\nindividuals_sum.set_index('patient_id', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.182464Z","iopub.execute_input":"2022-10-14T07:28:09.182996Z","iopub.status.idle":"2022-10-14T07:28:09.190823Z","shell.execute_reply.started":"2022-10-14T07:28:09.182857Z","shell.execute_reply":"2022-10-14T07:28:09.189230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patients = individuals_count.join(individuals_sum, lsuffix='_count', rsuffix='_sum')\npatients.drop(columns=['sex','age_approx_count','anatom_site_general_challenge','diagnosis', 'benign_malignant','target_count',\t'age_approx_sum'], inplace=True)\npatients.rename(columns={'image_name':'n_images'}, inplace=True)\npatients.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.192211Z","iopub.execute_input":"2022-10-14T07:28:09.192635Z","iopub.status.idle":"2022-10-14T07:28:09.212147Z","shell.execute_reply.started":"2022-10-14T07:28:09.192581Z","shell.execute_reply":"2022-10-14T07:28:09.210959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#join with train df to include sex and age data for each patient\npatients = patients.join(train.set_index('patient_id'), on='patient_id')","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.213143Z","iopub.execute_input":"2022-10-14T07:28:09.213463Z","iopub.status.idle":"2022-10-14T07:28:09.238478Z","shell.execute_reply.started":"2022-10-14T07:28:09.213433Z","shell.execute_reply":"2022-10-14T07:28:09.237184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#clean it up\npatients.reset_index(inplace=True)\npatients.drop_duplicates(subset='patient_id', inplace=True)\npatients.drop(columns=['image_name','anatom_site_general_challenge', 'diagnosis',\t'benign_malignant',\t'target'], inplace=True)\npatients.head()","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.240743Z","iopub.execute_input":"2022-10-14T07:28:09.241519Z","iopub.status.idle":"2022-10-14T07:28:09.274840Z","shell.execute_reply.started":"2022-10-14T07:28:09.241477Z","shell.execute_reply":"2022-10-14T07:28:09.273825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot n_images v. target_sum\nax1 = sns.scatterplot(data=patients, x='n_images', y ='target_sum', hue='sex')\nax1.set_title('n_images per patient v. target_sum')\n#ax1.tick_params(axis='x', labelrotation = 45, labelsize = 12)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.276516Z","iopub.execute_input":"2022-10-14T07:28:09.280152Z","iopub.status.idle":"2022-10-14T07:28:09.784113Z","shell.execute_reply.started":"2022-10-14T07:28:09.280106Z","shell.execute_reply":"2022-10-14T07:28:09.782690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax1 = sns.violinplot(data=patients, y='n_images', x ='target_sum')\nax1.set_title('n_images per patient v. target_sum')\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:09.785926Z","iopub.execute_input":"2022-10-14T07:28:09.786299Z","iopub.status.idle":"2022-10-14T07:28:10.197064Z","shell.execute_reply.started":"2022-10-14T07:28:09.786268Z","shell.execute_reply":"2022-10-14T07:28:10.195894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From these plots, we can see that the number of images overall for a patient is not correlated with a higher quantity of images positive for melanoma. Surprisingly, the number of positive targets does not necessarily increase with the total quantity of pictures for a patient. Nor does gender correlate with the number of positive targets.","metadata":{}},{"cell_type":"code","source":"ax1 = sns.stripplot(data=patients, y='target_sum', x ='sex')\nax1.set_title('target_sum distribution by gender')\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:10.198485Z","iopub.execute_input":"2022-10-14T07:28:10.198824Z","iopub.status.idle":"2022-10-14T07:28:10.460411Z","shell.execute_reply.started":"2022-10-14T07:28:10.198793Z","shell.execute_reply":"2022-10-14T07:28:10.458745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot age_approx v target_sum\nax1 = sns.stripplot(data=patients, x='age_approx', y ='target_sum')\nax1.set_title('age_approx v. target_sum')\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:10.463122Z","iopub.execute_input":"2022-10-14T07:28:10.463781Z","iopub.status.idle":"2022-10-14T07:28:10.923167Z","shell.execute_reply.started":"2022-10-14T07:28:10.463711Z","shell.execute_reply":"2022-10-14T07:28:10.921586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A spectrum of target_sum values were found across all the ages included in the image dataset, with the exception of the 10year olds.","metadata":{}},{"cell_type":"code","source":"#plot n_images v. target_sum\nax1 = sns.histplot(data=patients, x='age_approx', bins=16)\nax1.set_title('Ditribution of Patient Ages in Dataset')\n","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:10.925040Z","iopub.execute_input":"2022-10-14T07:28:10.925466Z","iopub.status.idle":"2022-10-14T07:28:11.285678Z","shell.execute_reply.started":"2022-10-14T07:28:10.925430Z","shell.execute_reply":"2022-10-14T07:28:11.284027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Visualizing Images**","metadata":{}},{"cell_type":"code","source":"#open one image\n\n#import ImageIO\nimport imageio.v2 as imageio\n\n#load an image\nim = imageio.imread('../input/siim-isic-melanoma-classification/jpeg/train/ISIC_0015719.jpg')\n\n#plot image\nplt.imshow(im)\nplt.axis('off')\nplt.show()\n\n#print metadata\nprint(im.meta.keys())","metadata":{"execution":{"iopub.status.busy":"2022-10-14T08:13:03.107567Z","iopub.execute_input":"2022-10-14T08:13:03.109231Z","iopub.status.idle":"2022-10-14T08:13:07.413441Z","shell.execute_reply.started":"2022-10-14T08:13:03.109147Z","shell.execute_reply":"2022-10-14T08:13:07.411808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Plot malignant lesions**","metadata":{}},{"cell_type":"code","source":"#Now, let's see if we can show all the malignant images\n\n#make a list of image names that have the malignant target value = 1\nmal_ims = train[train['target']==1].image_name\n\n#convert those image names into path names to access each photo\n#mal_im_paths = ['../input/siim-isic-melanoma-classification/jpeg/train/' + str(im) + '.jpg' for im in mal_ims]","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:15.199825Z","iopub.execute_input":"2022-10-14T07:28:15.200285Z","iopub.status.idle":"2022-10-14T07:28:15.210032Z","shell.execute_reply.started":"2022-10-14T07:28:15.200241Z","shell.execute_reply":"2022-10-14T07:28:15.208729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot all malignant images\n\nfig, axs = plt.subplots(nrows=146, ncols=4, tight_layout=True, figsize=(15,365))\nfig.suptitle('Malignant Lesions', fontsize=14)\n\nfor ax, image in zip(axs.ravel(), mal_ims):\n    path = '../input/siim-isic-melanoma-classification/jpeg/train/' + str(image) + '.jpg'\n    im = imageio.imread(path)\n    ax.imshow(im)\n    ax.axis('off')\n    ax.set_title(image)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T07:28:15.211629Z","iopub.execute_input":"2022-10-14T07:28:15.212892Z","iopub.status.idle":"2022-10-14T07:42:07.165417Z","shell.execute_reply.started":"2022-10-14T07:28:15.212816Z","shell.execute_reply":"2022-10-14T07:42:07.162810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that the images labeled as positive for being melanoma come in different shapes and sizes--some images are round, while most are rectangular. To the naked human eye, most of these malignant images have either:\n\n* irregular edges\n* irregular color intensity throughout the legion\n* a very 'dispersed' appearance, where edges are indistinct\n\nThere are several characteristics that are different throughout the malignant images:\n\n* some have hair, some do not\n* the amount of space the lesion takes up in the image varies\n* some have a scale included, some do not\n* the brightness of the images varies\n\n(note: all the images are taken from subjects with light colored skin--this indicates that our classifier will not be trained to handle images from subjects with melanated skin, which isn't great)","metadata":{}},{"cell_type":"markdown","source":"**Plot benign lesions**\nIt would be counterproductive to plot all of the negative target images, but it would be demonstrative to display a random subset of the benign lesion images.","metadata":{}},{"cell_type":"code","source":"import random\nrandom.seed(42)\n\n#make a list of image names that have the malignant target value = 0\nben_ims = train[train['target']==0].image_name\n\n#take a random sample of 100 of these images\nrand_ben_ims = random.sample(list(ben_ims),100)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T08:12:30.242443Z","iopub.execute_input":"2022-10-14T08:12:30.243054Z","iopub.status.idle":"2022-10-14T08:12:30.274230Z","shell.execute_reply.started":"2022-10-14T08:12:30.242996Z","shell.execute_reply":"2022-10-14T08:12:30.273020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot the subset of 100 benign images\n\nfig, axs = plt.subplots(nrows=25, ncols=4, tight_layout=True, figsize=(15,62.5))\nfig.suptitle('Subset of Benign Lesions', fontsize=14, y=0.98)\n\nfor ax, image in zip(axs.ravel(), rand_ben_ims):\n    path = '../input/siim-isic-melanoma-classification/jpeg/train/' + str(image) + '.jpg'\n    im = imageio.imread(path)\n    ax.imshow(im)\n    ax.axis('off')\n    ax.set_title(image)","metadata":{"execution":{"iopub.status.busy":"2022-10-14T08:13:43.522198Z","iopub.execute_input":"2022-10-14T08:13:43.522690Z","iopub.status.idle":"2022-10-14T08:17:59.680610Z","shell.execute_reply.started":"2022-10-14T08:13:43.522653Z","shell.execute_reply":"2022-10-14T08:17:59.678966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the random subset of benign images above, I don't necessarily see any real differences from the malignant set. Some are hairy, many aren't....done are light, some are dark, many have irregular edges, and once again, all are on non-melanated skin. It's a good thing we're going to train a neural network to spot the differences between benign and malignant lesions, because I certainly cannot with the naked eye!\n\n","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}