{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-24T19:59:25.959789Z","iopub.execute_input":"2023-10-24T19:59:25.960191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport os\n\n# sns.set_style('darkgrid')\n# plt.style.use('seaborn-notebook')\n\nsns.set_style('darkgrid')\nsns.set(context='notebook', style='darkgrid', palette='deep', font_scale=1.2)\n\nplt.show","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\n\ntest = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use plt.subplots() for better control over the figure and axes\nfig, axis1 = plt.subplots(figsize=(10, 6))\n\n# Use sns.countplot() directly on the axis\nsns.countplot(data=train, x='anatom_site_general_challenge', ax=axis1)\n\n# Set title and adjust font size\naxis1.set_title('Distribution of Anatomical Sites (Training Set)', fontsize=16)\n\n# Rotate x-axis labels and adjust font size\naxis1.tick_params(axis='x', labelrotation=45, labelsize=12)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Create a countplot and set the title in one line\naxis2 = sns.countplot(data=train, x='sex').set_title('Distribution of Gender')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The data shows that the dataset has less majority of melonoma diagnosis images comapre to other diagnosis images","metadata":{}},{"cell_type":"code","source":"# Create a countplot and set title, tick parameters in one line\naxis3 = sns.countplot(data=train, x='diagnosis')\naxis3.set(\n    title='Distribution of Diagnosis',\n    xlabel='Diagnosis',  \n    ylabel='Count',      \n)\naxis3.tick_params(axis='x', labelrotation=45, labelsize=12)\n\n\n#","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"checking out on the class benign_malignant  \n*it seems malignant that is melonoma disease label is less comapre to benign","metadata":{}},{"cell_type":"code","source":"axis4 = sns.countplot(data=train, x='benign_malignant')\naxis4.set(\n    title = 'benign_malignant',\n    xlabel='Diagnosis',  \n    ylabel='Count',  \n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The below histplot shows that more affected people comes under the age 40 to 60","metadata":{}},{"cell_type":"code","source":"# Create a histogram plot and set title, tick parameters in one line\naxis5 = sns.histplot(data=train, x='age_approx', bins=18)\naxis5.set(\n    title='Distribution of Age',\n    xlabel='Age',  \n    ylabel='Count',  \n)\naxis5.tick_params(axis='x', labelrotation=45, labelsize=12)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Chceking whether the melonama diagnosis and benign_malignant coming together in any rows\n\n*but the data shows there is no correspondance between the two dataframe columns*\n","metadata":{}},{"cell_type":"code","source":"condition1 = (train['target'] == 1) & (train['benign_malignant'] != 'malignant')\ncondition2 = (train['diagnosis'] != 'melanoma') & (train['benign_malignant'] == 'malignant')\n\n# Calculate and print the counts for each condition\nprint('n rows where target != benign_malignant: {}'.format(len(train.loc[condition1])))\nprint('n rows where benign_malignant != melanoma diagnosis: {}'.format(len(train.loc[condition2])))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the number of missing values for each column in train\nfor col in train.columns:\n    missing_values = train[col].isna().sum()\n    print(f\"{col} missing values: {missing_values}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filter the DataFrame to include only observations with 'target' equal to 1\nmalignant = train[train['target'] == 1]\n\n# Print both the first few rows and the shape of the subset\nprint(malignant.head(), '\\nShape:', malignant.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in malignant.columns:\n    print(col + ' missing values: ' + str(malignant[col].isna().sum()))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill missing anatomical site values with \"other or unknown\"\ntrain['anatom_site_general_challenge'].fillna('other or unknown', inplace=True)\n\n# Verify there are no more missing values in that column\nmissing_values_after_fill = train['anatom_site_general_challenge'].isna().sum()\nprint('anatom_site_general_challenge missing values after fill: ' + str(missing_values_after_fill))\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the new distribution\nfig, axis = plt.subplots(figsize=(10, 6))\nsns.countplot(data=train, x='anatom_site_general_challenge', ax=axis)\naxis.set_title('Distribution of Anatomical Sites (Training Set)')\naxis.tick_params(axis='x', labelrotation=45, labelsize=12)\n\n# Show the plot\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test set Exploration","metadata":{}},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the number of missing values for each column in the 'test' DataFrame\nmissing_values = test.isna().sum()\nfor col, count in missing_values.items():\n    print(f\"{col} missing values: {count}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count the number of images associated with each patient\nindividuals_count = train.groupby('patient_id').count()\n\n# Display the first few rows of the result\nprint(individuals_count.head())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the size of the figure\nfig, ax = plt.subplots(figsize=(12, 6))\n\n# Create a histogram using Seaborn\nsns.histplot(data=individuals_count, x='image_name', bins=range(0, 120, 10), kde=True)\n\n# Set title and labels\nax.set_title('Distribution of Images per Patient')\nax.set_xlabel('Number of Images per Patient')\nax.set_ylabel('Number of Patients')\n\n# Show the plot\nplt.show()\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate statistics on the number of images per patient\nmean_images = round(individuals_count['image_name'].mean(), 2)\nmedian_images = round(individuals_count['image_name'].median(), 2)\nstd_dev_images = round(individuals_count['image_name'].std(), 2)\n\n# Print the results using f-strings for better readability\nprint(f\"The mean number of images per patient is: {mean_images}\")\nprint(f\"The median number of images per patient is: {median_images}\")\nprint(f\"The standard deviation of the number of images per patient is: {std_dev_images}\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Calculate the percentage of positive cases per patient\nindividuals_percentage = train.groupby('patient_id')['target'].mean() * 100\n\n# Display descriptive statistics\nprint(individuals_percentage.describe())\n\n# Plot the histogram\nsns.histplot(data=individuals_percentage, bins=10)\nplt.title('Percentage of Malignant Tumors per Patient')\nplt.xlabel('Percentage of Positives')\nplt.ylabel('Number of Patients')\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Chceking the misssing age and sex","metadata":{}},{"cell_type":"code","source":"individuals_count.loc[\n    (individuals_count['sex'] < individuals_count['image_name']) | \n    (individuals_count['age_approx'] < individuals_count['image_name'])\n]\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Convert it to a DataFrame with 'patient_id' as a column\nindividuals_percentage_df = pd.DataFrame({'patient_id': individuals_percentage.index, 'percentage': individuals_percentage.values})\n\n# Now, reset the index\nindividuals_percentage_df.reset_index(drop=True, inplace=True)\n\n# Continue with the query\npats = ['IP_0550106', 'IP_5205991', 'IP_9835712']\nselected_individuals = individuals_percentage_df.query('patient_id in @pats')\n\n# Display the result\nprint(selected_individuals)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('percentage of missing patient data= {}%'.format(round((3+48+17)/len(train)*100,2)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of patient IDs to drop\npats_to_drop = ['IP_0550106', 'IP_5205991', 'IP_9835712']\n\n# Drop rows associated with the specified patients\ntrain.drop(train[train['patient_id'].isin(pats_to_drop)].index, inplace=True)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for missing values in each column\nmissing_values = train.isna().sum()\n\n# Print the results\nfor col, count in missing_values.items():\n    print(f\"{col} missing values: {count}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Relationship between cases and variabes","metadata":{}},{"cell_type":"code","source":"selected_individuals.set_index('patient_id', inplace=True)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'individuals_count' and 'individuals_sum' are DataFrames with 'patient_id' as the index\npatients = individuals_count.join(selected_individuals, lsuffix='_count', rsuffix='_sum')\n\n# Columns to drop\ncolumns_to_drop = ['sex', 'age_approx_count', 'anatom_site_general_challenge', 'diagnosis', 'benign_malignant', 'target_count', 'age_approx_sum']\n\n# Check if the columns exist before dropping them\nexisting_columns = set(patients.columns)\ncolumns_to_drop = [col for col in columns_to_drop if col in existing_columns]\n\n# Drop the existing columns\npatients.drop(columns=columns_to_drop, inplace=True)\n\n# Rename the 'image_name' column to 'n_images'\npatients.rename(columns={'image_name': 'n_images'}, inplace=True)\n\n# Display the result\npatients.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# join with train df to include sex and age data for each patient\npatients = patients.join(train.set_index('patient_id'), on='patient_id', rsuffix='_train')\n\n# clean it up\npatients.reset_index(inplace=True)\npatients.drop_duplicates(subset='patient_id', inplace=True)\npatients.drop(columns=['image_name', 'anatom_site_general_challenge', 'diagnosis', 'benign_malignant', 'target_train'], inplace=True)\npatients.rename(columns={'age_approx_train': 'age_approx', 'sex_train': 'sex'}, inplace=True)\npatients.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reset the index\npatients.reset_index(drop=True, inplace=True)\n\n# Remove duplicate rows based on 'patient_id'\npatients.drop_duplicates(subset='patient_id', inplace=True)\n\n# Specify columns to drop\ncolumns_to_drop = ['image_name', 'anatom_site_general_challenge', 'diagnosis', 'benign_malignant', 'target']\n\n# Drop specified columns if they exist\npatients = patients.drop(columns=columns_to_drop, errors='ignore')\n\n# Display the first few rows\npatients.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the sum of 'target' for each patient\ntarget_sum_per_patient = train.groupby('patient_id')['target'].sum().reset_index()\n\n# Merge with the 'patients' DataFrame\npatients = pd.merge(patients, target_sum_per_patient, on='patient_id', how='left')\n\n# Scatter plot\nax1 = sns.scatterplot(data=patients, x='n_images', y='target', hue='sex')\nax1.set_title('n_images per patient vs. target sum')\nax1.set_xlabel('Number of Images per Patient')\nax1.set_ylabel('Sum of Targets')\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'target_sums' is the DataFrame with patient_id and target_sum columns\ntarget_sums = train.groupby('patient_id')['target'].sum().reset_index()\n\n# Merging patients DataFrame with target_sums DataFrame, specifying suffixes\npatients = patients.merge(target_sums, on='patient_id', suffixes=('_patients', '_target_sums'))\n\n# Check column names and data types\nprint(patients.columns)\nprint(patients[['n_images', 'target_patients', 'target_target_sums']].dtypes)\n\n# Violin plot\nax1 = sns.violinplot(data=patients, y='n_images', x='target_target_sums')\nax1.set_title('n_images per patient vs. target_sum')\nax1.set_xlabel('Target Sum')\nax1.set_ylabel('Number of Images per Patient')\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'target_sums' is the DataFrame with patient_id and target_sum columns\ntarget_sums = train.groupby('patient_id')['target'].sum().reset_index()\n\n# Merging patients DataFrame with target_sums DataFrame, specifying suffixes\npatients = patients.merge(target_sums, on='patient_id', suffixes=('_patients', '_target_sums'))\n\n# Check column names and data types\nprint(patients.columns)\nprint(patients[['target_patients', 'target_target_sums']].dtypes)\n\n# Strip plot\nax1 = sns.stripplot(data=patients, y='target_target_sums', x='sex')\nax1.set_title('target_sum distribution by gender')\nax1.set_xlabel('Gender')\nax1.set_ylabel('Target Sum')\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import imageio\nimport matplotlib.pyplot as plt\n\n# Load an image\nim = imageio.imread('../input/siim-isic-melanoma-classification/jpeg/train/ISIC_0015719.jpg')\n\n# Plot image\nplt.imshow(im)\nplt.axis('off')\n\n# Get metadata using imageio.get_reader\nreader = imageio.get_reader('../input/siim-isic-melanoma-classification/jpeg/train/ISIC_0015719.jpg')\nmeta_data = reader.get_meta_data()\n\n# Print metadata\nprint(meta_data)\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import imageio\nimport matplotlib.pyplot as plt\n\n# Make a list of image names that have the malignant target value = 1\nmal_ims = train[train['target'] == 1].image_name\n\n# Convert those image names into path names to access each photo\nmal_im_paths = ['../input/siim-isic-melanoma-classification/jpeg/train/' + str(im) + '.jpg' for im in mal_ims]\n\n# Plot malignant images in a grid\nrows = 10\ncols = 5\n\nfig, axs = plt.subplots(nrows=rows, ncols=cols, tight_layout=True, figsize=(15, 20))\nfig.suptitle('Malignant Lesions', fontsize=14)\n\nfor ax, path, image in zip(axs.ravel(), mal_im_paths, mal_ims):\n    im = imageio.imread(path)\n    ax.imshow(im)\n    ax.axis('off')\n    ax.set_title(image)\n\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}