{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport pydicom\nimport warnings\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.metrics import accuracy_score, classification_report, roc_auc_score, roc_curve, confusion_matrix\nfrom sklearn.svm import SVC","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-21T13:55:32.868609Z","iopub.execute_input":"2025-01-21T13:55:32.868993Z","iopub.status.idle":"2025-01-21T13:55:34.602266Z","shell.execute_reply.started":"2025-01-21T13:55:32.868946Z","shell.execute_reply":"2025-01-21T13:55:34.601027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Suppress warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Setting up Plotly and Seaborn aesthetics\nimport plotly.offline as py\nimport cufflinks as cf\ncf.go_offline()\ncf.set_config_file(offline=True, theme='ggplot')\nsns.set_style(\"whitegrid\")\nplt.style.use('fivethirtyeight')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-21T13:55:43.119623Z","iopub.execute_input":"2025-01-21T13:55:43.120202Z","iopub.status.idle":"2025-01-21T13:55:43.823186Z","shell.execute_reply.started":"2025-01-21T13:55:43.120170Z","shell.execute_reply":"2025-01-21T13:55:43.821863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the dataset\ninput_dir = \"/kaggle/input/siim-isic-melanoma-classification\"\ntrain_path = f\"{input_dir}/train.csv\"\ntest_path = f\"{input_dir}/test.csv\"\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Loading necessary CSV files\ntrain_csv_path = os.path.join(input_dir, \"/kaggle/input/siim-isic-melanoma-classification/train.csv\")\ntest_csv_path = os.path.join(input_dir, \"/kaggle/input/siim-isic-melanoma-classification/test.csv\")\n\n# Load data into DataFrames\ntrain_df = pd.read_csv(train_csv_path)\ntest_df = pd.read_csv(test_csv_path)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data into DataFrames\ntrain_df = pd.read_csv(train_csv_path)\ntest_df = pd.read_csv(test_csv_path)\n\n# Display basic information about the datasets\nprint(\"Training DataFrame shape:\", train_df.shape)\nprint(\"Testing DataFrame shape:\", test_df.shape)\n\nprint(\"Training DataFrame sample:\")\nprint(train_df.head())\n\nprint(\"Testing DataFrame sample:\")\nprint(test_df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Checking distribution of classes in the training data\nclass_counts = train_df['benign_malignant'].value_counts()\nprint(\"\\nClass Distribution in Training Data:\")\nprint(class_counts)\n\n# Visualization of class distribution\nplt.figure(figsize=(8, 6))\nsns.barplot(x=class_counts.index, y=class_counts.values, palette='viridis')\nplt.title(\"Class Distribution in Training Data\")\nplt.xlabel(\"Class\")\nplt.ylabel(\"Count\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display training and testing data details\nprint(\"\\nTraining DataFrame Info:\")\ntrain_df.info()\n\nprint(\"\\nTesting DataFrame Info:\")\ntest_df.info()\n\n# Display first few rows of training data\nprint(\"\\nFirst 5 rows of Training Data:\")\nprint(train_df.head())\n\n# Display first few rows of testing data\nprint(\"\\nFirst 5 rows of Testing Data:\")\nprint(test_df.head())\n\n# Check for missing values\nprint(\"\\nMissing Values in Training Data:\")\nmissing_train = train_df.isnull().sum()\nprint(missing_train)\n\nprint(\"\\nMissing Values in Testing Data:\")\nmissing_test = test_df.isnull().sum()\nprint(missing_test)\n\n# Visualize missing values\nplt.figure(figsize=(10, 6))\nsns.heatmap(train_df.isnull(), cbar=False, cmap=\"viridis\")\nplt.title(\"Missing Values in Training Data\")\nplt.show()\n\nplt.figure(figsize=(10, 6))\nsns.heatmap(test_df.isnull(), cbar=False, cmap=\"viridis\")\nplt.title(\"Missing Values in Testing Data\")\nplt.show()\n\n# Check unique patients in training data\nunique_patients_train = train_df['patient_id'].nunique()\nprint(f\"\\nNumber of unique patients in Training Data: {unique_patients_train}\")\n\n# Check unique patients in testing data\nunique_patients_test = test_df['patient_id'].nunique()\nprint(f\"Number of unique patients in Testing Data: {unique_patients_test}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4.3.2 Total Images and Unique IDs\nprint(\"\\nTotal Images and Unique IDs\")\nprint(f\"Total training images: {len(train_df)}\")\nprint(f\"Total testing images: {len(test_df)}\")\nprint(f\"Unique patient IDs in training data: {train_df['patient_id'].nunique()}\")\nprint(f\"Unique patient IDs in testing data: {test_df['patient_id'].nunique()}\")\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### 4.3.3 Exploring the Target Column\nprint(\"\\nExploring the Target Column\")\nplt.figure(figsize=(6, 6))\nsns.countplot(x='benign_malignant', data=train_df, palette='Set2')\nplt.title(\"Target Column Distribution\")\nplt.xlabel(\"Benign/Malignant\")\nplt.ylabel(\"Count\")\nplt.show()\n\n# Additional Bar Plot with Percentages\npercentages = (class_counts / class_counts.sum()) * 100\nplt.figure(figsize=(6, 6))\nax = sns.barplot(x=percentages.index, y=percentages.values, palette='pastel')\nplt.xlabel(\"Benign/Malignant\")\nplt.ylabel(\"Percentage\")\n\n# Add percentage annotations on top of bars\nfor p, percentage in zip(ax.patches, percentages):\n    ax.annotate(f'{percentage:.2f}%', (p.get_x() + p.get_width() / 2, p.get_height()), \n                ha='center', va='bottom', fontsize=13)\n\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4.3.4 Gender-wise Distribution\nprint(\"\\nGender-wise Distribution\")\n# Plot gender count\nplt.figure(figsize=(6, 6))\nsns.countplot(x='sex', data=train_df, palette='coolwarm')\nplt.title(\"Gender Distribution\")\nplt.xlabel(\"Gender\")\nplt.ylabel(\"Count\")\nplt.show()\n\n# Plot gender percentages\ngender_counts = train_df['sex'].value_counts()\ngender_percentages = (gender_counts / gender_counts.sum()) * 100\nplt.figure(figsize=(8, 6))\nax = sns.barplot(x=gender_counts.index, y=gender_percentages.values, palette='Greens')\nplt.title(\"Gender Distribution (Percentage)\")\nplt.xlabel(\"Gender\")\nplt.ylabel(\"Percentage\")\n\n# Annotate percentages on the bars\nfor p, percentage in zip(ax.patches, gender_percentages):\n    ax.annotate(f'{percentage:.2f}%', (p.get_x() + p.get_width() / 2, p.get_height()), \n                ha='center', va='bottom', fontsize=12)\n\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4.3.5 Gender with Target\nprint(\"Gender with Target Distribution\")\ngender_target = train_df.groupby(['sex', 'benign_malignant']).size().unstack()\n\n# Prepare data for plotting\ncategories = ['Benign Cases (0)', 'Malignant Cases (1)']\ngender_labels = ['Female', 'Male']\nbenign_counts = [gender_target.loc['female', 'benign'], gender_target.loc['male', 'benign']]\nmalignant_counts = [gender_target.loc['female', 'malignant'], gender_target.loc['male', 'malignant']]\n\n# Create a bar plot\nx = np.arange(len(gender_labels))  # Label locations\nwidth = 0.35  # Width of the bars\n\nfig, ax = plt.subplots(figsize=(10, 6))\n\n# Add bars for benign and malignant cases\nbars1 = ax.bar(x - width/2, benign_counts, width, label='Benign Cases (0)', color='skyblue')\nbars2 = ax.bar(x + width/2, malignant_counts, width, label='Malignant Cases (1)', color='salmon')\n\n# Add text annotations on the top of bars\nfor bar in bars1:\n    height = bar.get_height()\n    ax.annotate(f'{int(height)}',\n                xy=(bar.get_x() + bar.get_width() / 2, height),\n                xytext=(0, 3),  # Offset text by 3 units above the bar\n                textcoords=\"offset points\",\n                ha='center', va='bottom')\n\nfor bar in bars2:\n    height = bar.get_height()\n    ax.annotate(f'{int(height)}',\n                xy=(bar.get_x() + bar.get_width() / 2, height),\n                xytext=(0, 3),\n                textcoords=\"offset points\",\n                ha='center', va='bottom')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4.3.6 Location of Imaged Site\nprint(\"Location of Imaged Site\")\nplt.figure(figsize=(10, 6))\nsns.countplot(y='anatom_site_general_challenge', data=train_df, order=train_df['anatom_site_general_challenge'].value_counts().index, palette='cubehelix')\nplt.title(\"Distribution of Imaged Site\")\nplt.xlabel(\"Count\")\nplt.ylabel(\"Imaged Site\")\nplt.show()\n\n# Percentage-based bar plot\nsite_counts = train_df['anatom_site_general_challenge'].value_counts()\nsite_percentages = (site_counts / site_counts.sum()) * 100\n\n# Create a bar plot for percentages\nplt.figure(figsize=(10, 6))\nax = sns.barplot(x=site_percentages.values, y=site_percentages.index, palette='cubehelix')\nplt.title(\"Distribution of Imaged Site (Percentage)\")\nplt.xlabel(\"Percentage\")\nplt.ylabel(\"Imaged Site\")\n\n# Add percentage annotations on the bars\nfor p, percentage in zip(ax.patches, site_percentages):\n    ax.annotate(f'{percentage:.2f}%', (p.get_width(), p.get_y() + p.get_height() / 2), \n                xytext=(5, 0), textcoords=\"offset points\",\n                ha='left', va='center', fontsize=10)\n\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4.3.7 Location of Imaged Site with Respect to Gender\nprint(\"Location of Imaged Site\")\n\n# Prepare data for plotting\nsite_gender_counts = train_df.groupby(['anatom_site_general_challenge', 'sex']).size().unstack(fill_value=0)\nsite_labels = site_gender_counts.index\nfemale_counts = site_gender_counts['female']\nmale_counts = site_gender_counts['male']\n\n# Create a grouped bar plot\nx = np.arange(len(site_labels))  # Label locations\nwidth = 0.4  # Width of the bars\n\nfig, ax = plt.subplots(figsize=(12, 8))\n\n# Add bars for female and male counts\nbars1 = ax.bar(x - width/2, female_counts, width, label='Female', color='skyblue')\nbars2 = ax.bar(x + width/2, male_counts, width, label='Male', color='salmon')\n\n# Add text annotations on the top of bars\nfor bar in bars1:\n    height = bar.get_height()\n    ax.annotate(f'{int(height)}',\n                xy=(bar.get_x() + bar.get_width() / 2, height),\n                xytext=(0, 3),  # Offset text by 3 units above the bar\n                textcoords=\"offset points\",\n                ha='center', va='bottom')\n\nfor bar in bars2:\n    height = bar.get_height()\n    ax.annotate(f'{int(height)}',\n                xy=(bar.get_x() + bar.get_width() / 2, height),\n                xytext=(0, 3),\n                textcoords=\"offset points\",\n                ha='center', va='bottom')\n\n# Customize the plot\nax.set_xlabel('Location of Imaged Site')\nax.set_ylabel('Count of Melanoma Cases')\nax.set_title('Location of Imaged Site')\nax.set_xticks(x)\nax.set_xticklabels(site_labels, rotation=45, ha='right')\nax.legend()\n\n# Show the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4.3.8 Age Distribution of Patients\nprint(\"\\nAge Distribution of Patients\")\nplt.figure(figsize=(10, 6))\nsns.histplot(train_df['age_approx'], kde=True, bins=30, color='blue')\nplt.title(\"Age Distribution\")\nplt.xlabel(\"Approximate Age\")\nplt.ylabel(\"Frequency\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4.3.9 Age Distribution with Respect to Target\nprint(\"\\nAge Distribution w.r.t Target\")\nplt.figure(figsize=(10, 6))\nsns.kdeplot(data=train_df, x='age_approx', hue='benign_malignant', fill=True, common_norm=False, palette='Set2')\nplt.title(\"Age Distribution by Target\")\nplt.xlabel(\"Approximate Age\")\nplt.ylabel(\"Density\")\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4.3.10 Diagnosis Distribution\nprint(\"\\nDiagnosis Distribution\")\nplt.figure(figsize=(10, 6))\nsns.countplot(y='diagnosis', data=train_df, order=train_df['diagnosis'].value_counts().index, palette='plasma')\nplt.title(\"Distribution of Diagnosis\")\nplt.xlabel(\"Count\")\nplt.ylabel(\"Diagnosis\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}