{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Import libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom scipy.stats import ttest_ind\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport matplotlib.patches as patches\nfrom PIL import Image\nimport random\n\n# Set global seeds for reproducibility\nSEED = 42\nnp.random.seed(SEED)\nrandom.seed(SEED)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:15:18.614638Z","iopub.execute_input":"2024-12-03T03:15:18.615151Z","iopub.status.idle":"2024-12-03T03:15:19.402988Z","shell.execute_reply.started":"2024-12-03T03:15:18.615097Z","shell.execute_reply":"2024-12-03T03:15:19.402040Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Prepare dataframe","metadata":{}},{"cell_type":"code","source":"# Define the file path for the directory containing training images\ntrain_images = '/kaggle/input/histopathologic-cancer-detection/train/'\n\n# Define the file path for the CSV file that contains the labels for the images\nlabel_csv = '/kaggle/input/histopathologic-cancer-detection/train_labels.csv'\n\n# Load the CSV file into a pandas DataFrame\nfull_df = pd.read_csv(label_csv)\n\n# Append the '.tif' extension to each image ID in the 'id' column\n# This is necessary because the actual image files in the directory have a '.tif' extension\nfull_df['id'] = full_df['id'] + '.tif'\n\n# Convert the 'label' column to a string type\nfull_df['label'] = full_df['label'].astype(str)\n\n# Add a new column 'image_path' to the DataFrame\nfull_df['image_path'] = full_df['id'].apply(lambda x: os.path.join(train_images, x))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:15:19.404673Z","iopub.execute_input":"2024-12-03T03:15:19.405151Z","iopub.status.idle":"2024-12-03T03:15:20.026498Z","shell.execute_reply.started":"2024-12-03T03:15:19.405116Z","shell.execute_reply":"2024-12-03T03:15:20.025536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preview dataframe\nfull_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:15:20.027956Z","iopub.execute_input":"2024-12-03T03:15:20.028296Z","iopub.status.idle":"2024-12-03T03:15:20.042354Z","shell.execute_reply.started":"2024-12-03T03:15:20.028256Z","shell.execute_reply":"2024-12-03T03:15:20.041300Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Determine distribution of labels in image dataframe","metadata":{}},{"cell_type":"code","source":"# Calculate frequency distribution\nfrequency_distribution = ((full_df.label.value_counts() / len(full_df)).to_frame() * 100)\n\n# Plotting the frequency distribution as a bar chart\nplt.figure(figsize=(6, 4))\ncolors = ['lightgreen', 'lightcoral']  # light green for benign, light red for malignant\n\n# Plotting bar chart with specified colors\nfrequency_distribution.iloc[:, 0].plot(kind='bar', color=colors)\n\n# Customizing chart\nplt.title('Frequency Distribution of Labels')\nplt.xlabel('Label')\nplt.ylabel('Percent Frequency')\nplt.xticks([0, 1], ['Benign', 'Malignant'], rotation=0)\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:15:20.045163Z","iopub.execute_input":"2024-12-03T03:15:20.045593Z","iopub.status.idle":"2024-12-03T03:15:20.296790Z","shell.execute_reply.started":"2024-12-03T03:15:20.045549Z","shell.execute_reply":"2024-12-03T03:15:20.295351Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Sample images for previewing","metadata":{}},{"cell_type":"code","source":"# Sample images with label '0' and label '1'\nsample_images_label_0 = full_df[full_df['label'] == '0'].sample(6)\nsample_images_label_1 = full_df[full_df['label'] == '1'].sample(6)\n\n# Combine the samples into a single DataFrame\nsample_images = pd.concat([sample_images_label_0, sample_images_label_1])\n\n# Set up the figure and axes\nfig, axes = plt.subplots(2, 6, figsize=(12, 4))\nfig.tight_layout(pad=1.0)\n\n# Loop through sampled images and mark feature area.\nfor i, ax in enumerate(axes.flat):\n    # Get the image and label\n    id = sample_images.iloc[i]['id']\n    label = sample_images.iloc[i]['label']\n    \n    # Load the image from file\n    img = mpimg.imread(os.path.join(train_images, id))\n\n    # Display the image\n    ax.imshow(img, cmap='gray')\n    ax.axis('off')\n\n    # Highlight the center 32x32 area\n    h, w = img.shape[:2]\n    center_x, center_y = w // 2, h // 2\n    rect_x, rect_y = center_x - 16, center_y - 16\n\n    # Correctly determine the box color based on label\n    box_color = 'green' if label == '0' else 'red'\n\n    rect = patches.Rectangle(\n        (rect_x, rect_y), 32, 32,\n        linewidth=2,\n        edgecolor=box_color,\n        linestyle='dotted',\n        facecolor='none'\n    )\n    ax.add_patch(rect)\n\n    # Set title for the image\n    ax.set_title(f\"Label: {label}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:15:20.298233Z","iopub.execute_input":"2024-12-03T03:15:20.298577Z","iopub.status.idle":"2024-12-03T03:15:21.795744Z","shell.execute_reply.started":"2024-12-03T03:15:20.298544Z","shell.execute_reply":"2024-12-03T03:15:21.794560Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. View histogram of RGB channel intensities for benign or malignant","metadata":{}},{"cell_type":"code","source":"# Filter DataFrame for benign and malignant samples\nbenign_df = full_df[full_df['label'] == '0']  # Label 0 = benign\nmalignant_df = full_df[full_df['label'] == '1']  # Label 1 = malignant\n\ndef load_image(image_path, target_size=(96, 96)):\n    \"\"\"\n    Load a single image, resize it, and convert to a NumPy array.\n    \"\"\"\n    try:\n        img = Image.open(image_path)  # Load the image\n        img = img.resize(target_size)  # Resize\n        img_array = np.array(img)  # Convert to NumPy array\n        return img_array\n    except Exception as e:\n        print(f\"Error loading image: {image_path}\")\n        print(e)\n        return None  # Return None if loading fails\n\ndef load_images(df, max_images=None):\n    \"\"\"\n    Load multiple images based on a DataFrame of paths.\n    \"\"\"\n    images = []\n    for path in df['image_path'][:max_images]:\n        img = load_image(path)\n        if img is not None:\n            images.append(img)\n    return np.array(images)\n\n# Limit the number of images to load for testing\nbenign_images = load_images(benign_df, max_images=100)\nmalignant_images = load_images(malignant_df, max_images=100)\n\n# Check the shapes of the loaded data\nprint(f\"Benign images shape: {benign_images.shape}\")\nprint(f\"Malignant images shape: {malignant_images.shape}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:15:21.797215Z","iopub.execute_input":"2024-12-03T03:15:21.797697Z","iopub.status.idle":"2024-12-03T03:15:22.262447Z","shell.execute_reply.started":"2024-12-03T03:15:21.797660Z","shell.execute_reply":"2024-12-03T03:15:22.261389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nr_of_bins = 256  # Each possible pixel value (0-255) gets a bin\nfig, axs = plt.subplots(4, 2, sharey=True, figsize=(8, 8), dpi=120)  # Create a 4x2 plot grid\nrgb_list = [\"Red\", \"Green\", \"Blue\", \"RGB\"]\n\n# Loop through rows (RGB channels and combined) and columns (benign and malignant)\nfor row_idx in range(0, 4):\n    for col_idx in range(0, 2):\n        if row_idx < 3:\n            axs[row_idx, 0].set_ylabel(\"Relative Frequency\")\n            axs[row_idx, 1].set_ylabel(rgb_list[row_idx], rotation=\"horizontal\",\n                                       labelpad=35, fontsize=12)\n            # Plot histograms for RGB channels\n            if col_idx == 0:  # Benign (label 0)\n                axs[row_idx, 0].hist(benign_images[:, :, :, row_idx].flatten(),\n                                     bins=nr_of_bins, density=True, color='green', alpha=0.7)\n            elif col_idx == 1:  # Malignant (label 1)\n                axs[row_idx, 1].hist(malignant_images[:, :, :, row_idx].flatten(),\n                                     bins=nr_of_bins, density=True, color='red', alpha=0.7)\n        else:\n            # Plot histograms for combined RGB intensities\n            if col_idx == 0:  # Benign (label 0)\n                axs[row_idx, 0].hist(benign_images.flatten(),\n                                     bins=nr_of_bins, density=True, color='green', alpha=0.7)\n            elif col_idx == 1:  # Malignant (label 1)\n                axs[row_idx, 1].hist(malignant_images.flatten(),\n                                     bins=nr_of_bins, density=True, color='red', alpha=0.7)\n\n# Add titles for columns\naxs[0, 0].set_title(\"Benign (Label 0)\")\naxs[0, 1].set_title(\"Malignant (Label 1)\")\n\n# Show the plots\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:15:22.263791Z","iopub.execute_input":"2024-12-03T03:15:22.264269Z","iopub.status.idle":"2024-12-03T03:15:26.238092Z","shell.execute_reply.started":"2024-12-03T03:15:22.264202Z","shell.execute_reply":"2024-12-03T03:15:26.237047Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Conclusion:  There appears to be a shift in pixel intensity distribution between the benign and malignant channels.","metadata":{}},{"cell_type":"markdown","source":"## 6. Statistical analysis comparing channel intensity between benign and malignant channels","metadata":{}},{"cell_type":"code","source":"def calculate_statistics_with_significance(benign_images, malignant_images):\n\n    stats = []\n    channels = [\"Red\", \"Green\", \"Blue\"]\n    \n    for i, channel in enumerate(channels):\n        # Flatten pixel intensities for the current channel\n        benign_flattened = benign_images[:, :, :, i].flatten()\n        malignant_flattened = malignant_images[:, :, :, i].flatten()\n        \n        # Calculate descriptive statistics for benign\n        benign_mean = benign_flattened.mean()\n        benign_median = np.median(benign_flattened)\n        benign_std = benign_flattened.std()\n        benign_min = benign_flattened.min()\n        benign_max = benign_flattened.max()\n        \n        # Calculate descriptive statistics for malignant\n        malignant_mean = malignant_flattened.mean()\n        malignant_median = np.median(malignant_flattened)\n        malignant_std = malignant_flattened.std()\n        malignant_min = malignant_flattened.min()\n        malignant_max = malignant_flattened.max()\n        \n        # Perform a two-sample t-test\n        t_stat, p_value = ttest_ind(benign_flattened, malignant_flattened, equal_var=False)\n        \n        # Append results to the stats list\n        stats.append({\n            \"Channel\": channel,\n            \"Benign Mean\": benign_mean,\n            \"Benign Median\": benign_median,\n            \"Benign Std Dev\": benign_std,\n            \"Malignant Mean\": malignant_mean,\n            \"Malignant Median\": malignant_median,\n            \"Malignant Std Dev\": malignant_std,\n            \"t-statistic\": t_stat,\n            \"p-value\": p_value\n        })\n    \n    # Convert stats list to a DataFrame\n    return pd.DataFrame(stats)\n\n# Compute statistics for all channels with significance\nstats_with_significance = calculate_statistics_with_significance(benign_images, malignant_images)\n\n# Display the statistics with t-tests\nprint(\"Statistics and T-Test Results for Benign vs Malignant Images\")\nprint(stats_with_significance)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:15:26.239321Z","iopub.execute_input":"2024-12-03T03:15:26.239640Z","iopub.status.idle":"2024-12-03T03:15:26.422409Z","shell.execute_reply.started":"2024-12-03T03:15:26.239609Z","shell.execute_reply":"2024-12-03T03:15:26.421368Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Conclusion:  Based on these results, only the green channel has a statistical difference in pixel value intensities.","metadata":{}},{"cell_type":"markdown","source":"## 7.  Overlay of green channel pixel intensity","metadata":{}},{"cell_type":"code","source":"# Compare Green channel pixel distributions\n\nplt.figure(figsize=(10, 6))\nplt.hist(benign_images[:, :, :, 1].flatten(), bins=50, alpha=0.5, label=\"Benign (Green)\", color=\"green\")\nplt.hist(malignant_images[:, :, :, 1].flatten(), bins=50, alpha=0.5, label=\"Malignant (Green)\", color=\"lime\")\nplt.title(\"Green Channel Pixel Intensity Distribution\")\nplt.xlabel(\"Pixel Intensity\")\nplt.ylabel(\"Frequency\")\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T03:15:26.424122Z","iopub.execute_input":"2024-12-03T03:15:26.424521Z","iopub.status.idle":"2024-12-03T03:15:26.865007Z","shell.execute_reply.started":"2024-12-03T03:15:26.424477Z","shell.execute_reply":"2024-12-03T03:15:26.863763Z"}},"outputs":[],"execution_count":null}]}