{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":11848,"databundleVersionId":862157,"isSourceIdPinned":false},{"sourceType":"kernelVersion","sourceId":314622147,"isSourceIdPinned":false}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"img00md","cell_type":"markdown","source":"# Histopathologic Cancer Detection — Image EDA","metadata":{}},{"id":"img01md","cell_type":"markdown","source":"## A · Setup","metadata":{}},{"id":"img01code","cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nimport seaborn as sns\nfrom PIL import Image\nimport warnings\nwarnings.filterwarnings('ignore')\n\nsns.set_theme(style='whitegrid', font_scale=1.1)\nCLR_IMG = {0: '#2ecc71', 1: '#e74c3c'}\n\nIMG_DIR   = '/kaggle/input/competitions/histopathologic-cancer-detection/train'\nLABEL_CSV = '/kaggle/input/competitions/histopathologic-cancer-detection/train_labels.csv'\n\nlabels_df = pd.read_csv(LABEL_CSV)\nlabels_df['path'] = labels_df['id'].apply(lambda x: os.path.join(IMG_DIR, x + '.tif'))\nprint(f'Total images : {len(labels_df):,}')\nprint(f'Normal  (0)  : {(labels_df.label==0).sum():,}  ({(labels_df.label==0).mean()*100:.1f}%)')\nprint(f'Tumor   (1)  : {(labels_df.label==1).sum():,}  ({(labels_df.label==1).mean()*100:.1f}%)')\nlabels_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:43:33.887070Z","iopub.execute_input":"2026-04-30T22:43:33.887355Z","iopub.status.idle":"2026-04-30T22:43:34.525192Z","shell.execute_reply.started":"2026-04-30T22:43:33.887330Z","shell.execute_reply":"2026-04-30T22:43:34.524294Z"}},"outputs":[],"execution_count":null},{"id":"img02md","cell_type":"markdown","source":"## B · Sample Images (Normal vs Tumor)","metadata":{}},{"id":"img02code","cell_type":"code","source":"N_SHOW = 8\nnormal_paths = labels_df[labels_df.label == 0]['path'].sample(N_SHOW, random_state=42).values\ntumor_paths  = labels_df[labels_df.label == 1]['path'].sample(N_SHOW, random_state=42).values\n\nfig, axes = plt.subplots(2, N_SHOW, figsize=(18, 5))\nfor i, (npath, tpath) in enumerate(zip(normal_paths, tumor_paths)):\n    axes[0, i].imshow(Image.open(npath))\n    axes[0, i].axis('off')\n    axes[1, i].imshow(Image.open(tpath))\n    axes[1, i].axis('off')\n\naxes[0, 0].set_ylabel('Normal', fontsize=12, color=CLR_IMG[0], fontweight='bold')\naxes[1, 0].set_ylabel('Tumor',  fontsize=12, color=CLR_IMG[1], fontweight='bold')\nfig.suptitle('Random Sample: Normal vs Tumor Patches', fontsize=14, fontweight='bold')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:43:39.151092Z","iopub.execute_input":"2026-04-30T22:43:39.151524Z","iopub.status.idle":"2026-04-30T22:43:40.115323Z","shell.execute_reply.started":"2026-04-30T22:43:39.151498Z","shell.execute_reply":"2026-04-30T22:43:40.113781Z"}},"outputs":[],"execution_count":null},{"id":"img03md","cell_type":"markdown","source":"## C · Image Size & Channel Check","metadata":{}},{"id":"img03code","cell_type":"code","source":"# Check a sample of 500 images for size consistency and channel count\nsample_paths = labels_df['path'].sample(500, random_state=0).values\nsizes, channels = [], []\nfor p in sample_paths:\n    img = np.array(Image.open(p))\n    sizes.append(img.shape[:2])\n    channels.append(img.shape[2] if img.ndim == 3 else 1)\n\nunique_sizes = pd.Series(sizes).value_counts()\nunique_chans = pd.Series(channels).value_counts()\nprint('Image sizes found:')\nprint(unique_sizes.to_string())\nprint(f'\\nChannels: {unique_chans.to_dict()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:43:53.074436Z","iopub.execute_input":"2026-04-30T22:43:53.074745Z","iopub.status.idle":"2026-04-30T22:43:58.452331Z","shell.execute_reply.started":"2026-04-30T22:43:53.074724Z","shell.execute_reply":"2026-04-30T22:43:58.451597Z"}},"outputs":[],"execution_count":null},{"id":"img04md","cell_type":"markdown","source":"## D · Mean RGB per Class","metadata":{}},{"id":"img04code","cell_type":"code","source":"SAMPLE_N = 1000  # images per class\nmean_rgb = {0: [], 1: []}\n\nfor lbl in [0, 1]:\n    paths = labels_df[labels_df.label == lbl]['path'].sample(SAMPLE_N, random_state=42).values\n    for p in paths:\n        img = np.array(Image.open(p)).astype(np.float32)\n        mean_rgb[lbl].append(img.mean(axis=(0, 1)))  # shape (3,)\n\nmean_rgb = {k: np.array(v) for k, v in mean_rgb.items()}\n\nfig, axes = plt.subplots(1, 3, figsize=(13, 4))\nchannel_names = ['Red', 'Green', 'Blue']\nfor ch, (ax, ch_name) in enumerate(zip(axes, channel_names)):\n    for lbl, label_name, color in [(0, 'Normal', CLR_IMG[0]), (1, 'Tumor', CLR_IMG[1])]:\n        ax.hist(mean_rgb[lbl][:, ch], bins=40, alpha=0.6, color=color, label=label_name, density=True)\n    ax.set_title(f'{ch_name} channel')\n    ax.set_xlabel('Mean pixel value')\n    ax.legend()\n\nfig.suptitle('Mean RGB Distributions: Normal vs Tumor (1k sample each)', fontweight='bold')\nplt.tight_layout()\nplt.show()\n\nfor lbl, name in [(0, 'Normal'), (1, 'Tumor')]:\n    r, g, b = mean_rgb[lbl].mean(axis=0)\n    print(f'{name:8s} — R: {r:.1f}  G: {g:.1f}  B: {b:.1f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:44:16.041071Z","iopub.execute_input":"2026-04-30T22:44:16.041876Z","iopub.status.idle":"2026-04-30T22:44:38.920939Z","shell.execute_reply.started":"2026-04-30T22:44:16.041828Z","shell.execute_reply":"2026-04-30T22:44:38.919687Z"}},"outputs":[],"execution_count":null},{"id":"img05md","cell_type":"markdown","source":"## E · Pixel Brightness Distribution (Grayscale)","metadata":{}},{"id":"img05code","cell_type":"code","source":"brightness = {0: [], 1: []}\n\nfor lbl in [0, 1]:\n    paths = labels_df[labels_df.label == lbl]['path'].sample(SAMPLE_N, random_state=7).values\n    for p in paths:\n        img = np.array(Image.open(p).convert('L')).astype(np.float32)\n        brightness[lbl].append(img.mean())\n\nfig, ax = plt.subplots(figsize=(8, 4))\nfor lbl, label_name, color in [(0, 'Normal', CLR_IMG[0]), (1, 'Tumor', CLR_IMG[1])]:\n    ax.hist(brightness[lbl], bins=50, alpha=0.6, color=color, label=label_name, density=True)\nax.set_xlabel('Mean grayscale brightness')\nax.set_ylabel('Density')\nax.set_title('Brightness Distribution: Normal vs Tumor')\nax.legend()\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:44:55.821595Z","iopub.execute_input":"2026-04-30T22:44:55.821880Z","iopub.status.idle":"2026-04-30T22:45:16.481782Z","shell.execute_reply.started":"2026-04-30T22:44:55.821843Z","shell.execute_reply":"2026-04-30T22:45:16.480395Z"}},"outputs":[],"execution_count":null},{"id":"img06md","cell_type":"markdown","source":"## F · Mean Image per Class","metadata":{}},{"id":"img06code","cell_type":"code","source":"MEAN_N = 500\nmean_imgs = {}\n\nfor lbl in [0, 1]:\n    paths = labels_df[labels_df.label == lbl]['path'].sample(MEAN_N, random_state=99).values\n    stack = np.stack([np.array(Image.open(p)).astype(np.float32) for p in paths])\n    mean_imgs[lbl] = stack.mean(axis=0).astype(np.uint8)\n\nfig, axes = plt.subplots(1, 3, figsize=(13, 4))\naxes[0].imshow(mean_imgs[0])\naxes[0].set_title('Mean Normal patch', color=CLR_IMG[0], fontweight='bold')\naxes[0].axis('off')\n\naxes[1].imshow(mean_imgs[1])\naxes[1].set_title('Mean Tumor patch', color=CLR_IMG[1], fontweight='bold')\naxes[1].axis('off')\n\ndiff = np.abs(mean_imgs[1].astype(int) - mean_imgs[0].astype(int)).astype(np.uint8)\naxes[2].imshow(diff)\naxes[2].set_title('Absolute difference', fontweight='bold')\naxes[2].axis('off')\n\nfig.suptitle(f'Mean Image per Class ({MEAN_N} samples each)', fontweight='bold')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:45:22.923513Z","iopub.execute_input":"2026-04-30T22:45:22.923878Z","iopub.status.idle":"2026-04-30T22:45:33.096025Z","shell.execute_reply.started":"2026-04-30T22:45:22.923834Z","shell.execute_reply":"2026-04-30T22:45:33.094759Z"}},"outputs":[],"execution_count":null},{"id":"img07md","cell_type":"markdown","source":"## G · Class Balance (Images)","metadata":{}},{"id":"img07code","cell_type":"code","source":"counts = labels_df['label'].value_counts().sort_index()\nfig, ax = plt.subplots(figsize=(4, 4))\nax.bar(['Normal (0)', 'Tumor (1)'], counts.values,\n       color=[CLR_IMG[0], CLR_IMG[1]], edgecolor='white')\nfor i, v in enumerate(counts.values):\n    ax.text(i, v + 300, f'{v:,}\\n({v/len(labels_df)*100:.1f}%)', ha='center', fontsize=10)\nax.set_ylabel('Count')\nax.set_title('Class Distribution (Image Dataset)')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:45:45.014813Z","iopub.execute_input":"2026-04-30T22:45:45.015099Z","iopub.status.idle":"2026-04-30T22:45:45.119751Z","shell.execute_reply.started":"2026-04-30T22:45:45.015080Z","shell.execute_reply":"2026-04-30T22:45:45.118657Z"}},"outputs":[],"execution_count":null},{"id":"img_sep_md","cell_type":"markdown","source":"---\n# Histopathologic Cancer Detection — Features EDA","metadata":{}},{"id":"b2c3d4e5","cell_type":"markdown","source":"## 1 · Imports","metadata":{}},{"id":"c3d4e5f6","cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nimport warnings\nwarnings.filterwarnings('ignore')\n\nsns.set_theme(style='whitegrid', font_scale=1.1)\n\nFEATURE_GROUPS = {\n    'DoG Blob':  (0,  11),\n    'Color HSV': (11, 17),\n    'GLCM':      (17, 22),\n    'LBP':       (22, 32),\n    'LBGLCM':    (32, 37),\n    'GLRLM':     (37, 42),\n    'SFTA':      (42, 51),\n}\n\nCLR = {0: '#2ecc71', 1: '#e74c3c'}\nprint('Imports OK')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:45:51.396848Z","iopub.execute_input":"2026-04-30T22:45:51.397157Z","iopub.status.idle":"2026-04-30T22:45:51.404805Z","shell.execute_reply.started":"2026-04-30T22:45:51.397136Z","shell.execute_reply":"2026-04-30T22:45:51.403619Z"}},"outputs":[],"execution_count":null},{"id":"d4e5f6g7","cell_type":"markdown","source":"## 2 · Load Data","metadata":{}},{"id":"e5f6g7h8","cell_type":"code","source":"NPZ_PATH = '/kaggle/input/notebooks/amr2054/feature-extraction/extracted_features_dog_color_glcm_lbp_lbglcm_glrlm_sfta.npz'\n\ndata = np.load(NPZ_PATH, allow_pickle=True)\nX = data['X'].astype(np.float64)\ny = data['Y'].astype(int)\n\nN, D = X.shape\nfeat_names = [f'f{i}' for i in range(D)]\ndf = pd.DataFrame(X, columns=feat_names)\ndf.insert(0, 'label', y)\n\nprint(f'Shape      : {X.shape}')\nprint(f'Normal (0) : {(y==0).sum():,}  ({(y==0).mean()*100:.1f}%)')\nprint(f'Tumor  (1) : {(y==1).sum():,}  ({(y==1).mean()*100:.1f}%)')\nprint(f'NaNs       : {np.isnan(X).sum()}')\nprint(f'Infs       : {np.isinf(X).sum()}')\ndf.describe().T.round(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:45:54.972353Z","iopub.execute_input":"2026-04-30T22:45:54.972609Z","iopub.status.idle":"2026-04-30T22:45:55.926686Z","shell.execute_reply.started":"2026-04-30T22:45:54.972591Z","shell.execute_reply":"2026-04-30T22:45:55.925776Z"}},"outputs":[],"execution_count":null},{"id":"f6g7h8i9","cell_type":"markdown","source":"## 3 · Class Balance (Features)","metadata":{}},{"id":"g7h8i9j0","cell_type":"code","source":"counts = pd.Series(y).value_counts().sort_index()\n\nfig, ax = plt.subplots(figsize=(4, 4))\nax.bar(['Normal (0)', 'Tumor (1)'], counts.values,\n       color=[CLR[0], CLR[1]], edgecolor='white')\nfor i, v in enumerate(counts.values):\n    ax.text(i, v + 500, f'{v:,}\\n({v/N*100:.1f}%)', ha='center', fontsize=10)\nax.set_ylabel('Count')\nax.set_title('Class Distribution')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:46:02.058315Z","iopub.execute_input":"2026-04-30T22:46:02.058670Z","iopub.status.idle":"2026-04-30T22:46:02.168743Z","shell.execute_reply.started":"2026-04-30T22:46:02.058648Z","shell.execute_reply":"2026-04-30T22:46:02.167244Z"}},"outputs":[],"execution_count":null},{"id":"h8i9j0k1","cell_type":"markdown","source":"## 4 · Feature Distributions by Group","metadata":{}},{"id":"i9j0k1l2","cell_type":"code","source":"for group, (start, end) in FEATURE_GROUPS.items():\n    cols = feat_names[start:end]\n    n_cols = len(cols)\n    fig, axes = plt.subplots(2, (n_cols + 1) // 2, figsize=(14, 5))\n    axes = axes.flatten()\n    for ax, col in zip(axes, cols):\n        for lbl, color in CLR.items():\n            subset = df.loc[df['label'] == lbl, col]\n            ax.hist(subset, bins=40, alpha=0.6, color=color,\n                    label='Normal' if lbl == 0 else 'Tumor', density=True)\n        ax.set_title(col, fontsize=9)\n        ax.set_yticks([])\n    # hide unused subplots\n    for ax in axes[n_cols:]:\n        ax.set_visible(False)\n    handles = [plt.Rectangle((0,0),1,1, color=CLR[0], alpha=0.6),\n               plt.Rectangle((0,0),1,1, color=CLR[1], alpha=0.6)]\n    fig.legend(handles, ['Normal', 'Tumor'], loc='upper right')\n    fig.suptitle(f'{group} Features', fontweight='bold')\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:46:05.819735Z","iopub.execute_input":"2026-04-30T22:46:05.820776Z","iopub.status.idle":"2026-04-30T22:46:13.116456Z","shell.execute_reply.started":"2026-04-30T22:46:05.820740Z","shell.execute_reply":"2026-04-30T22:46:13.115309Z"}},"outputs":[],"execution_count":null},{"id":"j0k1l2m3","cell_type":"markdown","source":"## 5 · Correlation Heatmap","metadata":{}},{"id":"k1l2m3n4","cell_type":"code","source":"corr = df[feat_names].corr()\n\nfig, ax = plt.subplots(figsize=(14, 12))\nsns.heatmap(corr, cmap='coolwarm', center=0, vmin=-1, vmax=1,\n            square=True, linewidths=0, ax=ax, cbar_kws={'shrink': 0.8})\n\n# draw group boundary lines\nboundaries = [0] + [end for _, (_, end) in FEATURE_GROUPS.items()]\nfor b in boundaries:\n    ax.axhline(b, color='black', lw=1.2)\n    ax.axvline(b, color='black', lw=1.2)\n\nax.set_title('Feature Correlation Matrix', fontsize=14, fontweight='bold')\nplt.tight_layout()\nplt.show()\n\n# highly correlated pairs\nhigh_corr = (\n    corr.abs()\n    .where(np.triu(np.ones(corr.shape), k=1).astype(bool))\n    .stack()\n    .reset_index()\n    .rename(columns={'level_0': 'f1', 'level_1': 'f2', 0: 'corr'})\n    .query('corr > 0.90')\n    .sort_values('corr', ascending=False)\n)\nprint(f'Pairs with |corr| > 0.90: {len(high_corr)}')\nhigh_corr.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:46:19.769575Z","iopub.execute_input":"2026-04-30T22:46:19.769909Z","iopub.status.idle":"2026-04-30T22:46:25.047892Z","shell.execute_reply.started":"2026-04-30T22:46:19.769887Z","shell.execute_reply":"2026-04-30T22:46:25.047024Z"}},"outputs":[],"execution_count":null},{"id":"l2m3n4o5","cell_type":"markdown","source":"## 6 · Feature–Label Correlation (Separability)","metadata":{}},{"id":"m3n4o5p6","cell_type":"code","source":"label_corr = df[feat_names].corrwith(df['label']).abs().sort_values(ascending=False)\n\n# color bars by group\ngroup_color_map = {\n    'DoG Blob': '#9b59b6', 'Color HSV': '#f39c12', 'GLCM': '#2980b9',\n    'LBP': '#1abc9c', 'LBGLCM': '#16a085', 'GLRLM': '#e67e22', 'SFTA': '#c0392b'\n}\ndef feat_to_color(fname):\n    idx = int(fname[1:])\n    for grp, (s, e) in FEATURE_GROUPS.items():\n        if s <= idx < e:\n            return group_color_map[grp]\n    return 'gray'\n\ncolors = [feat_to_color(f) for f in label_corr.index]\n\nfig, ax = plt.subplots(figsize=(14, 4))\nax.bar(range(len(label_corr)), label_corr.values, color=colors)\nax.set_xticks(range(len(label_corr)))\nax.set_xticklabels(label_corr.index, rotation=90, fontsize=8)\nax.set_ylabel('|Correlation with label|')\nax.set_title('Feature Separability (Pearson |r| with label)')\n\nfrom matplotlib.patches import Patch\nlegend_elements = [Patch(facecolor=c, label=g) for g, c in group_color_map.items()]\nax.legend(handles=legend_elements, loc='upper right', fontsize=8)\nplt.tight_layout()\nplt.show()\n\nprint('Top 10 most separating features:')\nprint(label_corr.head(10).to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:46:32.775563Z","iopub.execute_input":"2026-04-30T22:46:32.775827Z","iopub.status.idle":"2026-04-30T22:46:33.312881Z","shell.execute_reply.started":"2026-04-30T22:46:32.775806Z","shell.execute_reply":"2026-04-30T22:46:33.311388Z"}},"outputs":[],"execution_count":null},{"id":"n4o5p6q7","cell_type":"markdown","source":"## 7 · PCA (2 Components)","metadata":{}},{"id":"o5p6q7r8","cell_type":"code","source":"# subsample for speed\nIDX = np.random.choice(N, size=10_000, replace=False)\nX_sub = X[IDX]\ny_sub = y[IDX]\n\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X_sub)\n\npca = PCA(n_components=2, random_state=42)\nX_pca = pca.fit_transform(X_scaled)\n\nfig, ax = plt.subplots(figsize=(7, 6))\nfor lbl, label_name, color in [(0, 'Normal', CLR[0]), (1, 'Tumor', CLR[1])]:\n    mask = y_sub == lbl\n    ax.scatter(X_pca[mask, 0], X_pca[mask, 1],\n               c=color, label=label_name, alpha=0.3, s=8, rasterized=True)\nax.set_xlabel(f'PC1 ({pca.explained_variance_ratio_[0]*100:.1f}% var)')\nax.set_ylabel(f'PC2 ({pca.explained_variance_ratio_[1]*100:.1f}% var)')\nax.set_title('PCA — 10k sample')\nax.legend(markerscale=3)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:46:57.299979Z","iopub.execute_input":"2026-04-30T22:46:57.300407Z","iopub.status.idle":"2026-04-30T22:46:57.570338Z","shell.execute_reply.started":"2026-04-30T22:46:57.300369Z","shell.execute_reply":"2026-04-30T22:46:57.569124Z"}},"outputs":[],"execution_count":null},{"id":"aa1bb2cc3","cell_type":"markdown","source":"## 8 · Outlier Check (features beyond 3σ)","metadata":{}},{"id":"aa1bb2cc4","cell_type":"code","source":"means = df[feat_names].mean()\nstds  = df[feat_names].std()\noutlier_pct = (((df[feat_names] - means).abs() > 3 * stds).sum() / N * 100).sort_values(ascending=False)\n\nfig, ax = plt.subplots(figsize=(14, 4))\ncolors = [feat_to_color(f) for f in outlier_pct.index]\nax.bar(range(len(outlier_pct)), outlier_pct.values, color=colors)\nax.axhline(1, color='black', linestyle='--', linewidth=0.8, label='1% threshold')\nax.set_xticks(range(len(outlier_pct)))\nax.set_xticklabels(outlier_pct.index, rotation=90, fontsize=8)\nax.set_ylabel('% samples beyond 3σ')\nax.set_title('Outlier Rate per Feature')\nlegend_elements = [plt.matplotlib.patches.Patch(facecolor=c, label=g) for g, c in group_color_map.items()]\nax.legend(handles=legend_elements, loc='upper right', fontsize=8)\nplt.tight_layout()\nplt.show()\n\nprint('Features with >1% outliers:')\nprint(outlier_pct[outlier_pct > 1].to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:46:40.180029Z","iopub.execute_input":"2026-04-30T22:46:40.180348Z","iopub.status.idle":"2026-04-30T22:46:40.863115Z","shell.execute_reply.started":"2026-04-30T22:46:40.180326Z","shell.execute_reply":"2026-04-30T22:46:40.861721Z"}},"outputs":[],"execution_count":null},{"id":"bb2cc3dd5","cell_type":"markdown","source":"## 9 · Per-Group Mean Comparison (Normal vs Tumor)","metadata":{}},{"id":"bb2cc3dd6","cell_type":"code","source":"group_means = df.groupby('label')[feat_names].mean().T\ngroup_means.columns = ['Normal (0)', 'Tumor (1)']\ngroup_means['diff_%'] = ((group_means['Tumor (1)'] - group_means['Normal (0)']) / (group_means['Normal (0)'].abs() + 1e-9) * 100).round(1)\ngroup_means = group_means.sort_values('diff_%', key=abs, ascending=False)\n\nprint('Top 15 features by relative mean difference (Normal vs Tumor):')\ngroup_means.head(15).style.background_gradient(subset=['diff_%'], cmap='RdYlGn', vmin=-100, vmax=100).format({'Normal (0)': '{:.3f}', 'Tumor (1)': '{:.3f}', 'diff_%': '{:.1f}%'})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:46:47.049125Z","iopub.execute_input":"2026-04-30T22:46:47.049486Z","iopub.status.idle":"2026-04-30T22:46:47.094808Z","shell.execute_reply.started":"2026-04-30T22:46:47.049458Z","shell.execute_reply":"2026-04-30T22:46:47.093760Z"}},"outputs":[],"execution_count":null},{"id":"c5fdd51c-7c78-41a2-9121-ae81c67303da","cell_type":"markdown","source":"This table shows how much each of the 51 features changes on average between Normal and Tumor patches. The **`diff_%`** value captures this relative change, where large positive values mean the feature is higher in tumor tissue, large negative values mean it is lower, and values near zero indicate little difference between the two. Features from **GLRLM**, **SFTA**, and **GLCM** generally appear at the top, suggesting they better capture the differences between healthy and cancerous tissue.","metadata":{}},{"id":"cc3dd4ee7","cell_type":"markdown","source":"## 10 · Near-Zero Variance Check","metadata":{}},{"id":"cc3dd4ee8","cell_type":"code","source":"variances = df[feat_names].var().sort_values()\nlow_var = variances[variances < variances.quantile(0.10)]\n\nfig, ax = plt.subplots(figsize=(14, 4))\ncolors_var = [feat_to_color(f) for f in variances.index]\nax.bar(range(len(variances)), variances.values, color=colors_var)\nax.set_xticks(range(len(variances)))\nax.set_xticklabels(variances.index, rotation=90, fontsize=8)\nax.set_ylabel('Variance')\nax.set_title('Feature Variance (sorted ascending)')\nlegend_elements = [plt.matplotlib.patches.Patch(facecolor=c, label=g) for g, c in group_color_map.items()]\nax.legend(handles=legend_elements, loc='upper left', fontsize=8)\nplt.tight_layout()\nplt.show()\n\nprint(f'Bottom 10% lowest-variance features ({len(low_var)} features):')\nprint(low_var.round(6).to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-30T22:46:52.130444Z","iopub.execute_input":"2026-04-30T22:46:52.130730Z","iopub.status.idle":"2026-04-30T22:46:52.705824Z","shell.execute_reply.started":"2026-04-30T22:46:52.130706Z","shell.execute_reply":"2026-04-30T22:46:52.705015Z"}},"outputs":[],"execution_count":null},{"id":"9b4764d0-3844-4fdb-a078-54b5164b6503","cell_type":"markdown","source":"This plot ranks all 51 features by their variance across the dataset, from lowest to highest. Features with near-zero variance are almost constant and don’t add useful information, so they can be removed before training. The lowest 10% are especially safe to drop, while features with very high variance may need scaling or outlier handling. Along with the correlated features identified earlier, this helps guide feature selection by removing weak features and reducing redundancy.","metadata":{}},{"id":"8f3d6f32-e99e-4c83-aa11-d9e4a39ea7dd","cell_type":"markdown","source":"## Conclusion\n\n| | Finding |\n|---|---|\n| **Class Balance** | ~60/40 split — healthy enough to train without resampling |\n| **Visual Signal** | Tumor patches are visibly pinker and denser; color and brightness shift consistently across the dataset |\n| **Best Features** | GLCM, GLRLM, and SFTA features separate the classes most clearly — confirmed by both correlation (EDA #6) and mean shift (EDA #10) |\n| **Redundancy** | Several features are near-duplicates of each other — worth pruning before training |\n| **Weak Features** | A handful of features barely vary across images at all (EDA #11) — can be dropped |\n| **Outliers** | A few features have heavy tails (EDA #9) — to be scaled robustly or capped before training |\n| **PCA** | Classes overlap in 2D, meaning the signal is spread across many dimensions — linear models will struggle, tree-based ones won't |","metadata":{}}]}