{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":11848,"databundleVersionId":862157}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport cv2\n\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import classification_report, accuracy_score, ConfusionMatrixDisplay\nimport matplotlib.pyplot as plt\nfrom math import sqrt\nfrom skimage.feature import blob_dog, hog\nfrom scipy.spatial.distance import pdist\n\ndata = pd.read_csv('/kaggle/input/competitions/histopathologic-cancer-detection/train_labels.csv')\ntrain_path = '/kaggle/input/competitions/histopathologic-cancer-detection/train/'\ntest_path = '/kaggle/input/competitions/histopathologic-cancer-detection/test/'\n\ndata['label'].value_counts()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-20T23:26:32.025547Z","iopub.execute_input":"2026-03-20T23:26:32.026559Z","iopub.status.idle":"2026-03-20T23:26:32.392675Z","shell.execute_reply.started":"2026-03-20T23:26:32.026512Z","shell.execute_reply":"2026-03-20T23:26:32.391327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_files = sorted(os.listdir(train_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T23:26:32.394302Z","iopub.execute_input":"2026-03-20T23:26:32.394624Z","iopub.status.idle":"2026-03-20T23:26:39.805311Z","shell.execute_reply.started":"2026-03-20T23:26:32.394598Z","shell.execute_reply":"2026-03-20T23:26:39.804228Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"first_file_name = training_files[0]\nfirst_image_path = os.path.join(train_path, first_file_name)\nimg_bgr = cv2.imread(first_image_path)\nimg = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n\nplt.imshow(img)\nplt.title(f\"Loaded: {first_file_name}\")\nplt.axis('off') # Hides the axes for a cleaner look\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T23:26:39.807038Z","iopub.execute_input":"2026-03-20T23:26:39.807429Z","iopub.status.idle":"2026-03-20T23:26:40.066470Z","shell.execute_reply.started":"2026-03-20T23:26:39.807402Z","shell.execute_reply":"2026-03-20T23:26:40.065510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Parameters\nnum_samples = 10000","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T23:26:40.069168Z","iopub.execute_input":"2026-03-20T23:26:40.069503Z","iopub.status.idle":"2026-03-20T23:26:40.074694Z","shell.execute_reply.started":"2026-03-20T23:26:40.069475Z","shell.execute_reply":"2026-03-20T23:26:40.073758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\nfrom math import sqrt\nfrom scipy.spatial.distance import pdist\nfrom sklearn.preprocessing import StandardScaler\n\n# --- Feature Extraction Imports ---\nfrom skimage.feature import blob_dog\nfrom skimage.feature import graycomatrix, graycoprops\nfrom skimage.feature import local_binary_pattern\n\n# ==========================================\n# 1. DoG Blob Statistics (11 Features)\n# ==========================================\ndef extract_advanced_stats(img_gray, blobs):\n    if len(blobs) == 0:\n        return np.zeros(11)\n\n    radii = blobs[:, 2] * sqrt(2)\n    count = len(blobs)\n    mean_radius = np.mean(radii)\n    std_radius = np.std(radii)\n\n    coordinates = blobs[:, 0:2]\n    if count > 1:\n        distances = pdist(coordinates, metric='euclidean')\n        mean_dist = np.mean(distances)\n        min_dist = np.min(distances)\n        std_dist = np.std(distances)\n    else:\n        mean_dist, min_dist, std_dist = 0.0, 0.0, 0.0\n\n    intensities = []\n    for y, x, r in blobs:\n        y_int, x_int = int(y), int(x)\n        if 0 <= y_int < img_gray.shape[0] and 0 <= x_int < img_gray.shape[1]:\n            intensities.append(img_gray[y_int, x_int])\n\n    if len(intensities) > 0:\n        mean_intensity = np.mean(intensities)\n        std_intensity = np.std(intensities)\n        min_intensity = np.min(intensities)\n        max_intensity = np.max(intensities)\n    else:\n        mean_intensity, std_intensity, min_intensity, max_intensity = 0.0, 0.0, 0.0, 0.0\n\n    return np.array([\n        count, mean_radius, std_radius,\n        mean_dist, min_dist, std_dist,\n        mean_intensity, std_intensity, min_intensity, max_intensity,\n        np.median(intensities) if len(intensities) > 0 else 0.0\n    ])\n\n# ==========================================\n# 2. Color Statistics (6 Features)\n# ==========================================\ndef extract_color_stats(img_bgr):\n    img_hsv = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2HSV)\n    h_mean, h_std = np.mean(img_hsv[:,:,0]), np.std(img_hsv[:,:,0])\n    s_mean, s_std = np.mean(img_hsv[:,:,1]), np.std(img_hsv[:,:,1])\n    v_mean, v_std = np.mean(img_hsv[:,:,2]), np.std(img_hsv[:,:,2])\n    return np.array([h_mean, h_std, s_mean, s_std, v_mean, v_std])\n\n# ==========================================\n# 3. GLCM Texture Features (5 Features)\n# ==========================================\ndef extract_glcm_features(img_gray):\n    glcm = graycomatrix(img_gray, distances=[1], angles=[0], levels=256, symmetric=True, normed=True)\n    contrast = graycoprops(glcm, 'contrast')[0, 0]\n    dissimilarity = graycoprops(glcm, 'dissimilarity')[0, 0]\n    homogeneity = graycoprops(glcm, 'homogeneity')[0, 0]\n    energy = graycoprops(glcm, 'energy')[0, 0]\n    correlation = graycoprops(glcm, 'correlation')[0, 0]\n    return np.array([contrast, dissimilarity, homogeneity, energy, correlation])\n\n# ==========================================\n# 4. LBP Micro-Texture Features (10 Features)\n# ==========================================\ndef extract_lbp_features(img_gray):\n    radius = 1\n    n_points = 8 * radius\n    lbp = local_binary_pattern(img_gray, n_points, radius, method='uniform')\n    n_bins = int(lbp.max() + 1)\n    hist, _ = np.histogram(lbp.ravel(), bins=n_bins, range=(0, n_bins))\n    hist = hist.astype(\"float\")\n    hist /= (hist.sum() + 1e-7)\n    return hist\n\n# ==========================================\n# 5. LBGLCM Features (5 Features)\n# ==========================================\ndef extract_lbglcm_features(img_gray):\n    radius = 1\n    n_points = 8 * radius\n    lbp_img = local_binary_pattern(img_gray, n_points, radius, method='uniform').astype(np.uint8)\n    glcm = graycomatrix(lbp_img, distances=[1], angles=[0], levels=int(lbp_img.max()+1), symmetric=True, normed=True)\n    contrast = graycoprops(glcm, 'contrast')[0, 0]\n    dissimilarity = graycoprops(glcm, 'dissimilarity')[0, 0]\n    homogeneity = graycoprops(glcm, 'homogeneity')[0, 0]\n    energy = graycoprops(glcm, 'energy')[0, 0]\n    correlation = graycoprops(glcm, 'correlation')[0, 0]\n    return np.array([contrast, dissimilarity, homogeneity, energy, correlation])\n\n# ==========================================\n# 6. GLRLM Features (5 Features)\n# ==========================================\ndef extract_glrlm_features(img_gray):\n    img = (img_gray / 16).astype(np.uint8)\n    rows, cols = img.shape\n    run_lengths = []\n    total_runs = 0\n    for row in img:\n        run_val = row[0]\n        run_len = 1\n        for val in row[1:]:\n            if val == run_val:\n                run_len += 1\n            else:\n                run_lengths.append(run_len)\n                total_runs += 1\n                run_val = val\n                run_len = 1\n        run_lengths.append(run_len)\n        total_runs += 1\n    run_lengths = np.array(run_lengths)\n    if total_runs > 0:\n        SRE = np.sum(1 / (run_lengths ** 2)) / total_runs\n        LRE = np.sum(run_lengths ** 2) / total_runs\n        RP = total_runs / (rows * cols)\n        GLN = np.var(img.flatten())\n        RLN = np.var(run_lengths)\n    else:\n        SRE = LRE = RP = GLN = RLN = 0.0\n    return np.array([SRE, LRE, GLN, RLN, RP])\n\n# ==========================================\n# 7. SFTA Features (~9 Features)\n# ==========================================\ndef extract_sfta_features(img_gray, n_levels=3):\n    thresholds = np.linspace(img_gray.min(), img_gray.max(), n_levels+2)[1:-1]\n    features = []\n    for t in thresholds:\n        binary = (img_gray > t).astype(np.uint8)\n        area = binary.sum()\n        if area > 0:\n            mean_val = img_gray[binary==1].mean()\n            std_val = img_gray[binary==1].std()\n        else:\n            mean_val = std_val = 0.0\n        features.extend([area, mean_val, std_val])\n    return np.array(features)\n\n# ==========================================\n# Main Feature Extraction Loop\n# ==========================================\nX = []\nY = []\n\nfor i in tqdm(range(10000), desc=\"Extracting all features\"):\n    file_name = training_files[i]\n    Id = file_name.rsplit(\".\", 1)[0]\n    label = data['label'][np.where(data['id'] == Id)[0].item()].item()\n    image_path = os.path.join(train_path, file_name)\n    \n    img_bgr = cv2.imread(image_path)\n    img_gray = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2GRAY)\n    \n    dog_features = extract_advanced_stats(img_gray, blob_dog(img_gray, min_sigma=2, max_sigma=15, threshold=0.1))\n    color_features = extract_color_stats(img_bgr)\n    glcm_features = extract_glcm_features(img_gray)\n    lbp_features = extract_lbp_features(img_gray)\n    lbglcm_features = extract_lbglcm_features(img_gray)\n    glrlm_features = extract_glrlm_features(img_gray)\n    sfta_features = extract_sfta_features(img_gray, n_levels=3)\n    \n    combined_fd = np.concatenate([\n        dog_features, color_features, glcm_features, lbp_features,\n        lbglcm_features, glrlm_features, sfta_features\n    ])\n    \n    X.append(combined_fd)\n    Y.append(label)\n\nX_raw = np.array(X)\nY = np.array(Y)\n\nprint(f\"\\nRaw X shape: {X_raw.shape}\")\nprint(f\"Final Y shape: {Y.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T23:26:40.075917Z","iopub.execute_input":"2026-03-20T23:26:40.076251Z","iopub.status.idle":"2026-03-20T23:35:15.702892Z","shell.execute_reply.started":"2026-03-20T23:26:40.076225Z","shell.execute_reply":"2026-03-20T23:35:15.701913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport numpy as np\nfrom skimage.color import rgb2hed\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom imblearn.over_sampling import SMOTE\nfrom collections import Counter\n\n# ==========================================\n# Phase 1: Image Preprocessing (Example of HED)\n# (You would put this inside your extraction loop)\n# ==========================================\ndef preprocess_he_image(img_bgr):\n    \"\"\"\n    Separates the H&E image into its chemical stains.\n    Returns the pure Hematoxylin channel (Nuclei) for DoG/HOG extraction.\n    \"\"\"\n    # rgb2hed expects RGB, not BGR\n    img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n    \n    # Deconvolve into Hematoxylin, Eosin, and DAB channels\n    ihc_hed = rgb2hed(img_rgb)\n    \n    # Extract just the Hematoxylin channel (Index 0)\n    # Rescale to 0-255 uint8 so DoG and HOG can process it normally\n    h_channel = ihc_hed[:, :, 0]\n    h_channel_normalized = cv2.normalize(h_channel, None, 0, 255, cv2.NORM_MINMAX, dtype=cv2.CV_8U)\n    \n    return h_channel_normalized\n\n\n# ==========================================\n# Phase 2: Dataset Preprocessing (Run this AFTER your extraction loop)\n# Assuming you have X_raw and Y from the previous steps\n# ==========================================\n\nprint(f\"Original class distribution: {Counter(Y)}\")\n\n# 1. Train/Test Split (DO THIS FIRST to prevent data leakage!)\nX_train, X_test, Y_train, Y_test = train_test_split(X_raw, Y, test_size=0.2, random_state=42, stratify=Y)\n\nprint(f\"Training set class distribution before SMOTE: {Counter(Y_train)}\")\n\n# 2. Feature Scaling\n# Fit the scaler ONLY on the training data, then transform both train and test.\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test) # Never fit the scaler on test data!\n\n# 3. Apply SMOTE to the scaled training data\n# SMOTE interpolates between minority class instances to create synthetic examples\nsmote = SMOTE(random_state=42)\nX_train_final, Y_train_final = smote.fit_resample(X_train_scaled, Y_train)\n\nprint(f\"Training set class distribution AFTER SMOTE: {Counter(Y_train_final)}\")\nprint(f\"Final X_train shape: {X_train_final.shape}\")\nprint(f\"Final X_test shape: {X_test_scaled.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T23:35:15.704320Z","iopub.execute_input":"2026-03-20T23:35:15.704911Z","iopub.status.idle":"2026-03-20T23:35:17.171797Z","shell.execute_reply.started":"2026-03-20T23:35:15.704882Z","shell.execute_reply":"2026-03-20T23:35:17.170926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"svm_clf = SVC(class_weight='balanced')\nsvm_clf.fit(X_train_final, Y_train_final)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T23:35:17.172915Z","iopub.execute_input":"2026-03-20T23:35:17.173468Z","iopub.status.idle":"2026-03-20T23:35:20.363331Z","shell.execute_reply.started":"2026-03-20T23:35:17.173428Z","shell.execute_reply":"2026-03-20T23:35:20.362365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = svm_clf.predict(X_test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T23:35:20.364565Z","iopub.execute_input":"2026-03-20T23:35:20.364919Z","iopub.status.idle":"2026-03-20T23:35:21.031950Z","shell.execute_reply.started":"2026-03-20T23:35:20.364884Z","shell.execute_reply":"2026-03-20T23:35:21.031002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_score = svm_clf.score(X_train_final, Y_train_final)\ntest_score = svm_clf.score(X_test_scaled, Y_test)\n\nprint(\"Train accuracy:\", train_score)\nprint(\"Test accuracy:\", test_score)\nprint(\"\\nAccuracy:\", accuracy_score(Y_test, y_pred))\nprint(classification_report(Y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T23:35:21.033097Z","iopub.execute_input":"2026-03-20T23:35:21.033472Z","iopub.status.idle":"2026-03-20T23:35:24.855303Z","shell.execute_reply.started":"2026-03-20T23:35:21.033446Z","shell.execute_reply":"2026-03-20T23:35:24.854334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ConfusionMatrixDisplay.from_estimator(svm_clf, X_test_scaled, Y_test)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T23:35:24.857652Z","iopub.execute_input":"2026-03-20T23:35:24.858099Z","iopub.status.idle":"2026-03-20T23:35:25.669183Z","shell.execute_reply.started":"2026-03-20T23:35:24.858071Z","shell.execute_reply":"2026-03-20T23:35:25.668180Z"}},"outputs":[],"execution_count":null}]}