{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":11848,"databundleVersionId":862157}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport random\nfrom sklearn.utils import shuffle\nfrom tqdm import tqdm_notebook\nfrom skimage import color\nfrom skimage.feature import hog\nfrom skimage import data, exposure, io\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import svm\nfrom sklearn.metrics import classification_report,accuracy_score,roc_auc_score\nfrom skimage import io, color, feature, filters, measure\nfrom skimage.util import img_as_ubyte\nfrom sklearn.ensemble import RandomForestClassifier\n\nbase_dir = '/kaggle/input/competitions/histopathologic-cancer-detection' \ntrain_dir = os.path.join(base_dir, 'train')\nlabels_csv = os.path.join(base_dir, 'train_labels.csv')\ndf = pd.read_csv(labels_csv)\ndf['label'].value_counts()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-18T14:54:03.736201Z","iopub.execute_input":"2026-03-18T14:54:03.736604Z","iopub.status.idle":"2026-03-18T14:54:04.014764Z","shell.execute_reply.started":"2026-03-18T14:54:03.736575Z","shell.execute_reply":"2026-03-18T14:54:04.013735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_files = os.listdir(train_dir)\nfirst_file_name = training_files[3]\nfirst_image_path = os.path.join(train_dir, first_file_name)\nimg_bgr = cv2.imread(first_image_path)\nimg = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)\n\nplt.imshow(img)\nplt.title(f\"Loaded: {first_file_name}\")\nplt.axis('off') # Hides the axes for a cleaner look\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T14:54:04.016379Z","iopub.execute_input":"2026-03-18T14:54:04.016742Z","iopub.status.idle":"2026-03-18T14:54:08.191676Z","shell.execute_reply.started":"2026-03-18T14:54:04.016707Z","shell.execute_reply":"2026-03-18T14:54:08.190700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T14:54:08.192997Z","iopub.execute_input":"2026-03-18T14:54:08.193353Z","iopub.status.idle":"2026-03-18T14:54:08.199571Z","shell.execute_reply.started":"2026-03-18T14:54:08.193328Z","shell.execute_reply":"2026-03-18T14:54:08.198597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fd, hog_image = hog(img, orientations=8, pixels_per_cell=(16, 16), cells_per_block=(1, 1), visualize=True, channel_axis=-1)\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(8,4), sharex=True, sharey=True)\nfig.suptitle('Image with Histogram of Oriented Gradients (HOG)', fontsize=14, fontweight='bold')\nax1.imshow(img, cmap=plt.cm.gray)\nax1.set_title('Original Image')\nax1.axis('off')\n\n# Rescale histogram for better display\nhog_image_rescaled = exposure.rescale_intensity(hog_image, in_range=(0, 10))\n\n# HOG Features\nax2.imshow(hog_image_rescaled, cmap=plt.cm.gray)\nax2.set_title('HOG Features')\nax2.axis('off')\n\nplt.tight_layout()\nplt.subplots_adjust(top=0.85)  # Adjust the layout to accommodate the title\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T14:54:08.201383Z","iopub.execute_input":"2026-03-18T14:54:08.201752Z","iopub.status.idle":"2026-03-18T14:54:08.386054Z","shell.execute_reply.started":"2026-03-18T14:54:08.201726Z","shell.execute_reply":"2026-03-18T14:54:08.385237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_features(image_path):\n    \"\"\"\n    Reads an H&E image, isolates the center, extracts Hematoxylin, \n    and computes LBP, GLCM, and Nuclei Count features.\n    \"\"\"\n    # 1. Load the raw 96x96 image\n    img = io.imread(image_path)\n    \n    # 2. Crop strictly to the center 32x32 patch (where the label applies)\n    cropped = img[32:64, 32:64]\n    \n    # 3. Color Deconvolution\n    # Mathematically separate the RGB image into Hematoxylin, Eosin, and DAB channels\n    ihc_hed = color.separate_stains(cropped, color.hed_from_rgb)\n    h_channel = ihc_hed[:, :, 0] # Index 0 is Hematoxylin (nuclei)\n    \n    # Normalize the H channel to an 8-bit integer (0-255) for texture algorithms\n    h_min, h_max = h_channel.min(), h_channel.max()\n    if h_max - h_min > 0:\n        h_norm = (h_channel - h_min) / (h_max - h_min)\n    else:\n        h_norm = h_channel\n    h_uint8 = img_as_ubyte(h_norm)\n    \n    # 4. Feature: Local Binary Patterns (LBP)\n    radius = 2\n    n_points = 8 \n    lbp = feature.local_binary_pattern(h_uint8, n_points, radius, method='uniform')\n    \n    # Convert LBP matrix into a normalized 1D histogram of patterns\n    lbp_hist, _ = np.histogram(lbp.ravel(), bins=np.arange(0, n_points + 3), range=(0, n_points + 2))\n    lbp_hist = lbp_hist.astype(\"float\") / lbp_hist.sum()\n    \n    # 5. Feature: GLCM (Texture Contrast)\n    glcm = feature.graycomatrix(h_uint8, distances=[1], angles=[0, np.pi/4, np.pi/2, 3*np.pi/4], symmetric=True, normed=True)\n    contrast = feature.graycoprops(glcm, 'contrast').ravel()\n    \n    # 6. Feature: Nuclei Count\n    thresh = filters.threshold_otsu(h_channel)\n    binary = h_channel > thresh\n    _, num_nuclei = measure.label(binary, return_num=True)\n\n    # 7. Feature: Histogram of Oriented Gradients (HOG).\n    # hog_features = feature.hog(h_uint8, orientations=8, pixels_per_cell=(8, 8),\n    #                            cells_per_block=(1, 1), visualize=False, feature_vector=True)\n    \n    # final concatenation to include the HOG array\n    features = np.concatenate((lbp_hist, contrast, [num_nuclei], \n                               # hog_features\n                              ))\n    \n    return features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T14:54:08.387089Z","iopub.execute_input":"2026-03-18T14:54:08.387613Z","iopub.status.idle":"2026-03-18T14:54:08.396697Z","shell.execute_reply.started":"2026-03-18T14:54:08.387584Z","shell.execute_reply":"2026-03-18T14:54:08.395716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_pos = df[df['label'] == 1].sample(n=500, random_state=42)\n# df_neg = df[df['label'] == 0].sample(n=500, random_state=42)\n# df_subset = pd.concat([df_pos, df_neg]).sample(frac=1, random_state=42).reset_index(drop=True)\ndf_subset = df.sample(1000,random_state=42)\nX = []\ny = []\n\n# Iterate over the subset DataFrame\nfor index, row in df_subset.iterrows():\n    img_id = row['id']\n    label = row['label']\n    \n    # Reconstruct the full path to the .tif file\n    img_path = os.path.join(train_dir, f\"{img_id}.tif\")\n    \n    try:\n        features = extract_features(img_path)\n        X.append(features)\n        y.append(label)\n    except Exception as e:\n        print(f\"Failed to process {img_id}.tif - Error: {e}\")\n        \nX = np.array(X)\ny = np.array(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T14:54:08.397919Z","iopub.execute_input":"2026-03-18T14:54:08.398371Z","iopub.status.idle":"2026-03-18T14:54:30.151434Z","shell.execute_reply.started":"2026-03-18T14:54:08.398333Z","shell.execute_reply":"2026-03-18T14:54:30.150457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Total features extracted per image: {features.shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T14:56:57.291001Z","iopub.execute_input":"2026-03-18T14:56:57.291547Z","iopub.status.idle":"2026-03-18T14:56:57.297460Z","shell.execute_reply.started":"2026-03-18T14:56:57.291517Z","shell.execute_reply":"2026-03-18T14:56:57.296321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if len(X) > 0:\n  \n    # Split into training and validation sets\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n    \n    # Train the Random Forest\n    clf = RandomForestClassifier(n_estimators=100, random_state=42)\n    clf.fit(X_train, y_train)\n    \n    # Evaluate\n    preds = clf.predict(X_test)\n    probs = clf.predict_proba(X_test)[:, 1]\n\n    print(f\"Validation Accuracy: {accuracy_score(y_test, preds):.4f}\")\n    print(f\"Validation ROC AUC:  {roc_auc_score(y_test, probs):.4f}\")\n    print(classification_report(y_test, preds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-18T14:54:30.152646Z","iopub.execute_input":"2026-03-18T14:54:30.153357Z","iopub.status.idle":"2026-03-18T14:54:30.521538Z","shell.execute_reply.started":"2026-03-18T14:54:30.153319Z","shell.execute_reply":"2026-03-18T14:54:30.520773Z"}},"outputs":[],"execution_count":null}]}