{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport random\nimport warnings\nfrom skimage.feature import hog\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report, ConfusionMatrixDisplay\n\n# 1. CONFIGURATION & SETUP\n# -----------------------------------------------------------------------------\nwarnings.filterwarnings(\"ignore\") # Mute warnings\nIMG_SIZE = 256       \nPATCH_SIZE = 32      \nTRAIN_PATIENT_LIMIT = 800   # Number of patients to TRAIN on\nTEST_PATIENT_LIMIT = 200    # Number of patients to TEST on (for Accuracy/Confusion Matrix)\n\n# Install pydicom if missing\ntry:\n    import pydicom\nexcept ImportError:\n    os.system('pip install pydicom')\n    import pydicom\n\n# Paths\nlabels_path = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'\nimages_dir = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images'\n\n# =============================================================================\n# 2. FEATURE EXTRACTION ENGINE\n# =============================================================================\n\ndef get_hog_features(patch):\n    \"\"\"Extracts HOG features (Shape/Texture/Edges).\"\"\"\n    features = hog(patch, orientations=9, pixels_per_cell=(8, 8),\n                   cells_per_block=(2, 2), visualize=False, block_norm='L2-Hys')\n    return features\n\ndef process_image_patches(image, boxes=None, is_training=True):\n    \"\"\"Splits image into 32x32 patches and extracts features.\"\"\"\n    patch_data = []\n    steps = IMG_SIZE // PATCH_SIZE\n    \n    for r in range(steps):\n        for c in range(steps):\n            y_start, y_end = r*PATCH_SIZE, (r+1)*PATCH_SIZE\n            x_start, x_end = c*PATCH_SIZE, (c+1)*PATCH_SIZE\n            patch = image[y_start:y_end, x_start:x_end]\n            \n            # Features\n            hog_feats = get_hog_features(patch)\n            mean_val = np.mean(patch)\n            var_val = np.var(patch)\n            y_norm, x_norm = r / steps, c / steps\n            full_features = np.concatenate([hog_feats, [mean_val, var_val, y_norm, x_norm]])\n            \n            # Labeling\n            label = 0\n            if is_training and boxes is not None:\n                p_box = [x_start, y_start, PATCH_SIZE, PATCH_SIZE]\n                for box in boxes:\n                    gt_x, gt_y, gt_w, gt_h = box\n                    # Intersection logic\n                    x_left = max(p_box[0], gt_x)\n                    y_top = max(p_box[1], gt_y)\n                    x_right = min(p_box[0]+p_box[2], gt_x+gt_w)\n                    y_bottom = min(p_box[1]+p_box[3], gt_y+gt_h)\n                    if x_right > x_left and y_bottom > y_top:\n                        intersection = (x_right - x_left) * (y_bottom - y_top)\n                        if intersection > (PATCH_SIZE*PATCH_SIZE) * 0.15:\n                            label = 1; break\n            \n            patch_data.append((full_features, label))\n    return patch_data\n\n# =============================================================================\n# 3. DATA PREPARATION (TRAIN/TEST SPLIT)\n# =============================================================================\n\nprint(\"Mapping data and splitting into Train/Test sets...\")\ndf = pd.read_csv(labels_path)\nbox_map = {}\nfor _, row in df.iterrows():\n    pid = row['patientId']\n    if pid not in box_map: box_map[pid] = []\n    if row['Target'] == 1:\n        scale = IMG_SIZE / 1024\n        box_map[pid].append([int(row['x']*scale), int(row['y']*scale), \n                             int(row['width']*scale), int(row['height']*scale)])\n\n# Separate Healthy and Sick patients\nall_pids = list(box_map.keys())\nsick_pids = [pid for pid in all_pids if len(box_map[pid]) > 0]\nhealthy_pids = [pid for pid in all_pids if len(box_map[pid]) == 0]\n\n# Shuffle\nrandom.shuffle(sick_pids)\nrandom.shuffle(healthy_pids)\n\n# Create Training Lists (Mostly Sick to teach model features + some Healthy)\ntrain_pids = sick_pids[:TRAIN_PATIENT_LIMIT] \n\n# Create Testing Lists (50% Sick, 50% Healthy for fair evaluation)\ntest_sick = sick_pids[TRAIN_PATIENT_LIMIT : TRAIN_PATIENT_LIMIT + (TEST_PATIENT_LIMIT//2)]\ntest_healthy = healthy_pids[: (TEST_PATIENT_LIMIT//2)]\ntest_pids = test_sick + test_healthy\nrandom.shuffle(test_pids)\n\nprint(f\"Training on {len(train_pids)} patients.\")\nprint(f\"Testing on {len(test_pids)} separated patients.\")\n\n# =============================================================================\n# 4. TRAINING\n# =============================================================================\n\ndef load_training_data(pids):\n    print(\"Extracting features from Training Set...\")\n    positive_patches, negative_patches = [], []\n    \n    for i, pid in enumerate(pids):\n        dcm_path = os.path.join(images_dir, pid + '.dcm')\n        if not os.path.exists(dcm_path): continue\n        ds = pydicom.dcmread(dcm_path)\n        img = cv2.resize(ds.pixel_array, (IMG_SIZE, IMG_SIZE))\n        \n        patches = process_image_patches(img, box_map[pid], is_training=True)\n        for feats, label in patches:\n            if label == 1: positive_patches.append(feats)\n            else: negative_patches.append(feats)\n            \n    # Balance Data\n    random.shuffle(negative_patches)\n    balanced_negatives = negative_patches[:len(positive_patches)]\n    X = np.vstack(positive_patches + balanced_negatives)\n    y = np.concatenate([np.ones(len(positive_patches)), np.zeros(len(balanced_negatives))])\n    return X, y\n\nX_train, y_train = load_training_data(train_pids)\nprint(\"\\nTraining Random Forest Model...\")\nrf_model = RandomForestClassifier(n_estimators=100, max_depth=None, n_jobs=-1, random_state=42)\nrf_model.fit(X_train, y_train)\nprint(\"Training Complete!\")\n\n# =============================================================================\n# 5. EVALUATION (ACCURACY & CONFUSION MATRIX)\n# =============================================================================\n\nprint(f\"\\nEvaluating Model on {len(test_pids)} unseen patients...\")\ny_test_true = []\ny_test_pred = []\n\nSTRICT_THRESHOLD = 0.80 # Confidence threshold\n\nfor i, pid in enumerate(test_pids):\n    # 1. Ground Truth\n    is_actually_sick = 1 if len(box_map[pid]) > 0 else 0\n    y_test_true.append(is_actually_sick)\n    \n    # 2. Prediction\n    dcm_path = os.path.join(images_dir, pid + '.dcm')\n    ds = pydicom.dcmread(dcm_path)\n    img = cv2.resize(ds.pixel_array, (IMG_SIZE, IMG_SIZE))\n    \n    patches = process_image_patches(img, is_training=False)\n    X_test_feats = np.array([p[0] for p in patches])\n    probs = rf_model.predict_proba(X_test_feats)[:, 1]\n    \n    # Patient Diagnosis Logic\n    # If ANY patch in the image is > 80% confident, we mark patient as Sick\n    steps = IMG_SIZE // PATCH_SIZE\n    heatmap = probs.reshape(steps, steps)\n    heatmap_smooth = cv2.resize(heatmap, (IMG_SIZE, IMG_SIZE), interpolation=cv2.INTER_CUBIC)\n    \n    if np.max(heatmap_smooth) > STRICT_THRESHOLD:\n        y_test_pred.append(1) # Predicted Sick\n    else:\n        y_test_pred.append(0) # Predicted Healthy\n\n# --- CALCULATE METRICS ---\naccuracy = accuracy_score(y_test_true, y_test_pred)\ncm = confusion_matrix(y_test_true, y_test_pred)\n\nprint(\"\\n\" + \"=\"*40)\nprint(f\"FINAL MODEL ACCURACY: {accuracy*100:.2f}%\")\nprint(\"=\"*40)\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test_true, y_test_pred, target_names=['Healthy', 'Pneumonia']))\n\n# --- PLOT CONFUSION MATRIX ---\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', cbar=False,\n            xticklabels=['Pred: Healthy', 'Pred: Pneumonia'],\n            yticklabels=['Actual: Healthy', 'Actual: Pneumonia'])\nplt.title('Confusion Matrix (Patient Level)')\nplt.show()\n\n# =============================================================================\n# 6. VISUALIZATION (Clean Boxes)\n# =============================================================================\n\ndef visualize_prediction(pid):\n    dcm_path = os.path.join(images_dir, pid + '.dcm')\n    ds = pydicom.dcmread(dcm_path)\n    img = cv2.resize(ds.pixel_array, (IMG_SIZE, IMG_SIZE))\n    img_rgb = cv2.cvtColor(img, cv2.COLOR_GRAY2RGB)\n    \n    # Prediction\n    patches = process_image_patches(img, is_training=False)\n    X_feats = np.array([p[0] for p in patches])\n    probs = rf_model.predict_proba(X_feats)[:, 1]\n    \n    heatmap = cv2.resize(probs.reshape(IMG_SIZE//PATCH_SIZE, IMG_SIZE//PATCH_SIZE), \n                         (IMG_SIZE, IMG_SIZE), interpolation=cv2.INTER_CUBIC)\n    \n    max_conf = np.max(heatmap)\n    \n    # Draw Doctor Label (Red)\n    if len(box_map[pid]) > 0:\n        for box in box_map[pid]:\n            x, y, w, h = box\n            cv2.rectangle(img_rgb, (x, y), (x+w, y+h), (255, 0, 0), 2)\n            cv2.putText(img_rgb, \"Doctor\", (x, y-5), cv2.FONT_HERSHEY_SIMPLEX, 0.4, (255,0,0), 1)\n\n    # Draw AI Prediction (Blue)\n    if max_conf > STRICT_THRESHOLD:\n        mask = (heatmap > STRICT_THRESHOLD).astype(np.uint8) * 255\n        cnts, _ = cv2.findContours(mask, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n        for c in cnts:\n            x, y, w, h = cv2.boundingRect(c)\n            if w > 10 and h > 10:\n                cv2.rectangle(img_rgb, (x, y), (x+w, y+h), (0, 0, 255), 2)\n                cv2.putText(img_rgb, f\"AI {max_conf:.2f}\", (x, y+h+15), cv2.FONT_HERSHEY_SIMPLEX, 0.4, (0,0,255), 1)\n                \n    plt.figure(figsize=(6, 6))\n    plt.imshow(img_rgb)\n    status = \"PNEUMONIA\" if max_conf > STRICT_THRESHOLD else \"HEALTHY\"\n    plt.title(f\"AI Prediction: {status} ({max_conf:.2f})\")\n    plt.axis('off')\n    plt.show()\n\nprint(\"\\nVisualizing 3 Test Cases...\")\nfor i in range(3):\n    visualize_prediction(test_pids[i])","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-08T19:31:49.218316Z","iopub.execute_input":"2025-12-08T19:31:49.219035Z","iopub.status.idle":"2025-12-08T19:33:11.543384Z","shell.execute_reply.started":"2025-12-08T19:31:49.219005Z","shell.execute_reply":"2025-12-08T19:33:11.542249Z"}},"outputs":[],"execution_count":null}]}