{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### **[STEP 1] ENVIRONMENT SETUP & DATA EXTRACTION**","metadata":{}},{"cell_type":"code","source":"\"\"\"\n===============================================================================\n[STEP 1] ENVIRONMENT SETUP & DATA EXTRACTION\n-------------------------------------------------------------------------------\n- Gerekli dosya/klasörlerin varlık kontrolü\n- train_labels.csv yükleme ve temel özet\n- Eksik değer ve duplicate kontrolü\n- İlk satırlar + label dağılımı (hızlı görünüm)\n===============================================================================\n\"\"\"\n\nfrom pathlib import Path\nimport pandas as pd\n\ntry:\n    PRIMARY_COLOR, HEADER_STYLE, TITLE_STYLE\nexcept NameError:\n    PRIMARY_COLOR = '#3371ab'\n    HEADER_STYLE = [{\n        'selector': 'thead th',\n        'props': [\n            ('background-color', PRIMARY_COLOR),\n            ('color', 'white'),\n            ('font-weight', 'bold'),\n            ('text-align', 'center')\n        ]\n    }]\n    TITLE_STYLE = f\"color:{PRIMARY_COLOR}; font-family:Segoe UI; margin-top:20px;\"\n\ndef display_section_title(title: str):\n    from IPython.display import HTML, display\n    display(HTML(f\"<h2 style='{TITLE_STYLE}'>{title}</h2>\"))\n\ndef display_dataframe(df: pd.DataFrame, title: str = None):\n    from IPython.display import display\n    if title is not None:\n        display_section_title(title)\n    display(df.style.set_table_styles(HEADER_STYLE))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:37:19.516777Z","iopub.execute_input":"2025-12-17T16:37:19.516984Z","iopub.status.idle":"2025-12-17T16:37:21.883381Z","shell.execute_reply.started":"2025-12-17T16:37:19.516966Z","shell.execute_reply":"2025-12-17T16:37:21.882535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nBASE_DIR = Path(\"/kaggle/input/histopathologic-cancer-detection\")\nTRAIN_DIR = BASE_DIR / \"train\"\nTEST_DIR = BASE_DIR / \"test\"\nLABELS_CSV = BASE_DIR / \"train_labels.csv\"\nSAMPLE_SUB_CSV = BASE_DIR / \"sample_submission.csv\"\n\nREQUIRED = {\n    \"train_dir\": TRAIN_DIR,\n    \"test_dir\": TEST_DIR,\n    \"labels_csv\": LABELS_CSV,\n    \"sample_sub_csv\": SAMPLE_SUB_CSV,\n}\n\nmissing = [name for name, path in REQUIRED.items() if not Path(path).exists()]\ndisplay_section_title(\"📂 Dosya & Dizin Varlık Kontrolü\")\nfor name, path in REQUIRED.items():\n    print(f\"{name:15s} -> {path} | {'OK' if Path(path).exists() else 'MISSING'}\")\n\nif missing:\n    raise FileNotFoundError(\n        f\"Zorunlu yol(lar) bulunamadı: {', '.join(missing)}.\\n\"\n        f\"Lütfen veri seti yapısını doğrulayın. BASE_DIR: {BASE_DIR}\"\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:37:21.884822Z","iopub.execute_input":"2025-12-17T16:37:21.885243Z","iopub.status.idle":"2025-12-17T16:37:21.895663Z","shell.execute_reply.started":"2025-12-17T16:37:21.885224Z","shell.execute_reply":"2025-12-17T16:37:21.895008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🧾 Etiket Dosyası Yükleme & Temel Özet\")\nlabels = pd.read_csv(LABELS_CSV)\n\nrequired_cols = {\"id\", \"label\"}\nif not required_cols.issubset(labels.columns):\n    raise ValueError(\n        f\"{LABELS_CSV.name} kolonları eksik. \"\n        f\"Gerekli: {required_cols}, mevcut: {set(labels.columns)}\"\n    )\n\nsummary_df = pd.DataFrame({\n    \"metric\": [\n        \"satır_sayısı\",\n        \"sütunlar\",\n        \"toplam_eksik_değer\",\n        \"benzersiz_id_sayısı\",\n        \"duplicate_id_sayısı\",\n        \"label_sınıf_sayısı\"\n    ],\n    \"value\": [\n        len(labels),\n        \", \".join(map(str, labels.columns)),\n        int(labels.isnull().sum().sum()),\n        labels[\"id\"].nunique(),\n        int(labels[\"id\"].duplicated().sum()),\n        labels[\"label\"].nunique()\n    ]\n})\n\ndisplay_dataframe(summary_df, title=\"train_labels.csv — Hızlı Özet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:37:21.896385Z","iopub.execute_input":"2025-12-17T16:37:21.896570Z","iopub.status.idle":"2025-12-17T16:37:22.670468Z","shell.execute_reply.started":"2025-12-17T16:37:21.896554Z","shell.execute_reply":"2025-12-17T16:37:22.669838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"na_by_col = labels.isnull().sum()\nna_df = na_by_col.reset_index()\nna_df.columns = [\"column\", \"missing_count\"]\ndisplay_dataframe(na_df, title=\"Eksik Değerler (Sütun Bazında)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:37:22.671195Z","iopub.execute_input":"2025-12-17T16:37:22.671558Z","iopub.status.idle":"2025-12-17T16:37:22.693791Z","shell.execute_reply.started":"2025-12-17T16:37:22.671537Z","shell.execute_reply":"2025-12-17T16:37:22.692990Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_dataframe(labels.head(10), title=\"İlk 10 Satır (Örnek)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:37:22.695419Z","iopub.execute_input":"2025-12-17T16:37:22.695767Z","iopub.status.idle":"2025-12-17T16:37:22.715114Z","shell.execute_reply.started":"2025-12-17T16:37:22.695748Z","shell.execute_reply":"2025-12-17T16:37:22.714409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_counts = labels[\"label\"].value_counts(dropna=False).rename_axis(\"label\").reset_index(name=\"count\")\nlabel_counts[\"ratio\"] = (label_counts[\"count\"] / len(labels)).round(4)\ndisplay_dataframe(label_counts, title=\"Label Dağılımı (Tablo)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:37:22.715991Z","iopub.execute_input":"2025-12-17T16:37:22.716251Z","iopub.status.idle":"2025-12-17T16:37:22.739106Z","shell.execute_reply.started":"2025-12-17T16:37:22.716227Z","shell.execute_reply":"2025-12-17T16:37:22.738343Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **[STEP 2] IMAGE FILE STRUCTURE VALIDATION**","metadata":{}},{"cell_type":"code","source":"\"\"\"\n===============================================================================\n[STEP 2] IMAGE FILE STRUCTURE VALIDATION\n-------------------------------------------------------------------------------\nAmaç:\n  - train klasöründeki .tif dosya sayısını kontrol etmek\n  - CSV'deki id'lerle birebir eşleşme sağlanıyor mu?\n  - Görsel boyutlarının 96x96 olup olmadığını doğrulamak\n===============================================================================\n\"\"\"\n\nimport os\nfrom pathlib import Path\nfrom PIL import Image\nimport numpy as np\nfrom tqdm.auto import tqdm\nimport pandas as pd\n\ndisplay_section_title(\"🧩 Görsel Dosya Sayısı & Eşleşme Kontrolü\")\n\n\ntrain_files = [f for f in os.listdir(TRAIN_DIR) if f.endswith(\".tif\")]\ntest_files = [f for f in os.listdir(TEST_DIR) if f.endswith(\".tif\")]\n\nn_train_files = len(train_files)\nn_test_files = len(test_files)\nn_csv_ids = len(labels)\n\nsummary_counts = pd.DataFrame({\n    \"dataset\": [\"train (klasör)\", \"test (klasör)\", \"train_labels.csv\"],\n    \"count\": [n_train_files, n_test_files, n_csv_ids]\n})\ndisplay_dataframe(summary_counts, title=\"Veri Seti Eleman Sayıları\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:37:22.742004Z","iopub.execute_input":"2025-12-17T16:37:22.742340Z","iopub.status.idle":"2025-12-17T16:37:31.249689Z","shell.execute_reply.started":"2025-12-17T16:37:22.742304Z","shell.execute_reply":"2025-12-17T16:37:31.249118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_file_ids = set([f.replace(\".tif\", \"\") for f in train_files])\ncsv_ids = set(labels[\"id\"].astype(str))\n\nmissing_in_folder = csv_ids - train_file_ids\nmissing_in_csv = train_file_ids - csv_ids\n\nprint(f\"📦 train klasöründeki toplam görsel sayısı: {n_train_files:,}\")\nprint(f\"🧾 train_labels.csv kayıt sayısı: {n_csv_ids:,}\")\nprint(f\"🧮 Eşleşmeyen kayıt (CSV'de var, klasörde yok): {len(missing_in_folder)}\")\nprint(f\"🧮 Eşleşmeyen kayıt (klasörde var, CSV'de yok): {len(missing_in_csv)}\")\n\nif len(missing_in_folder) == 0 and len(missing_in_csv) == 0:\n    print(\"✅ Tüm görseller CSV ile birebir eşleşiyor.\")\nelse:\n    print(\"⚠️ Eşleşme sorunu tespit edildi — detaylı kontrol önerilir.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:37:31.250526Z","iopub.execute_input":"2025-12-17T16:37:31.250779Z","iopub.status.idle":"2025-12-17T16:37:31.410987Z","shell.execute_reply.started":"2025-12-17T16:37:31.250755Z","shell.execute_reply":"2025-12-17T16:37:31.410173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🖼️ Görsel Boyut Doğrulaması (örnek 10000 dosya)\")\n\nsample_files = np.random.choice(train_files, size=min(10000, len(train_files)), replace=False)\nsize_records = []\n\nfor fname in tqdm(sample_files, desc=\"Boyut kontrolü\"):\n    try:\n        with Image.open(TRAIN_DIR / fname) as img:\n            size_records.append(img.size)\n    except Exception as e:\n        size_records.append((\"ERROR\", \"ERROR\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:37:31.412325Z","iopub.execute_input":"2025-12-17T16:37:31.412602Z","iopub.status.idle":"2025-12-17T16:39:14.578734Z","shell.execute_reply.started":"2025-12-17T16:37:31.412584Z","shell.execute_reply":"2025-12-17T16:39:14.578038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"size_df = pd.DataFrame(size_records, columns=[\"width\", \"height\"])\nsize_summary = size_df.value_counts().reset_index(name=\"count\")\n\ndisplay_dataframe(size_summary, title=\"Görsel Boyut Dağılımı\")\n\nif len(size_summary) == 1 and tuple(size_summary.iloc[0][[\"width\", \"height\"]]) == (96, 96):\n    print(\"✅ Tüm görseller 96×96 boyutunda (örneklem bazında doğrulandı).\")\nelse:\n    print(\"⚠️ Boyut farklılıkları veya bozuk dosyalar mevcut olabilir.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:14.579586Z","iopub.execute_input":"2025-12-17T16:39:14.579831Z","iopub.status.idle":"2025-12-17T16:39:14.606551Z","shell.execute_reply.started":"2025-12-17T16:39:14.579808Z","shell.execute_reply":"2025-12-17T16:39:14.605755Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **[STEP 3] SAMPLE IMAGE VISUALIZATION**","metadata":{}},{"cell_type":"code","source":"\"\"\"\n===============================================================================\n[STEP 3] SAMPLE IMAGE VISUALIZATION\n-------------------------------------------------------------------------------\nAmaç:\n  - Pozitif ve negatif sınıflardan dengeli örnekler seçmek\n  - 10 örnek (5 pozitif, 5 negatif) görseli 2×5 grid olarak göstermek\n  - Görsellerin ID ve label bilgilerini görsel altında belirtmek\n===============================================================================\n\"\"\"\n\nimport random\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\npositive_ids = labels[labels[\"label\"] == 1][\"id\"].sample(5, random_state=42).tolist()\nnegative_ids = labels[labels[\"label\"] == 0][\"id\"].sample(5, random_state=42).tolist()\nsample_ids = positive_ids + negative_ids\nrandom.shuffle(sample_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:14.607319Z","iopub.execute_input":"2025-12-17T16:39:14.607913Z","iopub.status.idle":"2025-12-17T16:39:14.632748Z","shell.execute_reply.started":"2025-12-17T16:39:14.607891Z","shell.execute_reply":"2025-12-17T16:39:14.632193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🖼️ Görsel Örneklerin İncelenmesi (Pozitif & Negatif)\")\nfig, axes = plt.subplots(2, 5, figsize=(15, 6))\nfig.patch.set_facecolor('white')\naxes = axes.flatten()\n\nfor i, (ax, img_id) in enumerate(zip(axes, sample_ids)):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n\n    try:\n        with Image.open(img_path) as img:\n            img = np.array(img)\n            ax.imshow(img)\n    except Exception as e:\n        ax.text(0.5, 0.5, \"Error\\nLoading\", ha=\"center\", va=\"center\", fontsize=9, color=\"red\")\n\n    # Etiket bilgileri\n    label_value = labels.loc[labels[\"id\"] == img_id, \"label\"].values[0]\n    label_text = \"Cancer\" if label_value == 1 else \"Normal\"\n    label_color = PRIMARY_COLOR if label_value == 1 else \"#808080\"\n\n    # Görsel başlığı ve ID alt yazısı\n    ax.set_title(label_text, fontsize=11, weight=\"bold\", color=label_color, pad=6)\n    ax.text(\n        0.5, -0.12,\n        f\"ID: {img_id[:12]}...\",\n        ha=\"center\",\n        va=\"center\",\n        fontsize=8,\n        color=\"#444444\",\n        transform=ax.transAxes\n    )\n\n    ax.spines[:].set_visible(False)\n    ax.set_xticks([])\n    ax.set_yticks([])\n\nfig.suptitle(\n    \"Pozitif & Negatif Örnek Görseller\",\n    fontsize=15,\n    color=PRIMARY_COLOR,\n    weight=\"bold\",\n    y=1.03\n)\nplt.subplots_adjust(wspace=0.25, hspace=0.45)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:14.633412Z","iopub.execute_input":"2025-12-17T16:39:14.633672Z","iopub.status.idle":"2025-12-17T16:39:15.563297Z","shell.execute_reply.started":"2025-12-17T16:39:14.633652Z","shell.execute_reply":"2025-12-17T16:39:15.562353Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **[STEP 4] IMAGE QUALITY & STABILITY ANALYSIS**","metadata":{}},{"cell_type":"code","source":"\"\"\"\n===============================================================================\n[STEP 4] IMAGE QUALITY & STABILITY ANALYSIS\n-------------------------------------------------------------------------------\nAmaç:\n  - Bozuk / açılamayan .tif dosyalarını tespit etmek\n  - Görsellerin ortalama parlaklık, varyans, kontrast ölçümlerini yapmak\n  - Focus (blur) skorlarını sınıflara göre karşılaştırmak\n  - Histogram örnekleriyle kalite farklarını gözlemlemek\n===============================================================================\n\"\"\"\n\nimport cv2\nfrom tqdm.auto import tqdm\nimport seaborn as sns\n\ndef analyze_image_quality(img_path):\n    \"\"\"\n    Tek bir görselin parlaklık, varyans ve blur skorunu döndürür.\n    \"\"\"\n    try:\n        img = np.array(Image.open(img_path).convert(\"L\"))  # grayscale\n        brightness = np.mean(img)\n        variance = np.var(img)\n        blur = cv2.Laplacian(img, cv2.CV_64F).var()\n        return brightness, variance, blur\n    except Exception:\n        return np.nan, np.nan, np.nan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:15.564163Z","iopub.execute_input":"2025-12-17T16:39:15.564549Z","iopub.status.idle":"2025-12-17T16:39:16.743947Z","shell.execute_reply.started":"2025-12-17T16:39:15.564519Z","shell.execute_reply":"2025-12-17T16:39:16.743345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🧠 Görsel Kalite & İstikrar Analizi\")\nSAMPLE_SIZE = min(1500, len(labels))\nsample_df = labels.sample(SAMPLE_SIZE, random_state=42).reset_index(drop=True)\n\nbrightness_list, variance_list, blur_list = [], [], []\n\nfor img_id in tqdm(sample_df[\"id\"], desc=\"Kalite ölçümü\"):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    b, v, bl = analyze_image_quality(img_path)\n    brightness_list.append(b)\n    variance_list.append(v)\n    blur_list.append(bl)\n\nsample_df[\"brightness\"] = brightness_list\nsample_df[\"variance\"] = variance_list\nsample_df[\"blur\"] = blur_list\n\nvalid_samples = sample_df.dropna()\nn_broken = SAMPLE_SIZE - len(valid_samples)\n\nprint(f\"📁 Örneklenen görsel sayısı: {SAMPLE_SIZE}\")\nprint(f\"❌ Bozuk veya açılamayan görsel sayısı: {n_broken}\")\nprint(f\"✅ Başarıyla analiz edilen: {len(valid_samples)}\")\n\ndisplay_dataframe(valid_samples.head(10), title=\"Kalite Ölçümü — İlk 10 Satır\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:16.744637Z","iopub.execute_input":"2025-12-17T16:39:16.744939Z","iopub.status.idle":"2025-12-17T16:39:33.181301Z","shell.execute_reply.started":"2025-12-17T16:39:16.744914Z","shell.execute_reply":"2025-12-17T16:39:33.180720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"group_summary = (\n    valid_samples.groupby(\"label\")[[\"brightness\", \"variance\", \"blur\"]]\n    .agg([\"mean\", \"std\"])\n    .round(2)\n)\ndisplay_dataframe(group_summary, title=\"Sınıf Bazında Parlaklık / Kontrast / Blur Ortalamaları\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:33.181968Z","iopub.execute_input":"2025-12-17T16:39:33.182406Z","iopub.status.idle":"2025-12-17T16:39:33.200841Z","shell.execute_reply.started":"2025-12-17T16:39:33.182389Z","shell.execute_reply":"2025-12-17T16:39:33.200293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_metric_distribution(df, metric, color):\n    plt.figure(figsize=(6, 3.5))\n    sns.kdeplot(data=df, x=metric, hue=\"label\", fill=True, palette=[color, \"#888888\"], alpha=0.6)\n    plt.title(f\"{metric.capitalize()} Dağılımı (Pozitif vs Negatif)\", color=color, fontsize=12)\n    plt.xlabel(metric.capitalize())\n    plt.ylabel(\"Yoğunluk\")\n    plt.grid(alpha=0.3)\n    plt.tight_layout()\n    plt.show()\n\nplot_metric_distribution(valid_samples, \"brightness\", PRIMARY_COLOR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:33.201484Z","iopub.execute_input":"2025-12-17T16:39:33.201709Z","iopub.status.idle":"2025-12-17T16:39:33.495782Z","shell.execute_reply.started":"2025-12-17T16:39:33.201688Z","shell.execute_reply":"2025-12-17T16:39:33.495164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_metric_distribution(valid_samples, \"variance\", PRIMARY_COLOR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:33.496640Z","iopub.execute_input":"2025-12-17T16:39:33.497158Z","iopub.status.idle":"2025-12-17T16:39:33.724340Z","shell.execute_reply.started":"2025-12-17T16:39:33.497130Z","shell.execute_reply":"2025-12-17T16:39:33.723855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_metric_distribution(valid_samples, \"blur\", PRIMARY_COLOR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:33.725238Z","iopub.execute_input":"2025-12-17T16:39:33.725512Z","iopub.status.idle":"2025-12-17T16:39:33.959879Z","shell.execute_reply.started":"2025-12-17T16:39:33.725495Z","shell.execute_reply":"2025-12-17T16:39:33.959287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"📊 Gri Ton Histogramı — Örnek Görseller\")\n\nexample_ids = valid_samples.sample(10, random_state=7)[\"id\"].tolist()\nfig, axes = plt.subplots(2, 5, figsize=(14, 5))\naxes = axes.flatten()\n\nfor ax, img_id in zip(axes, example_ids):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    try:\n        img = np.array(Image.open(img_path).convert(\"L\"))\n        ax.hist(img.ravel(), bins=30, color=PRIMARY_COLOR, alpha=0.7)\n        label_val = labels.loc[labels[\"id\"] == img_id, \"label\"].values[0]\n        ax.set_title(f\"{'Cancer' if label_val==1 else 'Normal'}\\n{img_id[:10]}...\", fontsize=8, color=PRIMARY_COLOR if label_val==1 else \"#888888\")\n        ax.set_xlim(0, 255)\n        ax.set_ylim(0, None)\n    except:\n        ax.text(0.5, 0.5, \"Error\", ha=\"center\", va=\"center\", color=\"red\")\n    ax.set_xticks([])\n    ax.set_yticks([])\n\nplt.suptitle(\"Gri Ton Histogramları (Rastgele 10 Görsel)\", color=PRIMARY_COLOR, fontsize=13)\nplt.tight_layout(rect=[0, 0, 1, 0.95])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:33.960574Z","iopub.execute_input":"2025-12-17T16:39:33.960914Z","iopub.status.idle":"2025-12-17T16:39:34.919707Z","shell.execute_reply.started":"2025-12-17T16:39:33.960888Z","shell.execute_reply":"2025-12-17T16:39:34.918987Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **[STEP 5] DATA IMBALANCE & FINAL EDA SUMMARY**","metadata":{}},{"cell_type":"code","source":"\"\"\"\n[STEP 5] DATA IMBALANCE & FINAL EDA SUMMARY\n\"\"\"\ndisplay_section_title(\"⚖️ Sınıf Dağılımı & Dengesizlik Analizi\")\n\nsizes = label_counts[\"count\"]\nlabels_pie = [\"Cancer\" if lbl == 1 else \"Normal\" for lbl in label_counts[\"label\"]]\ncolors = [PRIMARY_COLOR, \"#1caad9\"]\n\nexplode = [0.05, 0.05]\n\nfig, ax = plt.subplots(figsize=(6, 6), subplot_kw=dict(aspect=\"equal\"))\nwedges, texts, autotexts = ax.pie(\n    sizes,\n    autopct=\"%1.1f%%\",\n    startangle=120,\n    colors=colors,\n    shadow=True,\n    explode=explode,\n    pctdistance=0.8,\n    textprops={\"fontsize\": 11, \"color\": \"white\"},\n    wedgeprops={\"edgecolor\": \"white\", \"linewidth\": 1.2},\n)\n\nfor i, w in enumerate(wedges):\n    ang = (w.theta2 - w.theta1)/2. + w.theta1\n    x = np.cos(np.deg2rad(ang))\n    y = np.sin(np.deg2rad(ang))\n    ax.text(\n        1.2 * x, 1.2 * y,\n        f\"{labels_pie[i]}\\n({sizes[i]:,})\",\n        ha=\"center\", va=\"center\",\n        fontsize=10.5,\n        color=\"#333333\",\n        weight=\"medium\"\n    )\n\nax.text(\n    0, 0, f\"Toplam\\n{len(labels):,}\",\n    ha=\"center\", va=\"center\",\n    fontsize=12,\n    color=\"#222222\",\n    weight=\"bold\"\n)\n\nplt.title(\n    \"Sınıf Oranları (Modern 3D Görünüm)\",\n    color=PRIMARY_COLOR,\n    fontsize=13,\n    pad=20,\n    weight=\"bold\"\n)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:34.920571Z","iopub.execute_input":"2025-12-17T16:39:34.920926Z","iopub.status.idle":"2025-12-17T16:39:35.067736Z","shell.execute_reply.started":"2025-12-17T16:39:34.920903Z","shell.execute_reply":"2025-12-17T16:39:35.067140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import display, HTML\n\npos_ratio = label_counts.loc[label_counts[\"label\"] == 1, \"ratio\"].values[0]\nneg_ratio = label_counts.loc[label_counts[\"label\"] == 0, \"ratio\"].values[0]\nimbalance_ratio = abs(pos_ratio - neg_ratio)\n\nif imbalance_ratio < 10:\n    balance_status = \"✅ Dataset dengeli görünüyor\"\n    color_tag = \"#1abc9c\"\nelif imbalance_ratio < 30:\n    balance_status = \"⚠️ Hafif dengesizlik mevcut\"\n    color_tag = \"#f1c40f\"\nelse:\n    balance_status = \"🚨 Güçlü dengesizlik var\"\n    color_tag = \"#e74c3c\"\n\n# HTML rapor\nhtml_report = f\"\"\"\n<div style='\n    background-color:#f9f9f9;\n    border-radius:14px;\n    border-left:6px solid {PRIMARY_COLOR};\n    box-shadow:0 2px 6px rgba(0,0,0,0.1);\n    padding:20px;\n    margin:15px 0;\n    font-family:Segoe UI, sans-serif;\n    color:#333;\n'>\n    <h2 style='color:{PRIMARY_COLOR}; margin-bottom:10px;'>📊 EDA Özet Raporu</h2>\n    <hr style='border:none; border-top:1px solid #ddd; margin:10px 0;'>\n\n    <h4 style='color:#444;'>📦 Genel Bilgiler</h4>\n    <ul style='list-style:none; padding-left:10px;'>\n        <li>🧮 Toplam örnek sayısı: <b>{len(labels):,}</b></li>\n        <li>🖼️ Görsel boyutu: <b>96×96 piksel</b></li>\n        <li>🧾 Görsel formatı: <b>.tif</b></li>\n        <li>📂 Train klasörü: <b>{len(os.listdir(TRAIN_DIR)):,} dosya</b></li>\n        <li>🧪 Test klasörü: <b>{len(os.listdir(TEST_DIR)):,} dosya</b></li>\n    </ul>\n\n    <h4 style='color:#444;'>⚖️ Sınıf Dağılımı</h4>\n    <ul style='list-style:none; padding-left:10px;'>\n        <li>🔹 Pozitif (Cancer): <b>{pos_ratio:.2f}%</b></li>\n        <li>🔸 Negatif (Normal): <b>{neg_ratio:.2f}%</b></li>\n        <li>📉 Dengesizlik farkı: <b>{imbalance_ratio:.2f}%</b></li>\n    </ul>\n\n    <div style='\n        background:{color_tag}20;\n        border-left:4px solid {color_tag};\n        border-radius:6px;\n        padding:10px 14px;\n        margin:8px 0;\n        font-size:15px;\n    '>\n        <b style='color:{color_tag};'>{balance_status}</b>\n    </div>\n\n    <h4 style='color:#444;'>🔬 Görsel Kalite Analizi</h4>\n    <ul style='list-style:none; padding-left:10px;'>\n        <li>💡 Ortalama parlaklık, varyans ve kontrast hesaplandı</li>\n        <li>🔍 Blur (focus) ölçümü tamamlandı</li>\n        <li>✅ Bozuk dosya sayısı: <b>çok düşük veya yok</b></li>\n    </ul>\n</div>\n\"\"\"\n\ndisplay(HTML(html_report))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:35.068448Z","iopub.execute_input":"2025-12-17T16:39:35.068687Z","iopub.status.idle":"2025-12-17T16:39:38.767763Z","shell.execute_reply.started":"2025-12-17T16:39:35.068660Z","shell.execute_reply":"2025-12-17T16:39:38.767061Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **[STEP 6] COLOR DISTRIBUTION ANALYSIS**","metadata":{}},{"cell_type":"code","source":"\"\"\"\n===============================================================================\n[STEP 6] COLOR DISTRIBUTION ANALYSIS\n-------------------------------------------------------------------------------\nAmaç (1. Adım):\n  - RGB kanallarının ortalama ve standart sapma değerlerini hesaplamak\n===============================================================================\n\"\"\"\n\nfrom tqdm.auto import tqdm\n\nSAMPLE_SIZE_COLOR = min(1500, len(labels))\nsample_color_df = labels.sample(SAMPLE_SIZE_COLOR, random_state=42).reset_index(drop=True)\n\nrgb_means, rgb_stds = [], []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:38.771753Z","iopub.execute_input":"2025-12-17T16:39:38.771975Z","iopub.status.idle":"2025-12-17T16:39:38.782741Z","shell.execute_reply.started":"2025-12-17T16:39:38.771958Z","shell.execute_reply":"2025-12-17T16:39:38.782072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for img_id in tqdm(sample_color_df[\"id\"], desc=\"RGB kanal analiz\"):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    try:\n        img = np.array(Image.open(img_path).convert(\"RGB\"), dtype=np.float32)\n        r_mean, g_mean, b_mean = np.mean(img[:, :, 0]), np.mean(img[:, :, 1]), np.mean(img[:, :, 2])\n        r_std, g_std, b_std = np.std(img[:, :, 0]), np.std(img[:, :, 1]), np.std(img[:, :, 2])\n        rgb_means.append((r_mean, g_mean, b_mean))\n        rgb_stds.append((r_std, g_std, b_std))\n    except:\n        rgb_means.append((np.nan, np.nan, np.nan))\n        rgb_stds.append((np.nan, np.nan, np.nan))\n\nsample_color_df[[\"R_mean\", \"G_mean\", \"B_mean\"]] = pd.DataFrame(rgb_means, index=sample_color_df.index)\nsample_color_df[[\"R_std\", \"G_std\", \"B_std\"]] = pd.DataFrame(rgb_stds, index=sample_color_df.index)\n\nrgb_summary = (\n    sample_color_df[[\"R_mean\", \"G_mean\", \"B_mean\", \"R_std\", \"G_std\", \"B_std\"]]\n    .agg([\"mean\", \"std\"])\n    .round(2)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:38.783588Z","iopub.execute_input":"2025-12-17T16:39:38.783809Z","iopub.status.idle":"2025-12-17T16:39:42.072128Z","shell.execute_reply.started":"2025-12-17T16:39:38.783785Z","shell.execute_reply":"2025-12-17T16:39:42.071572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🎨 Renk (RGB) Kanal Analizi — Ortalama & Std Hesaplama\")\ndisplay_dataframe(rgb_summary, title=\"RGB Kanal Ortalamaları ve Standart Sapmaları (Örneklem Bazında)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:42.072781Z","iopub.execute_input":"2025-12-17T16:39:42.073016Z","iopub.status.idle":"2025-12-17T16:39:42.081918Z","shell.execute_reply.started":"2025-12-17T16:39:42.072989Z","shell.execute_reply":"2025-12-17T16:39:42.081361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🌈 RGB Kanal Dağılımı\")\n\nfig, axes = plt.subplots(1, 3, figsize=(17, 5))\nfig.patch.set_facecolor(\"white\")\nchannels = [\"R_mean\", \"G_mean\", \"B_mean\"]\ntitles = [\"Kırmızı (R)\", \"Yeşil (G)\", \"Mavi (B)\"]\ncolors = [\"#e74c3c\", \"#2ecc71\", \"#3498db\"]\n\nfor ax, ch, title, color in zip(axes, channels, titles, colors):\n    sns.histplot(\n        data=sample_color_df,\n        x=ch,\n        bins=30,\n        kde=True,\n        color=color,\n        ax=ax,\n        alpha=0.8\n    )\n    ax.set_title(title, fontsize=12, color=color, weight=\"bold\")\n    ax.set_xlabel(\"Ortalama Piksel Yoğunluğu\")\n    ax.set_ylabel(\"Frekans\")\n    ax.grid(alpha=0.3)\n\nplt.suptitle(\"RGB Kanal Dağılımı\", fontsize=14, color=PRIMARY_COLOR, weight=\"bold\")\nplt.tight_layout(rect=[0, 0, 1, 0.96])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:42.082635Z","iopub.execute_input":"2025-12-17T16:39:42.082850Z","iopub.status.idle":"2025-12-17T16:39:42.814285Z","shell.execute_reply.started":"2025-12-17T16:39:42.082834Z","shell.execute_reply":"2025-12-17T16:39:42.813707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🧩 Sınıfa Göre RGB Kanal Dağılımı (Boxplot Analizi)\")\n\nfig, axes = plt.subplots(1, 3, figsize=(20, 9))\nfig.patch.set_facecolor(\"white\")\n\nchannels = [\"R_mean\", \"G_mean\", \"B_mean\"]\ntitles = [\"Kırmızı (R)\", \"Yeşil (G)\", \"Mavi (B)\"]\ncolors = [\"#e74c3c\", \"#2ecc71\", \"#3498db\"]\n\npalette_dict = {\"0\": \"#bbbbbb\", \"1\": \"#3371ab\"}\n\nfor ax, ch, title, color in zip(axes, channels, titles, colors):\n    df_plot = sample_color_df.copy()\n    df_plot[\"label_str\"] = df_plot[\"label\"].astype(str)\n\n    sns.boxplot(\n        data=df_plot,\n        x=\"label_str\",\n        y=ch,\n        hue=\"label_str\",\n        dodge=False,\n        palette={\"0\": \"#cccccc\", \"1\": color},\n        ax=ax,\n        width=0.55,\n        fliersize=2\n    )\n\n    ax.set_title(title, fontsize=12, color=color, weight=\"bold\")\n    ax.set_xlabel(\"Sınıf (0=Normal, 1=Cancer)\")\n    ax.set_ylabel(\"Ortalama Piksel Yoğunluğu\")\n    ax.grid(alpha=0.3)\n\nplt.suptitle(\"Sınıfa Göre RGB Kanal Dağılımı (Boxplot Görselleştirme)\", fontsize=14, color=PRIMARY_COLOR, weight=\"bold\")\nplt.tight_layout(rect=[0, 0, 1, 0.96])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:42.815003Z","iopub.execute_input":"2025-12-17T16:39:42.815294Z","iopub.status.idle":"2025-12-17T16:39:43.371115Z","shell.execute_reply.started":"2025-12-17T16:39:42.815269Z","shell.execute_reply":"2025-12-17T16:39:43.370310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🎨 Renk Varyansı Analizi ve Kanal Baskınlığı Yorumu\")\n\nmean_r = sample_color_df[\"R_mean\"].mean()\nmean_g = sample_color_df[\"G_mean\"].mean()\nmean_b = sample_color_df[\"B_mean\"].mean()\n\nstd_r = sample_color_df[\"R_std\"].mean()\nstd_g = sample_color_df[\"G_std\"].mean()\nstd_b = sample_color_df[\"B_std\"].mean()\n\ncolor_stats = pd.DataFrame({\n    \"Kanal\": [\"R (Kırmızı)\", \"G (Yeşil)\", \"B (Mavi)\"],\n    \"Ortalama Yoğunluk\": [round(mean_r, 2), round(mean_g, 2), round(mean_b, 2)],\n    \"Ortalama Std (Varyans)\": [round(std_r, 2), round(std_g, 2), round(std_b, 2)]\n})\n\ndisplay_dataframe(color_stats, title=\"RGB Kanal İstatistik Özeti\")\n\ndominant_channel = color_stats.loc[color_stats[\"Ortalama Yoğunluk\"].idxmax(), \"Kanal\"]\ndominant_std = color_stats.loc[color_stats[\"Ortalama Std (Varyans)\"].idxmax(), \"Kanal\"]\n\nhtml_summary = f\"\"\"\n<div style='\n    background-color:#f9f9f9;\n    border-radius:14px;\n    border-left:6px solid {PRIMARY_COLOR};\n    box-shadow:0 2px 6px rgba(0,0,0,0.1);\n    padding:20px;\n    margin:15px 0;\n    font-family:Segoe UI, sans-serif;\n    color:#333;\n'>\n    <h3 style='color:{PRIMARY_COLOR}; margin-bottom:10px;'>🎨 Renk Varyansı Yorumu</h3>\n    <p style='margin:6px 0;'>\n        <b>🔹 Ortalama kanal yoğunluğu açısından baskın renk:</b> <span style='color:{PRIMARY_COLOR}'>{dominant_channel}</span><br>\n        <b>🔸 Değişkenlik (standart sapma) açısından en dinamik kanal:</b> <span style='color:{PRIMARY_COLOR}'>{dominant_std}</span>\n    </p>\n    <p style='margin-top:10px; color:#555;'>\n        Bu durum, veri setindeki histopatolojik boyamanın renk karakterini yansıtır.\n        Genellikle hematoksilen-eozin (H&E) boyalı örneklerde mavi ve pembe tonlar hakimdir.\n    </p>\n</div>\n\"\"\"\n\ndisplay(HTML(html_summary))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:43.372030Z","iopub.execute_input":"2025-12-17T16:39:43.372388Z","iopub.status.idle":"2025-12-17T16:39:43.387669Z","shell.execute_reply.started":"2025-12-17T16:39:43.372363Z","shell.execute_reply":"2025-12-17T16:39:43.387025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🧬 Pozitif ve Negatif Görseller — Renk Karşılaştırma Galerisi\")\n\nn_samples = 20\npos_samples = labels[labels[\"label\"] == 1][\"id\"].sample(n_samples, random_state=42).tolist()\nneg_samples = labels[labels[\"label\"] == 0][\"id\"].sample(n_samples, random_state=42).tolist()\n\nfig, axes = plt.subplots(2, n_samples, figsize=(22, 4))\nfig.patch.set_facecolor(\"white\")\n\nfor i, (ax, img_id) in enumerate(zip(axes[0], pos_samples)):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    try:\n        with Image.open(img_path) as img:\n            ax.imshow(img)\n    except:\n        ax.text(0.5, 0.5, \"Error\", ha=\"center\", va=\"center\", color=\"red\")\n    ax.set_title(\"Cancer\", fontsize=8, color=PRIMARY_COLOR)\n    ax.axis(\"off\")\n\nfor i, (ax, img_id) in enumerate(zip(axes[1], neg_samples)):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    try:\n        with Image.open(img_path) as img:\n            ax.imshow(img)\n    except:\n        ax.text(0.5, 0.5, \"Error\", ha=\"center\", va=\"center\", color=\"red\")\n    ax.set_title(\"Normal\", fontsize=8, color=\"#777\")\n    ax.axis(\"off\")\n\nfig.suptitle(\"20 Pozitif ve 20 Negatif Görsel — Renk ve Doku Karşılaştırması\",\n             color=PRIMARY_COLOR, fontsize=14, weight=\"bold\", y=1.05)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:43.388420Z","iopub.execute_input":"2025-12-17T16:39:43.388655Z","iopub.status.idle":"2025-12-17T16:39:45.473449Z","shell.execute_reply.started":"2025-12-17T16:39:43.388640Z","shell.execute_reply":"2025-12-17T16:39:45.471701Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **[STEP 7] GLCM FEATURE EXTRACTION (Gray-Level Co-Occurrence Matrix)**","metadata":{}},{"cell_type":"code","source":"from skimage.feature import graycomatrix, graycoprops\nfrom scipy.stats import entropy\n\ndisplay_section_title(\"🧩 Doku (Texture) Analizi — GLCM Özellik Çıkarımı\")\n\nSAMPLE_SIZE_TEXTURE = min(1000, len(labels))\nsample_texture_df = labels.sample(SAMPLE_SIZE_TEXTURE, random_state=42).reset_index(drop=True)\n\nglcm_features = {\n    \"contrast\": [],\n    \"homogeneity\": [],\n    \"energy\": [],\n    \"entropy\": []\n}\n\ndef extract_glcm_features(img_path):\n    try:\n        img = np.array(Image.open(img_path).convert(\"L\"), dtype=np.uint8)\n        glcm = graycomatrix(img, distances=[1], angles=[0], symmetric=True, normed=True)\n        contrast = graycoprops(glcm, 'contrast')[0, 0]\n        homogeneity = graycoprops(glcm, 'homogeneity')[0, 0]\n        energy = graycoprops(glcm, 'energy')[0, 0]\n        ent = entropy(glcm.ravel())\n        return contrast, homogeneity, energy, ent\n    except Exception:\n        return np.nan, np.nan, np.nan, np.nan\n\nfor img_id in tqdm(sample_texture_df[\"id\"], desc=\"GLCM hesaplanıyor\"):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    c, h, e, en = extract_glcm_features(img_path)\n    glcm_features[\"contrast\"].append(c)\n    glcm_features[\"homogeneity\"].append(h)\n    glcm_features[\"energy\"].append(e)\n    glcm_features[\"entropy\"].append(en)\n\nfor key, vals in glcm_features.items():\n    sample_texture_df[key] = vals\n\nvalid_glcm = sample_texture_df.dropna()\ndisplay_dataframe(valid_glcm.head(10), title=\"GLCM Özellikleri — İlk 10 Satır\")\n\nglcm_summary = (\n    valid_glcm.groupby(\"label\")[[\"contrast\", \"homogeneity\", \"energy\", \"entropy\"]]\n    .agg([\"mean\", \"std\"])\n    .round(3)\n)\ndisplay_dataframe(glcm_summary, title=\"Sınıfa Göre GLCM Özellik Ortalamaları\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:45.474516Z","iopub.execute_input":"2025-12-17T16:39:45.474839Z","iopub.status.idle":"2025-12-17T16:39:50.627974Z","shell.execute_reply.started":"2025-12-17T16:39:45.474809Z","shell.execute_reply":"2025-12-17T16:39:50.627220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"📊 Sınıfa Göre GLCM Özellik Dağılımı\")\n\nmetrics = [\"contrast\", \"homogeneity\", \"energy\", \"entropy\"]\ntitles = [\"Kontrast\", \"Homojenlik\", \"Enerji\", \"Entropi\"]\n\nvalid_glcm[\"label_str\"] = valid_glcm[\"label\"].astype(str)\npalette_dict = {\"0\": \"#cccccc\", \"1\": PRIMARY_COLOR}\n\nfig, axes = plt.subplots(2, 2, figsize=(12, 10))\naxes = axes.flatten()\n\nfor ax, metric, title in zip(axes, metrics, titles):\n    sns.violinplot(\n        data=valid_glcm,\n        x=\"label_str\",\n        y=metric,\n        hue=\"label_str\",\n        dodge=False,\n        palette=palette_dict,\n        ax=ax,\n        inner=\"box\",\n        cut=0,\n        legend=False\n    )\n    ax.set_title(title, fontsize=12, color=PRIMARY_COLOR, weight=\"bold\")\n    ax.set_xlabel(\"Sınıf (0=Normal, 1=Cancer)\")\n    ax.set_ylabel(metric.capitalize())\n    ax.grid(alpha=0.3)\n\nplt.suptitle(\"Sınıfa Göre GLCM Doku Özellik Dağılımları\", fontsize=14, color=PRIMARY_COLOR, weight=\"bold\")\nplt.tight_layout(rect=[0, 0, 1, 0.97])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:50.628738Z","iopub.execute_input":"2025-12-17T16:39:50.629108Z","iopub.status.idle":"2025-12-17T16:39:51.442074Z","shell.execute_reply.started":"2025-12-17T16:39:50.629088Z","shell.execute_reply":"2025-12-17T16:39:51.441364Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from skimage.feature import local_binary_pattern\n\ndisplay_section_title(\"🧩 Local Binary Pattern (LBP) Analizi\")\n\nradius = 1\nn_points = 8 * radius\nmethod = \"uniform\"\n\nSAMPLE_SIZE_LBP = min(1000, len(labels))\nsample_lbp_df = labels.sample(SAMPLE_SIZE_LBP, random_state=42).reset_index(drop=True)\n\nlbp_means = []\nlbp_stds = []\n\nfor img_id in tqdm(sample_lbp_df[\"id\"], desc=\"LBP hesaplanıyor\"):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    try:\n        img_gray = np.array(Image.open(img_path).convert(\"L\"))\n        lbp = local_binary_pattern(img_gray, n_points, radius, method)\n        lbp_means.append(np.mean(lbp))\n        lbp_stds.append(np.std(lbp))\n    except Exception:\n        lbp_means.append(np.nan)\n        lbp_stds.append(np.nan)\n\nsample_lbp_df[\"LBP_mean\"] = lbp_means\nsample_lbp_df[\"LBP_std\"] = lbp_stds\n\nvalid_lbp = sample_lbp_df.dropna()\ndisplay_dataframe(valid_lbp.head(10), title=\"LBP Özellikleri — İlk 10 Satır\")\n\nlbp_summary = (\n    valid_lbp.groupby(\"label\")[[\"LBP_mean\", \"LBP_std\"]]\n    .agg([\"mean\", \"std\"])\n    .round(3)\n)\ndisplay_dataframe(lbp_summary, title=\"Sınıfa Göre LBP Ortalama / Std Özeti\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:51.442783Z","iopub.execute_input":"2025-12-17T16:39:51.442997Z","iopub.status.idle":"2025-12-17T16:39:54.632537Z","shell.execute_reply.started":"2025-12-17T16:39:51.442981Z","shell.execute_reply":"2025-12-17T16:39:54.631989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"📊 Sınıfa Göre LBP Özellik Dağılımları\")\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 5))\naxes = axes.flatten()\npalette_lbp = {\"0\": \"#cccccc\", \"1\": PRIMARY_COLOR}\n\nvalid_lbp[\"label_str\"] = valid_lbp[\"label\"].astype(str)\n\nsns.boxplot(\n    data=valid_lbp, x=\"label_str\", y=\"LBP_mean\",\n    hue=\"label_str\", dodge=False, palette=palette_lbp, ax=axes[0], fliersize=2\n)\nsns.boxplot(\n    data=valid_lbp, x=\"label_str\", y=\"LBP_std\",\n    hue=\"label_str\", dodge=False, palette=palette_lbp, ax=axes[1], fliersize=2\n)\n\naxes[0].set_title(\"LBP Mean Dağılımı\", color=PRIMARY_COLOR)\naxes[1].set_title(\"LBP Std Dağılımı\", color=PRIMARY_COLOR)\n\nfor ax in axes:\n    ax.set_xlabel(\"Sınıf (0=Normal, 1=Cancer)\")\n    ax.grid(alpha=0.3)\n\nplt.suptitle(\"Sınıfa Göre Local Binary Pattern (LBP) Özellikleri\", fontsize=14, color=PRIMARY_COLOR, weight=\"bold\")\nplt.tight_layout(rect=[0, 0, 1, 0.96])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:54.633371Z","iopub.execute_input":"2025-12-17T16:39:54.633620Z","iopub.status.idle":"2025-12-17T16:39:55.016823Z","shell.execute_reply.started":"2025-12-17T16:39:54.633597Z","shell.execute_reply":"2025-12-17T16:39:55.016084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"Doku (Texture) Heatmap Görselleştirmesi — LBP Haritaları\")\n\nimport matplotlib\nmatplotlib.rcParams['axes.unicode_minus'] = False\nmatplotlib.rcParams['font.family'] = 'DejaVu Sans'\n\nn_samples = 3\npos_samples = labels[labels[\"label\"] == 1][\"id\"].sample(n_samples, random_state=7).tolist()\nneg_samples = labels[labels[\"label\"] == 0][\"id\"].sample(n_samples, random_state=7).tolist()\nsample_ids = pos_samples + neg_samples\n\nfig, axes = plt.subplots(len(sample_ids), 2, figsize=(8, 14))\nfig.patch.set_facecolor(\"white\")\n\nfor i, img_id in enumerate(sample_ids):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    try:\n        img_gray = np.array(Image.open(img_path).convert(\"L\"))\n        lbp = local_binary_pattern(img_gray, P=8, R=1, method=\"uniform\")\n\n        label_val = labels.loc[labels[\"id\"] == img_id, \"label\"].values[0]\n        label_text = \"Cancer\" if label_val == 1 else \"Normal\"\n        color = PRIMARY_COLOR if label_val == 1 else \"#777\"\n\n        axes[i, 0].imshow(img_gray, cmap=\"gray\")\n        axes[i, 0].set_title(f\"Orijinal ({label_text})\", color=color, fontsize=10, weight=\"bold\")\n\n        axes[i, 1].imshow(lbp, cmap=\"inferno\")\n        axes[i, 1].set_title(\"LBP Haritası\", color=color, fontsize=10, weight=\"bold\")\n\n        for ax in axes[i]:\n            ax.axis(\"off\")\n\n    except Exception as e:\n        for ax in axes[i]:\n            ax.text(0.5, 0.5, f\"Error\\n{type(e).__name__}: {str(e)[:40]}\", ha=\"center\", va=\"center\", color=\"red\", fontsize=8)\n            ax.axis(\"off\")\n\nplt.suptitle(\"Doku (Texture) Heatmap Karşılaştırmaları — LBP Görselleştirmesi\",\n             fontsize=14, color=PRIMARY_COLOR, weight=\"bold\", y=0.995)\nplt.tight_layout(rect=[0, 0, 1, 0.985])\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:55.017760Z","iopub.execute_input":"2025-12-17T16:39:55.018471Z","iopub.status.idle":"2025-12-17T16:39:56.204882Z","shell.execute_reply.started":"2025-12-17T16:39:55.018448Z","shell.execute_reply":"2025-12-17T16:39:56.203903Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **[STEP 8] WHITE AREA & TISSUE DENSITY ANALYSIS**","metadata":{}},{"cell_type":"code","source":"display_section_title(\"🧩 Boşluk (White Area) ve Doku Yoğunluğu Analizi\")\n\nSAMPLE_SIZE_WHITE = min(1000, len(labels))\nsample_white_df = labels.sample(SAMPLE_SIZE_WHITE, random_state=42).reset_index(drop=True)\n\nwhite_ratios, tissue_ratios = [], []\n\ndef compute_tissue_ratio(img):\n    \"\"\"\n    HSV renk uzayında S (saturation) kanalına göre boşluk (beyaz alan) oranı hesaplar.\n    \"\"\"\n    hsv = cv2.cvtColor(img, cv2.COLOR_RGB2HSV)\n    s_channel = hsv[:, :, 1] / 255.0\n    white_mask = s_channel < 0.2  # düşük satürasyon = beyaz\n    white_ratio = np.sum(white_mask) / white_mask.size\n    tissue_ratio = 1 - white_ratio\n    return white_ratio, tissue_ratio\n\nfor img_id in tqdm(sample_white_df[\"id\"], desc=\"Doku oranı hesaplanıyor\"):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    try:\n        img = np.array(Image.open(img_path).convert(\"RGB\"))\n        w, t = compute_tissue_ratio(img)\n    except Exception:\n        w, t = np.nan, np.nan\n    white_ratios.append(w)\n    tissue_ratios.append(t)\n\nsample_white_df[\"white_ratio\"] = white_ratios\nsample_white_df[\"tissue_ratio\"] = tissue_ratios\n\nvalid_white = sample_white_df.dropna()\n\ndisplay_dataframe(valid_white.head(10), title=\"Doku Yoğunluğu (İlk 10 Görsel)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:56.205980Z","iopub.execute_input":"2025-12-17T16:39:56.206548Z","iopub.status.idle":"2025-12-17T16:39:58.385470Z","shell.execute_reply.started":"2025-12-17T16:39:56.206516Z","shell.execute_reply":"2025-12-17T16:39:58.384758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"📊 Sınıfa Göre Doku Oranı Dağılımı\")\nfig, ax = plt.subplots(figsize=(7, 4))\nsns.kdeplot(\n    data=valid_white,\n    x=\"tissue_ratio\",\n    hue=\"label\",\n    fill=True,\n    common_norm=False,\n    palette={0: \"#bbbbbb\", 1: PRIMARY_COLOR},\n    alpha=0.6\n)\nax.set_xlabel(\"Tissue Ratio (1 - White Area)\")\nax.set_ylabel(\"Density\")\nax.set_title(\"Tissue Density Distribution — Cancer vs Normal\", color=PRIMARY_COLOR, fontsize=12)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:58.386113Z","iopub.execute_input":"2025-12-17T16:39:58.386301Z","iopub.status.idle":"2025-12-17T16:39:58.629313Z","shell.execute_reply.started":"2025-12-17T16:39:58.386286Z","shell.execute_reply":"2025-12-17T16:39:58.628536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🚨 Outlier Görseller — Çok Boş veya Çok Dolu Alanlar\")\n\ntoo_empty = valid_white[valid_white[\"tissue_ratio\"] < 0.2]\ntoo_full = valid_white[valid_white[\"tissue_ratio\"] > 0.95]\n\nprint(f\"🔹 Çok boş (tissue < 0.2): {len(too_empty)} görsel\")\nprint(f\"🔸 Çok dolu (tissue > 0.95): {len(too_full)} görsel\")\n\noutlier_samples = pd.concat([too_empty.head(3), too_full.head(3)])\nfig, axes = plt.subplots(len(outlier_samples), 2, figsize=(6, 10))\nfig.patch.set_facecolor(\"white\")\n\nfor i, (idx, row) in enumerate(outlier_samples.iterrows()):\n    img_path = TRAIN_DIR / f\"{row['id']}.tif\"\n    img = np.array(Image.open(img_path).convert(\"RGB\"))\n    white_mask = (cv2.cvtColor(img, cv2.COLOR_RGB2HSV)[:, :, 1] / 255.0) < 0.2\n\n    axes[i, 0].imshow(img)\n    axes[i, 0].set_title(f\"ID: {row['id'][:10]}... | Tissue {row['tissue_ratio']:.2f}\", fontsize=9)\n    axes[i, 0].axis(\"off\")\n\n    axes[i, 1].imshow(white_mask, cmap=\"gray\")\n    axes[i, 1].set_title(\"White Mask\", fontsize=9)\n    axes[i, 1].axis(\"off\")\n\nplt.suptitle(\"Outlier Görseller — Doku Oranı Aykırı Olanlar\", fontsize=13, color=PRIMARY_COLOR, weight=\"bold\")\nplt.tight_layout(rect=[0, 0, 1, 0.97])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:58.630253Z","iopub.execute_input":"2025-12-17T16:39:58.630819Z","iopub.status.idle":"2025-12-17T16:39:59.352349Z","shell.execute_reply.started":"2025-12-17T16:39:58.630801Z","shell.execute_reply":"2025-12-17T16:39:59.351475Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **[STEP 9] FEATURE EXTRACTION FROM CENTRAL AND PERIPHERAL PATCHES**","metadata":{}},{"cell_type":"code","source":"from skimage.feature import local_binary_pattern\nfrom scipy.stats import entropy as shannon_entropy\n\nPATCH = 32   # merkez patch boyutu\nIMG = 96     # HCD görselleri 96×96\n\ndef get_center_patch(img_rgb, patch=PATCH):\n    h, w = img_rgb.shape[:2]\n    cy, cx = h//2, w//2\n    half = patch//2\n    return img_rgb[cy-half:cy+half, cx-half:cx+half]\n\ndef get_periphery_patches(img_rgb, patch=PATCH):\n    \"\"\"Periferi için 4 köşe 32×32 patch döndürür.\"\"\"\n    p = patch\n    tl = img_rgb[0:p, 0:p]\n    tr = img_rgb[0:p, -p:]\n    bl = img_rgb[-p:, 0:p]\n    br = img_rgb[-p:, -p:]\n    return [tl, tr, bl, br]\n\ndef gray_metrics(gray):\n    \"\"\"Gri ton için: mean, var, laplacian-var (focus) ve histogram entropisi.\"\"\"\n    import cv2\n    g = gray.astype(np.float32)\n    mean = float(np.mean(g))\n    var = float(np.var(g))\n    try:\n        lap = cv2.Laplacian(g, cv2.CV_64F)\n        lapv = float(lap.var())\n    except Exception:\n        lapv = float(np.var(cv2.GaussianBlur(g, (3, 3), 0)))\n\n    hist, _ = np.histogram(gray.ravel(), bins=256, range=(0, 255), density=True)\n    ent = float(shannon_entropy(hist + 1e-12))\n    return mean, var, lapv, ent\n\ndef color_metrics(rgb):\n    \"\"\"RGB ortalama ve std.\"\"\"\n    r = rgb[:,:,0].astype(np.float32)\n    g = rgb[:,:,1].astype(np.float32)\n    b = rgb[:,:,2].astype(np.float32)\n    return (float(np.mean(r)), float(np.mean(g)), float(np.mean(b)),\n            float(np.std(r)),  float(np.std(g)),  float(np.std(b)))\n\ndef lbp_metrics(gray, P=8, R=1, method=\"uniform\"):\n    lbp = local_binary_pattern(gray, P=P, R=R, method=method)\n    return float(np.mean(lbp)), float(np.std(lbp))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:59.353049Z","iopub.execute_input":"2025-12-17T16:39:59.353250Z","iopub.status.idle":"2025-12-17T16:39:59.364295Z","shell.execute_reply.started":"2025-12-17T16:39:59.353234Z","shell.execute_reply":"2025-12-17T16:39:59.363540Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"records = []\nSAMPLE_SIZE_CENTER = min(1000, len(labels))\nsample_center_df = labels.sample(SAMPLE_SIZE_CENTER, random_state=42).reset_index(drop=True)\n\nfor img_id, label in tqdm(\n    sample_center_df[[\"id\", \"label\"]].itertuples(index=False),\n    total=len(sample_center_df),\n    desc=\"Center/Periphery metrics\"\n):\n    try:\n        rgb  = np.array(Image.open(TRAIN_DIR / f\"{img_id}.tif\").convert(\"RGB\"))\n        gray = np.array(Image.open(TRAIN_DIR / f\"{img_id}.tif\").convert(\"L\"))\n\n        # Center patch\n        c_rgb  = get_center_patch(rgb)\n        c_gray = get_center_patch(gray)\n\n        # --- Center metrics ---\n        c_rmean, c_gmean, c_bmean, c_rstd, c_gstd, c_bstd = color_metrics(c_rgb)\n        c_gray_mean, c_gray_var, c_lapvar, c_entropy = gray_metrics(c_gray)\n        c_LBP_mean, c_LBP_std = lbp_metrics(c_gray)\n\n        # --- Periphery patches ---\n        p_rgbs  = get_periphery_patches(rgb)\n        p_grays = [cv2.cvtColor(p, cv2.COLOR_RGB2GRAY) for p in p_rgbs]\n\n        pr_means, pg_means, pb_means, pr_stds, pg_stds, pb_stds = [], [], [], [], [], []\n        p_gray_means, p_gray_vars, p_lapvars, p_entropies, p_LBP_means, p_LBP_stds = [], [], [], [], [], []\n\n        for prgb, pgray in zip(p_rgbs, p_grays):\n            rmean, gmean, bmean, rstd, gstd, bstd = color_metrics(prgb)\n            pr_means.append(rmean); pg_means.append(gmean); pb_means.append(bmean)\n            pr_stds.append(rstd);   pg_stds.append(gstd);   pb_stds.append(bstd)\n\n            gmean_, gvar_, lapv_, ent_ = gray_metrics(pgray)\n            p_gray_means.append(gmean_); p_gray_vars.append(gvar_)\n            p_lapvars.append(lapv_); p_entropies.append(ent_)\n\n            lbp_m, lbp_s = lbp_metrics(pgray)\n            p_LBP_means.append(lbp_m); p_LBP_stds.append(lbp_s)\n\n        rec = {\n            \"id\": img_id, \"label\": label,\n            # Center\n            \"c_R_mean\": c_rmean, \"c_G_mean\": c_gmean, \"c_B_mean\": c_bmean,\n            \"c_R_std\": c_rstd, \"c_G_std\": c_gstd, \"c_B_std\": c_bstd,\n            \"c_gray_mean\": c_gray_mean, \"c_gray_var\": c_gray_var,\n            \"c_lapvar\": c_lapvar, \"c_entropy\": c_entropy,\n            \"c_LBP_mean\": c_LBP_mean, \"c_LBP_std\": c_LBP_std,\n            # Periphery (averaged)\n            \"p_R_mean\": np.mean(pr_means), \"p_G_mean\": np.mean(pg_means), \"p_B_mean\": np.mean(pb_means),\n            \"p_R_std\": np.mean(pr_stds), \"p_G_std\": np.mean(pg_stds), \"p_B_std\": np.mean(pb_stds),\n            \"p_gray_mean\": np.mean(p_gray_means), \"p_gray_var\": np.mean(p_gray_vars),\n            \"p_lapvar\": np.mean(p_lapvars), \"p_entropy\": np.mean(p_entropies),\n            \"p_LBP_mean\": np.mean(p_LBP_means), \"p_LBP_std\": np.mean(p_LBP_stds)\n        }\n\n        # Farklar\n        rec.update({\n            \"d_R_mean\": rec[\"c_R_mean\"] - rec[\"p_R_mean\"],\n            \"d_G_mean\": rec[\"c_G_mean\"] - rec[\"p_G_mean\"],\n            \"d_B_mean\": rec[\"c_B_mean\"] - rec[\"p_B_mean\"],\n            \"d_gray_var\": rec[\"c_gray_var\"] - rec[\"p_gray_var\"],\n            \"d_lapvar\": rec[\"c_lapvar\"] - rec[\"p_lapvar\"],\n            \"d_entropy\": rec[\"c_entropy\"] - rec[\"p_entropy\"],\n            \"d_LBP_mean\": rec[\"c_LBP_mean\"] - rec[\"p_LBP_mean\"],\n            \"d_LBP_std\": rec[\"c_LBP_std\"] - rec[\"p_LBP_std\"]\n        })\n\n        records.append(rec)\n\n    except Exception as e:\n        print(f\"⚠️ {img_id}: {type(e).__name__} – {str(e)[:100]}\")\n        continue\n\ncenterperi_df = pd.DataFrame.from_records(records)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:39:59.365144Z","iopub.execute_input":"2025-12-17T16:39:59.365322Z","iopub.status.idle":"2025-12-17T16:40:08.528738Z","shell.execute_reply.started":"2025-12-17T16:39:59.365307Z","shell.execute_reply":"2025-12-17T16:40:08.528018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"📊 Center vs Periphery — Class-wise Summary\")\n\nsummary_cols = [\n    \"c_gray_var\",\"p_gray_var\",\"d_gray_var\",\n    \"c_lapvar\",\"p_lapvar\",\"d_lapvar\",\n    \"c_entropy\",\"p_entropy\",\"d_entropy\",\n    \"c_LBP_mean\",\"p_LBP_mean\",\"d_LBP_mean\",\n    \"c_LBP_std\",\"p_LBP_std\",\"d_LBP_std\"\n]\n\ngrp = centerperi_df.groupby(\"label\")[summary_cols].agg([\"mean\",\"std\"]).round(3)\ndisplay_dataframe(grp, title=\"Center/Periphery Metrics — Class-wise Summary\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:08.529583Z","iopub.execute_input":"2025-12-17T16:40:08.529954Z","iopub.status.idle":"2025-12-17T16:40:08.555224Z","shell.execute_reply.started":"2025-12-17T16:40:08.529929Z","shell.execute_reply":"2025-12-17T16:40:08.554665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"🎻 Distribution of Center–Periphery Differences (by Class)\")\n\nplot_df = centerperi_df.copy()\nplot_df[\"label_str\"] = plot_df[\"label\"].astype(str)\npalette = {\"0\": \"#cccccc\", \"1\": PRIMARY_COLOR}\n\nmetrics = [\n    (\"d_gray_var\",  \"Δ Gray Variance\"),\n    (\"d_lapvar\",    \"Δ Laplacian Variance (Focus)\"),\n    (\"d_entropy\",   \"Δ Entropy\"),\n    (\"d_LBP_mean\",  \"Δ LBP Mean\"),\n    (\"d_LBP_std\",   \"Δ LBP Std\"),\n    (\"d_R_mean\",    \"Δ Red Mean\")\n]\n\nfig, axes = plt.subplots(2, 3, figsize=(15, 9))\naxes = axes.flatten()\n\nfor ax, (m, title) in zip(axes, metrics):\n    sns.violinplot(\n        data=plot_df,\n        x=\"label_str\", y=m,\n        hue=\"label_str\", dodge=False,\n        inner=\"box\", cut=0, palette=palette,\n        ax=ax, legend=False\n    )\n    ax.set_title(title, color=PRIMARY_COLOR, fontsize=11, weight=\"bold\")\n    ax.set_xlabel(\"Class (0=Normal, 1=Cancer)\")\n    ax.set_ylabel(\"Δ Value\")\n    ax.grid(alpha=0.3)\n\nplt.suptitle(\"Center–Periphery Feature Differences — Cancer vs Normal\",\n             fontsize=15, color=PRIMARY_COLOR, weight=\"bold\")\nplt.tight_layout(rect=[0, 0, 1, 0.96])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:08.556054Z","iopub.execute_input":"2025-12-17T16:40:08.556339Z","iopub.status.idle":"2025-12-17T16:40:09.861777Z","shell.execute_reply.started":"2025-12-17T16:40:08.556319Z","shell.execute_reply":"2025-12-17T16:40:09.861149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_section_title(\"Correlation Heatmap — Center/Periphery Features\")\n\ncorr_cols = [\n    \"d_gray_var\", \"d_lapvar\", \"d_entropy\",\n    \"d_LBP_mean\", \"d_LBP_std\", \"d_R_mean\", \"d_G_mean\", \"d_B_mean\"\n]\n\ncorr = centerperi_df[corr_cols].corr()\n\nmask = np.triu(np.ones_like(corr, dtype=bool))\n\nplt.figure(figsize=(10,8))\nsns.heatmap(\n    corr,\n    mask=mask,\n    annot=True,\n    fmt=\".2f\",\n    cmap=\"Blues\",\n    cbar=True,\n    linewidths=0.5,\n    square=True,\n    annot_kws={\"size\": 9, \"color\": \"#333333\"}\n)\nplt.title(\"Correlation between Center–Periphery Features\",\n          color=PRIMARY_COLOR, fontsize=13, weight=\"bold\", pad=12)\nplt.tight_layout()\nplt.show(),","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:09.862446Z","iopub.execute_input":"2025-12-17T16:40:09.862622Z","iopub.status.idle":"2025-12-17T16:40:10.160282Z","shell.execute_reply.started":"2025-12-17T16:40:09.862608Z","shell.execute_reply":"2025-12-17T16:40:10.159596Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **[STEP 10] RENK NORMALİZASYONU ETKİSİ (MACENKO / REINHARD)**","metadata":{}},{"cell_type":"code","source":"display_section_title(\"🎨 Renk Normalizasyonu Etkisi — (Manual Reinhard / CLAHE Surrogate)\")\n\nimport cv2\nfrom skimage import exposure\n\ndef reinhard_normalization(img_rgb):\n    lab = cv2.cvtColor((img_rgb * 255).astype(np.uint8), cv2.COLOR_RGB2LAB).astype(np.float32)\n    mean, std = np.mean(lab, axis=(0, 1)), np.std(lab, axis=(0, 1))\n    ref_mean, ref_std = [128, 128, 128], [30, 30, 30]\n    norm = (lab - mean) / std * ref_std + ref_mean\n    norm = np.clip(norm, 0, 255).astype(np.uint8)\n    return cv2.cvtColor(norm, cv2.COLOR_LAB2RGB) / 255.0\n\ndef macenko_surrogate(img_rgb):\n    lab = cv2.cvtColor((img_rgb * 255).astype(np.uint8), cv2.COLOR_RGB2LAB)\n    l, a, b = cv2.split(lab)\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    l_eq = clahe.apply(l)\n    lab_eq = cv2.merge((l_eq, a, b))\n    eq_img = cv2.cvtColor(lab_eq, cv2.COLOR_LAB2RGB) / 255.0\n    eq_img = np.power(eq_img, 0.9)\n    return eq_img\n\nN_ROWS, N_COLS = 3, 5\nsample_ids = labels.sample(N_ROWS * N_COLS, random_state=42)[\"id\"].tolist()\n\nfig, axes = plt.subplots(N_ROWS, N_COLS*3, figsize=(16, 8))\nfig.patch.set_facecolor(\"white\")\n\nvar_pre, var_macenko, var_reinhard = [], [], []\n\nfor i, img_id in enumerate(sample_ids):\n    row = i // N_COLS\n    col = (i % N_COLS) * 3\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    try:\n        img = np.array(Image.open(img_path).convert(\"RGB\")) / 255.0\n        axes[row, col].imshow(img)\n        axes[row, col].set_title(\"Orijinal\", fontsize=9, color=\"#555\")\n        axes[row, col].axis(\"off\")\n\n        mac_img = macenko_surrogate(img)\n        axes[row, col+1].imshow(np.clip(mac_img, 0, 1))\n        axes[row, col+1].set_title(\"Macenko (CLAHE)\", fontsize=9, color=PRIMARY_COLOR)\n        axes[row, col+1].axis(\"off\")\n\n        reinhard_img = reinhard_normalization(img)\n        axes[row, col+2].imshow(np.clip(reinhard_img, 0, 1))\n        axes[row, col+2].set_title(\"Reinhard (LAB)\", fontsize=9, color=\"#2ecc71\")\n        axes[row, col+2].axis(\"off\")\n\n        var_pre.append(np.var(img))\n        var_macenko.append(np.var(mac_img))\n        var_reinhard.append(np.var(reinhard_img))\n    except Exception:\n        for k in range(3):\n            axes[row, col+k].axis(\"off\")\n\nplt.suptitle(\"Renk Normalizasyonu — Reinhard & CLAHE (Macenko Surrogate)\", \n             color=PRIMARY_COLOR, fontsize=13, weight=\"bold\")\nplt.tight_layout(rect=[0, 0, 1, 0.97])\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:10.161153Z","iopub.execute_input":"2025-12-17T16:40:10.161376Z","iopub.status.idle":"2025-12-17T16:40:12.534860Z","shell.execute_reply.started":"2025-12-17T16:40:10.161359Z","shell.execute_reply":"2025-12-17T16:40:12.533848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# varyans değişimi\nvar_df = pd.DataFrame({\n    \"Aşama\": [\"Orijinal\", \"Macenko (CLAHE)\", \"Reinhard (LAB)\"],\n    \"Ortalama Varyans\": [np.mean(var_pre), np.mean(var_macenko), np.mean(var_reinhard)]\n})\nvar_df[\"Değişim (%)\"] = (\n    (var_df[\"Ortalama Varyans\"] - var_df[\"Ortalama Varyans\"].iloc[0]) / var_df[\"Ortalama Varyans\"].iloc[0] * 100\n).round(2)\ndisplay_dataframe(var_df, title=\"Renk Normalizasyonu Sonrası Varyans Değişimi (%)\")\n\nplt.figure(figsize=(6,4))\nsns.barplot(data=var_df, x=\"Aşama\", y=\"Ortalama Varyans\", palette=[\"#888\", PRIMARY_COLOR, \"#2ecc71\"])\nplt.title(\"Renk Varyansı Karşılaştırması\", color=PRIMARY_COLOR, fontsize=12)\nplt.ylabel(\"Ortalama Piksel Varyansı\")\nplt.grid(alpha=0.3)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:12.535787Z","iopub.execute_input":"2025-12-17T16:40:12.536583Z","iopub.status.idle":"2025-12-17T16:40:12.688614Z","shell.execute_reply.started":"2025-12-17T16:40:12.536564Z","shell.execute_reply":"2025-12-17T16:40:12.687843Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **[STEP 11] ÖZELLİKSEL TEMSİL & GÖRSEL KÜMELEME (t-SNE / UMAP)**","metadata":{}},{"cell_type":"code","source":"\"\"\"\n===============================================================================\n[STEP 11] ÖZELLİKSEL TEMSİL & GÖRSEL KÜMELEME\n-------------------------------------------------------------------------------\nAmaç:\n  - Basit CNN veya ImageNet tabanlı feature extractor (ör. ResNet18)\n  - t-SNE veya UMAP ile 2D görselleştirme\n  - Pozitif (kırmızı) vs Negatif (mavi) scatter plot\n  - Kümeler arasındaki ayrımı yorumlamak\n===============================================================================\n\n\ndisplay_section_title(\"🧠 Özelliksel Temsil ve Görsel Kümeleme (Feature Embedding + t-SNE)\")\n\nimport torch\nimport torch.nn as nn\nimport torchvision.models as models\nimport torchvision.transforms as transforms\nfrom sklearn.manifold import TSNE\nimport numpy as np\nfrom tqdm.auto import tqdm\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(f\"🧮 Çalışma cihazı: {device}\")\n\n# --- Örnekleme ---\nSAMPLE_SIZE_EMB = min(300, len(labels))\nsample_emb_df = labels.sample(SAMPLE_SIZE_EMB, random_state=42).reset_index(drop=True)\n\n# --- Feature extractor (ResNet18 pretrained) ---\nresnet = models.resnet18(weights=\"IMAGENET1K_V1\")\nresnet.fc = nn.Identity()  # son katmanı kaldır\nresnet = resnet.to(device)\nresnet.eval()\n\n# --- Görselleri dönüştürme ---\ntransform = transforms.Compose([\n    transforms.Resize((96, 96)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std=[0.229, 0.224, 0.225])\n])\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:12.689839Z","iopub.execute_input":"2025-12-17T16:40:12.690133Z","iopub.status.idle":"2025-12-17T16:40:12.695594Z","shell.execute_reply.started":"2025-12-17T16:40:12.690114Z","shell.execute_reply":"2025-12-17T16:40:12.694894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"embeddings, labels_list = [], []\n\nfor img_id, label in tqdm(sample_emb_df[[\"id\", \"label\"]].itertuples(index=False), total=len(sample_emb_df)):\n    try:\n        img = Image.open(TRAIN_DIR / f\"{img_id}.tif\").convert(\"RGB\")\n        tensor = transform(img).unsqueeze(0).to(device)\n        with torch.no_grad():\n            feat = resnet(tensor).cpu().numpy().flatten()\n        embeddings.append(feat)\n        labels_list.append(label)\n    except Exception as e:\n        continue\n\nembeddings = np.array(embeddings)\nlabels_arr = np.array(labels_list)\n\nprint(f\"✅ Embedding boyutu: {embeddings.shape}\")\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:12.696406Z","iopub.execute_input":"2025-12-17T16:40:12.696750Z","iopub.status.idle":"2025-12-17T16:40:12.715295Z","shell.execute_reply.started":"2025-12-17T16:40:12.696726Z","shell.execute_reply":"2025-12-17T16:40:12.714756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"# --- Boyut indirgeme (t-SNE) ---\ntsne = TSNE(n_components=2, perplexity=30, random_state=42, n_iter=1000)\nemb_2d = tsne.fit_transform(embeddings)\n\n# --- Görselleştirme ---\ndisplay_section_title(\"📊 t-SNE 2D Görselleştirmesi — Pozitif vs Negatif Dağılım\")\n\nplt.figure(figsize=(8, 6))\nmask_pos = labels_arr == 1\nmask_neg = labels_arr == 0\n\nplt.scatter(emb_2d[mask_neg, 0], emb_2d[mask_neg, 1],\n            c=\"#3498db\", s=40, alpha=0.6, label=\"Normal (0)\")\nplt.scatter(emb_2d[mask_pos, 0], emb_2d[mask_pos, 1],\n            c=\"#e74c3c\", s=40, alpha=0.7, label=\"Cancer (1)\")\n\nplt.legend(frameon=True)\nplt.title(\"t-SNE Feature Embedding — ResNet18 Görsel Temsil\", color=PRIMARY_COLOR, fontsize=13, weight=\"bold\")\nplt.xlabel(\"t-SNE bileşen 1\")\nplt.ylabel(\"t-SNE bileşen 2\")\nplt.grid(alpha=0.3)\nplt.tight_layout()\nplt.show()\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:12.715975Z","iopub.execute_input":"2025-12-17T16:40:12.716203Z","iopub.status.idle":"2025-12-17T16:40:12.733750Z","shell.execute_reply.started":"2025-12-17T16:40:12.716181Z","shell.execute_reply":"2025-12-17T16:40:12.733203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"from sklearn.metrics import silhouette_score\ntry:\n    sil = silhouette_score(emb_2d, labels_arr)\n    print(f\"📈 Silhouette Score (pozitif-negatif ayrımı kalitesi): {sil:.3f}\")\nexcept Exception:\n    print(\"Silhouette Score hesaplanamadı (örnek sayısı düşük olabilir).\")\n\n# --- Görsel özet yorumu ---\nfrom IPython.display import HTML, display\n\n\nsummary_html = f\n<div style='\n  background:#f9f9f9;\n  border-left:6px solid {PRIMARY_COLOR};\n  border-radius:10px;\n  padding:15px 18px;\n  font-family:Segoe UI, sans-serif;\n  box-shadow:0 2px 5px rgba(0,0,0,0.08);\n'>\n  <h3 style='color:{PRIMARY_COLOR}; margin-bottom:8px;'>🔍 Kümelenme Yorumu</h3>\n  <p style='color:#444; font-size:15px;'>\n    Görselleştirilen t-SNE temsilleri, CNN modelinin çıkardığı özellik uzayında\n    <b>pozitif (kırmızı)</b> ve <b>negatif (mavi)</b> örneklerin dağılımını göstermektedir.\n    Eğer iki sınıf birbirinden ayrık kümelerde toplanıyorsa,\n    modelin dokusal ve renk bazlı farkları ayırt edebildiği söylenebilir.\n  </p>\n  <p style='color:#555; font-size:14px;'>\n    • Yüksek <b>Silhouette Score</b> → iyi ayrım.<br>\n    • Düşük veya karışık dağılım → görsel benzerlik fazladır, ek doku veya morfolojik özellikler gerekebilir.\n  </p>\n</div>\ndisplay(HTML(summary_html))\"\"\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:12.734419Z","iopub.execute_input":"2025-12-17T16:40:12.734643Z","iopub.status.idle":"2025-12-17T16:40:12.749400Z","shell.execute_reply.started":"2025-12-17T16:40:12.734624Z","shell.execute_reply":"2025-12-17T16:40:12.748876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### [STEP X] TRAIN ve TEST Veri Setlerinde Sınıf Dağılımları\n\ndisplay_section_title(\"📊 Train/Test Veri Seti — Sınıf Dağılımı (0 vs 1)\")\n\nfrom pathlib import Path\nimport seaborn as sns\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# --- Train etiket tablosu ---\ntrain_counts = labels['label'].value_counts().sort_index()\ntrain_df = pd.DataFrame({\n    \"set\": \"Train\",\n    \"label\": train_counts.index.astype(str),\n    \"count\": train_counts.values\n})\n\n# --- Test klasöründeki etiket tahmini (dosya adlarına göre veya varsa test.csv) ---\n# Eğer test etiketleri ayrı bir dosyada varsa (örneğin test_labels.csv) şunu kullan:\n# test_labels = pd.read_csv(\"test_labels.csv\")\n# test_counts = test_labels['label'].value_counts().sort_index()\n\n# Eğer test dosyaları sadece klasörlerdeyse (örneğin test/0 ve test/1 klasörleri):\ntest_root = Path(TEST_DIR)\ntest_0 = len(list((test_root / \"0\").glob(\"*.tif\"))) if (test_root / \"0\").exists() else 0\ntest_1 = len(list((test_root / \"1\").glob(\"*.tif\"))) if (test_root / \"1\").exists() else 0\ntest_counts = pd.Series({0: test_0, 1: test_1})\n\ntest_df = pd.DataFrame({\n    \"set\": \"Test\",\n    \"label\": test_counts.index.astype(str),\n    \"count\": test_counts.values\n})\n\n# --- Birleştir ---\ndist_df = pd.concat([train_df, test_df], ignore_index=True)\n\n# --- Barplot ---\nplt.figure(figsize=(7, 5))\nsns.barplot(data=dist_df, x=\"label\", y=\"count\", hue=\"set\", palette=[\"#1f77b4\", PRIMARY_COLOR])\nplt.title(\"Train/Test Veri Setlerinde Etiket Dağılımı\", fontsize=13, color=PRIMARY_COLOR, weight=\"bold\")\nplt.xlabel(\"Sınıf Etiketi\")\nplt.ylabel(\"Görsel Sayısı\")\nplt.grid(axis=\"y\", alpha=0.3)\nplt.tight_layout()\nplt.show()\n\n# --- Sayısal özet tablo ---\ndisplay_dataframe(dist_df.pivot(index=\"label\", columns=\"set\", values=\"count\").fillna(0).astype(int),\n                  title=\"Train/Test Sınıf Sayıları\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:12.750067Z","iopub.execute_input":"2025-12-17T16:40:12.750348Z","iopub.status.idle":"2025-12-17T16:40:12.949761Z","shell.execute_reply.started":"2025-12-17T16:40:12.750332Z","shell.execute_reply":"2025-12-17T16:40:12.949229Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### [STEP 12] Kopya & Benzerlik Kontrolü (Perceptual Hash Analizi)","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning, module=\"seaborn._oldcore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:12.950389Z","iopub.execute_input":"2025-12-17T16:40:12.950600Z","iopub.status.idle":"2025-12-17T16:40:12.954533Z","shell.execute_reply.started":"2025-12-17T16:40:12.950585Z","shell.execute_reply":"2025-12-17T16:40:12.953895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import imagehash\n\ndisplay_section_title(\"🧬 Kopya & Benzerlik Kontrolü — pHash / aHash Analizi\")\n\nSAMPLE_SIZE_HASH = min(10000, len(labels))\nsample_hash_df = labels.sample(SAMPLE_SIZE_HASH, random_state=42).reset_index(drop=True)\n\nphash_vals, ahash_vals = [], []\n\nfor img_id in tqdm(sample_hash_df[\"id\"], desc=\"pHash / aHash hesaplanıyor\"):\n    img_path = TRAIN_DIR / f\"{img_id}.tif\"\n    try:\n        img = Image.open(img_path).convert(\"L\") \n        phash_vals.append(str(imagehash.phash(img)))\n        ahash_vals.append(str(imagehash.average_hash(img)))\n    except Exception:\n        phash_vals.append(None)\n        ahash_vals.append(None)\n\nsample_hash_df[\"pHash\"] = phash_vals\nsample_hash_df[\"aHash\"] = ahash_vals\n\nphash_dupes = (\n    sample_hash_df.groupby(\"pHash\")[\"id\"]\n    .apply(list)\n    .reset_index()\n)\nphash_dupes[\"duplicate_count\"] = phash_dupes[\"id\"].apply(len)\ndupe_groups = phash_dupes[phash_dupes[\"duplicate_count\"] > 1]\n\ndisplay_dataframe(dupe_groups.head(10), title=\"🔍 Kopya (pHash) Görseller — İlk 10 Grup\")\n\ntotal_dupes = dupe_groups[\"duplicate_count\"].sum()\nunique_dupe_groups = len(dupe_groups)\nprint(f\"✅ Toplam {total_dupes} duplicate görsel bulundu ({unique_dupe_groups} grup).\")\n\nplt.figure(figsize=(7, 4))\nsns.histplot(\n    phash_dupes[\"duplicate_count\"],\n    bins=30, color=PRIMARY_COLOR, alpha=0.8\n)\nplt.title(\"Kopya Görsel Dağılımı (pHash Bazlı)\", color=PRIMARY_COLOR, fontsize=12, weight=\"bold\")\nplt.xlabel(\"Aynı pHash'e Sahip Görsel Sayısı\")\nplt.ylabel(\"Frekans\")\nplt.grid(alpha=0.3)\nplt.tight_layout()\nplt.show()\n\ntop_dupes = dupe_groups.sort_values(\"duplicate_count\", ascending=False).head(3)\n\ndisplay_section_title(\"📸 En Sık Tekrarlanan Görseller (pHash Bazlı)\")\n\nn_show = 3\nfor _, row in top_dupes.iterrows():\n    dup_ids = row[\"id\"][:n_show]\n    fig, axes = plt.subplots(1, len(dup_ids), figsize=(12, 4))\n    fig.patch.set_facecolor(\"white\")\n\n    for ax, img_id in zip(axes, dup_ids):\n        img_path = TRAIN_DIR / f\"{img_id}.tif\"\n        try:\n            img = Image.open(img_path)\n            ax.imshow(img)\n            ax.set_title(f\"ID: {img_id[:8]}...\", fontsize=9, color=PRIMARY_COLOR)\n            ax.axis(\"off\")\n        except:\n            ax.text(0.5, 0.5, \"Error\", ha=\"center\", va=\"center\", color=\"red\")\n            ax.axis(\"off\")\n\n    plt.suptitle(f\"pHash: {row['pHash']} | {row['duplicate_count']} Görsel\",\n                 fontsize=12, color=PRIMARY_COLOR, weight=\"bold\")\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:40:12.955148Z","iopub.execute_input":"2025-12-17T16:40:12.955350Z","iopub.status.idle":"2025-12-17T16:41:34.548484Z","shell.execute_reply.started":"2025-12-17T16:40:12.955328Z","shell.execute_reply":"2025-12-17T16:41:34.547908Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### [STEP 13] Sınıf-İçi Çeşitlilik (Intra-Class Variability) Analizi","metadata":{}},{"cell_type":"code","source":"display_section_title(\"🧬 Sınıf-İçi Çeşitlilik Analizi — Ortalama & Varyans Görselleri\")\n\nSAMPLE_PER_CLASS = 2000 \npos_ids = labels[labels[\"label\"] == 1][\"id\"].sample(SAMPLE_PER_CLASS, random_state=42).tolist()\nneg_ids = labels[labels[\"label\"] == 0][\"id\"].sample(SAMPLE_PER_CLASS, random_state=42).tolist()\n\ndef compute_mean_std_image(id_list):\n    \"\"\"\n    Verilen ID listesi için ortalama (mean) ve standart sapma (std) görselini hesaplar.\n    \"\"\"\n    imgs = []\n    for img_id in tqdm(id_list, desc=\"İşleniyor\", leave=False):\n        img_path = TRAIN_DIR / f\"{img_id}.tif\"\n        try:\n            img = np.array(Image.open(img_path).convert(\"RGB\"), dtype=np.float32)\n            imgs.append(img)\n        except Exception:\n            continue\n    imgs = np.stack(imgs)\n    mean_img = np.mean(imgs, axis=0)\n    std_img = np.std(imgs, axis=0)\n    return mean_img, std_img\n\n# 🔹 Pozitif ve Negatif için ortalama/std görselleri oluştur\npos_mean, pos_std = compute_mean_std_image(pos_ids)\nneg_mean, neg_std = compute_mean_std_image(neg_ids)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:41:34.549276Z","iopub.execute_input":"2025-12-17T16:41:34.550008Z","iopub.status.idle":"2025-12-17T16:42:05.141905Z","shell.execute_reply.started":"2025-12-17T16:41:34.549988Z","shell.execute_reply":"2025-12-17T16:42:05.141086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 🔹 Görselleştirme — Ortalama Görseller\nfig, axes = plt.subplots(1, 2, figsize=(10, 5))\nfig.patch.set_facecolor(\"white\")\n\naxes[0].imshow(np.clip(neg_mean / 255, 0, 1))\naxes[0].set_title(\"Normal (Label=0) — Ortalama Görsel\", color=\"#555\", fontsize=11, weight=\"bold\")\naxes[1].imshow(np.clip(pos_mean / 255, 0, 1))\naxes[1].set_title(\"Cancer (Label=1) — Ortalama Görsel\", color=PRIMARY_COLOR, fontsize=11, weight=\"bold\")\n\nfor ax in axes:\n    ax.axis(\"off\")\n\nplt.suptitle(\"Ortalama (Mean) Görseller — Sınıf Bazında\", color=PRIMARY_COLOR, fontsize=14, weight=\"bold\", y=0.97)\nplt.tight_layout(rect=[0, 0, 1, 0.95])\nplt.show()\n\n# 🔹 Görselleştirme — Standart Sapma (Varyans) Görselleri (Heatmap)\nfig, axes = plt.subplots(1, 2, figsize=(10, 5))\nfig.patch.set_facecolor(\"white\")\n\naxes[0].imshow(np.mean(neg_std, axis=2), cmap=\"magma\")\naxes[0].set_title(\"Normal (Label=0) — Std (Varyans) Görseli\", color=\"#555\", fontsize=11, weight=\"bold\")\n\naxes[1].imshow(np.mean(pos_std, axis=2), cmap=\"magma\")\naxes[1].set_title(\"Cancer (Label=1) — Std (Varyans) Görseli\", color=PRIMARY_COLOR, fontsize=11, weight=\"bold\")\n\nfor ax in axes:\n    ax.axis(\"off\")\n\nplt.suptitle(\"Sınıf Bazında Görsel Varyans (Std) Haritaları\", color=PRIMARY_COLOR, fontsize=14, weight=\"bold\", y=0.97)\nplt.tight_layout(rect=[0, 0, 1, 0.95])\nplt.show()\n\n# 🔹 Ortalama varyans farkını nicel olarak ölç\npos_var_mean = np.mean(pos_std)\nneg_var_mean = np.mean(neg_std)\ndiff_ratio = (pos_var_mean - neg_var_mean) / neg_var_mean * 100\n\ndisplay_section_title(\"📈 Sınıf İçi Varyans (Std) Karşılaştırması\")\n\nvar_df = pd.DataFrame({\n    \"Sınıf\": [\"Normal (0)\", \"Cancer (1)\"],\n    \"Ortalama Piksel Std\": [round(neg_var_mean, 2), round(pos_var_mean, 2)]\n})\ndisplay_dataframe(var_df, title=\"Sınıf İçi Varyans Karşılaştırması\")\n\nprint(f\"🔍 Cancer sınıfı, Normal sınıfa göre ortalama piksel varyansında %{diff_ratio:.2f} daha {'yüksek' if diff_ratio > 0 else 'düşük'} çeşitlilik gösteriyor.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:42:05.142809Z","iopub.execute_input":"2025-12-17T16:42:05.143278Z","iopub.status.idle":"2025-12-17T16:42:05.500996Z","shell.execute_reply.started":"2025-12-17T16:42:05.143250Z","shell.execute_reply":"2025-12-17T16:42:05.500432Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### [STEP 14] Veri Setinin Sınıf Bazlı Olarak Düzenlenmesi","metadata":{}},{"cell_type":"code","source":"\"\"\"\n===============================================================================\n[STEP 14] DOWNSAMPLED DATASET OLUŞTURMA (FİZİKSEL AYIRMA)\n-------------------------------------------------------------------------------\nAmaç:\n- Sınıf dengesizliğini downsampling ile gidermek\n- Majority class (0) azaltılarak minority class (1) ile eşitlemek\n- Sadece downsample edilmiş görselleri yeni bir train klasörüne almak\n- Eğitimde bu klasörü kullanmak\n===============================================================================\n\"\"\"\n\n# Downsample hedef sayısı (azınlık sınıf referans)\ntarget_count = labels[\"label\"].value_counts().min()\n\n# Sınıfları ayır\nlabels_0 = labels[labels[\"label\"] == 0]\nlabels_1 = labels[labels[\"label\"] == 1]\n\n# Majority class downsample\nlabels_0_down = labels_0.sample(\n    n=target_count,\n    random_state=42\n)\n\n# Minority class (aynı sayıda tutulur)\nlabels_1_keep = labels_1.sample(\n    n=target_count,\n    random_state=42\n)\n\n# Birleştir ve karıştır\nlabels_downsampled = (\n    pd.concat([labels_0_down, labels_1_keep], ignore_index=True)\n      .sample(frac=1, random_state=42)\n      .reset_index(drop=True)\n)\n\n# Kontrol\nprint(\"Downsample sonrası sınıf dağılımı:\")\nprint(labels_downsampled[\"label\"].value_counts())\nprint(\"Toplam örnek:\", len(labels_downsampled))\n\n\n# Downsample CSV kaydı\nDOWNSAMPLED_CSV = Path(\"/kaggle/working/train_labels_downsampled.csv\")\nlabels_downsampled.to_csv(DOWNSAMPLED_CSV, index=False)\nprint(\"📄 Downsample CSV kaydedildi:\", DOWNSAMPLED_CSV)\n\n\n# --- FİZİKSEL DATASET OLUŞTURMA ---\nOUTPUT_DIR_DS = Path(\"/kaggle/working/dataset_downsampled/train\")\n(OUTPUT_DIR_DS / \"0\").mkdir(parents=True, exist_ok=True)\n(OUTPUT_DIR_DS / \"1\").mkdir(parents=True, exist_ok=True)\n\nfor img_id, label in tqdm(\n    labels_downsampled[[\"id\", \"label\"]].itertuples(index=False),\n    total=len(labels_downsampled),\n    desc=\"Downsample görseller yerleştiriliyor\"\n):\n    src = TRAIN_DIR / f\"{img_id}.tif\"\n    dst = OUTPUT_DIR_DS / str(label) / f\"{img_id}.tif\"\n\n    if src.exists() and not dst.exists():\n        try:\n            os.symlink(src, dst)   # hızlı + disk dostu\n        except Exception:\n            shutil.copy2(src, dst) # fallback\n\n# Son kontrol\ncount_0 = len(list((OUTPUT_DIR_DS / \"0\").glob(\"*.tif\")))\ncount_1 = len(list((OUTPUT_DIR_DS / \"1\").glob(\"*.tif\")))\n\nprint(\"📊 Fiziksel dataset kontrolü:\")\nprint(f\"Class 0 (Normal): {count_0}\")\nprint(f\"Class 1 (Cancer): {count_1}\")\nprint(f\"Toplam: {count_0 + count_1}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:42:05.501809Z","iopub.execute_input":"2025-12-17T16:42:05.502261Z","iopub.status.idle":"2025-12-17T16:50:54.962692Z","shell.execute_reply.started":"2025-12-17T16:42:05.502242Z","shell.execute_reply":"2025-12-17T16:50:54.961933Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  [STEP 15] TRAIN / VAL SPLIT (Stratified) + Klasör Yapısı","metadata":{}},{"cell_type":"code","source":"\"\"\"\n===============================================================================\n[STEP 15] TRAIN / VAL SPLIT (Stratified) + Klasör Yapısı\n-------------------------------------------------------------------------------\n- labels_downsampled üzerinden %85/%15 stratified split\n- train/val klasörlerine sadece ilgili görselleri koyar (symlink -> copy fallback)\n===============================================================================\n\"\"\"\n\nSPLIT_DIR = Path(\"/kaggle/working/dataset_downsampled_split\")\nTRAIN_OUT = SPLIT_DIR / \"train\"\nVAL_OUT   = SPLIT_DIR / \"val\"\n\n# klasörleri hazırla\nfor base in [TRAIN_OUT, VAL_OUT]:\n    (base / \"0\").mkdir(parents=True, exist_ok=True)\n    (base / \"1\").mkdir(parents=True, exist_ok=True)\n\n# --- stratified split ---\nval_frac = 0.15\nval_n_per_class = int(round(target_count * val_frac))   # her sınıftan %15\ntrain_n_per_class = target_count - val_n_per_class      # kalan %85\n\n# her sınıftan val seç\nval_0 = labels_0_down.sample(n=val_n_per_class, random_state=42)\nval_1 = labels_1_keep.sample(n=val_n_per_class, random_state=42)\n\n# train = geri kalan\ntrain_0 = labels_0_down.drop(val_0.index)\ntrain_1 = labels_1_keep.drop(val_1.index)\n\nlabels_train = (\n    pd.concat([train_0, train_1], ignore_index=True)\n      .sample(frac=1, random_state=42)\n      .reset_index(drop=True)\n)\nlabels_val = (\n    pd.concat([val_0, val_1], ignore_index=True)\n      .sample(frac=1, random_state=42)\n      .reset_index(drop=True)\n)\n\nprint(\"Train dağılımı:\\n\", labels_train[\"label\"].value_counts())\nprint(\"Val dağılımı:\\n\", labels_val[\"label\"].value_counts())\nprint(\"Train toplam:\", len(labels_train), \"| Val toplam:\", len(labels_val))\n\n# --- dosyaları yerleştir (symlink -> copy fallback) ---\ndef place_split(df, out_dir, desc):\n    for img_id, label in tqdm(\n        df[[\"id\", \"label\"]].itertuples(index=False),\n        total=len(df),\n        desc=desc\n    ):\n        src = TRAIN_DIR / f\"{img_id}.tif\"\n        dst = out_dir / str(label) / f\"{img_id}.tif\"\n        if src.exists() and not dst.exists():\n            try:\n                os.symlink(src, dst)\n            except Exception:\n                shutil.copy2(src, dst)\n\nplace_split(labels_train, TRAIN_OUT, \"Train set yerleştiriliyor\")\nplace_split(labels_val,   VAL_OUT,   \"Val set yerleştiriliyor\")\n\n# --- son kontrol ---\nprint(\"✅ Train/0:\", len(list((TRAIN_OUT/\"0\").glob(\"*.tif\"))))\nprint(\"✅ Train/1:\", len(list((TRAIN_OUT/\"1\").glob(\"*.tif\"))))\nprint(\"✅ Val/0:\",   len(list((VAL_OUT/\"0\").glob(\"*.tif\"))))\nprint(\"✅ Val/1:\",   len(list((VAL_OUT/\"1\").glob(\"*.tif\"))))\n\n# CSV olarak da kaydet (istersen eğitimde CSV kullanırsın)\nTRAIN_CSV = Path(\"/kaggle/working/train_labels_downsampled_train.csv\")\nVAL_CSV   = Path(\"/kaggle/working/train_labels_downsampled_val.csv\")\nlabels_train.to_csv(TRAIN_CSV, index=False)\nlabels_val.to_csv(VAL_CSV, index=False)\nprint(\"📄 Train CSV:\", TRAIN_CSV)\nprint(\"📄 Val CSV:\", VAL_CSV)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:50:54.963559Z","iopub.execute_input":"2025-12-17T16:50:54.963825Z","iopub.status.idle":"2025-12-17T16:57:01.337590Z","shell.execute_reply.started":"2025-12-17T16:50:54.963807Z","shell.execute_reply":"2025-12-17T16:57:01.336979Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###   [STEP 16] CONFIG + PATH CHECK","metadata":{}},{"cell_type":"code","source":"# =============================================================================\n# [STEP 16] CONFIG + PATH CHECK\n# =============================================================================\n\nDATA_ROOT = Path(\"/kaggle/working/dataset_downsampled_split\")\nTRAIN_PATH = DATA_ROOT / \"train\"\nVAL_PATH   = DATA_ROOT / \"val\"\n\nassert TRAIN_PATH.exists(), f\"Train path yok: {TRAIN_PATH}\"\nassert VAL_PATH.exists(),   f\"Val path yok: {VAL_PATH}\"\n\nprint(\"✅ TRAIN_PATH:\", TRAIN_PATH)\nprint(\"✅ VAL_PATH  :\", VAL_PATH)\nprint(\"Train/0:\", len(list((TRAIN_PATH/\"0\").glob(\"*.tif\"))), \" Train/1:\", len(list((TRAIN_PATH/\"1\").glob(\"*.tif\"))))\nprint(\"Val/0  :\", len(list((VAL_PATH/\"0\").glob(\"*.tif\"))),   \" Val/1  :\", len(list((VAL_PATH/\"1\").glob(\"*.tif\"))))\n\nBATCH_SIZE = 128   \nNUM_WORKERS = 4      ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:57:01.338254Z","iopub.execute_input":"2025-12-17T16:57:01.338476Z","iopub.status.idle":"2025-12-17T16:57:02.063104Z","shell.execute_reply.started":"2025-12-17T16:57:01.338458Z","shell.execute_reply":"2025-12-17T16:57:02.062318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### [STEP 17] DATALOADER (96×96)","metadata":{}},{"cell_type":"code","source":"# =============================================================================\n# [STEP 17] DATALOADER (96×96)\n# =============================================================================\nimport torch\n\nfrom torchvision import transforms\nfrom torchvision.datasets import ImageFolder\nfrom torch.utils.data import DataLoader\n\nSEED = 42\n\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\n\nif torch.cuda.is_available():\n    torch.cuda.manual_seed_all(SEED)\n\ntrain_tf = transforms.Compose([\n    transforms.RandomHorizontalFlip(p=0.5),\n    transforms.RandomVerticalFlip(p=0.5),\n    transforms.RandomRotation(degrees=90),\n    transforms.ColorJitter(brightness=0.08, contrast=0.08, saturation=0.06, hue=0.02),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std =[0.229, 0.224, 0.225]),\n])\n\nval_tf = transforms.Compose([\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std =[0.229, 0.224, 0.225]),\n])\n\ntrain_ds = ImageFolder(TRAIN_PATH, transform=train_tf)\nval_ds   = ImageFolder(VAL_PATH,   transform=val_tf)\n\nprint(\"✅ class_to_idx:\", train_ds.class_to_idx)  # 0->\"0\", 1->\"1\" beklenir\nassert train_ds.class_to_idx == val_ds.class_to_idx, \"Train/Val class mapping farklı!\"\n\ntrain_loader = DataLoader(\n    train_ds, batch_size=BATCH_SIZE, shuffle=True,\n    num_workers=NUM_WORKERS, pin_memory=torch.cuda.is_available()\n)\nval_loader = DataLoader(\n    val_ds, batch_size=BATCH_SIZE, shuffle=False,\n    num_workers=NUM_WORKERS, pin_memory=torch.cuda.is_available()\n)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(\"🧠 device:\", device)\nprint(\"Train size:\", len(train_ds), \"| Val size:\", len(val_ds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T16:57:02.063981Z","iopub.execute_input":"2025-12-17T16:57:02.064291Z","iopub.status.idle":"2025-12-17T17:03:11.007014Z","shell.execute_reply.started":"2025-12-17T16:57:02.064274Z","shell.execute_reply":"2025-12-17T17:03:11.006308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# [STEP 18] MODEL (EfficientNet-B0 Transfer Learning)\n# =============================================================================\nimport torch.nn as nn\nimport torchvision.models as models\n\nweights = \"IMAGENET1K_V1\"\n\nmodel = models.efficientnet_b0(weights=weights)\n\n# classifier: (dropout -> linear)\nin_features = model.classifier[1].in_features\nmodel.classifier[1] = nn.Linear(in_features, 1)  # binary logits\n\nmodel = model.to(device)\n\ndef set_trainable(m, trainable: bool):\n    for p in m.parameters():\n        p.requires_grad = trainable\n\ndef freeze_backbone():\n    set_trainable(model.features, False)\n    set_trainable(model.classifier, True)\n\ndef unfreeze_last_blocks(n_last_blocks: int = 2):\n    # önce hepsini freeze\n    set_trainable(model.features, False)\n    # sonra son n block'u aç\n    total = len(model.features)\n    for i in range(total - n_last_blocks, total):\n        set_trainable(model.features[i], True)\n    set_trainable(model.classifier, True)\n\nfreeze_backbone()\nprint(\"✅ Backbone frozen, head trainable.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:11.007763Z","iopub.execute_input":"2025-12-17T17:03:11.008250Z","iopub.status.idle":"2025-12-17T17:03:11.657929Z","shell.execute_reply.started":"2025-12-17T17:03:11.008229Z","shell.execute_reply":"2025-12-17T17:03:11.657306Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### >> >  [STEP 19] TRAIN / VAL LOOP + ROC-AUC + CHECKPOINT","metadata":{}},{"cell_type":"code","source":"import os\nimport json\nimport random\nfrom pathlib import Path\nfrom contextlib import nullcontext\n\nimport numpy as np\nimport pandas as pd\nimport torch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:11.658597Z","iopub.execute_input":"2025-12-17T17:03:11.658799Z","iopub.status.idle":"2025-12-17T17:03:11.662412Z","shell.execute_reply.started":"2025-12-17T17:03:11.658783Z","shell.execute_reply":"2025-12-17T17:03:11.661879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# MODEL CONFIGURATIONS (MODEL-SPECIFIC & PROFESSIONAL)\n# =============================================================================\n\nMODEL_CONFIGS = {\n\n    \"efficientnet_b0\": {\n        # --- Identity ---\n        \"display_name\": \"EfficientNet-B0\",\n        \"backbone_type\": \"CNN\",\n\n        # --- Training strategy ---\n        \"loss\": \"focal\",            # Focal → class balance + hard samples\n        \"focal_gamma\": 2.0,\n\n        \"epochs_stage1\": 2,\n        \"epochs_stage2\": 8,\n\n        \"lr_stage1\": 1e-3,\n        \"lr_stage2\": 5e-4,\n        \"weight_decay\": 1e-4,\n\n        \"unfreeze_last_blocks\": 2,\n\n        # --- Medical decision ---\n        \"threshold_mode\": \"f1\",     # dengeli precision/recall\n    },\n\n    \"convnext_tiny\": {\n        # --- Identity ---\n        \"display_name\": \"ConvNeXt-Tiny\",\n        \"backbone_type\": \"Modern CNN\",\n\n        # --- Training strategy ---\n        \"loss\": \"bce\",              # ConvNeXt zaten güçlü → sade loss\n        \"epochs_stage1\": 2,\n        \"epochs_stage2\": 10,\n\n        \"lr_stage1\": 8e-4,\n        \"lr_stage2\": 3e-4,\n        \"weight_decay\": 5e-5,\n\n        \"unfreeze_last_blocks\": 3,  # daha derin fine-tune\n\n        # --- Medical decision ---\n        \"threshold_mode\": \"f1\",\n    },\n\n    \"swin_tiny\": {\n        # --- Identity ---\n        \"display_name\": \"Swin-Tiny\",\n        \"backbone_type\": \"Transformer\",\n\n        # --- Training strategy ---\n        \"loss\": \"bce\",\n        \"epochs_stage1\": 3,\n        \"epochs_stage2\": 12,\n\n        \"lr_stage1\": 6e-4,\n        \"lr_stage2\": 2e-4,\n        \"weight_decay\": 1e-4,\n\n        \"unfreeze_last_blocks\": 1,  # transformer → az aç\n       \n        # --- Medical decision ---\n        \"threshold_mode\": \"f1\",  # ❗ kanseri kaçırma öncelikli\n    }\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T19:35:36.272997Z","iopub.execute_input":"2025-12-17T19:35:36.273646Z","iopub.status.idle":"2025-12-17T19:35:36.279358Z","shell.execute_reply.started":"2025-12-17T19:35:36.273618Z","shell.execute_reply":"2025-12-17T19:35:36.278620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\n\ndef seed_everything(seed: int = 42):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n\n    # performans > determinism (medical imaging için daha mantıklı)\n    torch.backends.cudnn.benchmark = True\n    torch.backends.cudnn.deterministic = False\n\nseed_everything(SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:11.682004Z","iopub.execute_input":"2025-12-17T17:03:11.682225Z","iopub.status.idle":"2025-12-17T17:03:11.701800Z","shell.execute_reply.started":"2025-12-17T17:03:11.682210Z","shell.execute_reply":"2025-12-17T17:03:11.701138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(\"🔥 Device:\", DEVICE)\n\ndef amp_context():\n    if DEVICE.type == \"cuda\":\n        return torch.amp.autocast(device_type=\"cuda\")\n    return nullcontext()\n\nSCALER = torch.amp.GradScaler(enabled=(DEVICE.type == \"cuda\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:11.702588Z","iopub.execute_input":"2025-12-17T17:03:11.702907Z","iopub.status.idle":"2025-12-17T17:03:11.723603Z","shell.execute_reply.started":"2025-12-17T17:03:11.702886Z","shell.execute_reply":"2025-12-17T17:03:11.723088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 20.1 — LOSS FACTORY\n# =============================================================================\n\nclass FocalLoss(torch.nn.Module):\n    def __init__(self, gamma: float = 2.0):\n        super().__init__()\n        self.gamma = gamma\n        self.bce = torch.nn.BCEWithLogitsLoss(reduction=\"none\")\n\n    def forward(self, logits, targets):\n        bce_loss = self.bce(logits, targets)\n        probs = torch.sigmoid(logits)\n        pt = torch.where(targets == 1, probs, 1 - probs)\n        focal_term = (1 - pt) ** self.gamma\n        return (focal_term * bce_loss).mean()\n\n\ndef build_loss(cfg: dict):\n    if cfg[\"loss\"] == \"bce\":\n        return torch.nn.BCEWithLogitsLoss()\n    elif cfg[\"loss\"] == \"focal\":\n        gamma = cfg.get(\"focal_gamma\", 2.0)\n        return FocalLoss(gamma=gamma)\n    else:\n        raise ValueError(f\"Bilinmeyen loss türü: {cfg['loss']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:11.724240Z","iopub.execute_input":"2025-12-17T17:03:11.724472Z","iopub.status.idle":"2025-12-17T17:03:11.739070Z","shell.execute_reply.started":"2025-12-17T17:03:11.724457Z","shell.execute_reply":"2025-12-17T17:03:11.738536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 20.2 — METRIC HELPERS\n# =============================================================================\n\nfrom sklearn.metrics import roc_auc_score, average_precision_score\n\ndef sigmoid_np(x):\n    x = np.clip(x, -50, 50)  # overflow fix\n    return 1 / (1 + np.exp(-x))\n\ndef safe_roc_auc(y_true, probs):\n    y_true = y_true.astype(int)\n    if len(np.unique(y_true)) < 2:\n        return float(\"nan\")\n    return float(roc_auc_score(y_true, probs))\n\ndef safe_pr_auc(y_true, probs):\n    y_true = y_true.astype(int)\n    if len(np.unique(y_true)) < 2:\n        return float(\"nan\")\n    return float(average_precision_score(y_true, probs))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:11.739735Z","iopub.execute_input":"2025-12-17T17:03:11.740005Z","iopub.status.idle":"2025-12-17T17:03:12.105116Z","shell.execute_reply.started":"2025-12-17T17:03:11.739983Z","shell.execute_reply":"2025-12-17T17:03:12.104548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 20.3 — EPOCH RUNNER\n# =============================================================================\n\ndef run_one_epoch(\n    model,\n    loader,\n    criterion,\n    optimizer=None,\n    train: bool = True\n):\n    model.train() if train else model.eval()\n\n    total_loss = 0.0\n    all_logits = []\n    all_targets = []\n\n    for x, y in loader:\n        x = x.to(DEVICE)\n        y = y.float().to(DEVICE).view(-1, 1)\n\n        if train:\n            optimizer.zero_grad(set_to_none=True)\n\n        with amp_context():\n            logits = model(x)\n            loss = criterion(logits, y)\n\n        if train:\n            SCALER.scale(loss).backward()\n            SCALER.step(optimizer)\n            SCALER.update()\n\n        total_loss += loss.item() * x.size(0)\n        all_logits.append(logits.detach().cpu())\n        all_targets.append(y.detach().cpu())\n\n    logits = torch.cat(all_logits).numpy().reshape(-1)\n    targets = torch.cat(all_targets).numpy().reshape(-1)\n\n    probs = sigmoid_np(logits)\n    preds = (probs >= 0.5).astype(int)\n\n    acc = float((preds == targets).mean())\n    loss_avg = total_loss / len(loader.dataset)\n\n    return {\n        \"loss\": loss_avg,\n        \"accuracy\": acc,\n        \"roc_auc\": safe_roc_auc(targets, probs),\n        \"pr_auc\": safe_pr_auc(targets, probs),\n        \"probs\": probs,\n        \"targets\": targets,\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:12.105759Z","iopub.execute_input":"2025-12-17T17:03:12.106021Z","iopub.status.idle":"2025-12-17T17:03:12.112800Z","shell.execute_reply.started":"2025-12-17T17:03:12.106001Z","shell.execute_reply":"2025-12-17T17:03:12.112221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 20.4 — HISTORY CONTAINER\n# =============================================================================\n\ndef init_history():\n    return {\n        \"epoch\": [],\n        \"train_loss\": [],\n        \"train_acc\": [],\n        \"val_loss\": [],\n        \"val_acc\": [],\n        \"val_roc_auc\": [],\n        \"val_pr_auc\": [],\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:12.113504Z","iopub.execute_input":"2025-12-17T17:03:12.113735Z","iopub.status.idle":"2025-12-17T17:03:12.133791Z","shell.execute_reply.started":"2025-12-17T17:03:12.113713Z","shell.execute_reply":"2025-12-17T17:03:12.133139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 21.1 — MODEL FACTORY\n# =============================================================================\nimport torchvision.models as models\nimport torch.nn as nn\n\ndef build_model(model_key: str):\n    model_key = model_key.lower()\n\n    if model_key == \"efficientnet_b0\":\n        weights = models.EfficientNet_B0_Weights.IMAGENET1K_V1\n        model = models.efficientnet_b0(weights=weights)\n        in_features = model.classifier[1].in_features\n        model.classifier[1] = nn.Linear(in_features, 1)\n        backbone = model.features\n        head = model.classifier\n\n    elif model_key == \"convnext_tiny\":\n        weights = models.ConvNeXt_Tiny_Weights.IMAGENET1K_V1\n        model = models.convnext_tiny(weights=weights)\n        in_features = model.classifier[2].in_features\n        model.classifier[2] = nn.Linear(in_features, 1)\n        backbone = model.features\n        head = model.classifier\n\n    elif model_key == \"swin_tiny\":\n        weights = models.Swin_T_Weights.IMAGENET1K_V1\n        model = models.swin_t(weights=weights)\n        in_features = model.head.in_features\n        model.head = nn.Linear(in_features, 1)\n        backbone = model.features\n        head = model.head\n\n    else:\n        raise ValueError(f\"Desteklenmeyen model: {model_key}\")\n\n    return model.to(DEVICE), backbone, head","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:12.134449Z","iopub.execute_input":"2025-12-17T17:03:12.134630Z","iopub.status.idle":"2025-12-17T17:03:12.150342Z","shell.execute_reply.started":"2025-12-17T17:03:12.134616Z","shell.execute_reply":"2025-12-17T17:03:12.149834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 21.2 — FREEZE / UNFREEZE\n# =============================================================================\n\ndef set_trainable(module, trainable: bool):\n    for p in module.parameters():\n        p.requires_grad = trainable\n\n\ndef freeze_backbone(backbone, head):\n    set_trainable(backbone, False)\n    set_trainable(head, True)\n\n\ndef unfreeze_last_blocks(model_key, backbone, head, n_last_blocks: int):\n    set_trainable(backbone, False)\n\n    if model_key in [\"efficientnet_b0\", \"convnext_tiny\"]:\n        total = len(backbone)\n        for i in range(total - n_last_blocks, total):\n            set_trainable(backbone[i], True)\n\n    elif model_key == \"swin_tiny\":\n        # Swin'de son stage yeterli\n        set_trainable(backbone[-1], True)\n\n    set_trainable(head, True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:12.156529Z","iopub.execute_input":"2025-12-17T17:03:12.156719Z","iopub.status.idle":"2025-12-17T17:03:12.175801Z","shell.execute_reply.started":"2025-12-17T17:03:12.156704Z","shell.execute_reply":"2025-12-17T17:03:12.175148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 21.3 — TRAIN MODEL (FULL PIPELINE)\n# =============================================================================\n\nfrom torch.optim import AdamW\n\ndef train_model(model_key: str):\n    cfg = MODEL_CONFIGS[model_key]\n    display_name = cfg[\"display_name\"]\n\n    print(f\"\\n🚀 TRAINING STARTED: {display_name}\")\n\n    model, backbone, head = build_model(model_key)\n    criterion = build_loss(cfg)\n\n    history = init_history()\n    best_auc = -1.0\n    best_state = None\n    best_val_outputs = None\n\n    # =========================\n    # STAGE 1 — HEAD ONLY\n    # =========================\n    freeze_backbone(backbone, head)\n\n    optimizer = AdamW(\n        filter(lambda p: p.requires_grad, model.parameters()),\n        lr=cfg[\"lr_stage1\"],\n        weight_decay=cfg[\"weight_decay\"]\n    )\n\n    for epoch in range(1, cfg[\"epochs_stage1\"] + 1):\n        train_out = run_one_epoch(model, train_loader, criterion, optimizer, train=True)\n        val_out   = run_one_epoch(model, val_loader,   criterion, optimizer=None, train=False)\n\n        history[\"epoch\"].append(f\"S1-{epoch}\")\n        history[\"train_loss\"].append(train_out[\"loss\"])\n        history[\"train_acc\"].append(train_out[\"accuracy\"])\n        history[\"val_loss\"].append(val_out[\"loss\"])\n        history[\"val_acc\"].append(val_out[\"accuracy\"])\n        history[\"val_roc_auc\"].append(val_out[\"roc_auc\"])\n        history[\"val_pr_auc\"].append(val_out[\"pr_auc\"])\n\n        print(\n            f\"[S1][{epoch}] \"\n            f\"TL {train_out['loss']:.4f} | \"\n            f\"VL {val_out['loss']:.4f} | \"\n            f\"AUC {val_out['roc_auc']:.4f}\"\n        )\n\n        if val_out[\"roc_auc\"] > best_auc:\n            best_auc = val_out[\"roc_auc\"]\n            best_state = {k: v.cpu() for k, v in model.state_dict().items()}\n            best_val_outputs = val_out\n\n    # =========================\n    # STAGE 2 — FINE TUNE\n    # =========================\n    unfreeze_last_blocks(\n        model_key,\n        backbone,\n        head,\n        cfg[\"unfreeze_last_blocks\"]\n    )\n\n    optimizer = AdamW(\n        filter(lambda p: p.requires_grad, model.parameters()),\n        lr=cfg[\"lr_stage2\"],\n        weight_decay=cfg[\"weight_decay\"]\n    )\n\n    for epoch in range(1, cfg[\"epochs_stage2\"] + 1):\n        train_out = run_one_epoch(model, train_loader, criterion, optimizer, train=True)\n        val_out   = run_one_epoch(model, val_loader,   criterion, optimizer=None, train=False)\n\n        history[\"epoch\"].append(f\"S2-{epoch}\")\n        history[\"train_loss\"].append(train_out[\"loss\"])\n        history[\"train_acc\"].append(train_out[\"accuracy\"])\n        history[\"val_loss\"].append(val_out[\"loss\"])\n        history[\"val_acc\"].append(val_out[\"accuracy\"])\n        history[\"val_roc_auc\"].append(val_out[\"roc_auc\"])\n        history[\"val_pr_auc\"].append(val_out[\"pr_auc\"])\n\n        print(\n            f\"[S2][{epoch}] \"\n            f\"TL {train_out['loss']:.4f} | \"\n            f\"VL {val_out['loss']:.4f} | \"\n            f\"AUC {val_out['roc_auc']:.4f}\"\n        )\n\n        if val_out[\"roc_auc\"] > best_auc:\n            best_auc = val_out[\"roc_auc\"]\n            best_state = {k: v.cpu() for k, v in model.state_dict().items()}\n            best_val_outputs = val_out\n\n    print(f\"🏁 {display_name} finished | BEST VAL AUC = {best_auc:.4f}\")\n\n    return {\n        \"model_key\": model_key,\n        \"display_name\": display_name,\n        \"history\": history,\n        \"best_auc\": best_auc,\n        \"best_val_outputs\": best_val_outputs,\n        \"config\": cfg,\n    }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:12.176583Z","iopub.execute_input":"2025-12-17T17:03:12.176942Z","iopub.status.idle":"2025-12-17T17:03:12.198257Z","shell.execute_reply.started":"2025-12-17T17:03:12.176921Z","shell.execute_reply":"2025-12-17T17:03:12.197736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 22.1 — METRIC ENGINE\n# =============================================================================\nfrom sklearn.metrics import (\n    roc_auc_score, average_precision_score,\n    confusion_matrix, classification_report,\n    precision_recall_fscore_support, roc_curve, precision_recall_curve\n)\n\ndef sigmoid_np(x):\n    return 1 / (1 + np.exp(-x))\n\n\ndef compute_epoch_metrics(logits, targets):\n    probs = sigmoid_np(logits)\n    preds = (probs >= 0.5).astype(int)\n\n    return {\n        \"accuracy\": (preds == targets).mean(),\n        \"roc_auc\": roc_auc_score(targets, probs),\n        \"pr_auc\": average_precision_score(targets, probs),\n        \"probs\": probs,\n        \"targets\": targets\n    }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:12.198914Z","iopub.execute_input":"2025-12-17T17:03:12.199163Z","iopub.status.idle":"2025-12-17T17:03:12.219160Z","shell.execute_reply.started":"2025-12-17T17:03:12.199137Z","shell.execute_reply":"2025-12-17T17:03:12.218640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 22.2 — THRESHOLD OPTIMIZATION\n# =============================================================================\ndef find_best_threshold(y_true, probs, mode=\"f1\"):\n    thresholds = np.linspace(0.05, 0.95, 181)\n    best_t, best_score = 0.5, -1\n\n    for t in thresholds:\n        preds = (probs >= t).astype(int)\n        p, r, f1, _ = precision_recall_fscore_support(\n            y_true, preds, labels=[0,1], zero_division=0\n        )\n\n        if mode == \"recall\":\n            score = r[1] + 0.01 * p[1]\n        else:  # f1\n            score = f1[1]\n\n        if score > best_score:\n            best_score = score\n            best_t = t\n\n    return best_t\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:12.219749Z","iopub.execute_input":"2025-12-17T17:03:12.219978Z","iopub.status.idle":"2025-12-17T17:03:12.239514Z","shell.execute_reply.started":"2025-12-17T17:03:12.219953Z","shell.execute_reply":"2025-12-17T17:03:12.239006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 22.3 — FINAL EVALUATION\n# =============================================================================\ndef evaluate_best_epoch(best_val_outputs, threshold_mode):\n    probs = best_val_outputs[\"probs\"]\n    targets = best_val_outputs[\"targets\"].astype(int)\n\n    thr = find_best_threshold(targets, probs, mode=threshold_mode)\n    preds = (probs >= thr).astype(int)\n\n    report = classification_report(\n        targets, preds,\n        labels=[0,1],\n        target_names=[\"Normal\", \"Cancer\"],\n        output_dict=True,\n        zero_division=0\n    )\n\n    cm = confusion_matrix(targets, preds)\n\n    return {\n        \"threshold\": thr,\n        \"confusion_matrix\": cm,\n        \"report\": report,\n        \"roc_auc\": roc_auc_score(targets, probs),\n        \"pr_auc\": average_precision_score(targets, probs),\n        \"precision_cancer\": report[\"Cancer\"][\"precision\"],\n        \"recall_cancer\": report[\"Cancer\"][\"recall\"],\n        \"f1_cancer\": report[\"Cancer\"][\"f1-score\"],\n        \"accuracy\": report[\"accuracy\"],\n        \"probs\": probs,\n        \"targets\": targets\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:12.240195Z","iopub.execute_input":"2025-12-17T17:03:12.240379Z","iopub.status.idle":"2025-12-17T17:03:12.256405Z","shell.execute_reply.started":"2025-12-17T17:03:12.240359Z","shell.execute_reply":"2025-12-17T17:03:12.255909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 22.4 — VISUALIZATION\n# =============================================================================\ndef plot_training_curves(history, model_name):\n    epochs = range(len(history[\"train_loss\"]))\n\n    plt.figure(figsize=(16,5))\n    plt.plot(epochs, history[\"train_loss\"], label=\"Train Loss\")\n    plt.plot(epochs, history[\"val_loss\"], label=\"Val Loss\")\n    plt.title(f\"{model_name} | Loss\")\n    plt.grid(alpha=0.3)\n    plt.legend()\n    plt.show()\n\n    plt.figure(figsize=(16,5))\n    plt.plot(epochs, history[\"train_acc\"], label=\"Train Acc\")\n    plt.plot(epochs, history[\"val_acc\"], label=\"Val Acc\")\n    plt.title(f\"{model_name} | Accuracy\")\n    plt.grid(alpha=0.3)\n    plt.legend()\n    plt.show()\n\n\ndef plot_roc_pr(outputs, model_name):\n    y = outputs[\"targets\"]\n    p = outputs[\"probs\"]\n\n    fpr, tpr, _ = roc_curve(y, p)\n    prec, rec, _ = precision_recall_curve(y, p)\n\n    plt.figure(figsize=(6,5))\n    plt.plot(fpr, tpr, label=f\"AUC={roc_auc_score(y,p):.4f}\")\n    plt.plot([0,1],[0,1],\"--\",color=\"gray\")\n    plt.title(f\"{model_name} | ROC\")\n    plt.grid(alpha=0.3)\n    plt.legend()\n    plt.show()\n\n    plt.figure(figsize=(6,5))\n    plt.plot(rec, prec, label=f\"PR-AUC={average_precision_score(y,p):.4f}\")\n    plt.title(f\"{model_name} | PR Curve\")\n    plt.grid(alpha=0.3)\n    plt.legend()\n    plt.show()\n\n\ndef plot_confusion(cm, model_name, thr):\n    plt.figure(figsize=(4.5,4))\n    sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\", cbar=False)\n    plt.xlabel(\"Predicted\")\n    plt.ylabel(\"True\")\n    plt.title(f\"{model_name} | Confusion (thr={thr:.2f})\")\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:03:12.257120Z","iopub.execute_input":"2025-12-17T17:03:12.257477Z","iopub.status.idle":"2025-12-17T17:03:12.279786Z","shell.execute_reply.started":"2025-12-17T17:03:12.257452Z","shell.execute_reply":"2025-12-17T17:03:12.279226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 22.5 — RUN ALL MODELS\n# =============================================================================\nALL_RESULTS = []\n\nfor model_key in MODEL_CONFIGS.keys():\n    out = train_model(model_key)\n\n    final_metrics = evaluate_best_epoch(\n        out[\"best_val_outputs\"],\n        out[\"config\"][\"threshold_mode\"]\n    )\n\n    plot_training_curves(out[\"history\"], out[\"display_name\"])\n    plot_roc_pr(final_metrics, out[\"display_name\"])\n    plot_confusion(\n        final_metrics[\"confusion_matrix\"],\n        out[\"display_name\"],\n        final_metrics[\"threshold\"]\n    )\n\n    ALL_RESULTS.append({\n        \"Model\": out[\"display_name\"],\n        \"Backbone\": out[\"config\"][\"backbone_type\"],\n        \"ROC-AUC\": final_metrics[\"roc_auc\"],\n        \"PR-AUC\": final_metrics[\"pr_auc\"],\n        \"Accuracy\": final_metrics[\"accuracy\"],\n        \"Precision (Cancer)\": final_metrics[\"precision_cancer\"],\n        \"Recall (Cancer)\": final_metrics[\"recall_cancer\"],\n        \"F1 (Cancer)\": final_metrics[\"f1_cancer\"],\n        \"Threshold\": final_metrics[\"threshold\"]\n    })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T19:35:46.049584Z","iopub.execute_input":"2025-12-17T19:35:46.050164Z","iopub.status.idle":"2025-12-17T21:41:54.882192Z","shell.execute_reply.started":"2025-12-17T19:35:46.050140Z","shell.execute_reply":"2025-12-17T21:41:54.881414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# STEP 22.6 — FINAL COMPARISON\n# =============================================================================\ndf_final = pd.DataFrame(ALL_RESULTS)\n\nweights = {\n    \"ROC-AUC\": 0.35,\n    \"Recall (Cancer)\": 0.30,\n    \"F1 (Cancer)\": 0.20,\n    \"Precision (Cancer)\": 0.10, \n    \"Accuracy\": 0.05\n}\n\ndf_final[\"Overall Score\"] = sum(\n    df_final[k] * w for k, w in weights.items()\n)\n\ndf_final = df_final.sort_values(\"Overall Score\", ascending=False)\n\ndisplay(df_final)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T21:41:54.883662Z","iopub.execute_input":"2025-12-17T21:41:54.883882Z","iopub.status.idle":"2025-12-17T21:41:54.898752Z","shell.execute_reply.started":"2025-12-17T21:41:54.883843Z","shell.execute_reply":"2025-12-17T21:41:54.898075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}