{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nimport os\nfrom PIL import Image\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, roc_auc_score \nimport numpy as np \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm.notebook import tqdm ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:19:01.958421Z","iopub.execute_input":"2025-07-10T20:19:01.958946Z","iopub.status.idle":"2025-07-10T20:19:01.964243Z","shell.execute_reply.started":"2025-07-10T20:19:01.958920Z","shell.execute_reply":"2025-07-10T20:19:01.963437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        pass\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:00:12.781895Z","iopub.execute_input":"2025-07-10T20:00:12.782375Z","iopub.status.idle":"2025-07-10T20:07:33.727431Z","shell.execute_reply.started":"2025-07-10T20:00:12.782350Z","shell.execute_reply":"2025-07-10T20:07:33.726822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Device used: {device}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:07:33.728803Z","iopub.execute_input":"2025-07-10T20:07:33.729144Z","iopub.status.idle":"2025-07-10T20:07:33.822321Z","shell.execute_reply.started":"2025-07-10T20:07:33.729125Z","shell.execute_reply":"2025-07-10T20:07:33.821575Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"data_dir = '/kaggle/input/histopathologic-cancer-detection'\ntrain_image_dir = os.path.join(data_dir, 'train')\ntest_image_dir = os.path.join(data_dir, 'test')\n\n\ntrain_df = pd.read_csv(os.path.join(data_dir, 'train_labels.csv'))\nprint(f\"Eğitim veri seti boyutu: {len(train_df)}\")\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:08:00.254873Z","iopub.execute_input":"2025-07-10T20:08:00.255429Z","iopub.status.idle":"2025-07-10T20:08:00.621450Z","shell.execute_reply.started":"2025-07-10T20:08:00.255405Z","shell.execute_reply":"2025-07-10T20:08:00.620685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(6, 4))\nsns.countplot(x='label', data=train_df)\nplt.title('Kanser Etiketi Dağılımı')\nplt.xlabel('Etiket (0: Kansersiz, 1: Kanserli)')\nplt.ylabel('Sayı')\nplt.show()\n\nprint(train_df['label'].value_counts(normalize=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:09:15.681789Z","iopub.execute_input":"2025-07-10T20:09:15.682110Z","iopub.status.idle":"2025-07-10T20:09:15.909833Z","shell.execute_reply.started":"2025-07-10T20:09:15.682087Z","shell.execute_reply":"2025-07-10T20:09:15.909097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 5, figsize=(15, 5))\nfor i, idx in enumerate(train_df[train_df['label'] == 1].sample(5).index):\n    img_id = train_df.loc[idx, 'id']\n    img_path = os.path.join(train_image_dir, f\"{img_id}.tif\") # .tif uzantısına dikkat\n    img = Image.open(img_path)\n    axes[i].imshow(img)\n    axes[i].set_title(f\"Label: {train_df.loc[idx, 'label']}\")\n    axes[i].axis('off')\nplt.suptitle('Kanserli Örnek Görüntüler')\nplt.show()\n\nfig, axes = plt.subplots(1, 5, figsize=(15, 5))\nfor i, idx in enumerate(train_df[train_df['label'] == 0].sample(5).index):\n    img_id = train_df.loc[idx, 'id']\n    img_path = os.path.join(train_image_dir, f\"{img_id}.tif\")\n    img = Image.open(img_path)\n    axes[i].imshow(img)\n    axes[i].set_title(f\"Label: {train_df.loc[idx, 'label']}\")\n    axes[i].axis('off')\nplt.suptitle('Kansersiz Örnek Görüntüler')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:10:50.248751Z","iopub.execute_input":"2025-07-10T20:10:50.249286Z","iopub.status.idle":"2025-07-10T20:10:51.269263Z","shell.execute_reply.started":"2025-07-10T20:10:50.249258Z","shell.execute_reply":"2025-07-10T20:10:51.268422Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Custom Dataset","metadata":{}},{"cell_type":"code","source":"class CancerDataset(Dataset):\n    def __init__(self, df, img_dir, transform=None):\n        self.df = df\n        self.img_dir = img_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        # Görüntü ID'sini ve etiketini al\n        img_name = os.path.join(self.img_dir, self.df.iloc[idx]['id'] + '.tif')\n        label = self.df.iloc[idx]['label']\n\n        # Görüntüyü yükle\n        image = Image.open(img_name).convert('RGB') # .tif dosyaları için bazen 'RGB'ye çevirmek gerekebilir\n\n        # Dönüşümleri uygula (eğer varsa)\n        if self.transform:\n            image = self.transform(image)\n\n        # Etiketi PyTorch tensörüne çevir (float32 çünkü BCEWithLogitsLoss float bekler)\n        label = torch.tensor(label, dtype=torch.float32)\n\n        return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:16:18.276461Z","iopub.execute_input":"2025-07-10T20:16:18.276749Z","iopub.status.idle":"2025-07-10T20:16:18.282807Z","shell.execute_reply.started":"2025-07-10T20:16:18.276729Z","shell.execute_reply":"2025-07-10T20:16:18.281900Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Valid","metadata":{}},{"cell_type":"code","source":"data_dir = '/kaggle/input/histopathologic-cancer-detection'\ntrain_labels_path = os.path.join(data_dir, 'train_labels.csv')\ntrain_image_dir = os.path.join(data_dir, 'train')\ntest_image_dir = os.path.join(data_dir, 'test')\n\ntrain_df = pd.read_csv(train_labels_path)\n\nIMAGE_SIZE = 96\n\n\nNORM_MEAN = [0.485, 0.456, 0.406]\nNORM_STD = [0.229, 0.224, 0.225]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:16:20.147144Z","iopub.execute_input":"2025-07-10T20:16:20.147758Z","iopub.status.idle":"2025-07-10T20:16:20.387940Z","shell.execute_reply.started":"2025-07-10T20:16:20.147733Z","shell.execute_reply":"2025-07-10T20:16:20.387294Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Transformations","metadata":{}},{"cell_type":"code","source":"train_transform = transforms.Compose([\n    transforms.RandomResizedCrop(IMAGE_SIZE, scale=(0.8, 1.0)), # Rastgele kırpma ve boyutlandırma\n    transforms.RandomHorizontalFlip(), # Rastgele yatay çevirme\n    transforms.RandomVerticalFlip(),   # Rastgele dikey çevirme\n    transforms.RandomRotation(90),     # Rastgele 90 derece döndürme\n    transforms.ToTensor(),             # Görüntüyü PyTorch tensörüne çevir (0-1 aralığına otomatik ölçekler)\n    transforms.Normalize(NORM_MEAN, NORM_STD) # Normalizasyon\n])\n\n# Doğrulama (validation) ve Test verisi için dönüşümler (Sadece boyutlandırma, Tensör'e çevirme, normalizasyon)\n# Veri artırma test/doğrulama setlerinde uygulanmaz\nval_test_transform = transforms.Compose([\n    transforms.CenterCrop(IMAGE_SIZE), # Görüntünün ortasından kırpma\n    transforms.ToTensor(),\n    transforms.Normalize(NORM_MEAN, NORM_STD)\n])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, val_df = train_test_split(train_df, test_size=0.15, random_state=42, stratify=train_df['label'])\n\nprint(f\"\\nEğitim veri seti boyutu: {len(train_df)} görüntü\")\nprint(f\"Doğrulama veri seti boyutu: {len(val_df)} görüntü\")\n\n# Dataset nesnelerini oluştur\ntrain_dataset = CancerDataset(df=train_df, img_dir=train_image_dir, transform=train_transform)\nval_dataset = CancerDataset(df=val_df, img_dir=train_image_dir, transform=val_test_transform)\n\n# DataLoader nesnelerini oluştur\nBATCH_SIZE = 32 # Batch büyüklüğünü ayarlayabilirsiniz\n\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True, num_workers=2) # num_workers CPU çekirdeği kullanımı\nval_loader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=2)\n\nprint(f\"\\nEğitim DataLoader: {len(train_loader)} batch, her batch {BATCH_SIZE} görüntü\")\nprint(f\"Doğrulama DataLoader: {len(val_loader)} batch, her batch {BATCH_SIZE} görüntü\")\n\n# Bir örnek batch alıp kontrol edelim\nfor images, labels in train_loader:\n    print(f\"\\nİlk eğitim batch'i: Resimler boyutu {images.shape}, Etiketler boyutu {labels.shape}\")\n    break # Sadece ilk batch'i kontrol edip çık\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:16:28.102574Z","iopub.execute_input":"2025-07-10T20:16:28.102900Z","iopub.status.idle":"2025-07-10T20:16:28.824248Z","shell.execute_reply.started":"2025-07-10T20:16:28.102874Z","shell.execute_reply":"2025-07-10T20:16:28.823236Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define the model","metadata":{}},{"cell_type":"code","source":"class CNN(nn.Module):\n    def __init__(self):\n        super(CNN, self).__init__()\n\n        self.cnn1 = nn.Conv2d(in_channels=3, out_channels=32, kernel_size=3, stride=1, padding=1)\n        self.bn1 = nn.BatchNorm2d(32)\n\n        self.cnn2 = nn.Conv2d(in_channels=32, out_channels=64, kernel_size=3, stride=1, padding=1)\n        self.bn2 = nn.BatchNorm2d(64)\n\n        \n        self.fc1 = nn.Linear(in_features=64*24*24, out_features=600)\n        self.dropout = nn.Dropout(p=0.3)\n       \n        self.fc2 = nn.Linear(in_features=600, out_features=1)\n        \n\n    def forward(self, x):\n        # İlk blok\n        x = F.relu(self.bn1(self.cnn1(x)))\n        x = F.max_pool2d(x, kernel_size=2) # nn.MaxPool2d yerine F.max_pool2d kullanabiliriz\n\n        # İkinci blok\n        x = F.relu(self.bn2(self.cnn2(x)))\n        x = F.max_pool2d(x, kernel_size=2) # nn.MaxPool2d yerine F.max_pool2d kullanabiliriz\n\n        # Düzleştirme\n        x = x.view(x.size(0), -1) # Flatten\n\n        # Tam bağlı katmanlar\n        x = self.fc1(x)\n        x = F.relu(x)\n        x = self.dropout(x)\n\n        # Son çıktı (logit)\n        x = self.fc2(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:18:44.265288Z","iopub.execute_input":"2025-07-10T20:18:44.265972Z","iopub.status.idle":"2025-07-10T20:18:44.273870Z","shell.execute_reply.started":"2025-07-10T20:18:44.265940Z","shell.execute_reply":"2025-07-10T20:18:44.273168Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define loss function and optimizer","metadata":{}},{"cell_type":"code","source":"model = CNN().to(device)\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:19:57.869811Z","iopub.execute_input":"2025-07-10T20:19:57.870612Z","iopub.status.idle":"2025-07-10T20:19:58.103108Z","shell.execute_reply.started":"2025-07-10T20:19:57.870583Z","shell.execute_reply":"2025-07-10T20:19:58.102551Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Run the training loop","metadata":{}},{"cell_type":"code","source":"# Metrikleri kaydetmek için listeler\ntrain_losses = []\ntrain_accuracies = []\nval_losses = []\nval_accuracies = []\nval_aucs = [] # AUC skoru da önemli bir metrik\n\nbest_val_auc = 0.0 # En iyi modeli kaydetmek için\n\n\n\nnum_epochs = 5 \n\nprint(\"\\n--- Model Eğitimi Başlıyor ---\")\nfor epoch in range(num_epochs):\n   \n    model.train() \n    running_loss = 0.0\n    all_train_preds = []\n    all_train_labels = []\n\n    for images, labels in tqdm(train_loader, desc=f\"Epoch {epoch+1} Training\"):\n        images, labels = images.to(device), labels.to(device).unsqueeze(1) # Etiketlerin boyutunu (N,1) yap\n\n        \n        optimizer.zero_grad()\n\n        \n        outputs = model(images)\n\n       \n        loss = criterion(outputs, labels)\n\n        \n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item() * images.size(0) \n\n       \n        probabilities = torch.sigmoid(outputs)\n        all_train_preds.extend(probabilities.cpu().detach().numpy())\n        all_train_labels.extend(labels.cpu().detach().numpy())\n\n    epoch_train_loss = running_loss / len(train_loader.dataset)\n    \n    train_preds_binary = (np.array(all_train_preds) >= 0.5).astype(int)\n    epoch_train_accuracy = accuracy_score(all_train_labels, train_preds_binary)\n    epoch_train_auc = roc_auc_score(all_train_labels, np.array(all_train_preds))\n\n\n    train_losses.append(epoch_train_loss)\n    train_accuracies.append(epoch_train_accuracy)\n\n  \n    model.eval() \n    val_running_loss = 0.0\n    all_val_preds = []\n    all_val_labels = []\n\n    with torch.no_grad(): # Gradyan hesaplamayı kapat, bellekten tasarruf et\n        for images, labels in tqdm(val_loader, desc=f\"Epoch {epoch+1} Validation\"):\n            images, labels = images.to(device), labels.to(device).unsqueeze(1)\n\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n\n            val_running_loss += loss.item() * images.size(0)\n\n            probabilities = torch.sigmoid(outputs)\n            all_val_preds.extend(probabilities.cpu().detach().numpy())\n            all_val_labels.extend(labels.cpu().detach().numpy())\n\n    epoch_val_loss = val_running_loss / len(val_loader.dataset)\n    val_preds_binary = (np.array(all_val_preds) >= 0.5).astype(int)\n    epoch_val_accuracy = accuracy_score(all_val_labels, val_preds_binary)\n    epoch_val_auc = roc_auc_score(all_val_labels, np.array(all_val_preds))\n\n    val_losses.append(epoch_val_loss)\n    val_accuracies.append(epoch_val_accuracy)\n    val_aucs.append(epoch_val_auc)\n\n\n    print(f\"Epoch {epoch+1}/{num_epochs}:\")\n    print(f\"  Train Loss: {epoch_train_loss:.4f}, Train Acc: {epoch_train_accuracy:.4f}, Train AUC: {epoch_train_auc:.4f}\")\n    print(f\"  Val Loss:   {epoch_val_loss:.4f}, Val Acc:   {epoch_val_accuracy:.4f}, Val AUC:   {epoch_val_auc:.4f}\")\n\n    \n    if epoch_val_auc > best_val_auc:\n        best_val_auc = epoch_val_auc\n        torch.save(model.state_dict(), 'best_cnn_model.pth')\n        print(f\"  En iyi model kaydedildi! Val AUC: {best_val_auc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T20:19:58.143494Z","iopub.execute_input":"2025-07-10T20:19:58.144144Z","execution_failed":"2025-07-10T20:24:23.359Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Graphs Test","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(range(1, num_epochs + 1), train_losses, label='Eğitim Kaybı')\nplt.plot(range(1, num_epochs + 1), val_losses, label='Doğrulama Kaybı')\nplt.title('Kayıp (Loss) Grafiği')\nplt.xlabel('Epoch')\nplt.ylabel('Kayıp')\nplt.legend()\nplt.grid(True)\n\nplt.subplot(1, 2, 2)\nplt.plot(range(1, num_epochs + 1), train_accuracies, label='Eğitim Doğruluğu')\nplt.plot(range(1, num_epochs + 1), val_accuracies, label='Doğrulama Doğruluğu')\nplt.plot(range(1, num_epochs + 1), val_aucs, label='Doğrulama AUC', linestyle='--')\nplt.title('Doğruluk (Accuracy) ve AUC Grafiği')\nplt.xlabel('Epoch')\nplt.ylabel('Değer')\nplt.legend()\nplt.grid(True)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-10T20:24:23.360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}