{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pydicom\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nfrom sklearn.metrics import confusion_matrix, accuracy_score, precision_score, recall_score, f1_score, roc_auc_score, roc_curve\nfrom tensorflow.keras import layers, models, regularizers\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T06:21:50.623203Z","iopub.execute_input":"2025-06-04T06:21:50.624007Z","iopub.status.idle":"2025-06-04T06:22:09.567657Z","shell.execute_reply.started":"2025-06-04T06:21:50.623978Z","shell.execute_reply":"2025-06-04T06:22:09.566711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 227 \nBATCH_SIZE = 32\nEPOCHS = 15\nLEARNING_RATE = 1e-5 \n\nDICOM_DATA_DIR = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images\"\nLABELS_CSV = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\"\nPNG_OUTPUT_DIR = \"/kaggle/working/rsna_pneumonia_png_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T01:03:17.115468Z","iopub.execute_input":"2025-06-04T01:03:17.116120Z","iopub.status.idle":"2025-06-04T01:03:17.121451Z","shell.execute_reply.started":"2025-06-04T01:03:17.116094Z","shell.execute_reply":"2025-06-04T01:03:17.120501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.makedirs(PNG_OUTPUT_DIR, exist_ok=True) # Çıktı dizinini oluştur\n\n# Etiket DataFrame'ini yükle\ndf_labels = pd.read_csv(LABELS_CSV)\npatient_ids = df_labels['patientId'].unique()\n\nprint(f\"{len(patient_ids)} adet DICOM dosyası PNG'ye dönüştürülüyor...\")\n\nfor i, patient_id in enumerate(patient_ids):\n    dicom_path = os.path.join(DICOM_DATA_DIR, patient_id + \".dcm\")\n    output_path = os.path.join(PNG_OUTPUT_DIR, patient_id + \".png\")\n\n    # Eğer dosya zaten dönüştürülmüşse atla\n    if os.path.exists(output_path):\n        continue\n\n    try:\n        dicom = pydicom.dcmread(dicom_path)\n        img = dicom.pixel_array.astype(np.float32)\n\n        # DICOM Pencereleme Uygula\n        if 'WindowCenter' in dicom and 'WindowWidth' in dicom:\n            window_center = dicom.WindowCenter\n            window_width = dicom.WindowWidth\n\n            # Birden fazla pencere ayarı varsa ilkini kullan\n            if isinstance(window_center, pydicom.multival.MultiValue):\n                window_center = window_center[0]\n            if isinstance(window_width, pydicom.multival.MultiValue):\n                window_width = window_width[0]\n\n            min_val = window_center - window_width / 2\n            max_val = window_center + window_width / 2\n\n            img = np.clip(img, min_val, max_val)\n            # 0-255 aralığına ölçekle (PNG olarak kaydetmek için)\n            img = ((img - min_val) / (max_val - min_val + 1e-5)) * 255\n        else:\n            # Pencere bilgisi yoksa veya tanımsızsa basit min-max normalizasyona geri dön\n            img = (img - np.min(img)) / (np.max(img) - np.min(img) + 1e-5) * 255\n\n        img = img.astype(np.uint8)\n        \n        # Görüntüyü yeniden boyutlandır \n        img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n\n        # PNG olarak kaydet\n        cv2.imwrite(output_path, img)\n\n    except Exception as e:\n        print(f\"Hata: {patient_id}.dcm dönüştürülürken hata oluştu: {e}\")\n        continue\n    \n    if (i + 1) % 1000 == 0:\n        print(f\"{i + 1} görüntü dönüştürüldü.\")\n\nprint(\"Tüm DICOM dosyaları PNG'ye dönüştürüldü.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T01:03:17.122520Z","iopub.execute_input":"2025-06-04T01:03:17.122888Z","iopub.status.idle":"2025-06-04T01:14:37.112289Z","shell.execute_reply.started":"2025-06-04T01:03:17.122856Z","shell.execute_reply":"2025-06-04T01:14:37.111238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(LABELS_CSV)\ndf = df.drop_duplicates(subset=\"patientId\")[['patientId', 'Target']]\ndf[\"filename\"] = df[\"patientId\"] + \".png\"\n\nsns.countplot(x=df[\"Target\"])\nplt.title(\"Sınıf Dağılımı\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T01:14:37.113304Z","iopub.execute_input":"2025-06-04T01:14:37.113634Z","iopub.status.idle":"2025-06-04T01:14:37.411102Z","shell.execute_reply.started":"2025-06-04T01:14:37.113604Z","shell.execute_reply":"2025-06-04T01:14:37.409931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Eğitim-Doğrulama-Test Ayırma\ntrain_val_df, test_df = train_test_split(df, test_size=0.2, stratify=df[\"Target\"], random_state=42)\ntrain_df, val_df = train_test_split(train_val_df, test_size=0.1, stratify=train_val_df[\"Target\"], random_state=42)\n\ntrain_df['Target'] = train_df['Target'].astype(str)\nval_df['Target'] = val_df['Target'].astype(str)\ntest_df['Target'] = test_df['Target'].astype(str)\n\nprint(f\"Eğitim seti boyutu: {len(train_df)}\")\nprint(f\"Doğrulama seti boyutu: {len(val_df)}\")\nprint(f\"Test seti boyutu: {len(test_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T01:14:37.411854Z","iopub.execute_input":"2025-06-04T01:14:37.412196Z","iopub.status.idle":"2025-06-04T01:14:37.463622Z","shell.execute_reply.started":"2025-06-04T01:14:37.412176Z","shell.execute_reply":"2025-06-04T01:14:37.462589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sınıf ağırlıkları\nclass_weights = class_weight.compute_class_weight('balanced', classes=np.unique(train_df[\"Target\"]), y=train_df[\"Target\"])\nclass_weights = dict(enumerate(class_weights))\nprint(\"Sınıf Ağırlıkları (Class Weights):\", class_weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T01:14:37.464650Z","iopub.execute_input":"2025-06-04T01:14:37.464940Z","iopub.status.idle":"2025-06-04T01:14:37.492657Z","shell.execute_reply.started":"2025-06-04T01:14:37.464918Z","shell.execute_reply":"2025-06-04T01:14:37.491384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=15,       # Maksimum 15 derece döndürme\n    zoom_range=0.15,         # %15'e kadar yakınlaştırma\n    width_shift_range=0.1,   # %10 genişlik kaydırma\n    height_shift_range=0.1,  # %10 yükseklik kaydırma\n    horizontal_flip=True,    # Yatay çevirme \n    fill_mode='nearest'      # Boş kalan pikselleri en yakın değerle doldur\n)\n\n# Doğrulama ve test için sadece yeniden ölçeklendirme yapılır, veri artırma uygulanmaz.\nval_test_datagen = ImageDataGenerator(rescale=1./255)\n\n# GENERATORLAR (flow_from_dataframe kullanılarak)\n# directory: Dönüştürülmüş PNG'lerin bulunduğu dizini gösterir.\n# x_col: DataFrame'deki dosya adlarını içeren sütun adı.\n# y_col: DataFrame'deki etiketleri içeren sütun adı.\n# target_size: Görüntülerin yeniden boyutlandırılacağı hedef boyut.\n# class_mode: İkili sınıflandırma (0 veya 1) olduğu için 'binary'.\n# color_mode: AlexNet 3 kanal beklediği için 'rgb' kullanıldı. Gri tonlamalı PNG'leri 3 kanala kopyalayacak.\n# shuffle: Eğitim ve doğrulama için karıştırma açılır, test için kapatılır.\n\ntrain_gen = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=PNG_OUTPUT_DIR,\n    x_col=\"filename\",\n    y_col=\"Target\",\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode=\"binary\",\n    color_mode=\"rgb\", \n    shuffle=True\n)\n\nval_gen = val_test_datagen.flow_from_dataframe(\n    dataframe=val_df,\n    directory=PNG_OUTPUT_DIR, \n    x_col=\"filename\",\n    y_col=\"Target\",\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode=\"binary\",\n    color_mode=\"rgb\", \n    shuffle=False # Doğrulama için karıştırma kapalı\n)\n\ntest_gen = val_test_datagen.flow_from_dataframe(\n    dataframe=test_df,\n    directory=PNG_OUTPUT_DIR, \n    x_col=\"filename\",\n    y_col=\"Target\",\n    target_size=(IMG_SIZE, IMG_SIZE),\n    batch_size=BATCH_SIZE,\n    class_mode=\"binary\",\n    color_mode=\"rgb\", \n    shuffle=False # Test için karıştırma kapalı (önemli: y_true ile eşleşmesi için)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T01:14:37.495628Z","iopub.execute_input":"2025-06-04T01:14:37.495990Z","iopub.status.idle":"2025-06-04T01:14:37.820857Z","shell.execute_reply.started":"2025-06-04T01:14:37.495936Z","shell.execute_reply":"2025-06-04T01:14:37.819508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef build_alexnet_classic(input_shape=(227, 227, 3)):\n    model = models.Sequential([\n        layers.Conv2D(96, (11, 11), strides=4, activation='relu', input_shape=input_shape),\n        layers.MaxPooling2D(pool_size=(3, 3), strides=2),\n\n        layers.Conv2D(256, (5, 5), padding='same', activation='relu'),\n        layers.MaxPooling2D(pool_size=(3, 3), strides=2),\n\n        layers.Conv2D(384, (3, 3), padding='same', activation='relu'),\n        layers.Conv2D(384, (3, 3), padding='same', activation='relu'),\n        layers.Conv2D(256, (3, 3), padding='same', activation='relu'),\n        layers.MaxPooling2D(pool_size=(3, 3), strides=2),\n\n        layers.Flatten(),\n        layers.Dense(4096, activation='relu'),\n        layers.Dropout(0.5),\n        layers.Dense(4096, activation='relu'),\n        layers.Dropout(0.5),\n        layers.Dense(1, activation='sigmoid', dtype='float32')  # İkili sınıflandırma için sigmoid\n    ])\n    \n    return model\n\nmodel = build_alexnet_classic()\noptimizer = Adam(learning_rate=LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T01:14:37.822036Z","iopub.execute_input":"2025-06-04T01:14:37.822399Z","iopub.status.idle":"2025-06-04T01:14:39.444360Z","shell.execute_reply.started":"2025-06-04T01:14:37.822364Z","shell.execute_reply":"2025-06-04T01:14:39.440831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"early_stop = EarlyStopping(monitor='val_loss', patience=4, restore_best_weights=True, verbose=1)\n\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, # Öğrenme oranını %20'ye düşür\n                               patience=3, \n                               min_lr=1e-7, \n                               verbose=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T01:14:39.447791Z","iopub.execute_input":"2025-06-04T01:14:39.449041Z","iopub.status.idle":"2025-06-04T01:14:39.464409Z","shell.execute_reply.started":"2025-06-04T01:14:39.448942Z","shell.execute_reply":"2025-06-04T01:14:39.461136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=EPOCHS,\n    class_weight=class_weights, \n    callbacks=[early_stop, reduce_lr]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T01:14:39.468238Z","iopub.execute_input":"2025-06-04T01:14:39.470514Z","execution_failed":"2025-06-04T02:33:41.477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(history.history['accuracy'], label='Eğitim Doğruluğu')\nplt.plot(history.history['val_accuracy'], label='Doğrulama Doğruluğu')\nplt.legend()\nplt.title(\"Model Doğruluğu\")\nplt.xlabel(\"Epok\")\nplt.ylabel(\"Doğruluk\")\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='Eğitim Kaybı')\nplt.plot(history.history['val_loss'], label='Doğrulama Kaybı')\nplt.legend()\nplt.title(\"Model Kaybı\")\nplt.xlabel(\"Epok\")\nplt.ylabel(\"Kayıp\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-04T02:33:41.478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_steps = int(np.ceil(len(test_df) / BATCH_SIZE))\ny_pred_probs = model.predict(test_gen, steps=test_steps)\ny_pred = (y_pred_probs > 0.6).astype(int).flatten() \ny_true = test_gen.labels[:len(y_pred)] \n\n# Karışıklık Matrisi\ncm = confusion_matrix(y_true, y_pred)\naccuracy = accuracy_score(y_true, y_pred)\nprecision = precision_score(y_true, y_pred)\nsensitivity = recall_score(y_true, y_pred) \nspecificity = cm[0, 0] / (cm[0, 0] + cm[0, 1]) \nf1 = f1_score(y_true, y_pred)\n\nprint(pd.DataFrame({\n    \"Metrik\": [\"Doğruluk (Accuracy)\", \"Kesinlik (Precision)\", \"Duyarlılık (Sensitivity)\", \"Özgüllük (Specificity)\", \"F1 Skoru\"],\n    \"Değer\": [accuracy, precision, sensitivity, specificity, f1]\n}))","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-04T02:33:41.478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=['Tahmin: Normal', 'Tahmin: Pnömoni'],\n            yticklabels=['Gerçek: Normal', 'Gerçek: Pnömoni'])\nplt.title(\"Karışıklık Matrisi\")\nplt.xlabel(\"Tahmin Edilen Etiket\")\nplt.ylabel(\"Gerçek Etiket\")\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-04T02:33:41.478Z"}},"outputs":[],"execution_count":null}]}