{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport pydicom\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras import layers, models, regularizers\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nfrom sklearn.utils.class_weight import compute_class_weight","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T07:23:16.733745Z","iopub.execute_input":"2025-06-04T07:23:16.734082Z","iopub.status.idle":"2025-06-04T07:23:16.739743Z","shell.execute_reply.started":"2025-06-04T07:23:16.734059Z","shell.execute_reply":"2025-06-04T07:23:16.739016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 227\nBATCH_SIZE = 32\nEPOCHS = 16\nLEARNING_RATE = 1e-5\n\nDICOM_DATA_DIR = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images\"\nLABELS_CSV = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\"\nPNG_OUTPUT_DIR = \"/kaggle/working/rsna_pneumonia_png_images\"\n\nos.makedirs(PNG_OUTPUT_DIR, exist_ok=True)\n\n# --- DICOM'dan PNG'ye dönüşüm (senin verdiğin kodla aynı) ---\ndf_labels = pd.read_csv(LABELS_CSV)\npatient_ids = df_labels['patientId'].unique()\n\nprint(f\"{len(patient_ids)} adet DICOM dosyası PNG'ye dönüştürülüyor...\")\n\nfor i, patient_id in enumerate(patient_ids):\n    dicom_path = os.path.join(DICOM_DATA_DIR, patient_id + \".dcm\")\n    output_path = os.path.join(PNG_OUTPUT_DIR, patient_id + \".png\")\n\n    if os.path.exists(output_path):\n        continue\n\n    try:\n        dicom = pydicom.dcmread(dicom_path)\n        img = dicom.pixel_array.astype(np.float32)\n\n        if 'WindowCenter' in dicom and 'WindowWidth' in dicom:\n            window_center = dicom.WindowCenter\n            window_width = dicom.WindowWidth\n            if isinstance(window_center, pydicom.multival.MultiValue):\n                window_center = window_center[0]\n            if isinstance(window_width, pydicom.multival.MultiValue):\n                window_width = window_width[0]\n\n            min_val = window_center - window_width / 2\n            max_val = window_center + window_width / 2\n\n            img = np.clip(img, min_val, max_val)\n            img = ((img - min_val) / (max_val - min_val + 1e-5)) * 255\n        else:\n            img = (img - np.min(img)) / (np.max(img) - np.min(img) + 1e-5) * 255\n\n        img = img.astype(np.uint8)\n        img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n        cv2.imwrite(output_path, img)\n\n    except Exception as e:\n        print(f\"Hata: {patient_id}.dcm dönüştürülürken hata oluştu: {e}\")\n        continue\n\n    if (i + 1) % 1000 == 0:\n        print(f\"{i + 1} görüntü dönüştürüldü.\")\n\nprint(\"Tüm DICOM dosyaları PNG'ye dönüştürüldü.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T07:23:16.740843Z","iopub.execute_input":"2025-06-04T07:23:16.741055Z","iopub.status.idle":"2025-06-04T07:23:16.945712Z","shell.execute_reply.started":"2025-06-04T07:23:16.741040Z","shell.execute_reply":"2025-06-04T07:23:16.945083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Veri setini hazırla ---\ndf_labels['filename'] = df_labels['patientId'] + \".png\"\ndf_labels['Target'] = df_labels['Target'].astype(str)\n\ntrain_val_df, test_df = train_test_split(df_labels, test_size=0.2, stratify=df_labels[\"Target\"], random_state=42)\ntrain_df, val_df = train_test_split(train_val_df, test_size=0.1, stratify=train_val_df[\"Target\"], random_state=42)\n\nprint(f\"Eğitim seti boyutu: {len(train_df)}\")\nprint(f\"Doğrulama seti boyutu: {len(val_df)}\")\nprint(f\"Test seti boyutu: {len(test_df)}\")\n\nclass_weights = class_weight.compute_class_weight('balanced', classes=np.unique(train_df[\"Target\"]), y=train_df[\"Target\"])\nclass_weights = dict(enumerate(class_weights))\nprint(\"Sınıf Ağırlıkları (Class Weights):\", class_weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T07:23:16.946486Z","iopub.execute_input":"2025-06-04T07:23:16.946798Z","iopub.status.idle":"2025-06-04T07:23:17.046924Z","shell.execute_reply.started":"2025-06-04T07:23:16.946779Z","shell.execute_reply":"2025-06-04T07:23:17.046205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- TF Dataset fonksiyonu ---\ndef process_path(filename, label):\n    img_path = tf.strings.join([PNG_OUTPUT_DIR, \"/\", filename])\n    img = tf.io.read_file(img_path)\n    img = tf.io.decode_png(img, channels=1)  # Gri tonlama\n    img = tf.image.grayscale_to_rgb(img)    # 3 kanal yapıyoruz\n    img = tf.image.convert_image_dtype(img, tf.float32)  # 0-1 arası normalize\n\n    # Veri artırma (sadece eğitim için)\n    img = tf.image.random_flip_left_right(img)\n    img = tf.image.random_brightness(img, max_delta=0.1)\n    img = tf.image.random_zoom(img, (0.85, 1.15)) if hasattr(tf.image, \"random_zoom\") else img  # TF 2.10 ve üzeri için, değilse silebilirsin\n    img = tf.image.resize(img, [IMG_SIZE, IMG_SIZE])\n\n    return img, label\n\ndef process_path_no_aug(filename, label):\n    img_path = tf.strings.join([PNG_OUTPUT_DIR, \"/\", filename])\n    img = tf.io.read_file(img_path)\n    img = tf.io.decode_png(img, channels=1)\n    img = tf.image.grayscale_to_rgb(img)\n    img = tf.image.convert_image_dtype(img, tf.float32)\n    img = tf.image.resize(img, [IMG_SIZE, IMG_SIZE])\n    return img, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T07:23:17.048319Z","iopub.execute_input":"2025-06-04T07:23:17.048515Z","iopub.status.idle":"2025-06-04T07:23:17.054925Z","shell.execute_reply.started":"2025-06-04T07:23:17.048499Z","shell.execute_reply":"2025-06-04T07:23:17.054181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Label'ları int yapıyoruz\ntrain_labels = train_df['Target'].astype(int).values\nval_labels = val_df['Target'].astype(int).values\ntest_labels = test_df['Target'].astype(int).values\n\ntrain_ds = tf.data.Dataset.from_tensor_slices((train_df['filename'].values, train_labels))\nval_ds = tf.data.Dataset.from_tensor_slices((val_df['filename'].values, val_labels))\ntest_ds = tf.data.Dataset.from_tensor_slices((test_df['filename'].values, test_labels))\n\ntrain_ds = train_ds.shuffle(1000).map(process_path, num_parallel_calls=tf.data.AUTOTUNE).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\nval_ds = val_ds.map(process_path_no_aug, num_parallel_calls=tf.data.AUTOTUNE).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\ntest_ds = test_ds.map(process_path_no_aug, num_parallel_calls=tf.data.AUTOTUNE).batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T07:23:17.055763Z","iopub.execute_input":"2025-06-04T07:23:17.056323Z","iopub.status.idle":"2025-06-04T07:23:17.320347Z","shell.execute_reply.started":"2025-06-04T07:23:17.056300Z","shell.execute_reply":"2025-06-04T07:23:17.319755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_vgg16_from_scratch(input_shape=(IMG_SIZE, IMG_SIZE, 3)):\n    l2_reg = regularizers.l2(0.0005)\n\n    model = models.Sequential()\n\n    # Block 1\n    model.add(layers.Conv2D(64, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg, input_shape=input_shape))\n    model.add(layers.Conv2D(64, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.MaxPooling2D((2,2), strides=(2,2)))\n\n    # Block 2\n    model.add(layers.Conv2D(128, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.Conv2D(128, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.MaxPooling2D((2,2), strides=(2,2)))\n\n    # Block 3\n    model.add(layers.Conv2D(256, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.Conv2D(256, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.Conv2D(256, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.MaxPooling2D((2,2), strides=(2,2)))\n\n    # Block 4\n    model.add(layers.Conv2D(512, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.Conv2D(512, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.Conv2D(512, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.MaxPooling2D((2,2), strides=(2,2)))\n\n    # Block 5\n    model.add(layers.Conv2D(512, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.Conv2D(512, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.Conv2D(512, (3,3), activation='relu', padding='same', kernel_regularizer=l2_reg))\n    model.add(layers.MaxPooling2D((2,2), strides=(2,2)))\n\n    model.add(layers.Flatten())\n\n    model.add(layers.Dense(4096, activation='relu', kernel_regularizer=l2_reg))\n    model.add(layers.Dropout(0.5))\n\n    model.add(layers.Dense(4096, activation='relu', kernel_regularizer=l2_reg))\n    model.add(layers.Dropout(0.5))\n\n    model.add(layers.Dense(1, activation='sigmoid', dtype='float32'))\n\n    return model\n\nmodel = build_vgg16_from_scratch()\n\noptimizer = tf.keras.optimizers.Adam(learning_rate=LEARNING_RATE)\nmodel.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n\n# Callbacklar\nearly_stop = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=4, restore_best_weights=True, verbose=1)\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=2, verbose=1)\n\n# Model özet\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T07:23:17.321006Z","iopub.execute_input":"2025-06-04T07:23:17.321237Z","iopub.status.idle":"2025-06-04T07:23:17.606049Z","shell.execute_reply.started":"2025-06-04T07:23:17.321220Z","shell.execute_reply":"2025-06-04T07:23:17.605375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Modeli eğit ---\nhistory = model.fit(\n    train_ds,\n    epochs=EPOCHS,\n    validation_data=val_ds,\n    class_weight=class_weights,\n    callbacks=[early_stop, reduce_lr]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T07:23:17.606948Z","iopub.execute_input":"2025-06-04T07:23:17.607214Z","iopub.status.idle":"2025-06-04T08:09:08.221096Z","shell.execute_reply.started":"2025-06-04T07:23:17.607183Z","shell.execute_reply":"2025-06-04T08:09:08.220531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ndef plot_history(hist):\n    acc = hist.history['accuracy']\n    val_acc = hist.history['val_accuracy']\n    loss = hist.history['loss']\n    val_loss = hist.history['val_loss']\n    epochs = range(1, len(acc) + 1)\n\n    plt.figure(figsize=(14,5))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(epochs, acc, 'b-', label='Train Acc')\n    plt.plot(epochs, val_acc, 'r-', label='Val Acc')\n    plt.title('Accuracy')\n    plt.legend()\n\n    plt.subplot(1, 2, 2)\n    plt.plot(epochs, loss, 'b-', label='Train Loss')\n    plt.plot(epochs, val_loss, 'r-', label='Val Loss')\n    plt.title('Loss')\n    plt.legend()\n    \n    plt.show()\n\nplot_history(history)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T08:09:08.222548Z","iopub.execute_input":"2025-06-04T08:09:08.222734Z","iopub.status.idle":"2025-06-04T08:09:08.553581Z","shell.execute_reply.started":"2025-06-04T08:09:08.222720Z","shell.execute_reply":"2025-06-04T08:09:08.552778Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Gerçek etiketleri ve tahminleri toplamak için listeler\ny_true = []\ny_pred_probs = []\n\n# Dataset'ten verileri çek\nfor batch in val_ds:\n    X_batch, y_batch = batch\n    y_true.extend(y_batch.numpy())  # Gerçek etiketleri topla\n    preds = model.predict(X_batch, verbose=0)  # Tahmin olasılıkları\n    y_pred_probs.extend(preds)\n\n# Listeyi numpy dizisine çevir\ny_true = np.array(y_true)\ny_pred_probs = np.array(y_pred_probs).flatten()\ny_pred = (y_pred_probs > 0.5).astype(int)\n\n# Metrikleri hesapla\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\n\nacc = accuracy_score(y_true, y_pred)\nprec = precision_score(y_true, y_pred)\nrec = recall_score(y_true, y_pred)\nf1 = f1_score(y_true, y_pred)\nroc_auc = roc_auc_score(y_true, y_pred_probs)\n\nprint(f\"🔹 Accuracy     : {acc:.4f}\")\nprint(f\"🔹 Precision    : {prec:.4f}\")\nprint(f\"🔹 Recall       : {rec:.4f}\")\nprint(f\"🔹 F1-Score     : {f1:.4f}\")\nprint(f\"🔹 ROC AUC      : {roc_auc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-04T08:09:08.555145Z","iopub.execute_input":"2025-06-04T08:09:08.555433Z","iopub.status.idle":"2025-06-04T08:09:21.582838Z","shell.execute_reply.started":"2025-06-04T08:09:08.555416Z","shell.execute_reply":"2025-06-04T08:09:21.582112Z"}},"outputs":[],"execution_count":null}]}