{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# import biblioteque\n","metadata":{}},{"cell_type":"code","source":"from pathlib import Path  # Gestion des chemins de fichiers\nimport pydicom  # Lecture des fichiers médicaux DICOM\nimport numpy as np  # Calcul numérique\nimport pandas as pd  # Manipulation de données\nimport cv2  # Traitement d'image\nimport matplotlib.pyplot as plt  # Visualisation\nfrom tqdm.notebook import tqdm  # Barre de progression\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:30:29.794377Z","iopub.execute_input":"2026-03-26T14:30:29.794747Z","iopub.status.idle":"2026-03-26T14:30:29.799934Z","shell.execute_reply.started":"2026-03-26T14:30:29.794710Z","shell.execute_reply":"2026-03-26T14:30:29.798796Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# CHARGEMENT DES DONNÉES","metadata":{}},{"cell_type":"markdown","source":"# Chemin vers le dossier des images","metadata":{}},{"cell_type":"code","source":"chemin_images = Path(\"/kaggle/input/competitions/rsna-pneumonia-detection-challenge/stage_2_train_images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:30:29.801650Z","iopub.execute_input":"2026-03-26T14:30:29.802179Z","iopub.status.idle":"2026-03-26T14:30:29.806655Z","shell.execute_reply.started":"2026-03-26T14:30:29.802120Z","shell.execute_reply":"2026-03-26T14:30:29.805750Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Chargement du fichier CSV contenant les labels\n","metadata":{}},{"cell_type":"code","source":"labels = pd.read_csv(\"/kaggle/input/competitions/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\")\nlabels .head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:30:29.807781Z","iopub.execute_input":"2026-03-26T14:30:29.808105Z","iopub.status.idle":"2026-03-26T14:30:29.854791Z","shell.execute_reply.started":"2026-03-26T14:30:29.808058Z","shell.execute_reply":"2026-03-26T14:30:29.853840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels.count()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:30:29.857061Z","iopub.execute_input":"2026-03-26T14:30:29.857412Z","iopub.status.idle":"2026-03-26T14:30:29.867851Z","shell.execute_reply.started":"2026-03-26T14:30:29.857377Z","shell.execute_reply":"2026-03-26T14:30:29.866835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nlabels = labels.drop_duplicates(\"patientId\")\nlabels.count()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:30:29.868857Z","iopub.execute_input":"2026-03-26T14:30:29.869152Z","iopub.status.idle":"2026-03-26T14:30:29.884824Z","shell.execute_reply.started":"2026-03-26T14:30:29.869119Z","shell.execute_reply":"2026-03-26T14:30:29.883854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Setting the raw data path to the directory containing the RSNA Pneumonia Detection Challenge training images\nr_path = Path(\"/kaggle/input/competitions/rsna-pneumonia-detection-challenge/stage_2_train_images\")\n\n# Setting the processed data path to a directory named \"Processed\" for storing preprocessed or augmented data\ns_path = Path(\"Preprocessed_Data\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:30:29.885915Z","iopub.execute_input":"2026-03-26T14:30:29.886236Z","iopub.status.idle":"2026-03-26T14:30:29.890634Z","shell.execute_reply.started":"2026-03-26T14:30:29.886204Z","shell.execute_reply":"2026-03-26T14:30:29.889642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pydicom\n\n# إنشاء شبكة 9x9 من الـ subplots\nfig, axis = plt.subplots(9, 9, figsize=(25, 25))\n\n# عدّاد للوصول إلى الـ patient IDs في الـ DataFrame\nk = 0\n\n# المرور على جميع الـ subplots\nfor i in range(9):\n    for j in range(9):\n        # استخراج رقم المريض من الـ DataFrame\n        patient_ID = labels.patientId.iloc[k]\n\n        # مسار ملف الـ DICOM\n        dcm_path = r_path / f\"{patient_ID}.dcm\"\n\n        # قراءة ملف الـ DICOM والحصول على مصفوفة البكسلات\n        ds = pydicom.dcmread(str(dcm_path))\n        dcm = ds.pixel_array\n\n        # استخراج الـ label الخاص بالمريض الحالي\n        label = labels[\"Target\"].iloc[k]\n\n        # عرض الصورة في الـ subplot الحالي\n        axis[i, j].imshow(dcm, cmap=\"bone\")\n        axis[i, j].set_title(str(label))\n\n        # الانتقال إلى المريض التالي في الـ DataFrame\n        k += 1\n\n# ترتيب الواجهة لمنع تداخل الـ subplots\nplt.tight_layout()\n\n# عرض الشكل\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:37:28.461809Z","iopub.execute_input":"2026-03-26T14:37:28.462718Z","iopub.status.idle":"2026-03-26T14:37:56.072651Z","shell.execute_reply.started":"2026-03-26T14:37:28.462669Z","shell.execute_reply":"2026-03-26T14:37:56.071428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\n\n# تهيئة المتغيرات الخاصة بالمجاميع التراكمية\nsums, sums_sqr = 0.0, 0.0\n\n# المرور على البيانات مع شريط التقدّم\nfor k, patient_ID in enumerate(tqdm(labels.patientId)):\n    # قراءة ومعالجة صورة الـ DICOM\n    dcm_path = r_path / f\"{patient_ID}.dcm\"\n    \n    # استخدم dcmread بدلاً من read_file\n    ds = pydicom.dcmread(str(dcm_path))          # يقرأ ملف الـ DICOM [web:27][web:28]\n    dcm = ds.pixel_array.astype(np.float32) / 255.0   # تحويل لقيم بين 0 و 1 [web:27]\n\n    # تغيير حجم الصورة وتحويل النوع لـ float16\n    dcm_arr = cv2.resize(dcm, (224, 224)).astype(np.float16)\n\n    # استخراج الـ label وتحديد train أو val\n    label = labels.Target.iloc[k]\n    train_or_val = \"train\" if k < 24000 else \"val\"\n\n    # مسار الحفظ للبيانات المعالجة\n    current_save_path = s_path / train_or_val / str(label)\n    current_save_path.mkdir(parents=True, exist_ok=True)\n\n    # حفظ الصورة كـ NumPy array\n    np.save(current_save_path / patient_ID, dcm_arr)\n\n    # حساب المجاميع التراكمية لبيانات التدريب فقط\n    normalizer = 224 * 224\n    if train_or_val == \"train\":\n        sums += np.sum(dcm_arr) / normalizer\n        sums_sqr += (dcm_arr ** 2).sum() / normalizer\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:42:42.973646Z","iopub.execute_input":"2026-03-26T14:42:42.974024Z","iopub.status.idle":"2026-03-26T14:48:16.624279Z","shell.execute_reply.started":"2026-03-26T14:42:42.973986Z","shell.execute_reply":"2026-03-26T14:48:16.623288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len_train = (labels.patientId[:24000]).shape[0]\nlen_val   = (labels.patientId[24000:]).shape[0]\nprint(\"train images:\", len_train)\nprint(\"val images:\", len_val)\nprint(\"sums:\", sums)\nprint(\"sums_sqr:\", sums_sqr)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:52:09.908833Z","iopub.execute_input":"2026-03-26T14:52:09.909187Z","iopub.status.idle":"2026-03-26T14:52:09.915446Z","shell.execute_reply.started":"2026-03-26T14:52:09.909150Z","shell.execute_reply":"2026-03-26T14:52:09.914525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Importing the PyTorch library for deep learning functionality\nimport torch\n\n# Importing the torchvision library for computer vision tasks, including pre-trained models and datasets\nimport torchvision\n\n# Importing the transforms module from torchvision for data augmentation, normalization, and other image transformations\nfrom torchvision import transforms, datasets\n\n# Importing the torchmetrics library for additional metrics beyond what PyTorch provides\nimport torchmetrics\n\n# Importing the pytorch_lightning library for simplifying the training phase of PyTorch models\nimport pytorch_lightning as pl\n\n# Importing the ModelCheckpoint callback from pytorch_lightning.callbacks for saving the best model during training\nfrom pytorch_lightning.callbacks import ModelCheckpoint\n\n# Importing the tqdm library for displaying progress bars during training and evaluation\nfrom tqdm.notebook import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T15:00:05.687196Z","iopub.execute_input":"2026-03-26T15:00:05.687632Z","iopub.status.idle":"2026-03-26T15:00:05.693637Z","shell.execute_reply.started":"2026-03-26T15:00:05.687584Z","shell.execute_reply":"2026-03-26T15:00:05.692672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_file(path):\n    return np.load(path).astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T15:00:09.282482Z","iopub.execute_input":"2026-03-26T15:00:09.282830Z","iopub.status.idle":"2026-03-26T15:00:09.287408Z","shell.execute_reply.started":"2026-03-26T15:00:09.282794Z","shell.execute_reply":"2026-03-26T15:00:09.286418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_trans = transforms.Compose([transforms.ToTensor(),\n                                     transforms.Normalize(.5, .25),\n                                     transforms.RandomAffine(degrees=(-10,10), translate = (0, 0.1), scale = (0.8, 1.2)),\n                                     transforms.RandomResizedCrop((224, 224), scale=(0.4, 1))])\nval_trans = transforms.Compose([transforms.ToTensor(),\n                                     transforms.Normalize(.5, .25)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T15:00:12.127015Z","iopub.execute_input":"2026-03-26T15:00:12.127390Z","iopub.status.idle":"2026-03-26T15:00:12.134151Z","shell.execute_reply.started":"2026-03-26T15:00:12.127356Z","shell.execute_reply":"2026-03-26T15:00:12.133094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = torchvision.datasets.DatasetFolder(\"Preprocessed_Data/train/\", loader=load_file, extensions=\"npy\", transform=train_trans)\nvalidation = torchvision.datasets.DatasetFolder(\"Preprocessed_Data/val/\", loader=load_file, extensions=\"npy\", transform=val_trans)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T15:00:50.530515Z","iopub.execute_input":"2026-03-26T15:00:50.530884Z","iopub.status.idle":"2026-03-26T15:00:50.613761Z","shell.execute_reply.started":"2026-03-26T15:00:50.530847Z","shell.execute_reply":"2026-03-26T15:00:50.612924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(train), len(validation))\nsample, target = train[0]\nprint(sample.shape, target)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T15:06:18.210964Z","iopub.execute_input":"2026-03-26T15:06:18.211348Z","iopub.status.idle":"2026-03-26T15:06:18.355960Z","shell.execute_reply.started":"2026-03-26T15:06:18.211285Z","shell.execute_reply":"2026-03-26T15:06:18.354623Z"}},"outputs":[],"execution_count":null}]}