{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importing the Path class from the pathlib module for handling file paths in a platform-independent way\nfrom pathlib import Path\n\n# Importing the pydicom library for reading and working with DICOM (Digital Imaging and Communications in Medicine) files\nimport pydicom\n\n# Importing the NumPy library for numerical operations and array manipulations\nimport numpy as np\n\n# Importing the pandas library for data manipulation and analysis\nimport pandas as pd\n\n# Importing the OpenCV library for computer vision tasks, including image processing and computer vision algorithms\nimport cv2\n\n# Importing the pyplot module from the matplotlib library for creating visualizations and plots\nimport matplotlib.pyplot as plt\n\n# Importing the tqdm library for displaying progress bars in a notebook environment\nfrom tqdm.notebook import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-26T22:06:53.092850Z","iopub.execute_input":"2024-01-26T22:06:53.093185Z","iopub.status.idle":"2024-01-26T22:06:53.920181Z","shell.execute_reply.started":"2024-01-26T22:06:53.093155Z","shell.execute_reply":"2024-01-26T22:06:53.919239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:06:53.922194Z","iopub.execute_input":"2024-01-26T22:06:53.922706Z","iopub.status.idle":"2024-01-26T22:06:53.984464Z","shell.execute_reply.started":"2024-01-26T22:06:53.922673Z","shell.execute_reply":"2024-01-26T22:06:53.983727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:06:53.985867Z","iopub.execute_input":"2024-01-26T22:06:53.986350Z","iopub.status.idle":"2024-01-26T22:06:54.011051Z","shell.execute_reply.started":"2024-01-26T22:06:53.986325Z","shell.execute_reply":"2024-01-26T22:06:54.009847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = labels.drop_duplicates(\"patientId\")","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:06:54.013130Z","iopub.execute_input":"2024-01-26T22:06:54.013472Z","iopub.status.idle":"2024-01-26T22:06:54.030586Z","shell.execute_reply.started":"2024-01-26T22:06:54.013437Z","shell.execute_reply":"2024-01-26T22:06:54.029352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting the raw data path to the directory containing the RSNA Pneumonia Detection Challenge training images\nr_path = Path(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/\")\n\n# Setting the processed data path to a directory named \"Processed\" for storing preprocessed or augmented data\ns_path = Path(\"Preprocessed_Data\")","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:06:54.031841Z","iopub.execute_input":"2024-01-26T22:06:54.032544Z","iopub.status.idle":"2024-01-26T22:06:54.036058Z","shell.execute_reply.started":"2024-01-26T22:06:54.032516Z","shell.execute_reply":"2024-01-26T22:06:54.035437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a 3x3 grid of subplots\nfig, axis = plt.subplots(9, 9, figsize=(25, 25))\n# Initialize a counter for accessing patient IDs\nk = 0\n\n# Loop through the subplots\nfor i in range(9):\n    for j in range(9):\n        # Extract the patient ID from the DataFrame\n        patient_ID = labels.patientId.iloc[k]\n        \n        # Construct the path to the DICOM file\n        dcm_path = r_path / f\"{patient_ID}.dcm\"\n        # Read the DICOM file and get the pixel array\n        dcm = pydicom.read_file(dcm_path).pixel_array\n        \n        # Extract the label associated with the current patient\n        label = labels[\"Target\"].iloc[k]\n\n        # Display the DICOM image on the current subplot\n        axis[i, j].imshow(dcm, cmap=\"bone\")\n        # Set the title of the subplot to the extracted label\n        axis[i, j].set_title(label)\n        \n        # Move to the next patient in the DataFrame\n        k += 1\n\n# Adjust layout to prevent subplot overlap\nplt.tight_layout()\n# Display the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:06:54.037175Z","iopub.execute_input":"2024-01-26T22:06:54.037455Z","iopub.status.idle":"2024-01-26T22:07:11.659067Z","shell.execute_reply.started":"2024-01-26T22:06:54.037431Z","shell.execute_reply":"2024-01-26T22:07:11.657438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize variables to keep track of cumulative sums\nsums, sums_sqr = 0, 0\n\n# Loop through data with progress bar\nfor k, patient_ID in enumerate(tqdm(labels.patientId)):\n    # Read and preprocess DICOM image\n    dcm_path = r_path / f\"{patient_ID}.dcm\"\n    dcm = pydicom.read_file(dcm_path).pixel_array / 255\n\n    # Resize and convert image to float\n    dcm_arr = cv2.resize(dcm, (224, 224)).astype(np.float16)\n\n    # Extract label and determine training/validation\n    label = labels.Target.iloc[k]\n    train_or_val = \"train\" if k < 24000 else \"val\"\n\n    # Create save path for processed data\n    current_save_path = s_path / train_or_val / str(label)\n    current_save_path.mkdir(parents=True, exist_ok=True)\n\n    # Save processed data as NumPy array\n    np.save(current_save_path / patient_ID, dcm_arr)\n\n    # Calculate cumulative sums for training data\n    normalizer = 224 * 224\n    if train_or_val == 'train':\n        sums += np.sum(dcm_arr) / normalizer\n        sums_sqr += (dcm_arr ** 2).sum() / normalizer","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:07:11.660716Z","iopub.execute_input":"2024-01-26T22:07:11.661127Z","iopub.status.idle":"2024-01-26T22:15:34.937338Z","shell.execute_reply.started":"2024-01-26T22:07:11.661088Z","shell.execute_reply":"2024-01-26T22:15:34.935713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pytorch-lightning","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:15:34.939992Z","iopub.execute_input":"2024-01-26T22:15:34.940651Z","iopub.status.idle":"2024-01-26T22:15:46.910449Z","shell.execute_reply.started":"2024-01-26T22:15:34.940612Z","shell.execute_reply":"2024-01-26T22:15:46.908646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the PyTorch library for deep learning functionality\nimport torch\n\n# Importing the torchvision library for computer vision tasks, including pre-trained models and datasets\nimport torchvision\n\n# Importing the transforms module from torchvision for data augmentation, normalization, and other image transformations\nfrom torchvision import transforms, datasets\n\n# Importing the torchmetrics library for additional metrics beyond what PyTorch provides\nimport torchmetrics\n\n# Importing the pytorch_lightning library for simplifying the training phase of PyTorch models\nimport pytorch_lightning as pl\n\n# Importing the ModelCheckpoint callback from pytorch_lightning.callbacks for saving the best model during training\nfrom pytorch_lightning.callbacks import ModelCheckpoint\n\n# Importing the tqdm library for displaying progress bars during training and evaluation\nfrom tqdm.notebook import tqdm\n","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:15:46.912426Z","iopub.execute_input":"2024-01-26T22:15:46.912765Z","iopub.status.idle":"2024-01-26T22:15:55.719491Z","shell.execute_reply.started":"2024-01-26T22:15:46.912735Z","shell.execute_reply":"2024-01-26T22:15:55.717838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Creation of train and validation dataset\n","metadata":{}},{"cell_type":"code","source":"def load_file(path):\n    return np.load(path).astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:15:55.722964Z","iopub.execute_input":"2024-01-26T22:15:55.724078Z","iopub.status.idle":"2024-01-26T22:15:55.728736Z","shell.execute_reply.started":"2024-01-26T22:15:55.724035Z","shell.execute_reply":"2024-01-26T22:15:55.727490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_trans = transforms.Compose([transforms.ToTensor(),\n                                     transforms.Normalize(.5, .25),\n                                     transforms.RandomAffine(degrees=(-10,10), translate = (0, 0.1), scale = (0.8, 1.2)),\n                                     transforms.RandomResizedCrop((224, 224), scale=(0.4, 1))])\nval_trans = transforms.Compose([transforms.ToTensor(),\n                                     transforms.Normalize(.5, .25)])","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:15:55.730006Z","iopub.execute_input":"2024-01-26T22:15:55.730869Z","iopub.status.idle":"2024-01-26T22:15:55.749561Z","shell.execute_reply.started":"2024-01-26T22:15:55.730830Z","shell.execute_reply":"2024-01-26T22:15:55.748545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = torchvision.datasets.DatasetFolder(\"Preprocessed_Data/train/\", loader=load_file, extensions=\"npy\", transform=train_trans)\nvalidation = torchvision.datasets.DatasetFolder(\"Preprocessed_Data/val/\", loader=load_file, extensions=\"npy\", transform=val_trans)","metadata":{"execution":{"iopub.status.busy":"2024-01-26T22:18:48.316639Z","iopub.execute_input":"2024-01-26T22:18:48.317070Z","iopub.status.idle":"2024-01-26T22:18:48.408526Z","shell.execute_reply.started":"2024-01-26T22:18:48.317034Z","shell.execute_reply":"2024-01-26T22:18:48.407258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}