{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# End to End Pneumonia Classification","metadata":{}},{"cell_type":"markdown","source":"## Preprocessing","metadata":{"execution":{"iopub.status.busy":"2023-03-25T10:46:36.961179Z","iopub.execute_input":"2023-03-25T10:46:36.962777Z","iopub.status.idle":"2023-03-25T10:46:36.995606Z","shell.execute_reply.started":"2023-03-25T10:46:36.962708Z","shell.execute_reply":"2023-03-25T10:46:36.994387Z"}}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport pydicom \nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nimport cv2\nROOT_PATH=\"/kaggle/input/rsna-pneumonia-detection-challenge\"\nTEST_FOLDER=\"stage_2_test_images/\"\nTRAIN_FOLDER=\"stage_2_train_images/\"\nSAVE_PATH=Path(\"/kaggle/working/\")","metadata":{"execution":{"iopub.status.busy":"2023-03-27T08:33:12.159449Z","iopub.execute_input":"2023-03-27T08:33:12.159951Z","iopub.status.idle":"2023-03-27T08:33:12.333359Z","shell.execute_reply.started":"2023-03-27T08:33:12.159908Z","shell.execute_reply":"2023-03-27T08:33:12.332049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detailed_class_csv=pd.read_csv(os.path.join(ROOT_PATH,'stage_2_detailed_class_info.csv'))\ndetailed_class_csv.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T06:51:38.996878Z","iopub.execute_input":"2023-03-27T06:51:38.997386Z","iopub.status.idle":"2023-03-27T06:51:39.113573Z","shell.execute_reply.started":"2023-03-27T06:51:38.997337Z","shell.execute_reply":"2023-03-27T06:51:39.112163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels=pd.read_csv(os.path.join(ROOT_PATH,'stage_2_train_labels.csv'))\nlabels.drop_duplicates(inplace=True)\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-27T06:51:40.165852Z","iopub.execute_input":"2023-03-27T06:51:40.166240Z","iopub.status.idle":"2023-03-27T06:51:40.286125Z","shell.execute_reply.started":"2023-03-27T06:51:40.166206Z","shell.execute_reply":"2023-03-27T06:51:40.284763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.iloc[0]['patientId']","metadata":{"execution":{"iopub.status.busy":"2023-03-27T06:51:41.495181Z","iopub.execute_input":"2023-03-27T06:51:41.496416Z","iopub.status.idle":"2023-03-27T06:51:41.506316Z","shell.execute_reply.started":"2023-03-27T06:51:41.496359Z","shell.execute_reply":"2023-03-27T06:51:41.504645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig,ax=plt.subplots(3,3,figsize=(10,8))\nlabel_dict={0:'No Pneumonia',1:'Pneumonia'}\nc=0\nfor i in range(3):\n    for j in range(3):\n        patient_id=labels.iloc[c]['patientId']\n        dcm_path=os.path.join(ROOT_PATH,TRAIN_FOLDER)+patient_id+'.dcm'\n        dcm_file=pydicom.read_file(dcm_path).pixel_array\n        ax[i][j].imshow(dcm_file,cmap='gray')\n        ax[i][j].set_title(label_dict[labels.iloc[c]['Target']])\n        ax[i][j].axis('off')\n        c+=1\n","metadata":{"execution":{"iopub.status.busy":"2023-03-27T06:51:42.884716Z","iopub.execute_input":"2023-03-27T06:51:42.885128Z","iopub.status.idle":"2023-03-27T06:51:44.703830Z","shell.execute_reply.started":"2023-03-27T06:51:42.885093Z","shell.execute_reply":"2023-03-27T06:51:44.702689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### calculating mean std without having memory error","metadata":{}},{"cell_type":"code","source":"sums,sums_sq=0,0\nfor x ,pid in enumerate(tqdm(labels.patientId)):\n    dcm_path=os.path.join(ROOT_PATH,TRAIN_FOLDER)+pid+'.dcm'\n    dcm_file=pydicom.read_file(dcm_path).pixel_array/255\n    dcm_array=cv2.resize(dcm_file,(224,224)).astype(np.float16)\n    \n    label=labels.Target.iloc[x]\n    train_or_val='train' if x<28000 else 'val'\n    current_save_path=SAVE_PATH/train_or_val/str(label)\n    current_save_path.mkdir(parents=True,exist_ok=True)\n    np.save(current_save_path/pid,dcm_array)\n    \n    normalizer=224*224\n    if train_or_val=='train':\n        sums+=np.sum(dcm_array)/normalizer\n        sums_sq+=(dcm_array**2).sum() / normalizer","metadata":{"execution":{"iopub.status.busy":"2023-03-27T07:12:52.264998Z","iopub.execute_input":"2023-03-27T07:12:52.265451Z","iopub.status.idle":"2023-03-27T07:25:11.225744Z","shell.execute_reply.started":"2023-03-27T07:12:52.265410Z","shell.execute_reply":"2023-03-27T07:25:11.223980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean=sums/28000\nstd=np.sqrt((sums_sq/28000)-mean**2)\nprint(mean,std)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T08:29:31.524327Z","iopub.execute_input":"2023-03-27T08:29:31.524882Z","iopub.status.idle":"2023-03-27T08:29:31.566935Z","shell.execute_reply.started":"2023-03-27T08:29:31.524834Z","shell.execute_reply":"2023-03-27T08:29:31.564931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Loading","metadata":{}},{"cell_type":"code","source":"import torch\nimport torchvision\nfrom torchvision import transforms\nimport torchmetrics\nimport pytorch_lightning as pl\nfrom pytorch_lightning.callbacks import ModelCheckpoint\nfrom pytorch_lightning.loggers import TensorBoardLogger","metadata":{"execution":{"iopub.status.busy":"2023-03-27T08:24:40.001403Z","iopub.execute_input":"2023-03-27T08:24:40.002812Z","iopub.status.idle":"2023-03-27T08:24:57.994254Z","shell.execute_reply.started":"2023-03-27T08:24:40.002739Z","shell.execute_reply":"2023-03-27T08:24:57.992792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_file(path):\n    return np.load(path).astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-03-27T08:26:44.157891Z","iopub.execute_input":"2023-03-27T08:26:44.158395Z","iopub.status.idle":"2023-03-27T08:26:44.165700Z","shell.execute_reply.started":"2023-03-27T08:26:44.158271Z","shell.execute_reply":"2023-03-27T08:26:44.163899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_transform=transforms.Compose([\n                            transforms.ToTensor(),\n                            transforms.Normalize(0.49,0.242),\n                            transforms.RandomAffine(degrees=(-5,5),translate=(0,0.05),scale=(0.9,1.1)),\n                            transforms.RandomResizedCrop((224,224),scale=(0.35,1))\n])\nval_transform=transforms.Compose([\n                            transforms.ToTensor(),\n                            transforms.Normalize(0.49,0.242)\n])","metadata":{"execution":{"iopub.status.busy":"2023-03-27T08:30:24.230675Z","iopub.execute_input":"2023-03-27T08:30:24.231804Z","iopub.status.idle":"2023-03-27T08:30:24.240505Z","shell.execute_reply.started":"2023-03-27T08:30:24.231749Z","shell.execute_reply":"2023-03-27T08:30:24.239348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset=torchvision.datasets.DatasetFolder(SAVE_PATH/'train',loader=load_file,transform=train_transform,extensions=\"npy\")\nval_dataset=torchvision.datasets.DatasetFolder(SAVE_PATH/'val',loader=load_file,transform=val_transform,extensions=\"npy\")","metadata":{"execution":{"iopub.status.busy":"2023-03-27T08:33:17.078119Z","iopub.execute_input":"2023-03-27T08:33:17.078613Z","iopub.status.idle":"2023-03-27T08:33:17.134103Z","shell.execute_reply.started":"2023-03-27T08:33:17.078548Z","shell.execute_reply":"2023-03-27T08:33:17.132125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}