{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<p style=\"font-family: monospace; \n          font-weight: bold; \n          letter-spacing: 2px; \n          color: black; \n          font-size: 200%; \n          text-align: left;\n          padding: 0px; \n          border-bottom: 4px solid #a2d2ff\" >Pneumonia Classification EDA</p>\n          \nThis notebook provides a brief exploratory data analysis. Previously this notebook included some PyTorch code, but after reviewing it I wasn't satisfied with the quality and decided to delete everything except the EDA section.","metadata":{}},{"cell_type":"markdown","source":"<p style=\"font-family: monospace; \n          font-weight: bold; \n          letter-spacing: 2px; \n          color: black; \n          font-size: 200%; \n          text-align: left;\n          padding: 0px; \n          border-bottom: 4px solid #a2d2ff\" >Table of Contents</p>\n\n* [Data and Cohort Characteristics](#section-one)\n* [Pre-processing](#section-two)\n* [Dataloader and Model](#section-three)\n* [Training Loop](#section-four)\n* [Evaluation](#section-five)\n* [Conclusions](#section-six)","metadata":{}},{"cell_type":"code","source":"# Installs\n!pip install polars\n!pip install lets-plot\n\n# Imports\nimport os\nimport cv2\nimport glob\nimport torch\nimport pylab\nimport plotly\nimport random\nimport pydicom\nimport torchvision\nimport numpy as np\nimport polars as pl\nimport torch.nn as nn\nimport statistics as stats\nimport matplotlib.pyplot as plt\nimport torch.nn.functional as F\n\nfrom lets_plot import *\nfrom torchvision import transforms\nfrom sklearn.metrics import roc_auc_score\nfrom lets_plot.mapping import as_discrete\n\n# So the plots look nice\nLetsPlot.setup_html()\nplotly.offline.init_notebook_mode(connected = True)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-03T07:08:01.275574Z","iopub.execute_input":"2023-07-03T07:08:01.276218Z","iopub.status.idle":"2023-07-03T07:08:27.691175Z","shell.execute_reply.started":"2023-07-03T07:08:01.276185Z","shell.execute_reply":"2023-07-03T07:08:27.689755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n<p style=\"font-family: monospace; \n          font-weight: bold; \n          letter-spacing: 2px; \n          color: black; \n          font-size: 200%; \n          text-align: left;\n          padding: 0px; \n          border-bottom: 4px solid #a2d2ff\" >Data and Cohort Characteristics</p>\n          \nGiven that my motivation for making this notebook was to get extra practice using PyTorch and that my free time to develop this notebook was limited, I decided to reduce the size of the data set. The data were downsampled to 3000 images split with an even class balance.","metadata":{}},{"cell_type":"code","source":"# Read in data, drop duplicates\n#df_class_info = pl.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\ndf_labels = pl.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\ndf_labels = df_labels.unique('patientId')","metadata":{"execution":{"iopub.status.busy":"2023-07-03T07:08:37.411590Z","iopub.execute_input":"2023-07-03T07:08:37.412014Z","iopub.status.idle":"2023-07-03T07:08:37.652266Z","shell.execute_reply.started":"2023-07-03T07:08:37.411977Z","shell.execute_reply":"2023-07-03T07:08:37.651252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels.glimpse()","metadata":{"execution":{"iopub.status.busy":"2023-07-03T07:08:40.057201Z","iopub.execute_input":"2023-07-03T07:08:40.057676Z","iopub.status.idle":"2023-07-03T07:08:40.070694Z","shell.execute_reply.started":"2023-07-03T07:08:40.057640Z","shell.execute_reply":"2023-07-03T07:08:40.069245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-family: monospace; \n          font-weight: bold; \n          letter-spacing: 1px; \n          color: black; \n          font-size: 150%; \n          text-align: left;\n          padding: 0px; \n          border-bottom: 1px solid #a2d2ff\" >Target Distribution</p>","metadata":{}},{"cell_type":"code","source":"# Initialize colors\ncolor1='#a5ffd6'\ncolor2='#ff686b'\n\n# Make labels readable\ndf_plt = df_labels.with_columns(\n    pl.when(pl.col('Target') == 1).then('Yes').otherwise('No').alias('Target')\n)\n    \n# Target variable plot\nvar = 'Target'\ntitle = 'Does the Patient have Pneumonia?'\nlegend_title = ''\n\nn_yes = len(df_plt.filter(pl.col(\"Target\") == \"Yes\").get_column(\"Target\"))\nn_no = len(df_plt.filter(pl.col(\"Target\") == \"No\").get_column(\"Target\"))\nn = len(df_plt)\nyes_tag = f'{n_yes} ({round((n_yes/n)*100, 2)}%)'\nno_tag = f'{n_no} ({round((n_no/n)*100, 2)}%)'\n\nplt1 = \\\n    ggplot(df_plt)+\\\n    geom_bar(aes(x = as_discrete(var),\n                fill = as_discrete(var)),\n            color = 'black',\n            size = 0.5)+\\\n    geom_text(x = 0, y = 4400, label = yes_tag)+\\\n    geom_text(x = 1, y = 18000, label = no_tag)+\\\n    scale_fill_manual(values = [color1, color2])+\\\n    theme_minimal()+\\\n    theme(\n        plot_title = element_text(hjust = 0.5, face = 'bold'),\n        legend_position = \"top\",\n        panel_grid_major = element_blank(),\n        panel_grid_minor = element_blank(),\n        legend_title = element_blank(),\n        axis_title_x = element_blank(),\n        axis_line_y = element_line(size = 1))+\\\n    coord_flip()+\\\n    labs(y = \"Count\", title = title)\n\nbunch = GGBunch()\nbunch.add_plot(plt1, 0, 0, 700, 250)\nbunch","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-03T07:09:58.241302Z","iopub.execute_input":"2023-07-03T07:09:58.241772Z","iopub.status.idle":"2023-07-03T07:09:58.616733Z","shell.execute_reply.started":"2023-07-03T07:09:58.241735Z","shell.execute_reply":"2023-07-03T07:09:58.615613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-family: monospace; \n          font-weight: bold; \n          letter-spacing: 1px; \n          color: black; \n          font-size: 150%; \n          text-align: left;\n          padding: 0px; \n          border-bottom: 1px solid #a2d2ff\" >Target Examples - Patients WITHOUT pneumonia</p>","metadata":{}},{"cell_type":"code","source":"# Get IDs of everyone without pneumonia\npatientIDs_0 = df_labels\\\n    .filter(pl.col('Target') == 0)\\\n    .get_column('patientId')\n\n# Plot 9 examples for those without pneumonia\nfig, axis = plt.subplots(3, 3, figsize=(10, 10))\n\ncounter = 0\nfor i in range(3):\n    for j in range(3):\n        patientID = patientIDs_0[counter]\n        image_path = f'/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/{patientID}.dcm'\n        dcm = pydicom.read_file(image_path).pixel_array\n        axis[i][j].imshow(dcm, cmap=\"bone\")\n        axis[i][j].tick_params(axis='both', which='both', length=0, labelbottom=False, labelleft=False)\n        counter+=1","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-03T07:10:04.665119Z","iopub.execute_input":"2023-07-03T07:10:04.665849Z","iopub.status.idle":"2023-07-03T07:10:07.868068Z","shell.execute_reply.started":"2023-07-03T07:10:04.665807Z","shell.execute_reply":"2023-07-03T07:10:07.867122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style=\"font-family: monospace; \n          font-weight: bold; \n          letter-spacing: 1px; \n          color: black; \n          font-size: 150%; \n          text-align: left;\n          padding: 0px; \n          border-bottom: 1px solid #a2d2ff\" >Target Examples - Patients WITH pneumonia</p>","metadata":{}},{"cell_type":"code","source":"# Get IDs of everyone with pneumonia\npatientIDs_1 = df_labels\\\n    .filter(pl.col('Target') == 1)\\\n    .get_column('patientId')\n\n# Plot 9 examples for those with pneumonia\nfig, axis = plt.subplots(3, 3, figsize=(10, 10))\n\ncounter = 0\nfor i in range(3):\n    for j in range(3):\n        patientID = patientIDs_1[counter]\n        image_path = f'/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/{patientID}.dcm'\n        dcm = pydicom.read_file(image_path).pixel_array\n        axis[i][j].imshow(dcm, cmap=\"bone\")\n        axis[i][j].tick_params(axis='both', which='both', length=0, labelbottom=False, labelleft=False)\n        counter+=1","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-07-03T07:10:16.575858Z","iopub.execute_input":"2023-07-03T07:10:16.576300Z","iopub.status.idle":"2023-07-03T07:10:19.642328Z","shell.execute_reply.started":"2023-07-03T07:10:16.576266Z","shell.execute_reply":"2023-07-03T07:10:19.640986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-six\"></a>\n<p style=\"font-family: monospace; \n          font-weight: bold; \n          letter-spacing: 2px; \n          color: black; \n          font-size: 200%; \n          text-align: left;\n          padding: 0px; \n          border-bottom: 4px solid #a2d2ff\" > Conclusions</p>\n          \nThis notebook provided a brief EDA of the target and a few images, if you made it this far thanks for viewing the notebook! ","metadata":{}}]}