{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### In this kernel, I perform some basic EDA. The idea is to understand the **statistics of the train** data to help us choose the best hyperparameters for our models.","metadata":{}},{"cell_type":"markdown","source":"# **Getting Started**\n---\n* Load libraries and modules\n* Peak at train data","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport random\nfrom tqdm import tqdm\n\nimport numpy as np\nimport pandas as pd\n\nimport pydicom\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":false,"execution":{"iopub.status.busy":"2021-08-15T18:48:31.894621Z","iopub.execute_input":"2021-08-15T18:48:31.895099Z","iopub.status.idle":"2021-08-15T18:48:33.215902Z","shell.execute_reply.started":"2021-08-15T18:48:31.894988Z","shell.execute_reply":"2021-08-15T18:48:33.214874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\")\nprint(\"Overview of data:\\n\")\ndisplay(train_df)\nprint(f\"Total number of train samples: {train_df.shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T18:48:33.218205Z","iopub.execute_input":"2021-08-15T18:48:33.218790Z","iopub.status.idle":"2021-08-15T18:48:33.258416Z","shell.execute_reply.started":"2021-08-15T18:48:33.218743Z","shell.execute_reply":"2021-08-15T18:48:33.255873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (8, 5))\nplt.title(\"MGMT_value (Target) Distribution\", fontsize = 16)\n\nax = sns.countplot(data = train_df, x = \"MGMT_value\", order = [0, 1]);\nax.set_xlabel(xlabel = \"MGMT_value\", fontsize = 12)\nax.set_ylabel(ylabel = \"Count\", fontsize = 12)\n\nfor p in ax.patches:\n    x = p.get_bbox().get_points()[:, 0]\n    y = p.get_bbox().get_points()[1, 1]\n    ax.annotate(f\"{int(y):,} ({100*y/train_df.shape[0]:.1f}%)\", (x.mean(), y),\n                ha = \"center\", va = \"bottom\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T18:48:33.259994Z","iopub.execute_input":"2021-08-15T18:48:33.260301Z","iopub.status.idle":"2021-08-15T18:48:33.458017Z","shell.execute_reply.started":"2021-08-15T18:48:33.260272Z","shell.execute_reply":"2021-08-15T18:48:33.457195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Images**\n---\n* See some smaple images\n* Look at some **key statistics** for out train images\n","metadata":{}},{"cell_type":"code","source":"def load_dicom(path):\n    \"\"\"\n    Input: path to a dicom file\n    Output: Image in numpy format\n    Image is min-max normalised\n    \"\"\"\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\n\ndef visualize_sample(brats21id, slice_i, mgmt_value, types = (\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\")):\n    \"\"\"\n    For a given id in train data, plots the\n    first image of each of the four types\n    \"\"\"\n    plt.figure(figsize = (16, 5))\n    patient_path = os.path.join(\n        \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/\", \n        str(brats21id).zfill(5),\n    )\n    \n    for i, t in enumerate(types, 1):\n        t_paths = sorted(glob.glob(os.path.join(patient_path, t, \"*\")),\n                         key = lambda x: int(x[:-4].split(\"-\")[-1]))\n        data = load_dicom(t_paths[int(len(t_paths) * slice_i)])\n        \n        plt.subplot(1, 4, i)\n        plt.imshow(data, cmap = \"gray\")\n        plt.title(f\"{t}\", fontsize = 16)\n        plt.axis(\"off\")\n    plt.suptitle(f\"id: {_brats21id}  MGMT_value: {mgmt_value}\", fontsize = 16)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-15T18:48:33.459906Z","iopub.execute_input":"2021-08-15T18:48:33.460504Z","iopub.status.idle":"2021-08-15T18:48:33.471953Z","shell.execute_reply.started":"2021-08-15T18:48:33.460457Z","shell.execute_reply":"2021-08-15T18:48:33.471149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in random.sample(range(train_df.shape[0]), 2):\n    _brats21id = train_df.iloc[i][\"BraTS21ID\"]\n    _mgmt_value = train_df.iloc[i][\"MGMT_value\"]\n    visualize_sample(brats21id = _brats21id, mgmt_value = _mgmt_value, slice_i = 0.5)","metadata":{"execution":{"iopub.status.busy":"2021-08-15T18:48:33.473344Z","iopub.execute_input":"2021-08-15T18:48:33.473663Z","iopub.status.idle":"2021-08-15T18:48:34.634565Z","shell.execute_reply.started":"2021-08-15T18:48:33.473632Z","shell.execute_reply":"2021-08-15T18:48:34.633520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FILES = glob.glob(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/*/*/*\")\nprint(f\"Total number of Image files: {len(FILES):,}\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T18:48:34.636140Z","iopub.execute_input":"2021-08-15T18:48:34.636739Z","iopub.status.idle":"2021-08-15T18:49:24.595882Z","shell.execute_reply.started":"2021-08-15T18:48:34.636692Z","shell.execute_reply":"2021-08-15T18:49:24.594670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_dict = {\"FLAIR\": [], \"T1w\": [], \"T1wCE\": [], \"T2w\": []}\nfor filename in FILES:\n    scan = filename.split(\"/\")[-2]\n    if scan == \"FLAIR\":\n        files_dict[\"FLAIR\"].append(filename)\n    elif scan == \"T1w\":\n        files_dict[\"T1w\"].append(filename)\n    elif scan == \"T1wCE\":\n        files_dict[\"T1wCE\"].append(filename)\n    else:\n        files_dict[\"T2w\"].append(filename)\n        \nkeys = list(files_dict.keys())\nvals = [len(files_dict[k]) for k in keys]\n\nplt.figure(figsize = (12, 5))\nplt.title(\"Total files vs MRI type\", fontsize = 16)\n\nax = sns.barplot(x = keys, y = vals)\nax.set_xlabel(xlabel = \"Type\", fontsize = 12)\nax.set_ylabel(ylabel = \"Total Count\", fontsize = 12)\n\nfor p in ax.patches:\n    x = p.get_bbox().get_points()[:, 0]\n    y = int(p.get_bbox().get_points()[1, 1])\n    ax.annotate(f\"{y:,} ({100*y/len(FILES):.1f}%)\", (x.mean(), y),\n                ha = \"center\", va = \"bottom\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T18:49:24.597411Z","iopub.execute_input":"2021-08-15T18:49:24.597718Z","iopub.status.idle":"2021-08-15T18:49:25.306363Z","shell.execute_reply.started":"2021-08-15T18:49:24.597687Z","shell.execute_reply":"2021-08-15T18:49:25.305017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_dim_freq(type_name):\n    \"\"\"\n    type_name = [\"FLAIR\", \"T1w\", \"T2w\", \"T1wCE\"]\n    Plots frequency distribution of dimensions\n    of images for the given \"type_name\"\n    \"\"\"\n    def image_size(path):\n        dicom = pydicom.read_file(path)\n        data = dicom.pixel_array\n        return data.shape\n    \n    img_size_dict = dict()\n    for fpath in tqdm(files_dict[type_name]):\n        dims = image_size(fpath)\n        key = str(dims[0]) + \"x\" + str(dims[1])\n        if key in img_size_dict:\n            img_size_dict[key] += 1\n        else:\n            img_size_dict[key] = 1\n    img_size_dict = {k: v for k, v in sorted(img_size_dict.items(), key = lambda item: -1*item[1])}\n    \n    keys = list(img_size_dict.keys())\n    vals = [img_size_dict[k] for k in keys]\n    \n    plt.figure(figsize = (24, 5))\n    plt.title(f\"Image dimension Distribution for \\\"{type_name}\\\" (top 10)\", fontsize = 16)\n    \n    ax = sns.barplot(x = keys[:10], y = vals[:10], order = keys[:10])\n    ax.set_xlabel(xlabel = \"Dimensions\", fontsize = 12)\n    ax.set_ylabel(ylabel = \"Count\", fontsize = 12)\n    \n    for p in ax.patches:\n        x = p.get_bbox().get_points()[:, 0]\n        y = int(p.get_bbox().get_points()[1, 1])\n        ax.annotate(f\"{y:,} ({100*y/len(files_dict[type_name]):.1f}%)\", (x.mean(), y),\n                    ha = \"center\", va = \"bottom\", fontsize = 8)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-15T18:49:25.309993Z","iopub.execute_input":"2021-08-15T18:49:25.310382Z","iopub.status.idle":"2021-08-15T18:49:25.322972Z","shell.execute_reply.started":"2021-08-15T18:49:25.310344Z","shell.execute_reply":"2021-08-15T18:49:25.322027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_dim_freq(\"FLAIR\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T18:49:25.324925Z","iopub.execute_input":"2021-08-15T18:49:25.325381Z","iopub.status.idle":"2021-08-15T18:59:57.026125Z","shell.execute_reply.started":"2021-08-15T18:49:25.325335Z","shell.execute_reply":"2021-08-15T18:59:57.025111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_dim_freq(\"T2w\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T18:59:57.027590Z","iopub.execute_input":"2021-08-15T18:59:57.027907Z","iopub.status.idle":"2021-08-15T19:15:21.586622Z","shell.execute_reply.started":"2021-08-15T18:59:57.027873Z","shell.execute_reply":"2021-08-15T19:15:21.584925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_dim_freq(\"T1w\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T19:15:21.589199Z","iopub.execute_input":"2021-08-15T19:15:21.589692Z","iopub.status.idle":"2021-08-15T19:25:19.887664Z","shell.execute_reply.started":"2021-08-15T19:15:21.589637Z","shell.execute_reply":"2021-08-15T19:25:19.886341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_dim_freq(\"T1wCE\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T19:25:19.889604Z","iopub.execute_input":"2021-08-15T19:25:19.889962Z","iopub.status.idle":"2021-08-15T19:37:37.478617Z","shell.execute_reply.started":"2021-08-15T19:25:19.889929Z","shell.execute_reply":"2021-08-15T19:37:37.477319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_num_files(type_name):\n    \"\"\"\n    type_name = [\"FLAIR\", \"T1w\", \"T2w\", \"T1wCE\"]\n    Plots frequency distribution of number of files\n    for a patient for the given \"type_name\"\n    \"\"\"\n    all_freqs = []\n    num_files_dict = dict()\n    for i, data in train_df.iterrows():\n        brats21id = data[0]\n        patient_path = os.path.join(\n            \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/\", \n            str(brats21id).zfill(5),\n        )\n        \n        key = len(glob.glob(os.path.join(patient_path, type_name, \"*\")))\n        all_freqs.append(key)\n        if key in num_files_dict:\n                num_files_dict[key] += 1\n        else:\n            num_files_dict[key] = 1\n\n    num_files_dict = {k: v for k, v in sorted(num_files_dict.items(), key = lambda item: -1*item[1])}\n    keys = list(num_files_dict.keys())\n    vals = [num_files_dict[k] for k in keys]\n    \n    plt.figure(figsize = (24, 5))\n    plt.title(f\"Number of images for type \\\"{type_name}\\\" per Patient distribution (top 10)\", fontsize = 16)\n    \n    ax = sns.barplot(x = keys[:10], y = vals[:10], order = keys[:10])\n    ax.set_xlabel(xlabel = \"Number of files\", fontsize = 12)\n    ax.set_ylabel(ylabel = \"count\", fontsize = 12)\n    \n    for p in ax.patches:\n        x = p.get_bbox().get_points()[:, 0]\n        y = int(p.get_bbox().get_points()[1, 1])\n        ax.annotate(f\"{y:,} ({100*y/train_df.shape[0]:.2f}%)\", (x.mean(), y),\n                    ha = \"center\", va = \"bottom\", fontsize = 10)\n    plt.show()\n    \n    print(f\"Total ids in top 10: {sum(vals)} ({1.0*sum(vals[:10])/train_df.shape[0]:.2f}%)\")\n    print(f\"Minimum and Maximum number of files for a patient is {min(keys)} and {max(keys)}, respectively.\")\n    print(f\"Mean and Median of number of files is: {np.mean(all_freqs):.2f} and {np.median(all_freqs):.0f}, respectively.\")\n    print(f\"Varinace of number of files is {np.sqrt(np.var(all_freqs)):.2f}\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T19:37:37.480575Z","iopub.execute_input":"2021-08-15T19:37:37.480997Z","iopub.status.idle":"2021-08-15T19:37:37.498872Z","shell.execute_reply.started":"2021-08-15T19:37:37.480952Z","shell.execute_reply":"2021-08-15T19:37:37.497558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_num_files(\"FLAIR\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T19:37:37.500442Z","iopub.execute_input":"2021-08-15T19:37:37.500747Z","iopub.status.idle":"2021-08-15T19:37:46.485650Z","shell.execute_reply.started":"2021-08-15T19:37:37.500715Z","shell.execute_reply":"2021-08-15T19:37:46.484642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_num_files(\"T2w\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T19:37:46.487332Z","iopub.execute_input":"2021-08-15T19:37:46.487638Z","iopub.status.idle":"2021-08-15T19:37:57.845128Z","shell.execute_reply.started":"2021-08-15T19:37:46.487606Z","shell.execute_reply":"2021-08-15T19:37:57.844202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_num_files(\"T1w\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T19:37:57.846413Z","iopub.execute_input":"2021-08-15T19:37:57.846693Z","iopub.status.idle":"2021-08-15T19:38:09.510971Z","shell.execute_reply.started":"2021-08-15T19:37:57.846664Z","shell.execute_reply":"2021-08-15T19:38:09.509845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_num_files(\"T1wCE\")","metadata":{"execution":{"iopub.status.busy":"2021-08-15T19:38:09.512460Z","iopub.execute_input":"2021-08-15T19:38:09.512771Z","iopub.status.idle":"2021-08-15T19:38:23.200988Z","shell.execute_reply.started":"2021-08-15T19:38:09.512739Z","shell.execute_reply":"2021-08-15T19:38:23.199825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Final Thoughts**\n* **FLAIR and T2w have very similar image data statistics**\n* **T1w and T1wCE have very similar image data statistics**\n* Number of images for a patient per type differ significantly\n* Images have large blank areas on all sides\n\n\n# **.... Work in Progress**","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}