{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Key frame\n\nThe key frame is generated by comparing the frame the the frame before it. If the sum of the byte difference between frame n and n + 1 is diff_n, the key frames are the highest n frames of the video.\n\nFirst, we use [ihelon's code](https://www.kaggle.com/ihelon/brain-tumor-eda-with-animations-and-modeling) to read images from each data, we can pre-caculate the key frames.\n\n### DataSet\n\nData set [rsna-miccai-key-frames](https://www.kaggle.com/gordonwang141/rsnamiccaikeyframes) contains two sets of data:\n\n- keyframes.pkl: the ordered key frames for each data sample.\n- keyframes20.pkl: the average value of top 20 key frames for each data sample resized to 256 * 256\n","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\n\nimport numpy as np\nimport pandas as pd\n\nimport pydicom","metadata":{"execution":{"iopub.status.busy":"2021-07-31T19:21:20.117217Z","iopub.execute_input":"2021-07-31T19:21:20.117578Z","iopub.status.idle":"2021-07-31T19:21:20.122824Z","shell.execute_reply.started":"2021-07-31T19:21:20.117541Z","shell.execute_reply":"2021-07-31T19:21:20.122021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dicom(path):\n    dicom = pydicom.read_file(path)\n    data = dicom.pixel_array\n    data = data - np.min(data)\n    if np.max(data) != 0:\n        data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n\ndef load_dicom_line(path):\n    t_paths = sorted(\n        glob.glob(os.path.join(path, \"*\")), \n        key=lambda x: int(x[:-4].split(\"-\")[-1]),\n    )\n    images = []\n    for filename in t_paths:\n        data = load_dicom(filename)\n        if data.max() == 0:\n            continue\n        images.append(data)\n        \n    return images\n\ndef data_diff(d1, d2):\n    diff = np.sum(np.abs(d1 - d2))\n    return diff\n    \ndef get_key_frames(braTS21ID, t):\n    braTS21ID = int(braTS21ID)\n    print (braTS21ID, t)\n    patient_path = f\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/{str(braTS21ID).zfill(5)}/\"\n\n    isdir = os.path.isdir(patient_path) \n    if not isdir:\n        patient_path = f\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/{str(braTS21ID).zfill(5)}/\"\n\n    path = os.path.join(patient_path, t)\n    images = load_dicom_line(path)\n    diff_values = {}\n    for i in range(len(images) - 1):\n        diff = data_diff(images[i], images[i+1])\n        diff_values[i+1] = diff\n\n    return list(dict(sorted(diff_values.items(), key=lambda item: item[1], reverse=True)).keys())\n    ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-31T19:21:21.182212Z","iopub.execute_input":"2021-07-31T19:21:21.18296Z","iopub.status.idle":"2021-07-31T19:21:21.198924Z","shell.execute_reply.started":"2021-07-31T19:21:21.182881Z","shell.execute_reply":"2021-07-31T19:21:21.197504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, we calculate the key frames and save as a dataset. You don't need to run this part of the code unless you need to change the way we calculate key frames.","metadata":{}},{"cell_type":"code","source":"'''\nsamp_subm = pd.read_csv(\"/kaggle/input/rsna-miccai-brain-tumor-radiogenomic-classification/sample_submission.csv\")\n\ntrain_df = pd.read_csv(\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train_labels.csv\")\ndata_df = pd.concat([train_df, samp_subm])\n\nfor t in (\"FLAIR\", \"T1w\", \"T1wCE\", \"T2w\"):\n    data_df[t] = data_df.apply(lambda row: get_key_frames(row['BraTS21ID'], t), axis = 1)\ndata_df.to_pickle(\"keyframes.pkl\")\n'''","metadata":{"execution":{"iopub.status.busy":"2021-07-31T19:21:22.685351Z","iopub.execute_input":"2021-07-31T19:21:22.685799Z","iopub.status.idle":"2021-07-31T19:21:22.698176Z","shell.execute_reply.started":"2021-07-31T19:21:22.685755Z","shell.execute_reply":"2021-07-31T19:21:22.697085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Example code","metadata":{}},{"cell_type":"code","source":"data_df = pd.read_pickle('../input/rsnamiccaikeyframes/keyframes.pkl')","metadata":{"execution":{"iopub.status.busy":"2021-07-31T19:21:24.919735Z","iopub.execute_input":"2021-07-31T19:21:24.920115Z","iopub.status.idle":"2021-07-31T19:21:35.640825Z","shell.execute_reply.started":"2021-07-31T19:21:24.920081Z","shell.execute_reply":"2021-07-31T19:21:35.63965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_keyframe(braTS21ID, t, num_frames):\n    braTS21ID = int(braTS21ID)\n    if braTS21ID in data_df['BraTS21ID'].values:\n        # print('found ' + str(braTS21ID) + ' ' + t)\n        return data_df[data_df['BraTS21ID'] == braTS21ID][t].values[0][:num_frames]\n    \n    return []\n\ndef get_data_key_frames(braTS21ID, t, num_frames):\n    braTS21ID = int(braTS21ID)\n    print (braTS21ID, t)\n    patient_path = f\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/{str(braTS21ID).zfill(5)}/\"\n\n    isdir = os.path.isdir(patient_path) \n    if not isdir:\n        patient_path = f\"../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/{str(braTS21ID).zfill(5)}/\"\n    print(patient_path)\n    t_paths = sorted(\n        glob.glob(os.path.join(patient_path, t, \"*\")), \n        key=lambda x: int(x[:-4].split(\"-\")[-1]),\n    )\n    path = os.path.join(patient_path, t)\n    channel = []\n    for i in get_key_frames(path, num_frames):\n        channel.append(cv2.resize(load_dicom(t_paths[i]), (256, 256)) / 255)\n    channel = np.mean(channel, axis=0)\n    return channel\n\ndef get_keyframe20_data(braTS21ID, t):\n    braTS21ID = int(braTS21ID)\n    if braTS21ID in key_frame_20_data_df['BraTS21ID'].values:\n        return key_frame_20_data_df[key_frame_20_data_df['BraTS21ID'] == braTS21ID][t].values[0]\n    return get_data_key_frames(braTS21ID, t, 20)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T19:31:29.450194Z","iopub.execute_input":"2021-07-31T19:31:29.450586Z","iopub.status.idle":"2021-07-31T19:31:29.463528Z","shell.execute_reply.started":"2021-07-31T19:31:29.450548Z","shell.execute_reply":"2021-07-31T19:31:29.462243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_keyframe(0, 'FLAIR', 5)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T19:22:00.728499Z","iopub.execute_input":"2021-07-31T19:22:00.729087Z","iopub.status.idle":"2021-07-31T19:22:00.774007Z","shell.execute_reply.started":"2021-07-31T19:22:00.729025Z","shell.execute_reply":"2021-07-31T19:22:00.772697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visulize some key frames","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ndef visualize_key_frames(brats21id, t, frames):\n    plt.figure(figsize=(16, 5))\n    patient_path = os.path.join(\n        \"../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/\", \n        str(brats21id).zfill(5),\n    )\n    t_paths = sorted(\n        glob.glob(os.path.join(patient_path, t, \"*\")), \n        key=lambda x: int(x[:-4].split(\"-\")[-1]),\n    )\n    for i, f in enumerate(frames, 1):\n        data = load_dicom(t_paths[f])\n        plt.subplot(1, len(frames), i)\n        plt.imshow(data, cmap=\"gray\")\n        plt.title(f\"{f}\", fontsize=16)\n        plt.axis(\"off\")\n\n    plt.show()\n\nvisualize_key_frames(0, 'FLAIR', get_keyframe(0, 'FLAIR', 5))","metadata":{"execution":{"iopub.status.busy":"2021-07-31T19:22:02.226825Z","iopub.execute_input":"2021-07-31T19:22:02.227286Z","iopub.status.idle":"2021-07-31T19:22:02.833102Z","shell.execute_reply.started":"2021-07-31T19:22:02.227248Z","shell.execute_reply":"2021-07-31T19:22:02.831806Z"},"trusted":true},"execution_count":null,"outputs":[]}]}