{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\n# import os\n# print(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import math\nimport os\nimport shutil\nimport sys\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport glob\nimport pydicom\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9bf71dc651b7c2ed878ad8e58892d92cd19919d"},"cell_type":"code","source":"random_stat = 123\nnp.random.seed(random_stat)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d2fb8d5126b9b62ba7638068b321177b372e15a4"},"cell_type":"code","source":"!git clone https://github.com/pjreddie/darknet.git\n\n# Build gpu version darknet\n!cd darknet && sed '1 s/^.*$/GPU=1/; 2 s/^.*$/CUDNN=1/' -i Makefile\n\n# -j <The # of cpu cores to use>. Chang 999 to fit your environment. Actually i used '-j 50'.\n!cd darknet && make -j 999 -s\n!cp darknet/darknet darknet_gpu","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"57bbb8046e72ece933475a995bdfebe2cb5ace86"},"cell_type":"code","source":"DATA_DIR = \"../input\"\n\ntrain_dcm_dir = os.path.join(DATA_DIR, \"stage_2_train_images\")\ntest_dcm_dir = os.path.join(DATA_DIR, \"stage_2_test_images\")\n\nimg_dir = os.path.join(os.getcwd(), \"images\")  # .jpg\nlabel_dir = os.path.join(os.getcwd(), \"labels\")  # .txt\nmetadata_dir = os.path.join(os.getcwd(), \"metadata\") # .txt\n\n# YOLOv3 config file directory\ncfg_dir = os.path.join(os.getcwd(), \"cfg\")\n# YOLOv3 training checkpoints will be saved here\nbackup_dir = os.path.join(os.getcwd(), \"backup\")\n\nfor directory in [img_dir, label_dir, metadata_dir, cfg_dir, backup_dir]:\n    if os.path.isdir(directory):\n        continue\n    os.mkdir(directory)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"72d4e1241aa2d38a8cfcb68521ab50a39bd8064b"},"cell_type":"code","source":"!ls -shtl","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"375819be7764e475628f3224586266863e87be38"},"cell_type":"code","source":"annots = pd.read_csv(os.path.join(DATA_DIR, \"stage_2_train_labels.csv\"))\nannots.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4def8cbe73eb9c2bf1b09bbc673ac976c55544a2"},"cell_type":"code","source":"def save_img_from_dcm(dcm_dir, img_dir, patient_id):\n    img_fp = os.path.join(img_dir, \"{}.jpg\".format(patient_id))\n    if os.path.exists(img_fp):\n        return\n    dcm_fp = os.path.join(dcm_dir, \"{}.dcm\".format(patient_id))\n    img_1ch = pydicom.read_file(dcm_fp).pixel_array\n    img_3ch = np.stack([img_1ch]*3, -1)\n\n    img_fp = os.path.join(img_dir, \"{}.jpg\".format(patient_id))\n    cv2.imwrite(img_fp, img_3ch)\n    \ndef save_label_from_dcm(label_dir, patient_id, row=None):\n    # rsna defualt image size\n    img_size = 1024\n    label_fp = os.path.join(label_dir, \"{}.txt\".format(patient_id))\n    \n    f = open(label_fp, \"a\")\n    if row is None:\n        f.close()\n        return\n\n    top_left_x = row[1]\n    top_left_y = row[2]\n    w = row[3]\n    h = row[4]\n    \n    # 'r' means relative. 'c' means center.\n    rx = top_left_x/img_size\n    ry = top_left_y/img_size\n    rw = w/img_size\n    rh = h/img_size\n    rcx = rx+rw/2\n    rcy = ry+rh/2\n    \n    line = \"{} {} {} {} {}\\n\".format(0, rcx, rcy, rw, rh)\n    \n    f.write(line)\n    f.close()\n        \ndef save_yolov3_data_from_rsna(dcm_dir, img_dir, label_dir, annots):\n    for row in tqdm(annots.values):\n        patient_id = row[0]\n\n        img_fp = os.path.join(img_dir, \"{}.jpg\".format(patient_id))\n        if os.path.exists(img_fp):\n            save_label_from_dcm(label_dir, patient_id, row)\n            continue\n\n        target = row[5]\n        # Since kaggle kernel have samll volume (5GB ?), I didn't contain files with no bbox here.\n        if target == 0:\n            continue\n        save_label_from_dcm(label_dir, patient_id, row)\n        save_img_from_dcm(dcm_dir, img_dir, patient_id)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05a00ff64d903eaaf23bc3b390789011163ae8f8"},"cell_type":"code","source":"save_yolov3_data_from_rsna(train_dcm_dir, img_dir, label_dir, annots)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8c4f8abe4e6dd62ef7533d00232c3e2f415f573e"},"cell_type":"code","source":"!du -sh images labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f5a806f9c106d83067bd5e76f1dc730565b4aad2"},"cell_type":"code","source":"ex_patient_id = annots[annots.Target == 1].patientId.values[0]\nex_img_path = os.path.join(img_dir, \"{}.jpg\".format(ex_patient_id))\nex_label_path = os.path.join(label_dir, \"{}.txt\".format(ex_patient_id))\n\nplt.imshow(cv2.imread(ex_img_path))\n\nimg_size = 1014\nwith open(ex_label_path, \"r\") as f:\n    for line in f:\n        print(line)\n        class_id, rcx, rcy, rw, rh = list(map(float, line.strip().split()))\n        x = (rcx-rw/2)*img_size\n        y = (rcy-rh/2)*img_size\n        w = rw*img_size\n        h = rh*img_size\n        plt.plot([x, x, x+w, x+w, x], [y, y+h, y+h, y, y])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"950b9ca4b2372fa13a5b8aff16746e68e84dec80"},"cell_type":"code","source":"def write_train_list(metadata_dir, img_dir, name, series):\n    list_fp = os.path.join(metadata_dir, name)\n    with open(list_fp, \"w\") as f:\n        for patient_id in series:\n            line = \"{}\\n\".format(os.path.join(img_dir, \"{}.jpg\".format(patient_id)))\n            f.write(line)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b4a4ded5bb7042e01e1f69c0c4227f90163a9e6a"},"cell_type":"code","source":"patient_id_series = annots[annots.Target == 1].patientId.drop_duplicates()\n\ntr_series, val_series = train_test_split(patient_id_series, test_size=0.1, random_state=random_stat)\nprint(\"The # of train set: {}, The # of validation set: {}\".format(tr_series.shape[0], val_series.shape[0]))\n\n# train image path list\nwrite_train_list(metadata_dir, img_dir, \"tr_list.txt\", tr_series)\n# validation image path list\nwrite_train_list(metadata_dir, img_dir, \"val_list.txt\", val_series)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c21e5f2859700f700f2593494afc2f19d71a51ef"},"cell_type":"code","source":"def save_yolov3_test_data(test_dcm_dir, img_dir, metadata_dir, name, series):\n    list_fp = os.path.join(metadata_dir, name)\n    with open(list_fp, \"w\") as f:\n        for patient_id in series:\n            save_img_from_dcm(test_dcm_dir, img_dir, patient_id)\n            line = \"{}\\n\".format(os.path.join(img_dir, \"{}.jpg\".format(patient_id)))\n            f.write(line)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f039b24d6978aab746922fc4859d96b0a5b2ee63"},"cell_type":"code","source":"test_dcm_fps = list(set(glob.glob(os.path.join(test_dcm_dir, '*.dcm'))))\ntest_dcm_fps = pd.Series(test_dcm_fps).apply(lambda dcm_fp: dcm_fp.strip().split(\"/\")[-1].replace(\".dcm\",\"\"))\n\nsave_yolov3_test_data(test_dcm_dir, img_dir, metadata_dir, \"te_list.txt\", test_dcm_fps)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"856cabf31f97a71a3857b5fc56d80700e9d3a0d0"},"cell_type":"code","source":"ex_patient_id = test_dcm_fps[0]\nex_img_path = os.path.join(img_dir, \"{}.jpg\".format(ex_patient_id))\n\nplt.imshow(cv2.imread(ex_img_path))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"974bb29b80648222ee880621326e9d11d4e06ec0"},"cell_type":"code","source":"data_extention_file_path = os.path.join(cfg_dir, 'rsna.data')\nwith open(data_extention_file_path, 'w') as f:\n    contents = \"\"\"classes= 1\ntrain  = {}\nvalid  = {}\nnames  = {}\nbackup = {}\n    \"\"\".format(os.path.join(metadata_dir, \"tr_list.txt\"),\n               os.path.join(metadata_dir, \"val_list.txt\"),\n               os.path.join(cfg_dir, 'rsna.names'),\n               backup_dir)\n    f.write(contents)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"358775cc6bad655517c26acc9f981e0657390845"},"cell_type":"code","source":"!cat cfg/rsna.data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"116fee9134f14fe4f243792abd8f631446038588"},"cell_type":"code","source":"!echo \"pneumonia\" > cfg/rsna.names","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c5aaf93899248d8585892ddd51a6c559c8a11d2"},"cell_type":"code","source":"!wget -q https://pjreddie.com/media/files/darknet53.conv.74\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"09a882caeb3499e004c764718ee16e9b7179743e"},"cell_type":"code","source":"!wget --no-check-certificate -q \"https://docs.google.com/uc?export=download&id=18ptTK4Vbeokqpux8Onr0OmwUP9ipmcYO\" -O cfg/rsna_yolov3.cfg_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"849bf9fb367389efb9f1784802181589c42ffae2"},"cell_type":"code","source":"!wget --no-check-certificate -q \"https://docs.google.com/uc?export=download&id=1OhnlV3s7r6xsEme6DKkNYjcYjsl-C_Av\" -O train_log.txt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76b80dfd8683a91664ceee9c4c8ae64ecc73ed7b"},"cell_type":"code","source":"iters = []\nlosses = []\ntotal_losses = []\nwith open(\"train_log.txt\", 'r') as f:\n    for i,line in enumerate(f):\n        if \"images\" in line:\n            iters.append(int(line.strip().split()[0].split(\":\")[0]))\n            losses.append(float(line.strip().split()[2]))        \n            total_losses.append(float(line.strip().split()[1].split(',')[0]))\n\nplt.figure(figsize=(20, 5))\nplt.subplot(1,2,1)\nsns.lineplot(iters, total_losses, label=\"totla loss\")\nsns.lineplot(iters, losses, label=\"avg loss\")\nplt.xlabel(\"Iteration\")\nplt.ylabel(\"Loss\")\n\nplt.subplot(1,2,2)\nsns.lineplot(iters, total_losses, label=\"totla loss\")\nsns.lineplot(iters, losses, label=\"avg loss\")\nplt.xlabel(\"Iteration\")\nplt.ylabel(\"Loss\")\nplt.ylim([0, 4.05])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"198dc083aecd7e6c5f795624af31e900b71a3b41"},"cell_type":"code","source":"ex_patient_id = annots[annots.Target == 1].patientId.values[2]\nshutil.copy(ex_img_path, \"test.jpg\")\nprint(ex_patient_id)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9234029354578496f53f21b357fb347318193617"},"cell_type":"code","source":"!wget --load-cookies /tmp/cookies.txt -q \"https://docs.google.com/uc?export=download&confirm=$(wget --quiet --save-cookies /tmp/cookies.txt --keep-session-cookies --no-check-certificate 'https://docs.google.com/uc?export=download&id=1FDzMN-kGVYCvBeDKwemAazldSVkAEFyd' -O- | sed -rn 's/.*confirm=([0-9A-Za-z_]+).*/\\1\\n/p')&id=1FDzMN-kGVYCvBeDKwemAazldSVkAEFyd\" -O backup/rsna_yolov3_15300.weights && rm -rf /tmp/cookies.txt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8596e3fa4b150cd30fa066f26b57e70736f2c2ec"},"cell_type":"code","source":"!ls -alsth backup\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c2b50c4dbedc3b86912b665e4543690d5f86243e"},"cell_type":"code","source":"!wget --no-check-certificate -q \"https://docs.google.com/uc?export=download&id=10Yk6ZMAKGz5LeBbikciALy82aK3lX-57\" -O cfg/rsna_yolov3.cfg_test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6733acbb3b665e60bbf40b0315c83db1f69a88a1"},"cell_type":"code","source":"!cd darknet && ./darknet detector test ../cfg/rsna.data ../cfg/rsna_yolov3.cfg_test ../backup/rsna_yolov3_15300.weights ../test.jpg -thresh 0.005","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ccd2c8ffe16f69bd54e23873a055dc6569a491b2"},"cell_type":"code","source":"plt.imshow(cv2.imread(\"./darknet/predictions.jpg\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1f5c9d0638fd5418e266555f873084920d2f1e1e"},"cell_type":"code","source":"!wget --no-check-certificate -q \"https://docs.google.com/uc?export=download&id=1-KTV7K9G1bl3SmnLnzmpkDyNt6tDmH7j\" -O darknet.py","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af9788684e0e42ad411cc3f9c218289991a3ba1b"},"cell_type":"code","source":"from darknet import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b3a04916243f8145ba21ac8902c55ea92e6d748"},"cell_type":"code","source":"threshold = 0.2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6041e48901ea0320404f29df5f60296405695c4d"},"cell_type":"code","source":"submit_file_path = \"submission.csv\"\ncfg_path = os.path.join(cfg_dir, \"rsna_yolov3.cfg_test\")\nweight_path = os.path.join(backup_dir, \"rsna_yolov3_15300.weights\")\n\ntest_img_list_path = os.path.join(metadata_dir, \"te_list.txt\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"446383eee5c50b9c03c4e6a60e82cc4bb574266e"},"cell_type":"code","source":"gpu_index = 0\nnet = load_net(cfg_path.encode(),\n               weight_path.encode(), \n               gpu_index)\nmeta = load_meta(data_extention_file_path.encode())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17dc6a4034afe69262042793bc3d064e240452d3"},"cell_type":"code","source":"submit_dict = {\"patientId\": [], \"PredictionString\": []}\n\nwith open(test_img_list_path, \"r\") as test_img_list_f:\n    # tqdm run up to 1000(The # of test set)\n    for line in tqdm(test_img_list_f):\n        patient_id = line.strip().split('/')[-1].strip().split('.')[0]\n\n        infer_result = detect(net, meta, line.strip().encode(), thresh=threshold)\n\n        submit_line = \"\"\n        for e in infer_result:\n            confi = e[1]\n            w = e[2][2]\n            h = e[2][3]\n            x = e[2][0]-w/2\n            y = e[2][1]-h/2\n            submit_line += \"{} {} {} {} {} \".format(confi, x, y, w, h)\n\n        submit_dict[\"patientId\"].append(patient_id)\n        submit_dict[\"PredictionString\"].append(submit_line)\n\npd.DataFrame(submit_dict).to_csv(submit_file_path, index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9e2e24add942b61f1ff90f9189f39ce31364516"},"cell_type":"code","source":"# !ls -lsht\n!rm -rf darknet images labels metadata backup cfg\n!rm -rf train_log.txt darknet53.conv.74 darknet.py darknet_gpu\n!rm -rf test.jpg\n!rm -rf __pycache__ .ipynb_checkpoints","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d33268c19e8b31bd3b8cba0ddad012e4189c43d6"},"cell_type":"code","source":"!ls -alsht","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74a6dfcc3b2be03e05de2f278305f197f49a0a7f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}