{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"},{"sourceId":29653,"databundleVersionId":2420395,"sourceType":"competition"},{"sourceId":11713291,"sourceType":"datasetVersion","datasetId":7352392}],"dockerImageVersionId":30120,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os #ceate folders\nfrom glob import glob # get paths and use the to have floders name\nfrom tqdm.notebook import tqdm # get nice bar\nimport sys\nimport pydicom as pdc  # read dicom images\nimport numpy as np\nimport imageio    # save to PNG images\nimport matplotlib.pyplot as plt  # plot some PNG images\nimport cv2 as cv  # read PNG images\nfrom random import sample \nfrom joblib import Parallel,delayed\nimport subprocess\nfrom ast import literal_eval\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-05-04T05:08:42.866472Z","iopub.execute_input":"2025-05-04T05:08:42.866922Z","iopub.status.idle":"2025-05-04T05:08:43.653689Z","shell.execute_reply.started":"2025-05-04T05:08:42.866814Z","shell.execute_reply":"2025-05-04T05:08:43.652648Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get CPU informations \n\ndef run(command):\n    process = subprocess.Popen(command, shell=True, stdout=subprocess.PIPE)\n    out, err = process.communicate()\n    print(out.decode('utf-8').strip())\n    \nprint('# CPU')\nrun('cat /proc/cpuinfo | egrep -m 1 \"^model name\"')\nrun('cat /proc/cpuinfo | egrep -m 1 \"^cpu MHz\"')\nrun('cat /proc/cpuinfo | egrep -m 1 \"^cpu cores\"')","metadata":{"execution":{"iopub.status.busy":"2025-05-04T05:08:53.051308Z","iopub.execute_input":"2025-05-04T05:08:53.051831Z","iopub.status.idle":"2025-05-04T05:08:53.106026Z","shell.execute_reply.started":"2025-05-04T05:08:53.05178Z","shell.execute_reply":"2025-05-04T05:08:53.104619Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train and test folder names \ntrain_path_list=glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/*')\ntest_path_list= glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/*')\nkaggle_input_path= '../input/'\n\n#list of name of subdirectorys\ntrain_d=list(map(lambda path:path.split('/')[-1],train_path_list))\ntest_d=list(map(lambda path:path.split('/')[-1],test_path_list))\nmpMRI_scans=[\"FLAIR\",\"T1w\",\"T1wCE\",\"T2w\"] \n\n# sample of names\nprint(train_d[:4])","metadata":{"execution":{"iopub.status.busy":"2025-05-04T00:37:16.601274Z","iopub.execute_input":"2025-05-04T00:37:16.601754Z","iopub.status.idle":"2025-05-04T00:37:16.649719Z","shell.execute_reply.started":"2025-05-04T00:37:16.601689Z","shell.execute_reply":"2025-05-04T00:37:16.648563Z"},"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls ../input/rsna-pneumonia-detection-challenge/stage_2_train_images/ | head -n 10","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:08:55.906449Z","iopub.execute_input":"2025-05-04T05:08:55.906795Z","iopub.status.idle":"2025-05-04T05:08:57.497262Z","shell.execute_reply.started":"2025-05-04T05:08:55.906761Z","shell.execute_reply":"2025-05-04T05:08:57.496101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train and test folder names \ntrain_path_list=glob('../input/rsna-pneumonia-detection-challenge/stage_2_train_images/*')\ntest_path_list= glob('../input/rsna-pneumonia-detection-challenge/stage_2_test_images/*')\nkaggle_input_path= '../input/'\n\n#list of name of subdirectorys\ntrain_d=list(map(lambda path:path.split('/')[-1],train_path_list))\ntest_d=list(map(lambda path:path.split('/')[-1],test_path_list))\n# mpMRI_scans=[\"FLAIR\",\"T1w\",\"T1wCE\",\"T2w\"] \n\n# sample of names\nprint(train_d[:4])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:09:00.092405Z","iopub.execute_input":"2025-05-04T05:09:00.092791Z","iopub.status.idle":"2025-05-04T05:09:00.430248Z","shell.execute_reply.started":"2025-05-04T05:09:00.092751Z","shell.execute_reply":"2025-05-04T05:09:00.429145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# read dico image and return the image array\n\ndef read_dcm(img):\n    \"\"\"reading a dicom image with preprocessing\"\"\"\n    dcm_img=pdc.dcmread(img)\n    img_array=dcm_img.pixel_array\n    img_array = img_array - np.min(img_array)\n    if np.max(img_array) != 0:\n        img_array = img_array / np.max(img_array)\n    img_array = (img_array * 255).astype(np.uint8)\n    return img_array","metadata":{"execution":{"iopub.status.busy":"2025-05-04T05:09:01.984672Z","iopub.execute_input":"2025-05-04T05:09:01.985143Z","iopub.status.idle":"2025-05-04T05:09:01.990641Z","shell.execute_reply.started":"2025-05-04T05:09:01.985095Z","shell.execute_reply":"2025-05-04T05:09:01.989606Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#delete old folder (if you run the code twice)\n\n!rm -rf  Png-rsna-miccai-brain-tumor-radiogenomic-classification/\n\n# create the main PNG folder with the test and train \n\npng_test_path='Png-rsna-miccai-brain-tumor-radiogenomic-classification/test'\npng_train_path='Png-rsna-miccai-brain-tumor-radiogenomic-classification/train'\n\nos.mkdir('Png-rsna-miccai-brain-tumor-radiogenomic-classification')\nos.mkdir(png_train_path)\nos.mkdir(png_test_path)\n\n\n# floders creation \n\nfor trfold in train_d:\n    os.mkdir(png_train_path+'/'+trfold)\n    for mp in mpMRI_scans:\n        os.mkdir(png_train_path+'/'+trfold+'/'+mp)\n        \nfor tsfold in test_d: \n    os.mkdir(png_test_path+'/'+tsfold)\n    for mp in mpMRI_scans:\n        os.mkdir(png_test_path+'/'+tsfold+'/'+mp)","metadata":{"execution":{"iopub.status.busy":"2025-05-04T00:37:16.661497Z","iopub.execute_input":"2025-05-04T00:37:16.662044Z","iopub.status.idle":"2025-05-04T00:37:17.855453Z","shell.execute_reply.started":"2025-05-04T00:37:16.661991Z","shell.execute_reply":"2025-05-04T00:37:17.853674Z"},"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Delete old folder (if you run the code twice)\n!rm -rf rsna-pneumonia-detection-challenge/\n\n# Create the main PNG folder with the test and train \npng_test_path = 'rsna-pneumonia-detection-challenge/stage_2_test_images'\npng_train_path = 'rsna-pneumonia-detection-challenge/stage_2_train_images'\n\nos.mkdir('rsna-pneumonia-detection-challenge')\nos.mkdir(png_train_path)\nos.mkdir(png_test_path)\n\n# Folders creation for training and testing data\nfor trfold in train_d:\n    os.mkdir(png_train_path + '/' + trfold)\n\nfor tsfold in test_d: \n    os.mkdir(png_test_path + '/' + tsfold)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:09:03.677258Z","iopub.execute_input":"2025-05-04T05:09:03.677695Z","iopub.status.idle":"2025-05-04T05:09:05.824882Z","shell.execute_reply.started":"2025-05-04T05:09:03.677659Z","shell.execute_reply":"2025-05-04T05:09:05.823724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# list the paths and name directory \n\npng_train_path_list=glob('Png-rsna-miccai-brain-tumor-radiogenomic-classification/test/*')\npng_test_path_list= glob('Png-rsna-miccai-brain-tumor-radiogenomic-classification/train/*')\n\npng_train_d=list(map(lambda path:path.split('/')[-1],png_train_path_list))\npng_test_d=list(map(lambda path:path.split('/')[-1],png_test_path_list))\n\n#print sample of Png path folders and there names\nprint(png_train_path_list[:3])\nprint(png_train_d[:3])\n\n\n# compare the folder result \n\nprint(train_d.sort()==png_train_d.sort())\nprint(test_d.sort()==png_test_d.sort())\n\n# list of all dicom iamges paths\ntrain_images_path=glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/train/*/*/*')\ntest_images_path=glob('../input/rsna-miccai-brain-tumor-radiogenomic-classification/test/*/*/*')\n","metadata":{"execution":{"iopub.status.busy":"2025-05-04T00:37:17.857158Z","iopub.execute_input":"2025-05-04T00:37:17.857518Z","iopub.status.idle":"2025-05-04T00:38:11.524032Z","shell.execute_reply.started":"2025-05-04T00:37:17.857476Z","shell.execute_reply":"2025-05-04T00:38:11.522745Z"},"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# list the paths and name directory \n\npng_train_path_list=glob('rsna-pneumonia-detection-challenge/stage_2_test_images/*')\npng_test_path_list= glob('rsna-pneumonia-detection-challenge/stage_2_train_images/*')\n\npng_train_d=list(map(lambda path:path.split('/')[-1],png_train_path_list))\npng_test_d=list(map(lambda path:path.split('/')[-1],png_test_path_list))\n\n#print sample of Png path folders and there names\nprint(png_train_path_list[:3])\nprint(png_train_d[:3])\n\n\n# compare the folder result \n\nprint(train_d.sort()==png_train_d.sort())\nprint(test_d.sort()==png_test_d.sort())\n\n# list of all dicom iamges paths\ntrain_images_path=glob('../input/rsna-pneumonia-detection-challenge/stage_2_train_images/*')\ntest_images_path=glob('../input/rsna-pneumonia-detection-challenge/stage_2_test_images/*')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:09:08.389923Z","iopub.execute_input":"2025-05-04T05:09:08.390282Z","iopub.status.idle":"2025-05-04T05:09:08.609472Z","shell.execute_reply.started":"2025-05-04T05:09:08.390246Z","shell.execute_reply":"2025-05-04T05:09:08.608504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The logic to create the new PNG path \n\n# Ensure the test image path is valid\nif len(test_images_path) > 0:\n    image_name = test_images_path[0].split('/')[-1].split('.')[0]\n    new_png_path = png_train_path + '/' + image_name + '.dcm'  # 保持为 .dcm\n    print(new_png_path)\nelse:\n    print(\"No test images found.\")","metadata":{"execution":{"iopub.status.busy":"2025-05-04T05:09:10.617524Z","iopub.execute_input":"2025-05-04T05:09:10.617871Z","iopub.status.idle":"2025-05-04T05:09:10.62382Z","shell.execute_reply.started":"2025-05-04T05:09:10.617841Z","shell.execute_reply":"2025-05-04T05:09:10.622771Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(train_images_path),len(test_images_path))\n","metadata":{"execution":{"iopub.status.busy":"2025-05-04T05:09:13.130168Z","iopub.execute_input":"2025-05-04T05:09:13.130548Z","iopub.status.idle":"2025-05-04T05:09:13.136554Z","shell.execute_reply.started":"2025-05-04T05:09:13.130512Z","shell.execute_reply":"2025-05-04T05:09:13.135439Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# PNG save function \n\ndef train_png(train_path):\n    train_array_img=read_dcm(train_path)                                             \n    train_image_name=train_path.split('/')[4:][-1].split('.')[0]                               \n    new_train_png_path=png_train_path+'/'+'/'.join(train_path.split('/')[4:-1])+'/'+train_image_name+'.PNG'    \n    imageio.imsave(new_train_png_path,train_array_img)\n    \ndef test_png(test_path):\n    test_array_path=read_dcm(test_path) \n    test_image_name=test_path.split('/')[4:][-1].split('.')[0]\n    new_test_png_path=png_test_path+'/'+'/'.join(test_path.split('/')[4:-1])+'/'+test_image_name+'.PNG'\n    imageio.imsave(new_test_png_path,test_array_path)","metadata":{"execution":{"iopub.status.busy":"2025-05-04T05:09:14.382157Z","iopub.execute_input":"2025-05-04T05:09:14.382541Z","iopub.status.idle":"2025-05-04T05:09:14.389887Z","shell.execute_reply.started":"2025-05-04T05:09:14.382507Z","shell.execute_reply":"2025-05-04T05:09:14.388528Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#FIXED : not all the training data was converted and the slow conversion \n# test for 20000 for tarain and 10000 for test \nfb_train=Parallel(n_jobs=4,verbose=1,prefer='threads')(delayed(train_png)(train_path) for train_path in tqdm(train_images_path[:20000],total=len(train_images_path[:20000])))\nfb_test=Parallel(n_jobs=4,verbose=1,prefer='threads') (delayed(test_png) (test_path) for test_path in tqdm(test_images_path[:20000],total=len(test_images_path[:20000])))\n","metadata":{"execution":{"iopub.status.busy":"2025-05-04T05:09:16.14338Z","iopub.execute_input":"2025-05-04T05:09:16.143758Z","iopub.status.idle":"2025-05-04T05:09:21.566551Z","shell.execute_reply.started":"2025-05-04T05:09:16.143727Z","shell.execute_reply":"2025-05-04T05:09:21.564283Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nfrom PIL import Image\nimport numpy as np\n\ndef resize_image(image_path, target_size=(256, 256)):\n    # Read the DICOM file\n    ds = pydicom.dcmread(image_path)\n    img = ds.pixel_array  # Get the pixel array\n    \n    # Convert the pixel array to a PIL image\n    img = Image.fromarray(img)\n    \n    # Resize the image\n    img = img.resize(target_size, Image.ANTIALIAS)\n    return img\n\ndef train_png(train_path):\n    img = resize_image(train_path)\n    # Continue with processing logic\n\ndef test_png(test_path):\n    img = resize_image(test_path)\n    # Continue with processing logic\n\n# Convert training images using parallel processing\nfb_train = Parallel(n_jobs=4, verbose=1, prefer='threads')(\n    delayed(train_png)(train_path) for train_path in tqdm(train_images_path[:2000], total=len(train_images_path[:2000]), mininterval=10)\n)\n\n# Convert testing images using parallel processing\nfb_test = Parallel(n_jobs=4, verbose=1, prefer='threads')(\n    delayed(test_png)(test_path) for test_path in tqdm(test_images_path[:100], total=len(test_images_path[:100]), mininterval=10)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T01:18:15.800415Z","iopub.execute_input":"2025-05-04T01:18:15.800698Z","iopub.status.idle":"2025-05-04T01:18:28.424795Z","shell.execute_reply.started":"2025-05-04T01:18:15.800671Z","shell.execute_reply":"2025-05-04T01:18:28.423902Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nfrom PIL import Image\nimport numpy as np\nfrom joblib import Parallel, delayed\nfrom tqdm import tqdm\nimport os\n\ndef resize_image(image_path, target_size=(256, 256)):\n    # Read the DICOM file\n    ds = pydicom.dcmread(image_path)\n    img = ds.pixel_array  # Get the pixel array\n    \n    # Convert the pixel array to a PIL image\n    img = Image.fromarray(img)\n    \n    # Resize the image\n    img = img.resize(target_size, Image.ANTIALIAS)\n    return img\n\ndef save_image(image, save_path):\n    # Save as PNG format\n    image.save(save_path, format='PNG')\n\ndef train_png(train_path, output_dir):\n    img = resize_image(train_path)\n    # Generate the save path\n    filename = os.path.basename(train_path).replace('.dcm', '.png')\n    save_image(img, os.path.join(output_dir, filename))\n\ndef test_png(test_path, output_dir):\n    img = resize_image(test_path)\n    # Generate the save path\n    filename = os.path.basename(test_path).replace('.dcm', '.png')\n    save_image(img, os.path.join(output_dir, filename))\n\n# Output directories\ntrain_output_dir = 'output/train_images'\ntest_output_dir = 'output/test_images'\n\n# Create output directories\nos.makedirs(train_output_dir, exist_ok=True)\nos.makedirs(test_output_dir, exist_ok=True)\n\n# Convert training images\nfb_train = Parallel(n_jobs=4, verbose=1, prefer='threads')(\n    delayed(train_png)(train_path, train_output_dir) for train_path in tqdm(train_images_path[:2000], total=len(train_images_path[:2000]), mininterval=10)\n)\n\n# Convert testing images\nfb_test = Parallel(n_jobs=4, verbose=1, prefer='threads')(\n    delayed(test_png)(test_path, test_output_dir) for test_path in tqdm(test_images_path[:100], total=len(test_images_path[:100]), mininterval=10)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:09:38.795547Z","iopub.execute_input":"2025-05-04T05:09:38.795903Z","iopub.status.idle":"2025-05-04T05:10:04.462068Z","shell.execute_reply.started":"2025-05-04T05:09:38.795872Z","shell.execute_reply":"2025-05-04T05:10:04.460947Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Image creation and saving part","metadata":{}},{"cell_type":"code","source":"!ls ../working/output/test_images/ | head -n 10","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T01:28:07.319216Z","iopub.execute_input":"2025-05-04T01:28:07.319503Z","iopub.status.idle":"2025-05-04T01:28:07.733121Z","shell.execute_reply.started":"2025-05-04T01:28:07.319472Z","shell.execute_reply":"2025-05-04T01:28:07.732197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# FIXED : printing some black images \n\n#print(PNG random PNG images)\npng_images=glob('working/output/test_images//*')\nprint(f\"Total images found: {len(png_images)}\")\nprint(png_images[:10])\n\n\n\nplt.figure(figsize=(15,15))\nplot_indicator=0\nsub_in=0\nfor index,image in tqdm(enumerate(sample(png_images,1000))) :\n    img = cv.imread(image,cv.IMREAD_GRAYSCALE)\n    img=cv.resize(img,(200,200))\n    if np.max(img)!= 0 and np.mean(img)>=30:  \n        plot_indicator+=1\n        if plot_indicator==5:\n            break\n        else :    \n            sub_in+=1\n            plt.subplot(2,2,sub_in)  \n            plt.imshow(img)\n            \n    else:\n        continue ","metadata":{"execution":{"iopub.status.busy":"2025-05-04T01:28:20.363633Z","iopub.execute_input":"2025-05-04T01:28:20.36391Z","iopub.status.idle":"2025-05-04T01:28:20.39409Z","shell.execute_reply.started":"2025-05-04T01:28:20.363867Z","shell.execute_reply":"2025-05-04T01:28:20.392399Z"},"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2 as cv\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom random import sample\nfrom glob import glob\nfrom tqdm import tqdm\n\n# Get PNG images\npng_images = glob('../working/output/test_images/*.png')  # Ensure the correct file extension\nprint(f\"Total images found: {len(png_images)}\")\nprint(png_images[:10])\n\nplt.figure(figsize=(15, 15))\nplot_indicator = 0\nsub_in = 0\n\n# Check if any images were found\nif len(png_images) == 0:\n    print(\"No images found. Please check the path.\")\nelse:\n    # Sample and process images\n    for index, image in tqdm(enumerate(sample(png_images, min(1000, len(png_images))))):\n        img = cv.imread(image, cv.IMREAD_GRAYSCALE)\n        img = cv.resize(img, (200, 200))\n        \n        if np.max(img) != 0 and np.mean(img) >= 30:\n            plot_indicator += 1\n            if plot_indicator == 5:\n                break\n            else:\n                sub_in += 1\n                plt.subplot(2, 2, sub_in)\n                plt.imshow(img, cmap='gray')\n        else:\n            continue\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:12:09.074621Z","iopub.execute_input":"2025-05-04T05:12:09.074997Z","iopub.status.idle":"2025-05-04T05:12:09.884934Z","shell.execute_reply.started":"2025-05-04T05:12:09.07496Z","shell.execute_reply":"2025-05-04T05:12:09.883943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nfrom glob import glob\n\n# Get PNG images\npng_images = glob('../working/output/test_images/*.png')\nprint(f\"Total images found: {len(png_images)}\")\n\n# Check the size of each image\nfor image_path in png_images:\n    img = Image.open(image_path)  # Open the image\n    width, height = img.size       # Get width and height\n    print(f\"Image: {image_path}, Size: {width}x{height}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T01:30:14.837062Z","iopub.execute_input":"2025-05-04T01:30:14.837392Z","iopub.status.idle":"2025-05-04T01:30:14.86857Z","shell.execute_reply.started":"2025-05-04T01:30:14.837359Z","shell.execute_reply":"2025-05-04T01:30:14.867532Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nfrom glob import glob\n\n# Get DICOM file paths\ndicom_images = glob('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_test_images/*.dcm')\nprint(f\"Total DICOM images found: {len(dicom_images)}\")\n\n# Read the shape of the first 10 images\nfor index, dicom_path in enumerate(dicom_images[:10]):\n    ds = pydicom.dcmread(dicom_path)\n    img = ds.pixel_array\n    height, width = img.shape\n    print(f\"Image {index + 1}: {dicom_path}, Size: {width}x{height}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T01:32:07.653164Z","iopub.execute_input":"2025-05-04T01:32:07.653451Z","iopub.status.idle":"2025-05-04T01:32:07.744788Z","shell.execute_reply.started":"2025-05-04T01:32:07.653424Z","shell.execute_reply":"2025-05-04T01:32:07.743796Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"ok. The size of the original image is 1024. Then I need to read the block diagram label /kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv because my image has been reduced to 256*256. So the x,y,width and height in the table all need to be reduced accordingly. Moreover, I can see that the values in the table are all integers similar to \"366.0 289.0 208.0 527.0\". Therefore, it is better to round them to the nearest whole number. Then, through kaggle, it can be seen that there are a total of 30,000 data points for training and testing, but only 9,555 block diagrams in the training set. Therefore, the number of patients that can be used for training will be even smaller. I hope to first put these people with x labels into a data group df, and then adjust the new block diagram information and concatenate it.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Read the labels file\nlabels_path = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'\nlabels_df = pd.read_csv(labels_path)\n\n# Define original and target image sizes\noriginal_size = 1024\ntarget_size = 256\nscale_factor = target_size / original_size\n\n# Filter valid bounding boxes (images with labels)\nvalid_labels_df = labels_df[labels_df['width'] > 0]  # Keep only boxes with width greater than 0\n\n# Create a new DataFrame to store adjusted bounding box information\nnew_boxes = []\n\n# Iterate through valid label data\nfor index, row in valid_labels_df.iterrows():\n    # Get original bounding box information and scale it\n    x = int(row['x'] * scale_factor)\n    y = int(row['y'] * scale_factor)\n    width = int(row['width'] * scale_factor)\n    height = int(row['height'] * scale_factor)\n    \n    # Store new bounding box information\n    new_boxes.append({\n        'patientId': row['patientId'],\n        'x': x,\n        'y': y,\n        'width': width,\n        'height': height\n    })\n\n# Create a new DataFrame\nnew_boxes_df = pd.DataFrame(new_boxes)\n\nprint(new_boxes_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:12:22.729676Z","iopub.execute_input":"2025-05-04T05:12:22.730016Z","iopub.status.idle":"2025-05-04T05:12:23.732488Z","shell.execute_reply.started":"2025-05-04T05:12:22.729988Z","shell.execute_reply":"2025-05-04T05:12:23.731462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_boxes_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T01:39:21.949555Z","iopub.execute_input":"2025-05-04T01:39:21.949846Z","iopub.status.idle":"2025-05-04T01:39:21.956032Z","shell.execute_reply.started":"2025-05-04T01:39:21.949813Z","shell.execute_reply":"2025-05-04T01:39:21.955262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Then I want to try to extract the feature map\n\nSo first, 9,500 boxes were judged. Then, only for these framed graphs [by matching the file names and patientId], it was determined which files to read, and then the block diagram areas were extracted, right","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm import tqdm\n\n# Output directory for extracted boxes\noutput_dir = 'output/extracted_boxes'\nos.makedirs(output_dir, exist_ok=True)\n\ndef resize_and_save_box(dicom_path, box, output_dir):\n    # Read the DICOM file\n    ds = pydicom.dcmread(dicom_path)\n    img = ds.pixel_array  # Get the pixel array\n    \n    # Extract the bounding box area\n    x = box['x']\n    y = box['y']\n    width = box['width']\n    height = box['height']\n    box_img = img[y:y + height, x:x + width]\n    \n    # Convert the pixel array to a PIL image\n    box_img = Image.fromarray(box_img)\n    \n    # Resize the image\n    box_img = box_img.resize((target_size, target_size), Image.ANTIALIAS)  # Ensure this is appropriate for your task\n    \n    # Generate the save path\n    filename = f\"{box['patientId']}_{box['x']}_{box['y']}.png\"\n    box_img.save(os.path.join(output_dir, filename), format='PNG')\n\n# Iterate through new_boxes_df\nfor index, row in tqdm(new_boxes_df.iterrows(), total=new_boxes_df.shape[0], desc=\"Processing\"):\n    # Check if x column is not empty\n    if not pd.isna(row['x']) and row['x'] > 0:\n        # Get DICOM file path\n        dicom_path = f\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/{row['patientId']}.dcm\"\n        \n        # Extract and save the bounding box area\n        resize_and_save_box(dicom_path, row, output_dir)\n\nprint(\"Extraction and saving completed.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T01:54:43.323452Z","iopub.execute_input":"2025-05-04T01:54:43.323715Z","iopub.status.idle":"2025-05-04T01:56:51.619772Z","shell.execute_reply.started":"2025-05-04T01:54:43.323691Z","shell.execute_reply":"2025-05-04T01:56:51.619116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Output directory\noutput_dir = 'output/extracted_boxes1'\nos.makedirs(output_dir, exist_ok=True)\n\ndef save_box(dicom_path, box, output_dir):\n    # Read the DICOM file\n    ds = pydicom.dcmread(dicom_path)\n    img = ds.pixel_array  # Get the pixel array\n    \n    # Extract the bounding box area\n    x = box['x']\n    y = box['y']\n    width = box['width']\n    height = box['height']\n\n    # Ensure the bounding box area does not exceed image boundaries\n    box_img = img[y:y + height, x:x + width]\n\n    # Convert the pixel array to a PIL image\n    box_img = Image.fromarray(box_img)\n    \n    # Generate the save path\n    filename = f\"{box['patientId']}_{box['x']}_{box['y']}.png\"\n    box_img.save(os.path.join(output_dir, filename), format='PNG')\n\n# Iterate through new_boxes_df\nfor index, row in tqdm(new_boxes_df.iterrows(), total=new_boxes_df.shape[0], desc=\"Processing\"):\n    # Check if x column is not empty\n    if not pd.isna(row['x']) and row['x'] > 0:\n        # Get DICOM file path\n        dicom_path = f\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/{row['patientId']}.dcm\"\n        \n        # Extract and save the bounding box area\n        save_box(dicom_path, row, output_dir)\n\nprint(\"Extraction and saving completed.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T02:01:58.611646Z","iopub.execute_input":"2025-05-04T02:01:58.611973Z","iopub.status.idle":"2025-05-04T02:03:14.851107Z","shell.execute_reply.started":"2025-05-04T02:01:58.611937Z","shell.execute_reply":"2025-05-04T02:03:14.850173Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now I encounter a problem. For instance, feature map 45 is obviously not a feature. Then I need to obtain the positions of the original image and the block diagram. For these images, can we only read the first 100 ones [because they are too large] without reducing them to 256*256, and then label the block diagram information on the images? I need to see exactly what causes the feature maps that are not captured as features.","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport pydicom\nfrom PIL import Image, ImageDraw\nfrom tqdm import tqdm\n\n# Output directory\noutput_dir = 'output/extracted_boxes_with_annotations2'\nos.makedirs(output_dir, exist_ok=True)\n\ndef annotate_and_save_box(dicom_path, box, output_dir):\n    # Read the DICOM file\n    ds = pydicom.dcmread(dicom_path)\n    img = ds.pixel_array  # Get the pixel array\n\n    # Convert the pixel array to a PIL image\n    img = Image.fromarray(img)\n\n    # Create a drawable copy of the image\n    draw = ImageDraw.Draw(img)\n\n    # Extract the bounding box area\n    x = box['x']\n    y = box['y']\n    width = box['width']\n    height = box['height']\n\n    # Annotate the bounding box position on the image\n    draw.rectangle([x, y, x + width, y + height], outline=\"red\", width=3)\n\n    # Generate the save path\n    # filename = f\"{box['patientId']}_{x}_{y}.png\"\n    filename = f\"{box['patientId']}.png\"\n    img.save(os.path.join(output_dir, filename), format='PNG')\n\n# Counter\ncount = 0\nmax_images = 9555\n\n# Iterate through bounding boxes, process up to the first 100 images\nfor index, row in tqdm(new_boxes_df.iterrows(), total=max_images, desc=\"Processing\", mininterval=10):\n    if count >= max_images:\n        break  # Stop iterating when the maximum count is reached\n\n    # Check if x column is not empty\n    if not pd.isna(row['x']) and row['x'] > 0:\n        # Get DICOM file path\n        dicom_path = f\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/{row['patientId']}.dcm\"\n        \n        # Annotate and save the bounding box area\n        annotate_and_save_box(dicom_path, row, output_dir)\n        count += 1  # Increment the counter\n\nprint(\"Extraction and saving completed.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:37:52.725683Z","iopub.execute_input":"2025-05-04T05:37:52.726047Z","iopub.status.idle":"2025-05-04T06:08:44.710528Z","shell.execute_reply.started":"2025-05-04T05:37:52.726016Z","shell.execute_reply":"2025-05-04T06:08:44.709446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\n\n# Get all PNG files\nimage_files = glob.glob('../working/output/extracted_boxes_with_annotations2/*.png')\n\n# Count and print the number\nimage_count = len(image_files)\nprint(f'Found image count: {image_count}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T06:13:57.973158Z","iopub.execute_input":"2025-05-04T06:13:57.973511Z","iopub.status.idle":"2025-05-04T06:13:57.998304Z","shell.execute_reply.started":"2025-05-04T06:13:57.973479Z","shell.execute_reply":"2025-05-04T06:13:57.997396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport glob\n\n# Get paths of annotated images\nannotated_images = glob.glob('output/extracted_boxes_with_annotations1/*.png')\n\n# Select the first 10 images to view\nselected_images = annotated_images[:10]\n\n# Create subplots\nplt.figure(figsize=(20, 10))\n\nfor i, img_path in enumerate(selected_images):\n    img = mpimg.imread(img_path)  # Read the image\n    plt.subplot(2, 5, i + 1)      # Create a subplot\n    plt.imshow(img)                # Show the image\n    plt.axis('off')                # Turn off axis\n    plt.title(f\"Image {i + 1}\")    # Add title\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:12:52.921717Z","iopub.execute_input":"2025-05-04T05:12:52.922083Z","iopub.status.idle":"2025-05-04T05:12:55.220789Z","shell.execute_reply.started":"2025-05-04T05:12:52.922043Z","shell.execute_reply":"2025-05-04T05:12:55.219716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls /kaggle/working/output/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T02:05:48.244707Z","iopub.execute_input":"2025-05-04T02:05:48.245033Z","iopub.status.idle":"2025-05-04T02:05:48.649845Z","shell.execute_reply.started":"2025-05-04T02:05:48.244985Z","shell.execute_reply":"2025-05-04T02:05:48.649019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get PNG images\npng_images = glob.glob('/kaggle/working/output/extracted_boxes1/*.png')\nprint(f\"Total images found: {len(png_images)}\")\n\n# Check the size of each image\nselected_images = png_images[:5]\nfor image_path in selected_images:\n    img = Image.open(image_path)  # Open the image\n    width, height = img.size       # Get width and height\n    print(f\"Image: {image_path}, Size: {width}x{height}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T02:09:08.142379Z","iopub.execute_input":"2025-05-04T02:09:08.142675Z","iopub.status.idle":"2025-05-04T02:09:08.172986Z","shell.execute_reply.started":"2025-05-04T02:09:08.142647Z","shell.execute_reply":"2025-05-04T02:09:08.171881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport glob\n\n# Get PNG image paths\npng_images = glob.glob('/kaggle/working/output/extracted_boxes1/*.png')\n\n# Select the first 5 images\nselected_images = png_images[:5]\n\n# Create subplots\nplt.figure(figsize=(15, 10))\n\nfor i, img_path in enumerate(selected_images):\n    img = mpimg.imread(img_path)  # Read the image\n    plt.subplot(1, 5, i + 1)      # Create a subplot\n    plt.imshow(img)                # Show the image\n    plt.axis('off')                # Turn off axis\n    plt.title(f\"Image {i + 1}\")    # Add title\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:13:00.52426Z","iopub.execute_input":"2025-05-04T05:13:00.524646Z","iopub.status.idle":"2025-05-04T05:13:00.53583Z","shell.execute_reply.started":"2025-05-04T05:13:00.524608Z","shell.execute_reply":"2025-05-04T05:13:00.534821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport glob\n\n# Get PNG image paths\npng_images = glob.glob('/kaggle/working/output/extracted_boxes1/*.png')\n\n# Select the first 100 images\nselected_images = png_images[:100]\n\n# Create subplots\nplt.figure(figsize=(20, 20))  # Adjust the figure size to fit more images\n\n# Calculate number of rows and columns\nrows = 10  # Display 10 images per row\ncols = 10  # Show 10 rows\n\nfor i, img_path in enumerate(selected_images):\n    img = mpimg.imread(img_path)  # Read the image\n    plt.subplot(rows, cols, i + 1)  # Create a subplot\n    plt.imshow(img)                  # Show the image\n    plt.axis('off')                  # Turn off axis\n    plt.title(f\"Image {i + 1}\")      # Add title\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:13:10.784134Z","iopub.execute_input":"2025-05-04T05:13:10.784531Z","iopub.status.idle":"2025-05-04T05:13:10.799219Z","shell.execute_reply.started":"2025-05-04T05:13:10.784494Z","shell.execute_reply":"2025-05-04T05:13:10.797957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!tar -zcf dcm_to_png.tar.gz -C \"/kaggle/working/Png-rsna-miccai-brain-tumor-radiogenomic-classification/\" .","metadata":{"execution":{"iopub.status.busy":"2025-05-04T00:43:52.858731Z","iopub.execute_input":"2025-05-04T00:43:52.859039Z","iopub.status.idle":"2025-05-04T00:44:16.041956Z","shell.execute_reply.started":"2025-05-04T00:43:52.85901Z","shell.execute_reply":"2025-05-04T00:44:16.040405Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python -V","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T00:44:19.392569Z","iopub.execute_input":"2025-05-04T00:44:19.393091Z","iopub.status.idle":"2025-05-04T00:44:20.485677Z","shell.execute_reply.started":"2025-05-04T00:44:19.393035Z","shell.execute_reply":"2025-05-04T00:44:20.484328Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Try again","metadata":{}},{"cell_type":"code","source":"# Sample DataFrames\ndet_class_path = '../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv'\nbbox_path = '../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'\ndicom_dir = '../input/rsna-pneumonia-detection-challenge/stage_2_train_images/'\n\n# Read bounding box and detailed class info data\nbbox_df = pd.read_csv(bbox_path)\ndet_class_df = pd.read_csv(det_class_path)\n\n# Combine DataFrames\ncomb_bbox_df = pd.concat([bbox_df, det_class_df.drop('patientId', axis=1)], axis=1)  # Remove 'patientId' from det_class_df before merging\nprint(comb_bbox_df.shape[0], 'combined cases') \ncomb_bbox_df.sample(3)\n\n# Calculate the number of boxes per patient\nbox_df = comb_bbox_df.groupby('patientId').size().reset_index(name='boxes')  # New column 'boxes' for box counts\ncomb_box_df = pd.merge(comb_bbox_df, box_df, on='patientId')  # Merge original data with box count on 'patientId'\n\n# Count patients by number of boxes\nbox_df.groupby('boxes').size().reset_index(name='patients')  # Classification count\ncomb_bbox_df.groupby(['class', 'Target']).size().reset_index(name='Patient Count')  # Group by target and class for counting\n\n# Create DataFrame for image paths\nimage_df = pd.DataFrame({'path': glob.glob(os.path.join(dicom_dir, '*.dcm'))})  # Use glob to get DICOM paths\nimage_df['patientId'] = image_df['path'].map(lambda x: os.path.splitext(os.path.basename(x))[0])  # Extract patient ID from file name\nprint(image_df.shape[0], 'images found')\n\n# Extract patient IDs and store as sets\nimg_pat_ids = set(image_df['patientId'].values.tolist())  # Extract patient IDs as a set\nbox_pat_ids = set(comb_box_df['patientId'].values.tolist())\n\n# Merge bounding box DataFrame with image paths\nimage_bbox_df = pd.merge(comb_box_df, image_df, on='patientId', how='left').sort_values('patientId')  # Add paths\n\n# DICOM tags to extract\nDCM_TAG_LIST = ['PatientAge', 'BodyPartExamined', 'ViewPosition', 'PatientSex']\ndef get_tags(in_path):\n    c_dicom = pydicom.dcmread(in_path, stop_before_pixels=False)  # Read DICOM file\n    tag_dict = {c_tag: getattr(c_dicom, c_tag, '') for c_tag in DCM_TAG_LIST}  # Extract specified tags\n    tag_dict['path'] = in_path  # Add file path to dictionary\n    return pd.Series(tag_dict)\n\n# Get metadata for images\nimage_meta_df = image_df.head(1000).apply(lambda x: get_tags(x['path']), axis=1)  # Apply function to get tags for the first 1000 images\nimage_meta_df['PatientAge'] = image_meta_df['PatientAge'].map(int)  # Convert 'PatientAge' to integer\n\n# Create a sample DataFrame\nsample_df = image_bbox_df.copy()\nsample_df = sample_df[sample_df['path'].isin(image_meta_df['path'])]  # Filter to include only paths present in metadata\n\n# Check for null values in the sample DataFrame\nprint(sample_df.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:13:15.786371Z","iopub.execute_input":"2025-05-04T05:13:15.786767Z","iopub.status.idle":"2025-05-04T05:13:18.305149Z","shell.execute_reply.started":"2025-05-04T05:13:15.786728Z","shell.execute_reply":"2025-05-04T05:13:18.304071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"new_df = image_bbox_df[image_bbox_df['x'].notna()]\nnew_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:13:20.69744Z","iopub.execute_input":"2025-05-04T05:13:20.697785Z","iopub.status.idle":"2025-05-04T05:13:20.72704Z","shell.execute_reply.started":"2025-05-04T05:13:20.697753Z","shell.execute_reply":"2025-05-04T05:13:20.72606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T02:28:29.013357Z","iopub.execute_input":"2025-05-04T02:28:29.013606Z","iopub.status.idle":"2025-05-04T02:28:29.02922Z","shell.execute_reply.started":"2025-05-04T02:28:29.013583Z","shell.execute_reply":"2025-05-04T02:28:29.028167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def parse_data(bbox_df):  # Read CSV into a dictionary\n    # --- Define lambda to extract coordinates in list [y, x, height, width]\n    extract_box = lambda row: [row['y'], row['x'], row['height'], row['width']]\n\n    parsed = {}\n    for n, row in bbox_df.iterrows():\n        # Initialize\n        pid = row['patientId']\n        if pid not in parsed:\n            parsed[pid] = {\n                # 'dicom': '../input/stage_2_train_images/%s.dcm' % pid,\n                'dicom': os.path.join(dicom_dir, f'{pid}.dcm'),\n                'label': row['Target'],\n                'boxes': []\n            }\n\n        # If label indicates presence of pneumonia, add the box\n        if parsed[pid]['label'] == 1:\n            parsed[pid]['boxes'].append(extract_box(row))\n\n    return parsed\n\nparsed = parse_data(bbox_df)\nprint(parsed['00436515-870c-4b36-a041-de91049b9ab4'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:13:28.791462Z","iopub.execute_input":"2025-05-04T05:13:28.79183Z","iopub.status.idle":"2025-05-04T05:13:31.596879Z","shell.execute_reply.started":"2025-05-04T05:13:28.791795Z","shell.execute_reply":"2025-05-04T05:13:31.595885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pylab\nimport numpy as np\nimport pydicom\n\ndef draw(data):  # Draw a single patient's DICOM image with bounding boxes\n    # --- Open DICOM file\n    d = pydicom.dcmread(data['dicom'])  # Use pydicom to read the DICOM file\n    im = d.pixel_array\n\n    # --- Convert from single-channel grayscale to 3-channel RGB\n    im = np.stack([im] * 3, axis=2)\n\n    # --- Add boxes with random color if present\n    for box in data['boxes']:\n        rgb = np.floor(np.random.rand(3) * 256).astype('int')  # Generate random color\n        im = overlay_box(im=im, box=box, rgb=rgb, stroke=6)  # Overlay the box on the image\n\n    pylab.imshow(im, cmap=pylab.cm.gist_gray)\n    pylab.axis('off')\n\ndef overlay_box(im, box, rgb, stroke=1):  # Overlay a single box on the image\n    # --- Convert coordinates to integers\n    box = [int(b) for b in box]\n    \n    # --- Extract coordinates\n    y1, x1, height, width = box\n    y2 = y1 + height\n    x2 = x1 + width\n\n    im[y1:y1 + stroke, x1:x2] = rgb\n    im[y2:y2 + stroke, x1:x2] = rgb\n    im[y1:y2, x1:x1 + stroke] = rgb\n    im[y1:y2, x2:x2 + stroke] = rgb\n\n    return im\n\n# Draw the DICOM image for the specified patient ID\ndraw(parsed['00436515-870c-4b36-a041-de91049b9ab4'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:18:49.281592Z","iopub.execute_input":"2025-05-04T05:18:49.281968Z","iopub.status.idle":"2025-05-04T05:18:49.54974Z","shell.execute_reply.started":"2025-05-04T05:18:49.281935Z","shell.execute_reply":"2025-05-04T05:18:49.548755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"c_path_tuple","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T02:32:55.907419Z","iopub.execute_input":"2025-05-04T02:32:55.907696Z","iopub.status.idle":"2025-05-04T02:32:55.913199Z","shell.execute_reply.started":"2025-05-04T02:32:55.907667Z","shell.execute_reply":"2025-05-04T02:32:55.912389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls ../working/output/extracted_boxes_with_annotations2/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:20:57.74248Z","iopub.execute_input":"2025-05-04T05:20:57.742842Z","iopub.status.idle":"2025-05-04T05:20:58.824745Z","shell.execute_reply.started":"2025-05-04T05:20:57.742812Z","shell.execute_reply":"2025-05-04T05:20:58.823579Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"If the item of patientId in image_bbox_df is the same as the image name, box the image according to the x,y, width, and height columns in image_bbox_df","metadata":{}},{"cell_type":"markdown","source":"# When piecing together the image names here, some suffixes were added at the end, so an error occurred","metadata":{}},{"cell_type":"code","source":"import glob\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport matplotlib.patches as patches\n\n# Get all PNG files\nimage_files = glob.glob('../working/output/extracted_boxes_with_annotations2/*.png')\n\n# Check if any files were found\nif image_files:\n    # Create subplots\n    fig, m_axs = plt.subplots(10, 10, figsize=(20, 10))  # Create a 10x10 subplot\n    for c_ax, c_path in zip(m_axs.flatten(), image_files[:100]):  # Only get the first 100 files\n        print(f'Loading image file from path: {c_path}')  # Debug information\n        im = mpimg.imread(c_path)  # Read PNG image\n        c_ax.imshow(im, cmap='bone')  # Display image\n        c_ax.set_title(os.path.basename(c_path))  # Use filename as title\n        \n        # Get image name (without path and extension)\n        image_name = os.path.basename(c_path).split('.')[0]\n\n        # Find matching rows in the DataFrame\n        matching_rows = image_bbox_df[image_bbox_df['patientId'] == image_name]\n\n        # Add rectangles for bounding boxes\n        for _, row in matching_rows.iterrows():\n            rect = patches.Rectangle(\n                (row['x'], row['y']),  # Top-left corner coordinates\n                row['width'],          # Width\n                row['height'],         # Height\n                linewidth=2,\n                edgecolor='red',\n                facecolor='none'\n            )\n            c_ax.add_patch(rect)  # Add rectangle to the axis\n\n    plt.tight_layout()  # Adjust subplot layout\n    plt.show()  # Display images\nelse:\n    print(\"No PNG files found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T05:35:31.641227Z","iopub.execute_input":"2025-05-04T05:35:31.641635Z","iopub.status.idle":"2025-05-04T05:35:42.545703Z","shell.execute_reply.started":"2025-05-04T05:35:31.641597Z","shell.execute_reply":"2025-05-04T05:35:42.544636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This script extracts feature maps from images with bounding boxes\nimport glob\nimport os\nimport matplotlib.image as mpimg\nimport matplotlib.patches as patches\nfrom PIL import Image\n\n# Get all PNG files\nimage_files = glob.glob('../working/output/extracted_boxes_with_annotations2/*.png')\n\n# Check if any files were found\nif image_files:\n    for c_path in image_files:  # Iterate through each file\n        print(f'Loading image file from path: {c_path}')  # Debug information\n        im = mpimg.imread(c_path)  # Read the PNG image\n\n        # Get image name (without path and extension)\n        image_name = os.path.basename(c_path).split('.')[0]\n\n        # Find matching rows in the DataFrame\n        matching_rows = image_bbox_df[image_bbox_df['patientId'] == image_name]\n\n        # Add rectangles and save cropped images for each bounding box\n        for i, (_, row) in enumerate(matching_rows.iterrows()):\n            rect = patches.Rectangle(\n                (row['x'], row['y']),  # Top-left corner coordinates\n                row['width'],          # Width\n                row['height'],         # Height\n                linewidth=2,\n                edgecolor='red',\n                facecolor='none'\n            )\n\n            # Crop the image based on the bounding box\n            cropped_im = im[int(row['y']):int(row['y'] + row['height']),\n                             int(row['x']):int(row['x'] + row['width'])]\n            cropped_image = Image.fromarray((cropped_im * 255).astype('uint8'))  # Convert to PIL image\n\n            # Generate save path\n            save_path = os.path.join('../working/output/extracted_boxes_with_annotation3/', f\"{image_name}_feature{i + 1}.png\")\n            cropped_image.save(save_path)  # Save the cropped image\n            print(f'Saved cropped image: {save_path}')\nelse:\n    print(\"No PNG files found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T06:24:21.748543Z","iopub.execute_input":"2025-05-04T06:24:21.748925Z","iopub.status.idle":"2025-05-04T06:29:02.379425Z","shell.execute_reply.started":"2025-05-04T06:24:21.748889Z","shell.execute_reply":"2025-05-04T06:29:02.378063Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\n# Specify the output directory where cropped images are saved\noutput_dir = '../working/output/extracted_boxes_with_annotation3'\n\n# Create a ZIP archive of the output directory\nzip_file_path = '../working/output/extracted_boxes_with_annotations3.zip'\nshutil.make_archive(zip_file_path.replace('.zip', ''), 'zip', output_dir)\n\nprint(f'Packaging complete, file saved at: {zip_file_path}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T06:32:15.505658Z","iopub.execute_input":"2025-05-04T06:32:15.506047Z","iopub.status.idle":"2025-05-04T06:32:24.968611Z","shell.execute_reply.started":"2025-05-04T06:32:15.506008Z","shell.execute_reply":"2025-05-04T06:32:24.967582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\nimport os\n\n# Output directory\noutput_dir = '../working/output/extracted_boxes_with_annotation3/'\n\n# Count the number of images\nimage_files = glob.glob(os.path.join(output_dir, '*.png'))\nimage_count = len(image_files)\nprint(f'Number of images found: {image_count}')\n\n# Check each image path\nif image_files:\n    for c_path in image_files:\n        print(f'Image file: {c_path}')\nelse:\n    print(\"No PNG files found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T06:31:50.335872Z","iopub.execute_input":"2025-05-04T06:31:50.336257Z","iopub.status.idle":"2025-05-04T06:31:51.316549Z","shell.execute_reply.started":"2025-05-04T06:31:50.336223Z","shell.execute_reply":"2025-05-04T06:31:51.315557Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\n\n# Get all PNG files\nimage_files = glob.glob('../working/output/extracted_boxes_with_annotations2/*.png')\n\n# Count and print the number of images\nimage_count = len(image_files)\nprint(f'Number of images found: {image_count}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T06:22:53.661617Z","iopub.execute_input":"2025-05-04T06:22:53.66201Z","iopub.status.idle":"2025-05-04T06:22:53.686022Z","shell.execute_reply.started":"2025-05-04T06:22:53.661971Z","shell.execute_reply":"2025-05-04T06:22:53.685054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport glob\n\n# Set file path\nimage_files = glob.glob('../working/output/extracted_boxes_with_annotations1/*.png')  # Get all PNG files\n\n# Check if any files were found\nif image_files:\n    image_path = image_files[0]  # Select the first file\n    im = mpimg.imread(image_path)  # Read the image\n\n    # Display the image\n    plt.imshow(im)  # Show the image\n    plt.axis('off')  # Turn off the axis\n    plt.show()  # Display the image\nelse:\n    print(\"No PNG files found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T02:42:41.170259Z","iopub.execute_input":"2025-05-04T02:42:41.170634Z","iopub.status.idle":"2025-05-04T02:42:41.323606Z","shell.execute_reply.started":"2025-05-04T02:42:41.170601Z","shell.execute_reply":"2025-05-04T02:42:41.322929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\nimport os\nimport matplotlib.pyplot as plt\nimport pydicom\nimport numpy as np\nfrom matplotlib.patches import Rectangle\n\n# Set DICOM file directory\ndicom_dir = '../input/rsna-pneumonia-detection-challenge/stage_2_train_images/'\ndicom_files = glob.glob(os.path.join(dicom_dir, '*.dcm'))[:6]  # Get the first 6 files\n\n# Create subplots\nfig, m_axs = plt.subplots(2, 3, figsize=(20, 10))  # Create a 2x3 subplot\nfor c_ax, c_path in zip(m_axs.flatten(), dicom_files):\n    print(f'Loading DICOM file from path: {c_path}')  # Debug information\n    c_dicom = pydicom.dcmread(c_path)  # Read the DICOM file\n    c_img_arr = c_dicom.pixel_array\n    \n    # Overlay\n    c_img = plt.cm.gray(c_img_arr)\n    c_img += 0.25 * plt.cm.hot(c_img_arr / c_img_arr.max())\n    c_img = np.clip(c_img, 0, 1)\n    \n    c_ax.imshow(c_img)\n    c_ax.set_title('{class}'.format(**c_rows.iloc[0, :]))  # Adjust title as needed\n\n    for i, (_, c_row) in enumerate(c_rows.dropna().iterrows()):  # Iterate over each row\n        c_ax.plot(c_row['x'], c_row['y'], 's', label='{class}'.format(**c_row))\n        c_ax.add_patch(Rectangle(xy=(c_row['x'], c_row['y']),\n                                 width=c_row['width'],\n                                 height=c_row['height'], \n                                 alpha=0.5,\n                                 fill=False))\n        if i == 0:\n            c_ax.legend()\n\n# Save image\nfig.savefig('overview.png', dpi=600)  # Save with specified DPI\n# plt.close(fig)  # Close figure to free memory","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-04T02:36:22.868027Z","iopub.execute_input":"2025-05-04T02:36:22.868308Z","iopub.status.idle":"2025-05-04T02:36:35.812795Z","shell.execute_reply.started":"2025-05-04T02:36:22.868283Z","shell.execute_reply":"2025-05-04T02:36:35.811857Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# On May 10th, the difference results between direct training and extracted graph training were conducted","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\n\n# Function to load DICOM image\ndef load_dicom_image(file_path):\n    ds = pydicom.dcmread(file_path)\n    img = ds.pixel_array\n    img = img.astype(np.float32)  # Convert to float32\n    img = (img - np.min(img)) / (np.max(img) - np.min(img))  # Normalize\n    return img\n\n# Load a sample image\nimage_path = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/0004cfab-14fd-4e49-80ba-63a80b6bddd6.dcm'\nimage = load_dicom_image(image_path)\n\n# Display the image\nplt.imshow(image, cmap='gray')\nplt.axis('off')\nplt.show()\n\n# Assume you have label data; here we use random data as an example\nnum_samples = 100  # Assume there are 100 samples\n# Load your actual DICOM files to create X\nX = np.array([load_dicom_image(image_path) for image_path in glob.glob('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/*.dcm')[:num_samples]])\ny = np.random.randint(2, size=num_samples)  # Random binary labels\n\n# Split the dataset\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Build a simple convolutional neural network model\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), activation='relu', input_shape=(X.shape[1], X.shape[2], 1)))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Flatten())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dense(1, activation='sigmoid'))  # For binary classification\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\nmodel.fit(X_train, y_train, epochs=10, batch_size=16, validation_data=(X_test, y_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-10T06:40:01.777815Z","iopub.execute_input":"2025-05-10T06:40:01.778094Z","iopub.status.idle":"2025-05-10T06:40:08.406473Z","shell.execute_reply.started":"2025-05-10T06:40:01.778031Z","shell.execute_reply":"2025-05-10T06:40:08.405058Z"},"jupyter":{"outputs_hidden":true},"collapsed":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\n\n# Function to load DICOM image\ndef load_dicom_image(file_path):\n    ds = pydicom.dcmread(file_path)\n    img = ds.pixel_array\n    img = img.astype(np.float32)\n    img = (img - np.min(img)) / (np.max(img) - np.min(img))  # Normalize\n    return img\n\n# Load label data\nlabels_df = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\ndetailed_info_df = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\n\n# Select target columns and merge information\nlabels_df = labels_df[['patientId', 'Target']]\ndetailed_info_df = detailed_info_df[['patientId', 'class']]\n\n# Merge data\nmerged_df = pd.merge(labels_df, detailed_info_df, on='patientId', how='left')\n\n# Load images and labels\nimages = []\ntargets = []\n\nfor index, row in merged_df.iterrows():\n    image_path = f'/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/{row[\"patientId\"]}.dcm'\n    try:\n        img = load_dicom_image(image_path)\n        images.append(img)\n        targets.append(row['Target'])\n    except Exception as e:\n        print(f\"Error loading {image_path}: {e}\")\n\n# Convert to NumPy arrays\nX = np.array(images)\ny = np.array(targets)\n\n# Split the dataset\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Build a simple convolutional neural network model\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), activation='relu', input_shape=(X.shape[1], X.shape[2], 1)))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Flatten())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dense(1, activation='sigmoid'))  # For binary classification\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\nmodel.fit(X_train, y_train, epochs=10, batch_size=16, validation_data=(X_test, y_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-10T06:50:17.747031Z","iopub.execute_input":"2025-05-10T06:50:17.747378Z","execution_failed":"2025-05-10T06:53:07.774Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Lightweight model MobileNetV2+AMP+ frozen layer + disabled gradient","metadata":{}},{"cell_type":"code","source":"import pydicom\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\nfrom keras.applications import MobileNetV2\nfrom keras.optimizers import Adam\nfrom keras import mixed_precision\n\n# Use mixed precision\npolicy = mixed_precision.Policy('mixed_float16')\nmixed_precision.set_global_policy(policy)\n\n# Function to load DICOM image\ndef load_dicom_image(file_path):\n    ds = pydicom.dcmread(file_path)\n    img = ds.pixel_array\n    img = img.astype(np.float32)\n    img = (img - np.min(img)) / (np.max(img) - np.min(img))  # Normalize\n    return img\n\n# Load label data\nlabels_df = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\ndetailed_info_df = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\n\n# Select target columns and merge information\nlabels_df = labels_df[['patientId', 'Target']]\ndetailed_info_df = detailed_info_df[['patientId', 'class']]\nmerged_df = pd.merge(labels_df, detailed_info_df, on='patientId', how='left')\n\n# Load images and labels\nimages = []\ntargets = []\n\nfor index, row in merged_df.iterrows():\n    image_path = f'/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/{row[\"patientId\"]}.dcm'\n    try:\n        img = load_dicom_image(image_path)\n        images.append(img)\n        targets.append(row['Target'])\n    except Exception as e:\n        print(f\"Error loading {image_path}: {e}\")\n\n# Convert to NumPy array and reshape\nX = np.array(images).reshape(-1, images[0].shape[0], images[0].shape[1], 1)\ny = np.array(targets)\n\n# Split the dataset\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Build a lightweight convolutional neural network model\nbase_model = MobileNetV2(input_shape=(X.shape[1], X.shape[2], 1), include_top=False, weights=None)\nbase_model.trainable = False  # Freeze the base model layers\n\nmodel = Sequential()\nmodel.add(base_model)\nmodel.add(Flatten())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dense(1, activation='sigmoid'))  # For binary classification\n\n# Compile the model\noptimizer = Adam(learning_rate=0.001)\nmodel.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\nmodel.fit(X_train, y_train, epochs=10, batch_size=16, validation_data=(X_test, y_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-10T07:17:34.406871Z","iopub.execute_input":"2025-05-10T07:17:34.407205Z","iopub.status.idle":"2025-05-10T07:17:34.435777Z","shell.execute_reply.started":"2025-05-10T07:17:34.407174Z","shell.execute_reply":"2025-05-10T07:17:34.434663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}