{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-13T11:12:39.932451Z","iopub.execute_input":"2023-09-13T11:12:39.932804Z","iopub.status.idle":"2023-09-13T11:12:39.93849Z","shell.execute_reply.started":"2023-09-13T11:12:39.932773Z","shell.execute_reply":"2023-09-13T11:12:39.937344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pydicom pillow","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:12:39.941199Z","iopub.execute_input":"2023-09-13T11:12:39.941907Z","iopub.status.idle":"2023-09-13T11:12:51.709648Z","shell.execute_reply.started":"2023-09-13T11:12:39.941864Z","shell.execute_reply":"2023-09-13T11:12:51.708384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pwd","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:12:51.713699Z","iopub.execute_input":"2023-09-13T11:12:51.714036Z","iopub.status.idle":"2023-09-13T11:12:51.721287Z","shell.execute_reply.started":"2023-09-13T11:12:51.714005Z","shell.execute_reply":"2023-09-13T11:12:51.720121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport shutil\nimport pydicom\nfrom PIL import Image\n\n# Paths and ratios\nsource_folder = \"/kaggle/input/rsna-pneumonia-detection-challenge\"\ndestination_folder = \"/kaggle/working/chest-xray-pneumonia/chest_xray\"\nos.makedirs(destination_folder, exist_ok=True)\ntrain_ratio = 0.7\ntest_ratio = 0.15\nval_ratio = 0.15\n\n# Destination folders\ntrain_folder = os.path.join(destination_folder, \"train\")\ntest_folder = os.path.join(destination_folder, \"test\")\nval_folder = os.path.join(destination_folder, \"val\")\npneumonia_train_folder = os.path.join(train_folder, \"PNEUMONIA\")\nnormal_train_folder = os.path.join(train_folder, \"NORMAL\")\npneumonia_test_folder = os.path.join(test_folder, \"PNEUMONIA\")\nnormal_test_folder = os.path.join(test_folder, \"NORMAL\")\npneumonia_val_folder = os.path.join(val_folder, \"PNEUMONIA\")\nnormal_val_folder = os.path.join(val_folder, \"NORMAL\")\n\n# Create destination directories if they don't exist\nos.makedirs(pneumonia_train_folder, exist_ok=True)\nos.makedirs(normal_train_folder, exist_ok=True)\nos.makedirs(pneumonia_test_folder, exist_ok=True)\nos.makedirs(normal_test_folder, exist_ok=True)\nos.makedirs(pneumonia_val_folder, exist_ok=True)\nos.makedirs(normal_val_folder, exist_ok=True)\n\n# Process images and organize\nfor root, _, files in os.walk(source_folder):\n    if files:\n        random.shuffle(files)  # Shuffle files before splitting\n        num_files = len(files)\n        num_train = int(num_files * train_ratio)\n        num_test = int(num_files * test_ratio)\n        num_val = num_files - num_train - num_test\n\n        for i, file in enumerate(files):\n            src_path = os.path.join(root, file)\n            \n            try:\n                dcm_data = pydicom.dcmread(src_path, force=True)\n                image_array = dcm_data.pixel_array\n                image = Image.fromarray(image_array)\n                image_path = os.path.join(destination_folder, file.replace(\".dcm\", \".jpg\"))\n                image.save(image_path)\n\n                if \"PNEUMONIA\" in src_path:\n                    if i < num_train:\n                        dest_folder = pneumonia_train_folder\n                    elif i < num_train + num_test:\n                        dest_folder = pneumonia_test_folder\n                    else:\n                        dest_folder = pneumonia_val_folder\n                else:\n                    if i < num_train:\n                        dest_folder = normal_train_folder\n                    elif i < num_train + num_test:\n                        dest_folder = normal_test_folder\n                    else:\n                        dest_folder = normal_val_folder\n\n                dest_path = os.path.join(dest_folder, file.replace(\".dcm\", \".jpg\"))\n                shutil.copy(image_path, dest_path)\n                \n            except Exception as e:\n                print(f\"Error processing file {src_path}: {str(e)}\")\n\n# Print summary\nprint(\"Conversion and organization complete.\")\nprint(f\"Total images: {num_files}\")\nprint(f\"Train images: {num_train}\")\nprint(f\"Test images: {num_test}\")\nprint(f\"Validation images: {num_val}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:12:51.722677Z","iopub.execute_input":"2023-09-13T11:12:51.72306Z","iopub.status.idle":"2023-09-13T11:17:48.503122Z","shell.execute_reply.started":"2023-09-13T11:12:51.723028Z","shell.execute_reply":"2023-09-13T11:17:48.501082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport shutil\nimport pydicom\nfrom PIL import Image\n\n# Paths and ratios\nsource_folder = \"/kaggle/input/rsna-pneumonia-detection-challenge\"\ndestination_folder = \"/kaggle/working/chest-xray-pneumonia/chest_xray\"\ntrain_ratio = 0.7\ntest_ratio = 0.15\nval_ratio = 0.15\n\n# Destination folders\ntrain_folder = os.path.join(destination_folder, \"train\")\ntest_folder = os.path.join(destination_folder, \"test\")\nval_folder = os.path.join(destination_folder, \"val\")\npneumonia_train_folder = os.path.join(train_folder, \"PNEUMONIA\")\nnormal_train_folder = os.path.join(train_folder, \"NORMAL\")\npneumonia_test_folder = os.path.join(test_folder, \"PNEUMONIA\")\nnormal_test_folder = os.path.join(test_folder, \"NORMAL\")\npneumonia_val_folder = os.path.join(val_folder, \"PNEUMONIA\")\nnormal_val_folder = os.path.join(val_folder, \"NORMAL\")\n\n# Create destination directories if they don't exist\nos.makedirs(pneumonia_train_folder, exist_ok=True)\nos.makedirs(normal_train_folder, exist_ok=True)\nos.makedirs(pneumonia_test_folder, exist_ok=True)\nos.makedirs(normal_test_folder, exist_ok=True)\nos.makedirs(pneumonia_val_folder, exist_ok=True)\nos.makedirs(normal_val_folder, exist_ok=True)\n\n# Process images and organize\nfor root, _, files in os.walk(source_folder):\n    if files:\n        random.shuffle(files)  # Shuffle files before splitting\n        num_files = len(files)\n        num_train = int(num_files * train_ratio)\n        num_test = int(num_files * test_ratio)\n        num_val = num_files - num_train - num_test\n\n        for i, file in enumerate(files):\n            src_path = os.path.join(root, file)\n            \n            try:\n                dcm_data = pydicom.dcmread(src_path)\n                image_array = dcm_data.pixel_array\n                image = Image.fromarray(image_array)\n                image_path = os.path.join(destination_folder, file.replace(\".dcm\", \".jpg\"))\n                image.save(image_path)\n\n                if \"PNEUMONIA\" in src_path:\n                    if i < num_train:\n                        dest_folder = pneumonia_train_folder\n                    elif i < num_train + num_test:\n                        dest_folder = pneumonia_test_folder\n                    else:\n                        dest_folder = pneumonia_val_folder\n                else:\n                    if i < num_train:\n                        dest_folder = normal_train_folder\n                    elif i < num_train + num_test:\n                        dest_folder = normal_test_folder\n                    else:\n                        dest_folder = normal_val_folder\n\n                dest_path = os.path.join(dest_folder, file.replace(\".dcm\", \".jpg\"))\n                shutil.copy(image_path, dest_path)\n            \n            except pydicom.errors.InvalidDicomError:\n                print(f\"Skipping {src_path} as it is not a valid DICOM file.\")\n\n# Print summary\nprint(\"Conversion and organization complete.\")\nprint(f\"Total images: {num_files}\")\nprint(f\"Train images: {num_train}\")\nprint(f\"Test images: {num_test}\")\nprint(f\"Validation images: {num_val}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:17:48.506612Z","iopub.execute_input":"2023-09-13T11:17:48.50698Z","iopub.status.idle":"2023-09-13T11:22:40.982049Z","shell.execute_reply.started":"2023-09-13T11:17:48.50695Z","shell.execute_reply":"2023-09-13T11:22:40.981106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install ultralytics","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:22:40.98335Z","iopub.execute_input":"2023-09-13T11:22:40.984274Z","iopub.status.idle":"2023-09-13T11:22:53.138683Z","shell.execute_reply.started":"2023-09-13T11:22:40.984237Z","shell.execute_reply":"2023-09-13T11:22:53.137378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import wandb\nimport random\nimport cv2 as cv\nfrom ultralytics import YOLO\nfrom matplotlib import pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:22:53.140697Z","iopub.execute_input":"2023-09-13T11:22:53.141091Z","iopub.status.idle":"2023-09-13T11:22:57.787611Z","shell.execute_reply.started":"2023-09-13T11:22:53.14105Z","shell.execute_reply":"2023-09-13T11:22:57.786617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = YOLO('yolov8n-cls.pt')","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:22:57.789197Z","iopub.execute_input":"2023-09-13T11:22:57.789876Z","iopub.status.idle":"2023-09-13T11:22:58.60072Z","shell.execute_reply.started":"2023-09-13T11:22:57.78984Z","shell.execute_reply":"2023-09-13T11:22:58.599739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = model.train(data='/kaggle/working/chest-xray-pneumonia/chest_xray', epochs = 15)","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:22:58.602283Z","iopub.execute_input":"2023-09-13T11:22:58.603367Z","iopub.status.idle":"2023-09-13T11:49:55.083784Z","shell.execute_reply.started":"2023-09-13T11:22:58.603327Z","shell.execute_reply":"2023-09-13T11:49:55.080469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to get the predictions on test images and to plot random predictions with their labels\n\ndef get_rgb_image(image):\n    rgb_image = cv.cvtColor(image, cv.COLOR_BGR2RGB)\n    return rgb_image\n\n\ndef test_accuracy_yolo(test_dir,plot_random=False, n_samples = 8):\n    #test_dir = '/kaggle/input/sports-classification/test'\n    import math\n    \n    test_folders = os.listdir(test_dir)\n    i = 0\n    v = 0\n    result_dict = {}\n\n    for folder in test_folders:\n        results = model(f'{test_dir}/{folder}', verbose=False)\n        result_dict.update({folder:results})\n        for result in results:\n            i += 1\n            top1 = result.probs.top1\n            classes = result.names\n            top1_class_name = classes[top1]\n\n            if top1_class_name == folder:\n                v+=1\n            #top5 = result.probs.top5\n            #top5_class_name = []\n            #[top5_class_name.append(classes[cnum]) for cnum in top5]\n            #top5cont = result.probs.top5conf.tolist()\n\n    test_accuracy = v/i\n\n    if plot_random == True:\n        c = 4\n        r = math.ceil(n_samples/c)\n        plt.figure(figsize=(20,5*r+1))\n        plt.suptitle(f'Visualizing random values with their labels.\\nThe accuracy on the data is passed {str(\"%.2f\" % (test_accuracy*100))}%')\n        for i in range(n_samples):\n\n            random_label = random.choice(list(result_dict.keys()))\n            random_path = f'{test_dir}/{random_label}'\n            pred_vals = result_dict[random_label]\n            random_result = random.choice(pred_vals)\n            random_top1 = random_result.probs.top1\n            classes = random_result.names\n            random_top1_class_name = classes[random_top1]\n            random_top1cont = random_result.probs.top1conf.tolist()\n\n            plt.subplot(r,c,i+1)\n            plt.imshow(get_rgb_image(random_result.orig_img))\n            plt.ylabel(f'Actual : {random_label}')\n            plt.xlabel(f'Predicted : {random_top1_class_name}')\n            plt.title(f'Confidence Interval : {str(\"%.2f\" % random_top1cont)}')\n\n    return test_accuracy, result_dict, classes\n","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:49:55.086899Z","iopub.status.idle":"2023-09-13T11:49:55.087385Z","shell.execute_reply.started":"2023-09-13T11:49:55.08713Z","shell.execute_reply":"2023-09-13T11:49:55.087153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting on validation set , getting the accuracy and plotting random predictions\nimport os\n\nval_dir = '/kaggle/input/chest-xray-pneumonia/chest_xray/val/'\nval_accuracy, predict_dict, classes = test_accuracy_yolo(val_dir,plot_random=True, n_samples=16)\nprint(f'The accuracy on Validation data is {str(\"%.2f\" % (val_accuracy*100))}%')","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:49:55.092949Z","iopub.status.idle":"2023-09-13T11:49:55.096968Z","shell.execute_reply.started":"2023-09-13T11:49:55.096671Z","shell.execute_reply":"2023-09-13T11:49:55.096699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting on test set , getting the accuracy and plotting random predictions\n\ntest_dir = '/kaggle/input/chest-xray-pneumonia/chest_xray/test/'\ntest_accuracy, predict_dict, classes = test_accuracy_yolo(test_dir,plot_random=True, n_samples=16)\nprint(f'The accuracy on test data is {str(\"%.2f\" % (test_accuracy*100))}%')","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:49:55.100066Z","iopub.status.idle":"2023-09-13T11:49:55.100866Z","shell.execute_reply.started":"2023-09-13T11:49:55.100617Z","shell.execute_reply":"2023-09-13T11:49:55.100642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting on test set , getting the accuracy and plotting random predictions\n\ntest_dir = '/kaggle/input/localdataset/localtest'\ntest_accuracy, predict_dict, classes = test_accuracy_yolo(test_dir,plot_random=True, n_samples=16)\nprint(f'The accuracy on test data is {str(\"%.2f\" % (test_accuracy*100))}%')","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:49:55.106159Z","iopub.status.idle":"2023-09-13T11:49:55.106955Z","shell.execute_reply.started":"2023-09-13T11:49:55.10671Z","shell.execute_reply":"2023-09-13T11:49:55.106734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let us plot confusion matrix to identify on which of the images is the model missing out. \n# Also, just checking the images where the model has missed out and ignornig the perfectly predicted categories.\nimport pandas as pd\n\ncm = model.metrics.confusion_matrix.matrix\ncm_df = pd.DataFrame(data = cm)\nnew_df = cm_df.set_axis(cm_df.columns.map(classes), axis=1)\n\nfor i in range(len(new_df)):\n    if cm_df[i][i] == 5 and cm_df[i].sum() == 5 and new_df.iloc[i].sum() == 5:\n        cm_df.drop(i, axis=0, inplace=True)\n        cm_df.drop(i, axis=1, inplace=True)\n        break\n\n\nmappers = cm_df.columns.map(classes)\nmapped_df = cm_df.set_axis(list(mappers), axis=1)\nmapped_df = mapped_df.set_axis(list(mappers), axis=0)\n\nimport seaborn as sns\nplt.figure(figsize=(10,8))\nsns.heatmap(mapped_df, annot=True, cmap='OrRd')\nplt.xlabel('Prediciton')\nplt.ylabel('Truth')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-13T11:49:55.108372Z","iopub.status.idle":"2023-09-13T11:49:55.109131Z","shell.execute_reply.started":"2023-09-13T11:49:55.108891Z","shell.execute_reply":"2023-09-13T11:49:55.108915Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}