{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Welcome to the SPR Chest X-Ray Age Prediction Challenge\n\nThis challenge aims to train a model capable of predicting the patient's age based on a chest X-ray.\n\nAlthough this is a simple notebook, you can use it as a base to improve it, testing new network architectures, including augmentations, changing learning rates, etc.\n\nGood competition for everyone.\n\n**_PS: Before you start, click on the tab Notebook options that is located at the right and select a GPU Accelerator. This is important to accelerate your training._**","metadata":{}},{"cell_type":"code","source":"from fastai.vision.all import *\nimport pandas as pd\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-03-05T01:55:33.733698Z","iopub.execute_input":"2023-03-05T01:55:33.734322Z","iopub.status.idle":"2023-03-05T01:55:37.148974Z","shell.execute_reply.started":"2023-03-05T01:55:33.734261Z","shell.execute_reply":"2023-03-05T01:55:37.147740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defining the paths to images and labels\ntrain_folder = '/kaggle/input/spr-x-ray-age/kaggle/kaggle/train'\ntest_folder = '/kaggle/input/spr-x-ray-age/kaggle/kaggle/test'\ncsv_path = '/kaggle/input/spr-x-ray-age/train_age.csv'","metadata":{"execution":{"iopub.status.busy":"2023-03-05T01:55:37.155608Z","iopub.execute_input":"2023-03-05T01:55:37.156309Z","iopub.status.idle":"2023-03-05T01:55:37.162210Z","shell.execute_reply.started":"2023-03-05T01:55:37.156251Z","shell.execute_reply":"2023-03-05T01:55:37.161133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reading the labels dataframe\ndf = pd.read_csv(csv_path, dtype=float, index_col=0)\n\n# defining the dataset paths\ntrain_path = Path(train_folder)\ntest_path = Path(test_folder)\n\n# reading the folders looking for images\ntrain_files = get_image_files(train_path)\ntest_files = sorted(get_image_files(test_path))\n\nprint(f'{len(train_files)} files were found for training and {len(test_files)} files were found for test')","metadata":{"execution":{"iopub.status.busy":"2023-03-05T01:55:37.165431Z","iopub.execute_input":"2023-03-05T01:55:37.166197Z","iopub.status.idle":"2023-03-05T01:55:41.289905Z","shell.execute_reply.started":"2023-03-05T01:55:37.166158Z","shell.execute_reply":"2023-03-05T01:55:41.288800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Age varies from {df.age.min()} to {df.age.max()}\\n')\n_=df.age.hist()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T01:55:41.296002Z","iopub.execute_input":"2023-03-05T01:55:41.296987Z","iopub.status.idle":"2023-03-05T01:55:41.667750Z","shell.execute_reply.started":"2023-03-05T01:55:41.296947Z","shell.execute_reply":"2023-03-05T01:55:41.666466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function used to retun the label from an image\ndef label_func(file):\n    # takes the file's base name w/o the extension\n    basename = int(os.path.basename(str(file)).split('.')[0])\n    \n    # searches for the label \n    label = int(df.age[basename])\n\n    return label","metadata":{"execution":{"iopub.status.busy":"2023-03-05T01:55:41.669659Z","iopub.execute_input":"2023-03-05T01:55:41.670460Z","iopub.status.idle":"2023-03-05T01:55:41.678355Z","shell.execute_reply.started":"2023-03-05T01:55:41.670418Z","shell.execute_reply":"2023-03-05T01:55:41.677085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Here we will define the data loader used to feed the model during the training session.\n\n### You can see some samples below.","metadata":{}},{"cell_type":"code","source":"# defines the image dataloader\ndtblk = DataBlock(blocks=(ImageBlock, RegressionBlock), get_items=get_image_files, get_y=label_func, item_tfms=Resize(224))\ndls = dtblk.dataloaders(train_folder)\n\n# shows some samples\ndls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T01:55:41.681857Z","iopub.execute_input":"2023-03-05T01:55:41.683152Z","iopub.status.idle":"2023-03-05T01:55:52.164405Z","shell.execute_reply.started":"2023-03-05T01:55:41.683108Z","shell.execute_reply":"2023-03-05T01:55:52.163286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now it is time to train our model.","metadata":{}},{"cell_type":"code","source":"# defining the model\nlearn = vision_learner(dls, resnet34, y_range=(18, 98), loss_func=L1LossFlat(), metrics=mae)\n# training\nlearn.fine_tune(1)","metadata":{"execution":{"iopub.status.busy":"2023-03-05T01:55:52.165429Z","iopub.execute_input":"2023-03-05T01:55:52.165809Z","iopub.status.idle":"2023-03-05T02:12:53.478132Z","shell.execute_reply.started":"2023-03-05T01:55:52.165773Z","shell.execute_reply":"2023-03-05T02:12:53.476092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's check out some predictions:","metadata":{}},{"cell_type":"code","source":"print(f'{\" \"*26}PREDICTIONS:\\n\\n{\" \"*29}Label\\n{\" \"*26}(predicted)')\nlearn.show_results()","metadata":{"execution":{"iopub.status.busy":"2023-03-05T02:12:53.485647Z","iopub.execute_input":"2023-03-05T02:12:53.488719Z","iopub.status.idle":"2023-03-05T02:13:00.937900Z","shell.execute_reply.started":"2023-03-05T02:12:53.488636Z","shell.execute_reply":"2023-03-05T02:13:00.936164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n### Great job! Time to create your submission file contaning the predictions","metadata":{}},{"cell_type":"code","source":"# creating a data loader for the testset\ntest_dl = learn.dls.test_dl(test_files, item_tfms=Resize(224))\n\n# running inferences to get predictions\npreds, _, decoded = learn.get_preds(dl=test_dl, with_decoded=True)\n\n# creating the submission file\ndf = pd.DataFrame([{\"imageId\": idx, \"age\": int(pred)} for idx, pred in enumerate(decoded)])\ndf.to_csv('submission.csv', index=False)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-03-05T02:13:00.940017Z","iopub.execute_input":"2023-03-05T02:13:00.940627Z","iopub.status.idle":"2023-03-05T02:25:46.414105Z","shell.execute_reply.started":"2023-03-05T02:13:00.940581Z","shell.execute_reply":"2023-03-05T02:25:46.411846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now it's your time. Try to tune your model, improving its performance.","metadata":{}},{"cell_type":"markdown","source":"### Kudos to the FastAI team for creating such a powerful and easy to use library\n\nhttps://docs.fast.ai/tutorial.vision.html","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}