{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Welcome to the SPR Gender Prediction Challenge\n\nThis challenge aims to train a model capable of predicting the patient's gender based on a chest X-ray.\n\nAlthough this is a simple notebook, you can use it as a base to improve it, testing new network architectures, including augmentations, changing learning rates, etc.\n\nGood competition for everyone.\n\n**_PS: Before you start, click on the tab Notebook options that is located at the right and select a GPU Accelerator. This is important to accelerate your training._**","metadata":{}},{"cell_type":"code","source":"from fastai.vision.all import *\nimport pandas as pd\nimport warnings\n\nwarnings.filterwarnings('ignore')\ntorch.cuda.empty_cache()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-03T02:36:13.600771Z","iopub.execute_input":"2023-03-03T02:36:13.601097Z","iopub.status.idle":"2023-03-03T02:36:18.213614Z","shell.execute_reply.started":"2023-03-03T02:36:13.601063Z","shell.execute_reply":"2023-03-03T02:36:18.212452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defining the paths to images and labels\ntrain_folder = '/kaggle/input/spr-x-ray-gender/kaggle/kaggle/train'\ntest_folder = '/kaggle/input/spr-x-ray-gender/kaggle/kaggle/test'\ncsv_path = '/kaggle/input/spr-x-ray-gender/train_gender.csv'","metadata":{"execution":{"iopub.status.busy":"2023-03-03T02:36:18.215910Z","iopub.execute_input":"2023-03-03T02:36:18.216352Z","iopub.status.idle":"2023-03-03T02:36:18.223281Z","shell.execute_reply.started":"2023-03-03T02:36:18.216310Z","shell.execute_reply":"2023-03-03T02:36:18.222224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reading the labels dataframe\ndf = pd.read_csv(csv_path, dtype=str, index_col=0)\n\n# defining the dataset paths\ntrain_path = Path(train_folder)\ntest_path = Path(test_folder)\n\n# reading the folders looking for images\ntrain_files = get_image_files(train_path)\ntest_files = sorted(get_image_files(test_path))\n\nprint(f'{len(train_files)} files were found for training and {len(test_files)} files were found for test')","metadata":{"execution":{"iopub.status.busy":"2023-03-03T02:36:18.224798Z","iopub.execute_input":"2023-03-03T02:36:18.225872Z","iopub.status.idle":"2023-03-03T02:36:37.367727Z","shell.execute_reply.started":"2023-03-03T02:36:18.225826Z","shell.execute_reply":"2023-03-03T02:36:37.366606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_=df.gender.value_counts().plot.bar(title='Frequency of Women (0) and Men (1)')","metadata":{"execution":{"iopub.status.busy":"2023-03-03T03:09:21.676448Z","iopub.execute_input":"2023-03-03T03:09:21.676843Z","iopub.status.idle":"2023-03-03T03:09:21.875117Z","shell.execute_reply.started":"2023-03-03T03:09:21.676805Z","shell.execute_reply":"2023-03-03T03:09:21.874117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function used to retun the label from an image\ndef label_func(file):\n    # takes the file's base name w/o the extension\n    basename = int(os.path.basename(str(file)).split('.')[0])\n    \n    # searches for the label \n    label = df.iloc[basename, 0]\n\n    return label","metadata":{"execution":{"iopub.status.busy":"2023-03-03T02:36:37.370426Z","iopub.execute_input":"2023-03-03T02:36:37.370879Z","iopub.status.idle":"2023-03-03T02:36:37.378342Z","shell.execute_reply.started":"2023-03-03T02:36:37.370838Z","shell.execute_reply":"2023-03-03T02:36:37.377218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Here we will define the data loader used to feed the model during the training session.\n\n### You can see some samples below.\n\n- Women were labeled as **0**\n- Men were labeled as **1**","metadata":{}},{"cell_type":"code","source":"# defines the imagem dataloader\nset_seed(42, True)\ndls = ImageDataLoaders.from_name_func(path='.', fnames=train_files, label_func=label_func, item_tfms=Resize(224))            # batch_tfms=aug_transforms(size=224)\n\n# shows some samples\ndls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2023-03-03T02:36:37.379985Z","iopub.execute_input":"2023-03-03T02:36:37.380380Z","iopub.status.idle":"2023-03-03T02:36:45.682346Z","shell.execute_reply.started":"2023-03-03T02:36:37.380342Z","shell.execute_reply":"2023-03-03T02:36:45.681334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Now it is time to train our model.","metadata":{}},{"cell_type":"code","source":"# defining the model\n#model = 'vit_base_patch16_224'\nmodel = 'resnet34'\nlearn = vision_learner(dls, model, metrics=RocAucBinary())\n\n# searching for a good learning rate\nlr=learn.lr_find()\n\nfrom decimal import Decimal\nprint(f'Suggested Learning Rate: {\"%.0e\" % Decimal(lr[0])}')","metadata":{"execution":{"iopub.status.busy":"2023-03-03T02:36:45.683449Z","iopub.execute_input":"2023-03-03T02:36:45.683866Z","iopub.status.idle":"2023-03-03T02:48:49.377579Z","shell.execute_reply.started":"2023-03-03T02:36:45.683826Z","shell.execute_reply":"2023-03-03T02:48:49.376463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training the model\nlearn.fine_tune(1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's check our model's performance:","metadata":{}},{"cell_type":"code","source":"# showing performance metrics\ninterp = ClassificationInterpretation.from_learner(learn)\ninterp.plot_confusion_matrix()\ninterp.print_classification_report()\n\n# shows the biggest mistakes\ninterp.plot_top_losses(9, figsize=(15,10))","metadata":{"execution":{"iopub.status.busy":"2023-03-03T02:48:49.379467Z","iopub.execute_input":"2023-03-03T02:48:49.380092Z","iopub.status.idle":"2023-03-03T02:52:10.094254Z","shell.execute_reply.started":"2023-03-03T02:48:49.380049Z","shell.execute_reply":"2023-03-03T02:52:10.093101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n### Great job! Time to create your submission file contaning the predictions","metadata":{}},{"cell_type":"code","source":"# creating a data loader for the testset\ntest_dl = learn.dls.test_dl(test_files, item_tfms=Resize(224))\n\n# running inferences to get predictions\npreds, _, decoded = learn.get_preds(dl=test_dl, with_decoded=True)\n\n# creating the submission file\ndf = pd.DataFrame([{\"imageId\": idx, \"gender\": int(pred)} for idx, pred in enumerate(decoded)])\ndf.to_csv('submission.csv', index=False)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-03-03T02:52:10.096407Z","iopub.execute_input":"2023-03-03T02:52:10.097034Z","iopub.status.idle":"2023-03-03T02:58:26.651117Z","shell.execute_reply.started":"2023-03-03T02:52:10.096997Z","shell.execute_reply":"2023-03-03T02:58:26.650122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Now it's your time. Try to tune your model, improving its performance.","metadata":{}},{"cell_type":"markdown","source":"### Kudos to the FastAI team for creating such a powerful and easy to use library\n\nhttps://docs.fast.ai/tutorial.vision.html","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}