{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## X-Ray Body Part Competition FastAI + PNG Starter\n\nThis notebook will show how to classify converted images from DICOM to PNG to create a classifier using FAST.AI\n\nThe dataset contains imagens in PNG and Metadata in CSV format.  \n\nhttps://www.kaggle.com/datasets/ibombonato/xray-body-images-in-png-unifesp-competion\n\nThis can speed up the train many times, since it takes a long time to load/read DICOM files.\n\nPlease, **upvote** the **dataset** and this **notebook** if it helps you in some manner.","metadata":{}},{"cell_type":"code","source":"!pip install --user ../input/fastaimaster/torch-1.9.0-cp37-cp37m-manylinux1_x86_64.whl\n!pip install -U fastai","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pathlib import Path\nimport fastai\nfrom fastai.vision.all import *\n\nINPUT_PATH = \"../input/xray-body-images-in-png-unifesp-competion/\"\nEPOCHS = 50","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image\n\nim = plt.imread(\"../input/xray-body-images-in-png-unifesp-competion/image_png.png\")\nplt.imshow(im, cmap='gray')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(Path(INPUT_PATH, \"train_df.csv\"))\ndf_train['Target'] = df_train['Target'].apply(lambda x: x.strip())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Simple split for training.\n\nA better split strategy should improve your score.\n","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import ShuffleSplit\n\nX = df_train\ny = df_train['Target']\nsss = ShuffleSplit(n_splits=1, test_size=.2, random_state=42)\ntrain_idx, val_idx = next(sss.split(X, y))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load and visualize the data","metadata":{}},{"cell_type":"code","source":"df_train['is_valid'] = False\ndf_train.loc[val_idx, 'is_valid'] = True\n\ndls = ImageDataLoaders.from_df(df_train, \n                               fn_col = 'image_path',\n                               label_col = 'Target',\n                               label_delim = ' ',\n                               valid_col='is_valid',\n                               path = INPUT_PATH,\n                               item_tfms=Resize(224), \n                               batch_tfms=aug_transforms(size=224))\n\ndls.show_batch()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training phase","metadata":{}},{"cell_type":"code","source":"!mkdir ./model","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_macro = F1ScoreMulti(thresh=0.5, average='macro')\nf1_samples = F1ScoreMulti(thresh=0.5, average='samples')\nf1_micro = F1ScoreMulti(thresh=0.5, average='micro')\nf1_weighted = F1ScoreMulti(thresh=0.5, average='weighted')\nlearn = cnn_learner(dls, resnet50, \n                    metrics=[partial(accuracy_multi, thresh=0.5), f1_macro, f1_samples, f1_micro, f1_weighted], \n                    model_dir='/kaggle/working/model')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = learn.lr_find()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fine_tune(EPOCHS, lr[0])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## View results","metadata":{}},{"cell_type":"code","source":"learn.show_results()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp = Interpretation.from_learner(learn)\ninterp.plot_top_losses(9)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict Test data","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv(Path(INPUT_PATH, \"test_df.csv\"))\ntest_dl_df = dls.test_dl(df_test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_test_df = learn.get_preds(dl=test_dl_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_preds_from_preds(pred_tensor):\n    has_tsh = any(pred_tensor >= 0.5)\n    if has_tsh: \n        return learn.dls.vocab[pred_tensor >= 0.5]\n    else:\n        return learn.dls.vocab[pred_tensor.argmax()]\n\ndef get_labels_from_preds(items):\n    r = []\n    for item in items:\n        if(item.isdigit()):\n            r.append(int(item))\n    r = [str(i) for i in sorted(r)]\n    return \" \".join(r) if any(r) else \"-1\"\n\nlabelled_preds = [get_preds_from_preds(pred) for pred in preds_test_df[0]]\nstr_preds = [get_labels_from_preds(items) for items in labelled_preds]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['Target'] = str_preds\ndf_submission = df_test[['SOPInstanceUID', 'Target']]\nprint(df_submission.shape)\ndf_submission.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save and submit the results","metadata":{}},{"cell_type":"code","source":"learn.save('./learner_save_resnet50_25_epochs')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.export('/kaggle/working/model/learner_export_resnet50_25_epochs.pkl')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}