{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Using images converted to PNG by ICARO BOMBONATO.\nLink: https://www.kaggle.com/datasets/ibombonato/xray-body-images-in-png-unifesp-competion\n\nThis dataset contains 1,738 train images of 22 categories. With 724 images of the Chest, other parts are represented by 1,014 images of 21 classes, less than 49 images per class in average, with some rare classes having just few images, like Clavicles with 9 images only, and several multi-lables classes with a single image.\n\nI am starting from the 20% split for validation, once I see methodology working, I will re-train on the whole set without validation in order to use all of the scarse samples. It is not possible to split validation set nicely, which ia annoying.\n\nIt seems not very difficult to get 0.91 using any method. But it is tricky to get better score, because confusion will go around 60 images that are multi-labeled - not within a pure class. Looking through those images provides no help - it is very subjective, because they contain portions of other categories, example: you can see almost complete chest with Thoracic Spine, complete neck, and bottom jaw. Or chest + abdomen + almost complete pelvis + Thoracic Spine + Lumbar Spine. Most probably it would be classified by the organisers in 2 or 3 categories, but which? Who knows...\n\nOnce I stopped separating labels like '9 21' into 2 classes, but treat those as a single class - score improved to 0.92255.\nThat makes 41 or 42 classes instead 21 with even fewer training samples per class, but it scores better, like extra 7 images guessed right.","metadata":{}},{"cell_type":"code","source":"from fastai.vision.all import *\nimport pandas as pd\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-12T05:06:41.37435Z","iopub.execute_input":"2022-06-12T05:06:41.374799Z","iopub.status.idle":"2022-06-12T05:06:44.728346Z","shell.execute_reply.started":"2022-06-12T05:06:41.374711Z","shell.execute_reply":"2022-06-12T05:06:44.727353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import albumentations as Alb\nclass AlbTransform(Transform):\n    def __init__(self, aug): self.aug = aug\n    def encodes(self, img: PILImage):\n        aug_img = self.aug(image=np.array(img))['image']\n        return PILImage.create(aug_img)\n    \ndef get_augs(): return Alb.Compose([\n    Alb.Transpose(),\n    Alb.Flip(),\n    Alb.HorizontalFlip(),\n    Alb.RandomRotate90(),\n    Alb.RandomScale(),\n    Alb.RandomBrightnessContrast(),\n])\n\nitem_tfms = [Resize(224), AlbTransform(get_augs())]\nbatch_tfms = Normalize.from_stats(*imagenet_stats) ","metadata":{"execution":{"iopub.status.busy":"2022-05-30T15:02:35.655237Z","iopub.execute_input":"2022-05-30T15:02:35.655987Z","iopub.status.idle":"2022-05-30T15:02:37.042715Z","shell.execute_reply.started":"2022-05-30T15:02:35.655938Z","shell.execute_reply":"2022-05-30T15:02:37.041806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path='../input/xray-body-images-in-png-unifesp-competion/images/train'\ntrain_df = pd.read_csv('../input/xray-body-images-in-png-unifesp-competion/train_df.csv')\ntrain_df['Target'] = train_df['Target'].apply(lambda x: x.strip())\n# for interactive DEBUG: reduce amount of train images : train_df = train_df[:512]\ndls = ImageDataLoaders.from_df(train_df, path=train_path, suff='-c.png', # label_delim = ' ',\n    item_tfms=item_tfms, batch_tfms=batch_tfms, shuffle=True, bs=64, valid_pct=0.)","metadata":{"execution":{"iopub.status.busy":"2022-05-30T15:04:13.403645Z","iopub.execute_input":"2022-05-30T15:04:13.404206Z","iopub.status.idle":"2022-05-30T15:04:13.788468Z","shell.execute_reply.started":"2022-05-30T15:04:13.404173Z","shell.execute_reply":"2022-05-30T15:04:13.787536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# try this to see dataloader working\n# dls.train.show_batch(max_n=12)\n# dls.valid.show_batch(max_n=12)","metadata":{"execution":{"iopub.status.busy":"2022-05-30T15:04:16.907951Z","iopub.execute_input":"2022-05-30T15:04:16.908475Z","iopub.status.idle":"2022-05-30T15:04:19.389911Z","shell.execute_reply.started":"2022-05-30T15:04:16.908443Z","shell.execute_reply":"2022-05-30T15:04:19.389216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = vision_learner(dls, densenet121, path='.', \n    loss_func=FocalLoss(),  \n    metrics=[F1ScoreMulti(average='macro'), F1ScoreMulti(average='weighted'), APScoreMulti()] )","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:32:21.746353Z","iopub.execute_input":"2022-05-29T07:32:21.746838Z","iopub.status.idle":"2022-05-29T07:32:22.641297Z","shell.execute_reply.started":"2022-05-29T07:32:21.746807Z","shell.execute_reply":"2022-05-29T07:32:22.640445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fine_tune(90, freeze_epochs=3) # learn is performed using fit_one_cycle()","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:32:22.642947Z","iopub.execute_input":"2022-05-29T07:32:22.643381Z","iopub.status.idle":"2022-05-29T07:41:38.505493Z","shell.execute_reply.started":"2022-05-29T07:32:22.643337Z","shell.execute_reply":"2022-05-29T07:41:38.503915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.export()\n# learn.save('my_super_model') ","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:41:38.507326Z","iopub.execute_input":"2022-05-29T07:41:38.507739Z","iopub.status.idle":"2022-05-29T07:41:38.69498Z","shell.execute_reply.started":"2022-05-29T07:41:38.507702Z","shell.execute_reply":"2022-05-29T07:41:38.694057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path='../input/xray-body-images-in-png-unifesp-competion/images/test'\ntest_df = pd.read_csv('../input/xray-body-images-in-png-unifesp-competion/test_df.csv')\n# for interactive DEBUG: test_df = test_df[:12]\n# new loader for the sake of path, which if different from the training path\ntdls = ImageDataLoaders.from_df(test_df, path=test_path, suff='-c.png',\n    item_tfms=item_tfms, batch_tfms=batch_tfms, shuffle=False)\ntst_dl = tdls.test_dl(test_df) ","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:41:38.696416Z","iopub.execute_input":"2022-05-29T07:41:38.696848Z","iopub.status.idle":"2022-05-29T07:41:38.885239Z","shell.execute_reply.started":"2022-05-29T07:41:38.696807Z","shell.execute_reply":"2022-05-29T07:41:38.88447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# try it working \ntst_dl.show_batch(max_n=12)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:41:38.886334Z","iopub.execute_input":"2022-05-29T07:41:38.886838Z","iopub.status.idle":"2022-05-29T07:41:40.74428Z","shell.execute_reply.started":"2022-05-29T07:41:38.886805Z","shell.execute_reply":"2022-05-29T07:41:40.743495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = learn.tta(dl=tst_dl, n=64, use_max=False)\n# preds = learn.get_preds(dl=tst_dl)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:41:40.746113Z","iopub.execute_input":"2022-05-29T07:41:40.747022Z","iopub.status.idle":"2022-05-29T07:48:01.24521Z","shell.execute_reply.started":"2022-05-29T07:41:40.746985Z","shell.execute_reply":"2022-05-29T07:48:01.242874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# simplest single label submission\n# use this when multilabel like '9 21' is considered as a single class\npredss = learn.dls.vocab[np.argmax(preds[0], axis=1)]\ntest_df['Target'] = predss\nsubmission_df = test_df[['SOPInstanceUID', 'Target']]\nsubmission_df.to_csv(f'submission_XRP_argmax.csv', header=True, index=False)\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:49:59.311034Z","iopub.execute_input":"2022-05-29T07:49:59.311513Z","iopub.status.idle":"2022-05-29T07:49:59.37565Z","shell.execute_reply.started":"2022-05-29T07:49:59.311476Z","shell.execute_reply":"2022-05-29T07:49:59.374579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Have a nice day!","metadata":{}},{"cell_type":"code","source":"train_df.Target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-05-30T15:05:17.155715Z","iopub.execute_input":"2022-05-30T15:05:17.156352Z","iopub.status.idle":"2022-05-30T15:05:17.164162Z","shell.execute_reply.started":"2022-05-30T15:05:17.156301Z","shell.execute_reply":"2022-05-30T15:05:17.163504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.Target.value_counts() ","metadata":{"execution":{"iopub.status.busy":"2022-05-29T05:03:40.302678Z","iopub.execute_input":"2022-05-29T05:03:40.303232Z","iopub.status.idle":"2022-05-29T05:03:40.314305Z","shell.execute_reply.started":"2022-05-29T05:03:40.30319Z","shell.execute_reply":"2022-05-29T05:03:40.313205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# primary submission, multilabel\n# use this when \"label_delim = ' '\" in data loader\n# and multilabel like '9 21' is considered as 2 separate single classes: '9' and '21'\n# need to adjust threshold\n# I don't like code below, need to rewrite for beauty reasons\nthreshold = 9.\ndef get_preds_from_preds(pred_tensor):\n    has_tsh = any(pred_tensor >= threshold)\n    if has_tsh: \n        return learn.dls.vocab[pred_tensor >= threshold]\n    else:\n        return learn.dls.vocab[pred_tensor.argmax()]\n\ndef get_labels_from_preds(items):\n    if type(items) == str: # single string, do nothing\n        return items\n    # below it may be a list\n    r = []\n    for item in items:\n        if(item.isnumeric()):\n            r.append(int(item))\n    r = [str(i) for i in sorted(r)]\n    return \" \".join(r) if any(r) else \"-1\"\n\nlabelled_preds = [get_preds_from_preds(pred) for pred in preds[0]]\nstr_preds = [get_labels_from_preds(items) for items in labelled_preds]\n\n# test_df['Target'] = str_preds\n# submission_df = test_df[['SOPInstanceUID', 'Target']]\n# submission_df.to_csv(f'submission_XRP.csv', header=True, index=False)\n# submission_df","metadata":{"execution":{"iopub.status.busy":"2022-05-29T07:50:04.420836Z","iopub.execute_input":"2022-05-29T07:50:04.421244Z","iopub.status.idle":"2022-05-29T07:50:04.77661Z","shell.execute_reply.started":"2022-05-29T07:50:04.421213Z","shell.execute_reply":"2022-05-29T07:50:04.775654Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create file with all prediction for dubugging results\n\n# d = {'index': range(0,22), 'BodyPart': ['0_Abdomen', '1_Ankle', '2_CervicalSpine',\n#     '3_Chest', '4_Clavicles', '5_Elbow', '6_Feet', '7_Finger', '8_Forearm',\n#     '9_Hand', '10_Hip', '11_Knee', '12_LowerLeg', '13_LumbarSpine',\n#     '14_Others', '15_Pelvis', '16_Shoulder', '17_Sinus', '18_Skull',\n#     '19_Thigh', '20_ThoracicSpine', '21_Wrist']}\n# BodyPart_df = pd.DataFrame(d)\n# tens_df = pd.DataFrame(preds[0])\n# predicts_df = submission_df.copy() # make a copy\n\n# for i in range(0, len(preds[0][0])):\n#     bodyIndex = int(learn.dls.vocab[i])\n#     bpartName = BodyPart_df.loc[bodyIndex].BodyPart\n#     predicts_df[bpartName] = tens_df[i]\n    \n# predicts_df.to_csv(f'testPredictions_XRP.csv', header=True, index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-29T08:41:03.776149Z","iopub.execute_input":"2022-05-29T08:41:03.776599Z","iopub.status.idle":"2022-05-29T08:41:03.828255Z","shell.execute_reply.started":"2022-05-29T08:41:03.776547Z","shell.execute_reply":"2022-05-29T08:41:03.827133Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]}]}