{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Kannada MNIST with fastai2 2022\nThis notebook loads my trained model, performs predictions and output submission file.\n\nNotebook that performs training is here: https://www.kaggle.com/code/alexanderyyy/kannada-mnist-fastai2","metadata":{}},{"cell_type":"code","source":"from fastai.vision.all import *\nimport pandas as pd\n%matplotlib inline\nset_seed(3865)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:29:35.031608Z","iopub.execute_input":"2022-07-15T08:29:35.032304Z","iopub.status.idle":"2022-07-15T08:29:35.040665Z","shell.execute_reply.started":"2022-07-15T08:29:35.032263Z","shell.execute_reply":"2022-07-15T08:29:35.039847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert data to image files\nAt the beginning I want to convert dataset into the typical vision flow for fastai.","metadata":{}},{"cell_type":"code","source":"# this need only for the image conversion\nimport numpy as np\nimport os\nimport cv2","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:30:33.499291Z","iopub.execute_input":"2022-07-15T08:30:33.499833Z","iopub.status.idle":"2022-07-15T08:30:33.973132Z","shell.execute_reply.started":"2022-07-15T08:30:33.499792Z","shell.execute_reply":"2022-07-15T08:30:33.972065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to convert csv file into PNG images inside of categorized file tree\n\ninpSize = 28 # input image size\ndef makeTree( tpath, csvpath, withlabel=True ): # make file tree at tpath from csv dataframe\n    tpd = pd.read_csv(csvpath)\n    # tpd = tpd[:30] # short test\n    os.makedirs(tpath, exist_ok=True) # make train or valid or test folder\n    whiteImg = np.ones((28,28))*255\n    \n    for i in range(len(tpd)):\n        img = np.array(tpd.iloc[i,1:]).reshape(28,28)\n        img2 = whiteImg - img\n        catpath = tpath\n        if withlabel:\n            categ = str(tpd.iloc[i].label)\n            catpath = tpath + '/' + categ\n            os.makedirs(catpath, exist_ok=True)\n        cv2.imwrite(catpath + '/' + str(i) + '.png', img2)\n    \n    return","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:30:41.540913Z","iopub.execute_input":"2022-07-15T08:30:41.541360Z","iopub.status.idle":"2022-07-15T08:30:41.555251Z","shell.execute_reply.started":"2022-07-15T08:30:41.541323Z","shell.execute_reply":"2022-07-15T08:30:41.554386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Perform conversion from the CSV file to the image files arranged in classification folders","metadata":{}},{"cell_type":"code","source":"makeTree('./Kannada-PNG/train', '../input/Kannada-MNIST/train.csv', withlabel=True)\n\n# test files are not labeled, \n# so we put them into upper folder, otherwise they will all be auto-labeled as 'test'\nmakeTree('./test', '../input/Kannada-MNIST/test.csv', withlabel=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:30:45.278748Z","iopub.execute_input":"2022-07-15T08:30:45.279188Z","iopub.status.idle":"2022-07-15T08:31:36.973906Z","shell.execute_reply.started":"2022-07-15T08:30:45.279151Z","shell.execute_reply":"2022-07-15T08:31:36.973023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create augmentations, dataloader, model, learner\nNote: augmentations will also work for the TTA (Test Time Augmentation) to improve results","metadata":{}},{"cell_type":"code","source":"import albumentations as Alb\nclass AlbTransform(Transform):\n    def __init__(self, aug): self.aug = aug\n    def encodes(self, img: PILImage):\n        aug_img = self.aug(image=np.array(img))['image']\n        return PILImage.create(aug_img)\n    \ndef get_augs(): return Alb.Compose([\n    # Alb.InvertImg(p=1.), # just because I like it white\n    Alb.ShiftScaleRotate(rotate_limit=20, border_mode=0, value=(255,255,255) ),\n    # Alb.RandomResizedCrop(28,28),\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:33:12.344192Z","iopub.execute_input":"2022-07-15T08:33:12.344677Z","iopub.status.idle":"2022-07-15T08:33:13.387470Z","shell.execute_reply.started":"2022-07-15T08:33:12.344636Z","shell.execute_reply":"2022-07-15T08:33:13.386574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create dataloader from the folder structure\ndls = ImageDataLoaders.from_folder( './Kannada-PNG', train='train', valid=None, valid_pct=0., \n    item_tfms = [AlbTransform(get_augs())], \n    batch_tfms = None, \n    bs=256, shuffle=True )\n# uncomment to test data loaders\n# dls.train.show_batch(max_n=12)\n# dls.valid.show_batch(max_n=12)\nprint('train items:', len(dls.train.items), 'validation items:', len(dls.valid.items))\ndls.vocab","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:33:16.898007Z","iopub.execute_input":"2022-07-15T08:33:16.898566Z","iopub.status.idle":"2022-07-15T08:33:29.868649Z","shell.execute_reply.started":"2022-07-15T08:33:16.898518Z","shell.execute_reply":"2022-07-15T08:33:29.867861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some newer tricks are used, like Mish activation, self-attention, \"ranger\" optimizer, LabelSmoothingCrossEntropy loss function","metadata":{}},{"cell_type":"code","source":"mynet = xresnext50(pretrained=False, \n   act_cls=Mish, \n   sa=True, \n   n_out=dls.c)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:34:23.738653Z","iopub.execute_input":"2022-07-15T08:34:23.739082Z","iopub.status.idle":"2022-07-15T08:34:24.700361Z","shell.execute_reply.started":"2022-07-15T08:34:23.739044Z","shell.execute_reply":"2022-07-15T08:34:24.699395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learnOrig = Learner(dls, model=mynet, path='.',\n    loss_func=LabelSmoothingCrossEntropy(), \n    opt_func=ranger,\n    metrics=[accuracy]\n    ) #.to_fp16()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:39:26.773121Z","iopub.execute_input":"2022-07-15T08:39:26.773695Z","iopub.status.idle":"2022-07-15T08:39:26.800177Z","shell.execute_reply.started":"2022-07-15T08:39:26.773639Z","shell.execute_reply":"2022-07-15T08:39:26.799434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load our trained model\nModel is trained in a separate notebook and added to this notebook as private dataset.\n\nReason: this notebook should run shorter time, because when we submit - they will re-run notebook to confirm the results file.","metadata":{}},{"cell_type":"code","source":"# copy model to the working folder, because input is read-only\n!cp -ar ../input/models .","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:39:29.367442Z","iopub.execute_input":"2022-07-15T08:39:29.367883Z","iopub.status.idle":"2022-07-15T08:39:32.117895Z","shell.execute_reply.started":"2022-07-15T08:39:29.367846Z","shell.execute_reply":"2022-07-15T08:39:32.116514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = learnOrig.load('kanMNIST3-xresnet50')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:39:36.304356Z","iopub.execute_input":"2022-07-15T08:39:36.304877Z","iopub.status.idle":"2022-07-15T08:39:36.595703Z","shell.execute_reply.started":"2022-07-15T08:39:36.304831Z","shell.execute_reply":"2022-07-15T08:39:36.594831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Perform prediction","metadata":{}},{"cell_type":"code","source":"tfnames = get_image_files('./test')\ntst_dl = dls.test_dl(tfnames, with_labels=False, shuffle=False)\n# uncomment to see if dataloader is working\n# tst_dl.show_batch(max_n=12)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:41:12.639232Z","iopub.execute_input":"2022-07-15T08:41:12.639744Z","iopub.status.idle":"2022-07-15T08:41:12.732150Z","shell.execute_reply.started":"2022-07-15T08:41:12.639711Z","shell.execute_reply":"2022-07-15T08:41:12.731094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" TTA (Test Time Augmentation) improves results","metadata":{}},{"cell_type":"code","source":"preds = learn.tta(dl=tst_dl, n=32, use_max=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:41:23.686455Z","iopub.execute_input":"2022-07-15T08:41:23.686903Z","iopub.status.idle":"2022-07-15T08:41:49.032686Z","shell.execute_reply.started":"2022-07-15T08:41:23.686868Z","shell.execute_reply":"2022-07-15T08:41:49.031721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predss = learn.dls.vocab[np.argmax(preds[0], axis=1)] # convert to our classes from probabilities\nidlist = [item.stem for item in tst_dl.items]\nsubm_df = pd.DataFrame(list(zip(idlist, predss)), columns =['id', 'label'])\nsubm_df.to_csv('submission.csv', header=True, index=False)\nsubm_df","metadata":{"execution":{"iopub.status.busy":"2022-07-15T08:46:47.353297Z","iopub.execute_input":"2022-07-15T08:46:47.353852Z","iopub.status.idle":"2022-07-15T08:46:47.488736Z","shell.execute_reply.started":"2022-07-15T08:46:47.353811Z","shell.execute_reply":"2022-07-15T08:46:47.487846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clean up image files\nOtherwise it will burn time comitting.","metadata":{}},{"cell_type":"code","source":"from shutil import rmtree\nrmtree('./Kannada-PNG', ignore_errors=True)\nrmtree('./test', ignore_errors=True)\nrmtree('./models', ignore_errors=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-15T10:26:17.469861Z","iopub.execute_input":"2022-06-15T10:26:17.470315Z","iopub.status.idle":"2022-06-15T10:26:19.672229Z","shell.execute_reply.started":"2022-06-15T10:26:17.470264Z","shell.execute_reply":"2022-06-15T10:26:19.670658Z"},"trusted":true},"execution_count":null,"outputs":[]}]}