{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **This is my very first Kaggle competition and notebook**","metadata":{}},{"cell_type":"markdown","source":"I re-worked this to use a custom resized dataset for training since the kernel was very slow on the full sized images.  I implemented on GCP with many different architectures including deeper resnets, ResNext, efficientnet and others both pretrained and not, but nothing performed any better than plain old resnet50","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom fastai.vision.all import *","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check gpu install and availability\nimport torch\nprint(torch.__version__)\nprint(torch.cuda.is_available())\nprint(torch.cuda.current_device())\n!nvidia-smi","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#we have no internet in the kernel so we will copy our pretrained resnet50 model to the torch cache folder fastai will search by default for use in training\n!rm -rf /root/.cache/torch/hub/checkpoints/resnet50-19c8e357.pth\nPath('/root/.cache/torch/hub/checkpoints/').mkdir(exist_ok=True, parents=True)\n!cp '../input/resnet50-no-internet/resnet50-19c8e357.pth' '/root/.cache/torch/hub/checkpoints/resnet50-19c8e357.pth'\n!ls -ltr /root/.cache/torch/hub/checkpoints/","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#setup the config perhaps more useful for plain python than a notebook but it's a habit\nINPUT_DIR = \"/kaggle/input/plant-pathology-2021-fgvc8\"\nOUTPUT_DIR = \"/kaggle/working\"\nPICKLE_LEARNER = \"/kaggle/working/96acc-resnet50-lr3e-2.pkl\"\nSUBMISSION = \"/kaggle/working/submission.csv\"\nTRAINING_DATA_DIRECTORY = \"/kaggle/input/plant-pathology-2021-fgvc8-resized-600x400\"\nTEST_DATA_DIRECTORY = str(INPUT_DIR) + \"/test_images\"\nLABELS_FILE = str(INPUT_DIR) + \"/train.csv\"\n\nprint(f\"config path: {INPUT_DIR}\")\n\nprint(f\"labels file: {LABELS_FILE}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(LABELS_FILE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check the labels\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load the data  we will use a custom resized dataset for faster training\ndls = ImageDataLoaders.from_csv(INPUT_DIR, LABELS_FILE, folder=TRAINING_DATA_DIRECTORY, delimiter=',', label_delim=' ',\n                               item_tfms=Resize(460), batch_tfms=[*aug_transforms(size=224),Normalize.from_stats(*imagenet_stats)], bs=32)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#look at our classes\ndls.vocab","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#look at a batch of images and labels (multiple classes)\ndls.show_batch()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create a trainer using resnet50 architecture using our no internet cached model\nlearn = cnn_learner(dls, resnet50, model_dir='/kaggle/working', metrics=partial(accuracy_multi, thresh=0.5))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find a reasonable learning rate via this helpful fastai learning rate finder \n#learn.lr_find()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train four epochs using fine_tune which trains the head initially and then all the other layers for four epochs\nlearn.fine_tune(4, 3e-2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save our model for inference later\nlearn.export(fname=PICKLE_LEARNER)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load the pickled learner for inference\ninf_learn = load_learner(PICKLE_LEARNER)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check our classes\ninf_learn.dls.vocab","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get our test data using the fastai convenience function\ntest_files = get_image_files(TEST_DATA_DIRECTORY)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#see how many test images we got\nlen(test_files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load our test dataset for inference\ntest_dl = inf_learn.dls.test_dl(test_files)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show a batch\ntest_dl.show_batch()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get our predictions on the test set\npreds,_,dec_preds = inf_learn.get_preds(dl=test_dl, with_decoded=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check out our predication and the decoded results\npreds, dec_preds","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#look at the first class predicted\n#inf_learn.dls.vocab[dec_preds[0]]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create a df for submission\nsub_df = pd.DataFrame()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save our predictions to the df for submission\nimg = []\nlab = []\n#loop through our test batch predictions to get our submission ready\nfor idx, item in enumerate(inf_learn.dl.items):\n    preds = '';\n    for pred in inf_learn.dls.vocab[dec_preds[idx]]:\n        preds = preds + pred + ' '\n    print(f\"{item.name} : {preds}\")\n    lab.append(preds)\n    img.append(item.name)\nsub_df['image'] = img\nsub_df['labels'] = lab\nsub_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save our submission csv\nsub_df.to_csv(SUBMISSION, index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}