{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import fastai\nfastai.__version__","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nfrom fastai.vision.all import *\nset_seed(123)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = Path(\"../input\")\ndata_path = path/'cassava-leaf-disease-classification'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_path","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(data_path/'train.csv')\ntrain_df['image_id'] = train_df['image_id'].apply(lambda x: f'train_images/{x}')\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_x(row): \n    \n    return data_path/row['image_id']\n\n\ndef get_y(row): \n    \n    return row['label']\n\n\n\nblock = DataBlock(blocks = (ImageBlock, CategoryBlock),\n                 get_x = get_x,\n                 get_y = get_y,\n                 splitter = RandomSplitter(valid_pct=0.2,seed=123),\n                 item_tfms = [Resize(224)], # starting with small images for efficient testing\n                 batch_tfms = [RandomResizedCropGPU(224), *aug_transforms(), Normalize.from_stats(*imagenet_stats)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dls = block.dataloaders(train_df, bs=64)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dls.show_batch(figsize=(12,12))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn = cnn_learner(dls, resnet50, loss_func=CrossEntropyLossFlat(),\n               metrics=[accuracy])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.fine_tune(3, freeze_epochs = 2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Format submission df in same way as training df\n#sample_df = pd.read_csv('./input/cassava-leaf-disease-classification/sample_submission.csv')\n#sample_copy = sample_df.copy()\n#sample_copy['image_id'] = sample_copy['image_id'].apply(lambda x: f'test_images/{x}')\n\n#test_dl = learn.dls.test_dl(sample_copy)\n#preds, _ = learn.tta(dl=test_dl) # test-time augmentation can improve accuracy somewhat\n\n#sample_df['label'] = preds.argmax(dim=-1).numpy()\n#sample_df.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.save('ac_weights')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn2 = cnn_learner(dls, resnet50, loss_func=CrossEntropyLossFlat(),\n               metrics=[accuracy])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"load_model(Path('./models/ac_weights.pth'), learn2.model, learn2.opt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_df = pd.read_csv(data_path/'sample_submission.csv')\nsample_df.head()\nsample_copy = sample_df.copy()\nsample_copy['image_id'] = sample_copy['image_id'].apply(lambda x: f'test_images/{x}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_dl = learn2.dls.test_dl(sample_copy)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds, _ = learn2.tta(dl=test_dl)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_df['label'] = preds.argmax(dim=-1).numpy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_df.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#import PIL.Image\n\n#fp = open(\"../input/cassava-leaf-disease-classification/train_images/1003442061.jpg\",\"rb\")\n#img = PIL.Image.open(fp)\n#print(img.size)\n\n#All images seem to be of the same size","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Used ImageDataLoaders"},{"metadata":{"trusted":true},"cell_type":"code","source":"#path = '/kaggle/input/cassava-leaf-disease-classification/train_images/'\n#dls = ImageDataLoaders.from_df(df, path,valid_pct=0.2,bs=64, item_tfms=[Resize(224)], \n#    batch_tfms=[*aug_transforms(size=224, max_warp=0.),RandomResizedCrop(224, min_scale=0.75),Normalize.from_stats(*imagenet_stats)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#dls.show_batch(max_n=15,nrows=3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#dls.valid.show_batch(max_n=6,nrows=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#learn = cnn_learner(dls,resnet34,metrics=accuracy,model_dir = '/tmp/models')\n#learn.to_native_fp16()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Learning rate finder - start with a very small learning rate, train one mini batch track the loss and increase \n#learning rate (doubling it each time). Train another mini batch, track the loss and repeat until learning rate gets\n#worse. Guidelines to choosing learning rate -\n#One order of magnitude less than where the minimum loss was achieved(min %10) or\n#last point where the loss was clearly decreasing\n\n#lr_min,lr_steep = learn.lr_find()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#print(f'Mininum/10: {lr_min:.2e}, steepest point: {lr_steep:.2e}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# default learning rate is 1e-3\n# Train final linear layers that initally have random weights for 3 epochs at a learning rate = 3e-3\n# fit_one_cycle starts training at a a low learning rate, gradually increase it for the first section\n# of the training and then gradually decrease it again for the last section of training\n#learn.fit_one_cycle(3,3e-2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Unfreeze the model to train all layers\n#learn.unfreeze()\n#learn.lr_find()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# curve above has no sharp descent as our model has already been trained, we choose a point before the sharp\n#increase in loss","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#learn.fit_one_cycle(3, lr_max=1e-4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#sample_df = pd.read_csv('../input/cassava-leaf-disease-classification/sample_submission.csv')\n#sample_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#sample_copy = sample_df.copy()#\n#sample_copy['image_id'] = sample_copy['image_id'].apply(lambda x: f'../input/cassava-leaf-disease-classification/test_images/{x}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#sample_copy","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#learn.dls.vocab","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Learner.tta(ds_idx=1, dl=None, n=4, item_tfms=None, batch_tfms=None, beta=0.25, use_max=False)\n\nReturn predictions on the ds_idx dataset or dl using Test Time Augmentation"},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"#accuracy(*learn.tta())#RuntimeError: \"softmax_lastdim_kernel_impl\" not implemented for 'Half'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#submission_df = pd.read_csv(path/'sample_submission.csv')\n#submission_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"#test_data_path = submission_df['image_id'].apply(lambda x: path/'test_images'/x)\n#tst_dl = learn.dls.test_dl(test_data_path)\n#predictions = learn.tta(dl = tst_dl)\n#submission_df['label'] = np.argmax(predictions[0],axis=1)\n#submission_df\n\n\n\n# encountered RuntimeError: \"softmax_lastdim_kernel_impl\" not implemented for 'Half'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"\n#preds = learn.get_preds(dl=dls.test_dl(test_images, shuffle=False, drop_last=False))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"#test_dl = learn.dls.test_dl(sample_copy)\n#test_dl = dls.test_dl(sample_copy)\n#test_dl.show_batch()\n\n#FileNotFoundError: [Errno 2] No such file or directory: '/kaggle/input/cassava-leaf-disease-classification/train_images/../input/cassava-leaf-disease-classification/test_images/2216849948.jpg'\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"#preds, _ = learn.tta(test_dl)#\n\n\n#TypeError: list indices must be integers or slices, not TfmdDL\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}