{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install comet_ml","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import comet_ml #at the top of your file\n# from comet_ml import Experiment\n\n# # Create an experiment with your api key:\n# experiment = Experiment(\n#     api_key=\"cjZUHKCBKcrudJIeYuUe1zaBT\",\n#     project_name=\"leaf-disease-classification\",\n#     workspace=\"kaggle\",\n#     log_code=True,\n# )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport os\nimport pathlib as pt\n\nimport matplotlib.pyplot as plt\nimport numpy as np \nimport pandas as pd \n\nfrom fastai.vision.all import *\nfrom fastai.data.core import DataLoaders\n\nfrom tqdm import tqdm\n\nimport torch.cuda\nif torch.cuda.is_available():\n    print('PyTorch found cuda')\nelse:\n    print('PyTorch could not find cuda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare data","metadata":{}},{"cell_type":"code","source":"ROOT = pt.Path('../input/cassava-leaf-disease-classification')\nLABEL_JSON = ROOT/\"label_num_to_disease_map.json\"\nTRAIN_CSV  = ROOT/\"train.csv\"\nTRAIN_DIR  = ROOT/\"train_images\"\nTEST_DIR   = ROOT/\"test_images\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_imgs = list(TRAIN_DIR.glob(\"*\"))\ntest_imgs = list(TEST_DIR.glob(\"*\"))\nprint(\"Train: # {}\".format(len(train_imgs)))\nprint(\"Test: # {}\".format(len(test_imgs)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(TRAIN_CSV)\ntrain_df.head(3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(LABEL_JSON) as fp:\n    label_dict = json.load(fp)\nlabel_dict","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_path(row):\n    return TRAIN_DIR/row\ndef get_label_name(row):\n    return label_dict[str(row)]\n\ntrain_df['img_path'] = train_df['image_id'].apply(create_path)\ntrain_df['disease_name'] = train_df['label'].apply(get_label_name)\ntrain_df.head(3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_df = train_df['disease_name'].value_counts().sort_index()\nfig = plt.figure()\nax = _df.plot(kind='bar')\nax.set_xlabel(\"Disease\")\nax.set_ylabel(\"Frequency\")\nax.set_title(\"Nr of samples / disease\")\n# experiment.log_figure(figure_name=\"Leaf Diseases\", figure=fig)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aug_tfms = aug_transforms(mult=1.5, \n                          do_flip=True, \n                          flip_vert=False, \n                          max_rotate=20.0, \n                          min_zoom=1.0, \n                          max_zoom=1.5, \n                          max_lighting=0.3, \n                          max_warp=0.2, \n                          p_affine=0.75, \n                          p_lighting=0.65, \n                          xtra_tfms=None, \n                          size=224, \n                          mode='bilinear', \n                          pad_mode='reflection', \n                          align_corners=True, \n                          batch=False, \n                          min_scale=1.0)\n\ndata_loaders = ImageDataLoaders.from_df(train_df, \n                                        path=\"\", \n                                        seed=42, \n                                        fn_col='img_path', \n                                        label_col='label', \n                                        valid_pct=0.2,\n                                        item_tfms=Resize(460), #RandomResizedCrop(460, min_scale=0.3),\n                                        batch_tfms=aug_tfms)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_loaders.show_batch(max_n=8, nrows=2, unique=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Setup the resnet architecture","metadata":{}},{"cell_type":"code","source":"# creating directories and copying the models to those directories\n!mkdir -p /root/.cache/torch/hub/checkpoints/\n!cp ../input/resnet34/resnet34.pth /root/.cache/torch/hub/checkpoints/resnet34-333f7ec4.pth\n!cp ../input/resnet50/resnet50.pth /root/.cache/torch/hub/checkpoints/resnet50-19c8e357.pth\n!cp ../input/resnet152/resnet152.pth /root/.cache/torch/hub/checkpoints/resnet152-b121ed2d.pth","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = cnn_learner(data_loaders, resnet152, metrics=[error_rate, accuracy], opt_func=Adam)\nlearn.lr_find()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# experiment.log_parameters(hyperparams)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`cnn_learner` call `.freeze()` by default. This means it freezes all the layers except the last which are added for the new classification task. When `.fit_one_cycle()` it's called, only these last layers are trained.","metadata":{}},{"cell_type":"code","source":"learn.fine_tune??","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_lr = 0.003\nlearn.fit_one_cycle(n_epoch=1, lr_max=slice(base_lr/100, base_lr))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.unfreeze()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.lr_find()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# learn.fit_one_cycle??","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(1)\nlearn.fit_one_cycle(n_epoch=20, lr_max=slice(1e-6, 1e-4))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# experiment.log_model(name=\"resnet34_model_v0\", file_or_folder=\"./resnet34_model.pkl\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Look at some predictions","metadata":{}},{"cell_type":"code","source":"learn.show_results()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp = ClassificationInterpretation.from_learner(learn)\ninterp.plot_top_losses(9, figsize=(15,10))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp.plot_confusion_matrix()\nplt.savefig(\"confusion_matrix.png\", bbox_inches='tight', padding=0)\n# experiment.log_image(\"confusion_matrix.png\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp.most_confused() # (actual, predicted, nr of occurences)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp.print_classification_report()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create submission file","metadata":{}},{"cell_type":"code","source":"def get_image_id(row):\n    p = pt.Path(row)\n    return p.name\ntest_df = pd.DataFrame()\ntest_df['img_path'] = test_imgs\ntest_df['image_id'] = test_df['img_path'].apply(get_image_id)\ntest_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dl = data_loaders.test_dl(test_df)\nres_preds = learn.get_preds(dl=test_dl, with_decoded=True) # returns (predictions, _, predicted label)\npreds_values = res_preds[0]\npreds_labels = res_preds[2]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Generating submission file...\")\nsubmission_data = {'image_id': [], 'label': []}\n\nfor idx, pred_label in enumerate(preds_labels):\n\n    submission_data['image_id'].append(test_df.iloc[idx]['image_id'])\n    submission_data['label'].append(pred_label.item())\n\n\nsubmission_df = pd.DataFrame(data=submission_data)\nsubmission_df.to_csv(\"submission.csv\", index=False)\n# experiment.log_table(\"submission.csv\")\n!head submission.csv","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# experiment.end()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}