{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import fastai\nfastai.__version__","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport torch\nfrom fastai import *\nfrom fastai.vision.all import *\nfrom fastai.callback.all import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def seed_everything(seed=0):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n\nseed_everything()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path = Path('/kaggle/input/cassava-leaf-disease-classification')\ntrain = path /\"train.csv\"\ntrain_df = pd.read_csv(train)\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.label.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import seaborn as sns\nsns.countplot(train_df.label)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"item_tfms = Resize(256)\nbatch_tfms = [RandomResizedCrop(224), *aug_transforms(mult=1.0, do_flip=True, max_rotate=30.0, max_zoom=1.5,\n                            max_lighting=.8, max_warp=0.3, p_lighting=.9)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dls = ImageDataLoaders.from_df(train_df, path/\"train_images\", \n                              item_tfms=item_tfms,\n                              batch_tfms = batch_tfms,\n                              bs=64, num_workers=4, \n                              label_col=\"label\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dls.show_batch(max_n=9,figsize=(12,8))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# category names, number of categories\nprint(dls.vocab); print(dls.c)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn = cnn_learner(dls, resnet34, metrics=[FBeta(beta=1, average='macro'),accuracy], cbs=MixUp, model_dir=\"/tmp/model/\").to_fp16()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.fit_one_cycle(15, lr_max=1e-2, cbs=EarlyStoppingCallback(patience=3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.fine_tune(5, cbs=[EarlyStoppingCallback(patience=3)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.lr_find()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.save('model_stage_1')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.unfreeze()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.fit_one_cycle(15, lr_max=slice(1e-7, 1e-3), cbs=EarlyStoppingCallback(patience=2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df = pd.read_csv(path/'sample_submission.csv')\nsub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds,targs = learn.tta()\nprint(accuracy(preds, targs).item())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**CrossValidation**"},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_dls(bs, size, val_idx):\n    item_tfms = Resize(256)\n    batch_tfms = [RandomResizedCrop(size), *aug_transforms(mult=1.0, do_flip=True, max_rotate=30.0, max_zoom=1.5,\n                            max_lighting=.8, max_warp=0.3, p_lighting=.9)]\n    dls = ImageDataLoaders.from_df(train_df, path/\"train_images\", \n                              splitter=IndexSplitter(val_idx),\n                              item_tfms=item_tfms,\n                              batch_tfms = batch_tfms,\n                              bs=bs, num_workers=4, \n                              label_col=\"label\")\n    return dls","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for train_idx, val_idx in skf.split(train_df['image_id'].values, train_df['label'].values):\n    print(train_idx, val_idx)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nval_pct = []\nskf = StratifiedKFold(n_splits=3, shuffle=True)\ni = 0\n\nfor _, val_idx in skf.split(train_df['image_id'].values, train_df['label'].values):\n  dls = get_dls(32, 128, val_idx)\n  learn = cnn_learner(dls, resnet34, metrics=[FBeta(beta=1, average='macro'),accuracy], cbs=MixUp, model_dir=\"/tmp/model/\").to_fp16()\n  learn.fine_tune(2, cbs=EarlyStoppingCallback(patience=2))\n  learn.dls = get_dls(32, 224, val_idx)\n  learn.fine_tune(5, 1e-3, cbs=EarlyStoppingCallback(patience=2))\n  preds,targs = learn.tta()\n  print(accuracy(preds, targs).item())\n  val_pct.append(accuracy(preds, targs).item())\n  i+=1","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Inference**"},{"metadata":{"trusted":true},"cell_type":"code","source":"get_image_files(path/\"test_images\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_items = get_image_files(path/\"test_images\")\ndl = learn.dls.test_dl(test_items, rm_type_tfms=1, bs=64)\ny_pred, _ = learn.get_preds(dl=dl)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_items = get_image_files(path/\"test_images\")\ndl2 = learn.dls.test_dl(test_items)\ntest_preds = learn.get_preds(dl=dl2)\npredictions = []\nfor pred in test_preds[0]:\n    predictions.append(pred.argmax().item())\n#\nimport seaborn as sns\nsns.countplot(predictions)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df = pd.read_csv(path/'sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df['label'] = y_pred.argmax(dim=-1).numpy()\nsub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_pred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"PATH = \"/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json\"\nPATH","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import json\n# Opening JSON file \nf = open(PATH,) \n  \n# returns JSON object as  \n# a dictionary \ndata = json.load(f) \n  \nprint(data)\nprint(type(data))\n  \n# Closing file \nf.close() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lb =y_pred.argmax(dim=-1).numpy().tolist()[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"The Label predicted for the image {sub_df['image_id'].values.tolist()[0]} is : {data[str(lb)]}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df.to_csv('/kaggle/working/submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}