{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from fastai.vision import *\nfrom fastai.metrics import error_rate","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from fastai.vision import *\nfrom fastai.callbacks import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Speical Thanks for getting ideas from this Kernel\nhttps://www.kaggle.com/tanlikesmath/intro-aptos-diabetic-retinopathy-eda-starter"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"PATH = Path('../input/aptos2019-blindness-detection')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"[x for x in PATH.iterdir() if x.is_dir()]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":""},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(PATH/'train.csv')\ntest = pd.read_csv(PATH/'test.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"_ = train.hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"SEED = 20192\ndef ret_percentage(column):\n    return round(column.value_counts(normalize=True) * 100,2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Holdout test set from training samples\nfrom sklearn.model_selection import StratifiedShuffleSplit\n\nsplit = StratifiedShuffleSplit(n_splits=1, test_size=0.15, random_state= SEED)\nfor train_index, test_index in split.split(train[\"id_code\"], train[\"diagnosis\"]):\n    df_train = train.iloc[train_index]\n    df_test = train.iloc[test_index]\n\n#print(\"Old Train Class Percentage Dist:\\n\", (df_train[\"diagnosis\"]))\n\n   \nprint(\"New Train Sample Size\", df_train.shape)\nprint(\"New Train Class Percentage Dist:\\n\", ret_percentage(df_train[\"diagnosis\"]))\n\nprint(\"New Test Sample Size\",df_test.shape)\nprint(\"New Test Class Percentage Dist:\\n\", ret_percentage(df_test[\"diagnosis\"]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Initialize batch processing size\nbs = 16  #64\n# bs = 16   # uncomment this line if you run out of memory even after clicking Kernel->Restart\nsz = 256 #224 #Image size\nn_folds = 2\nmodel_name = \"resnet50\"\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom tqdm import tqdm\n\nskf = StratifiedKFold(n_splits=n_folds, shuffle=True, random_state = SEED)\ntfms = get_transforms(do_flip=True,flip_vert=True,max_rotate=360,max_warp=0,max_zoom=1.1,max_lighting=0.1,p_lighting=0.5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tfms = get_transforms(do_flip=True,flip_vert=True,max_rotate=360,max_warp=0,max_zoom=1.1,max_lighting=0.1,p_lighting=0.5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score\ndef quadratic_kappa(y_hat, y):\n    return torch.tensor(cohen_kappa_score(torch.round(y_hat), y, weights='quadratic'),device='cuda:0')\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nPath('/tmp/.cache/torch/checkpoints/').mkdir(exist_ok=True, parents=True)\n!cp '../input/resetnet50-weight/resnet50-19c8e357.pth' '/tmp/.cache/torch/checkpoints/resnet50-19c8e357.pth'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nprint(os.listdir(\"../input\"))\n#del learn\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"kp_score = []\nerr_rate = []\nlosses = []\n#loss_func = FocalLoss(gamma=1.)\nlearn = \"\"\ndata_fold = \"\"\npredictions = torch.from_numpy(np.zeros((len(df_test))))\n\nfor fold, (train_index, val_index) in tqdm(enumerate(skf.split(df_train[\"id_code\"], df_train[\"diagnosis\"]))):\n    del learn, data_fold\n    gc.collect()\n    filename = '/tmp/' + model_name + \"fold_\" + str(fold)+\".pkl\"\n    print(\"Fold:\", filename)\n    print(\"TRAIN:\", train_index, \"VALIDATE:\", val_index)\n    data_fold = (ImageList.from_df(df_train, PATH, folder='train_images', cols=\"id_code\",suffix='.png')\n        .split_by_idxs(train_index, val_index)\n        .label_from_df(cols='diagnosis')\n        .transform(tfms,size=sz,resize_method=ResizeMethod.SQUISH,padding_mode='zeros') #Data transform\n        .databunch(bs=bs)).normalize(imagenet_stats)\n   # learn = cnn_learner(data_fold, base_arch=models.resnet50, loss_func = mse, metrics=mse)\n    learn = cnn_learner(data_fold, base_arch=models.resnet50, metrics=[accuracy, KappaScore(weights=\"quadratic\")],callback_fns=[BnFreeze,partial(SaveModelCallback, monitor='kappa_score')])\n    learn.model_dir = '/tmp/'\n    lr = 0.02\n    learn.fit_one_cycle(2, slice(lr))\n    learn.save('stage-1-rn50')\n    learn.unfreeze()\n    learn.fit_one_cycle(2, slice(1e-4, lr/5))\n    learn.save('stage-2-rn50')\n    learn.freeze()\n    lr=1e-2/2\n    learn.save('stage-1-256-rn50')\n    learn.unfreeze()\n    learn.fit_one_cycle(2, slice(1e-4, lr/5))\n    learn.save('stage-2-256-rn50')\n    learn.export(filename)\n    loss, err , kp = learn.validate()\n    kp_score.append(kp.numpy())\n    err_rate.append(err.numpy())\n    losses.append(loss)\n    learn.data.add_test(ImageList.from_df(df_test ,PATH ,folder='train_images',suffix='.png'))\n    preds, _ = learn.TTA(ds_type=DatasetType.Test)\n    predictions = predictions + preds.argmax(dim=-1).double()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = torch.round(predictions/n_folds)\ndf_test['diagnosis_pred'] = pd.Series(predictions.numpy().astype(int), index=df_test.index)\ndf_test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from scipy.spatial.distance import cosine\n\nprint(\"New Test Set Correlation:\", df_test['diagnosis'].corr(df_test['diagnosis_pred']))\nprint(\"New Test Set Cosine Similarity:\", 1 - cosine(df_test[\"diagnosis\"], df_test[\"diagnosis_pred\"]))\ndf_test.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.7"}},"nbformat":4,"nbformat_minor":1}