{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"    from fastai.vision import *\n    from fastai.metrics import *\n    PATH = Path(\"../input\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ann_file = \"../input/train2019.json\"\n\nwith open(ann_file) as data_file:\n    train_anns = json.load(data_file)\n    \ntrain_anns_df = pd.DataFrame(train_anns[\"annotations\"])[[\"id\", \"category_id\"]]\ntrain_img_df = pd.DataFrame(train_anns[\"images\"])[[\"id\", \"file_name\"]]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.merge(train_img_df, train_anns_df, on = \"id\")\ndf_train.drop([\"id\"], axis = 1, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_to = df_train.category_id.value_counts().max()\nres = None\n\nfor grp in df_train.groupby(\"category_id\"):\n    n = grp[1].shape[0]\n    additional_rows = grp[1].sample(0 if sample_to < n else sample_to - n, replace=True)\n    rows = pd.concat((grp[1], additional_rows))\n    res = pd.concat((res, rows))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_ann_file = PATH/\"test2019.json\"\n\nwith open(test_ann_file) as data_file:\n    test_anns = json.load(data_file)\n\ntest_img_df = pd.DataFrame(test_anns[\"images\"])[[\"file_name\", \"id\"]]\n#test_img_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train_sub = df_train[:10000]\n#print(df_train_sub)\n#print(df_train_sub.shape)\nres_sub = res[:10000]\n#res_sub.head()\ntest_img_df_sub = test_img_df[:1000]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"src = (ImageList.from_df(df=res, path=PATH/\"train_val2019\", cols = 0)\n    .use_partial_data(0.2)\n    .split_by_rand_pct(0.1)\n    .label_from_df(\"category_id\")\n    .add_test(ImageList.from_df(df=test_img_df, path=PATH/\"test2019\", cols = 0))\n    )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = (src\n       .transform(get_transforms(), size = 128)\n       .databunch(bs=32)\n       .normalize(imagenet_stats))\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.classes\ndata.show_batch(rows=3, figsize=(7,6))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#learn = cnn_learner(data, models.resnet34, metrics = accuracy, model_dir=\"/tmp/model/\")\n#learn.save(\"StaticWeights_resnet34_v1\")\n#learn.lr_find()\n#learn.recorder.plot()\n#learn.unfreeze()\n#learn.fit_one_cycle(2, max_lr=slice(1e-6,1e-1))\n#learn.save(\"FittedWeights_resnet34_v1\")\n#interp = ClassificationInterpretation.from_learner(learn)\n#losses,idxs = interp.top_losses()\n#interp.plot_top_losses(9, figsize=(15,11))\n#interp.plot_confusion_matrix(figsize=(12,12), dpi=60)\n#interp.most_confused(min_val=2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import re\nfor grp in res_sub.groupby(\"category_id\"):\n    n = grp[1].iloc[0,0]\n    n = re.search(\"(?<=/)(.*)(?=/.*/)\",n).group(0)\n    cat = grp[1].iloc[0,1]\n    print(\"Type {} has label {}\".format(n, cat))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn = cnn_learner(data, models.resnet50, metrics = accuracy, model_dir=\"/tmp/model/\")\n#learn.fit_one_cycle(2, max_lr=slice(1e-6,1e-1))\n#learn.save(\"StaticWeights_resnet50_v1\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.lr_find()\nlearn.recorder.plot()\nlearn.unfreeze()\nlearn.fit_one_cycle(10, max_lr=slice(1e-6,1e-1))\nlearn.save(\"FittedWeights_resnet50_v1\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"interp = ClassificationInterpretation.from_learner(learn)\nlosses,idxs = interp.top_losses()\ninterp.plot_top_losses(9, figsize=(15,11))\ninterp.plot_confusion_matrix(figsize=(12,12), dpi=60)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds, y = learn.get_preds(DatasetType.Test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#pred_t, _ = learn.TTA(ds_type=DatasetType.Test)\n#pred_t_max = np.argmax(pred_t, 1); pred_t_max[0:5]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#result = torch.topk(pred_t, 5)\nresults = torch.topk(preds, 5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = []\nfor i in results[1].numpy():\n    temp = \"\"\n    for j in i:\n        temp += (\" \"+str(data.classes[j]))\n    predictions.append(temp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df = pd.read_csv(PATH/\"kaggle_sample_submission.csv\")\n#submission_df_sub = submission_df[:1000]\n#submission_df_sub[\"predicted\"] = predictions\nsubmission_df[\"predicted\"] = predictions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#submission_df_sub.to_csv(\"submission_sub.csv\", index = False)\nsubmission_df.to_csv(\"submission.csv\", index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}