{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    i=0\n    #print(filenames)\n    #print(os.path.join(dirname, filenames))\n    #for filename in filenames:\n    #    if(i>10):\n    #        break\n    #    print(os.path.join(dirname, filename))\n    #    print(filename)\n    #    i+=1\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"bs = 250\n# bs = 16   # uncomment this line if you run out of memory even after clicking Kernel->Restart","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def readJSONFile(path):\n    import json\n    with open(path) as f:\n        data = json.load(f)\n    return data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from fastai.vision import *\nfrom fastai.metrics import error_rate","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dataPath=Path('/kaggle/input/iwildcam-2020-fgvc7')\n#dataPath=Path('c:/Users/manoj/PycharmProjects/data/iwildcam-2020/')\ndataPath.ls()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"jsonFilePath=dataPath/'iwildcam2020_train_annotations.json'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = readJSONFile(jsonFilePath)\n\nannotations = data[\"annotations\"]\nimages=data[\"images\"]\ncategories = data[\"categories\"]\ninfo = data[\"info\"]\n\n# Convert to Data frame\n\nannotations = pd.DataFrame.from_dict(annotations)\nimages = pd.DataFrame.from_dict(images)\ncategories = pd.DataFrame.from_dict(categories)\n\n\n#Remove data from memory\ndel data\n\n#Create column image_id to use for merging the two data frames\nimages[\"image_id\"]  = images[\"id\"]\n\n# Merge annotations and images on image_id\n\ntrainDf1 = (pd.merge(annotations, images, on='image_id'))\n#Remove Unnecessary fields\ntrainDf1.drop([\"id_y\",\"id_x\"], axis = 1, inplace=True)\n\n#print(trainDf1.columns)\n\ntrainDf1 = pd.merge(trainDf1, categories.rename(columns={\"id\":\"category_id\"}), on=\"category_id\" )\n# Unset annotations and images dataframe as they are no longer needed\ndel annotations\ndel images\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"categories[categories[\"id\"]==115 ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"categories","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trainDf1[[\"name\",\"category_id\"]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df=trainDf1[[\"file_name\",\"category_id\"]]\ndf=df.rename(columns={\"file_name\":\"name\",\"category_id\":\"label\"})\n\n#df=trainDf1[[\"file_name\",\"name\"]]\n#df=df.rename(columns={\"file_name\":\"name\",\"name\":\"label\"})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# To see if creating Duplicates improves the model\nI dont think this will improve the model by a lot. But I'm placing this here anyway"},{"metadata":{"trusted":true},"cell_type":"code","source":"minSamples=1000\nduplicateDf=pd.DataFrame()\n\nfor label in df[\"label\"].unique():\n    length=0\n    \n    x=min\n    y=len(df[df[\"label\"]==label])\n    multiplier=1\n    if(y<minSamples):\n        multiplier=int(minSamples/y)\n        y=y*multiplier\n    #length=y\n    #print(\"{} {} {}\".format(y, multiplier, length))\n    duplicateDf=duplicateDf.append([df[df[\"label\"]==label]]*multiplier,ignore_index=True)\n\ndf=duplicateDf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tfms = get_transforms(do_flip=False)\n\nfilePath=str(dataPath/\"train\")\n\nimport os\nprint(os.getcwd())\nprint(filePath)\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Just to make sure that the Image data bunch selected is proper\n'''\ni=0\nwhile 1:\n    try:\n        np.random.seed(i)\n        data=ImageDataBunch.from_df(filePath, df, ds_tfms=tfms, size=224, bs=100)\n    except:\n        i+=1\n        if(i%100==0):\n            print(str(i) + \" did not work\")\n        continue\n    else: \n        print('Seed '+str(i)+' works')\n        break\n    break\n'''    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nnp.random.seed(25)\n#data = ImageDataBunch.from_df(\"/home/manoj/Documents/data/data/iwildcam-2020/train/28X28\",df, \ndata = ImageDataBunch.from_df(filePath\n                              , pd.DataFrame(df)\n                              , ds_tfms=tfms\n                              , size=200\n                              , valid_pct=.2\n                              , bs=bs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"categories[categories[\"id\"].isin([257, 229, 420, 306, 296, 402, 408, 420, 412])]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data.show_batch(rows=3, figsize=(15,15))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(data.classes)\nlen(data.classes),data.c","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn = cnn_learner(data, models.resnet34, metrics=error_rate)\n#learn = cnn_learner(data, models.resnet50, metrics=error_rate)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.fit_one_cycle(1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.recorder.plot_losses()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"interp = ClassificationInterpretation.from_learner(learn)\n\nlosses,idxs = interp.top_losses()\n\nlen(data.valid_ds)==len(losses)==len(idxs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"interp.plot_top_losses(9, figsize=(15,11))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"interp.plot_confusion_matrix(figsize=(12,12), dpi=60)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.unfreeze()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"learn.fit_one_cycle(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#jsonTestFilePath='/home/manoj/Documents/data/data/iwildcam-2020/iwildcam2020_test_information.json'\njsonTestFilePath=dataPath/'iwildcam2020_test_information.json'\ntestData = readJSONFile(jsonTestFilePath)\n\ntestImages=testData[\"images\"]\ntestCategories = testData[\"categories\"]\ntestInfo = testData[\"info\"]\n\n# Convert to Data frame\n\ntestImages = pd.DataFrame.from_dict(testImages)\ntestCategories = pd.DataFrame.from_dict(testCategories)\n\n#Remove data from memory\ndel testData, testInfo\n\n# Remove Unnecessary fields from images\ntestDf1 = pd.DataFrame(testImages.file_name)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"testImages","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#testPath=Path(\"/home/manoj/Documents/data/data/iwildcam-2020/test/100X100\")\ntestPath=dataPath/'test'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#df=[ {\"file_name\":str(file).replace(str(testPath)+'/',''), \"name\": learn.predict(open_image(file))[0] }\ndf=[ {\"file_name\":str(file).replace(str(testPath)+'/',''), \"Id\": learn.predict(open_image(file))[0] }\n    for file in testPath.ls()[:]\n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df=pd.DataFrame(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df[\"file_name\"]=list(map(lambda x: os.path.basename(x), df[\"file_name\"]))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"jsonSubmissionFilePath=dataPath/'sample_submission.csv'\nsubmission=pd.read_csv(jsonSubmissionFilePath)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.drop(columns=[\"Category\"], inplace=True)\nsubmission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"testXref=testImages[[\"file_name\",\"id\"]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(testXref)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.merge(testXref, on='file_name')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#df1=df.merge(testImages, on='file_name')[[\"id\",\"name\"]]\ndf1=df.merge(testXref, on='file_name')[[\"id\",\"Id\"]]\n#df1=df1.rename(columns={\"id\":\"Id\"})\ndf1=df1.rename(columns={\"Id\":\"Category\", \"id\":\"Id\"})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#df2=submission.merge(df1, on=\"Id\")[[\"Id\",\"Category\"]]\ndf2=submission.merge(df1, on=\"Id\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df2.to_csv(\"submission.2020040217.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}