{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":34375,"databundleVersionId":3405142,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-07T16:18:31.679843Z","iopub.execute_input":"2024-07-07T16:18:31.680469Z","iopub.status.idle":"2024-07-07T16:19:03.158040Z","shell.execute_reply.started":"2024-07-07T16:18:31.680438Z","shell.execute_reply":"2024-07-07T16:19:03.157080Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sorghum-100 Cultivar ID**\n# Using fastai; with TTA, various architectures, and augmentations\n","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"If we call tta() then we'll get the average of predictions made for multiple different augmented versions of each image, along with the unaugmented original\n\nSee https://www.kaggle.com/code/jhoward/small-models-road-to-the-top-part-2/","metadata":{}},{"cell_type":"code","source":"!pip install timm","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:19:40.676768Z","iopub.execute_input":"2024-07-07T16:19:40.677757Z","iopub.status.idle":"2024-07-07T16:19:53.862612Z","shell.execute_reply.started":"2024-07-07T16:19:40.677724Z","shell.execute_reply":"2024-07-07T16:19:53.861635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import timm\nfrom fastai.vision.all import *","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:20:45.278723Z","iopub.execute_input":"2024-07-07T16:20:45.279097Z","iopub.status.idle":"2024-07-07T16:20:45.284377Z","shell.execute_reply.started":"2024-07-07T16:20:45.279067Z","shell.execute_reply":"2024-07-07T16:20:45.283474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"../input/sorghum-id-fgvc-9/\" #set path\npath","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:20:45.427497Z","iopub.execute_input":"2024-07-07T16:20:45.427785Z","iopub.status.idle":"2024-07-07T16:20:45.434390Z","shell.execute_reply.started":"2024-07-07T16:20:45.427760Z","shell.execute_reply":"2024-07-07T16:20:45.433484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = path + 'train_images/' #training images are here in train_dir\ntrain_dir","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:20:49.481172Z","iopub.execute_input":"2024-07-07T16:20:49.481516Z","iopub.status.idle":"2024-07-07T16:20:49.488004Z","shell.execute_reply.started":"2024-07-07T16:20:49.481489Z","shell.execute_reply":"2024-07-07T16:20:49.486918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir = path + 'test/' #testing images are here in test_dir\ntest_dir","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:20:55.438489Z","iopub.execute_input":"2024-07-07T16:20:55.439316Z","iopub.status.idle":"2024-07-07T16:20:55.445618Z","shell.execute_reply.started":"2024-07-07T16:20:55.439275Z","shell.execute_reply":"2024-07-07T16:20:55.444638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_all = pd.read_csv(path + 'train_cultivar_mapping.csv') #let's load the CSV to df_all, and look at it\ndf_all.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:21:02.024324Z","iopub.execute_input":"2024-07-07T16:21:02.024686Z","iopub.status.idle":"2024-07-07T16:21:02.081697Z","shell.execute_reply.started":"2024-07-07T16:21:02.024657Z","shell.execute_reply":"2024-07-07T16:21:02.080670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df_all)) #how many items?","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:21:10.010432Z","iopub.execute_input":"2024-07-07T16:21:10.010835Z","iopub.status.idle":"2024-07-07T16:21:10.016129Z","shell.execute_reply.started":"2024-07-07T16:21:10.010805Z","shell.execute_reply":"2024-07-07T16:21:10.014963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_cultivars = list(df_all[\"cultivar\"].unique()) #how many unique cultivars are there?","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:21:18.268545Z","iopub.execute_input":"2024-07-07T16:21:18.268938Z","iopub.status.idle":"2024-07-07T16:21:18.275614Z","shell.execute_reply.started":"2024-07-07T16:21:18.268907Z","shell.execute_reply":"2024-07-07T16:21:18.274014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(unique_cultivars))","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:21:23.083234Z","iopub.execute_input":"2024-07-07T16:21:23.084140Z","iopub.status.idle":"2024-07-07T16:21:23.089211Z","shell.execute_reply.started":"2024-07-07T16:21:23.084101Z","shell.execute_reply":"2024-07-07T16:21:23.088124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\na=pd.DataFrame({'cultivar':df_all['cultivar']})\nplt.figure(figsize=(30,10))\nsns.histplot(a,x='cultivar')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-04T12:08:19.094791Z","iopub.execute_input":"2024-07-04T12:08:19.095899Z","iopub.status.idle":"2024-07-04T12:08:20.170301Z","shell.execute_reply.started":"2024-07-04T12:08:19.095861Z","shell.execute_reply":"2024-07-04T12:08:20.169361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"------------------------------------------------------------------\n# Getting input data ready, and resizing","metadata":{}},{"cell_type":"code","source":"test_files = get_image_files(test_dir) #testing data to test_files","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:21:28.087032Z","iopub.execute_input":"2024-07-07T16:21:28.087847Z","iopub.status.idle":"2024-07-07T16:21:35.625632Z","shell.execute_reply.started":"2024-07-07T16:21:28.087808Z","shell.execute_reply":"2024-07-07T16:21:35.624689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files = get_image_files(train_dir) #training data to train_files","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:21:43.605193Z","iopub.execute_input":"2024-07-07T16:21:43.606155Z","iopub.status.idle":"2024-07-07T16:21:53.441627Z","shell.execute_reply.started":"2024-07-07T16:21:43.606106Z","shell.execute_reply":"2024-07-07T16:21:53.440643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train_files)) #how many training images/","metadata":{"execution":{"iopub.status.busy":"2024-07-07T12:58:00.097499Z","iopub.execute_input":"2024-07-07T12:58:00.098485Z","iopub.status.idle":"2024-07-07T12:58:00.103872Z","shell.execute_reply.started":"2024-07-07T12:58:00.098435Z","shell.execute_reply":"2024-07-07T12:58:00.102916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(test_files)) #how many testing images?","metadata":{"execution":{"iopub.status.busy":"2024-07-07T12:58:13.968901Z","iopub.execute_input":"2024-07-07T12:58:13.969732Z","iopub.status.idle":"2024-07-07T12:58:13.975123Z","shell.execute_reply.started":"2024-07-07T12:58:13.969689Z","shell.execute_reply":"2024-07-07T12:58:13.974040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = PILImage.create(train_files[0]) #let's look at an image from training data!\nimg","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:22:03.074375Z","iopub.execute_input":"2024-07-07T16:22:03.074742Z","iopub.status.idle":"2024-07-07T16:22:03.904923Z","shell.execute_reply.started":"2024-07-07T16:22:03.074713Z","shell.execute_reply":"2024-07-07T16:22:03.903853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.size #how big is it?","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:22:21.125874Z","iopub.execute_input":"2024-07-07T16:22:21.126257Z","iopub.status.idle":"2024-07-07T16:22:21.132563Z","shell.execute_reply.started":"2024-07-07T16:22:21.126227Z","shell.execute_reply":"2024-07-07T16:22:21.131508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from fastcore.parallel import * #get all sizes in parallel\n\ndef f(o): return PILImage.create(o).size #let's check all the sizes; doing it in parallel","metadata":{"execution":{"iopub.status.busy":"2024-07-02T16:42:54.839455Z","iopub.execute_input":"2024-07-02T16:42:54.839747Z","iopub.status.idle":"2024-07-02T16:42:54.850356Z","shell.execute_reply.started":"2024-07-02T16:42:54.839716Z","shell.execute_reply":"2024-07-02T16:42:54.849575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sizes = parallel(f, train_files, n_workers=8) #let's see if they are all the same size in the train_files!","metadata":{"execution":{"iopub.status.busy":"2024-07-02T16:42:54.851515Z","iopub.execute_input":"2024-07-02T16:42:54.852030Z","iopub.status.idle":"2024-07-02T16:50:32.252265Z","shell.execute_reply.started":"2024-07-02T16:42:54.852002Z","shell.execute_reply":"2024-07-02T16:50:32.251165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(sizes).value_counts() #check sizes","metadata":{"execution":{"iopub.status.busy":"2024-07-02T16:50:32.253712Z","iopub.execute_input":"2024-07-02T16:50:32.254086Z","iopub.status.idle":"2024-07-02T16:50:32.317835Z","shell.execute_reply.started":"2024-07-02T16:50:32.254052Z","shell.execute_reply":"2024-07-02T16:50:32.316955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**You can try resizing if wanted!**","metadata":{}},{"cell_type":"code","source":"resize_size = 512\n#let's see if we can resize the images, and send the small ones to the small_trn_dir\nresize_images(train_dir, dest=\"tmp/small_trn_dir\", max_size=resize_size, recurse=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-07T12:59:09.198836Z","iopub.execute_input":"2024-07-07T12:59:09.199711Z","iopub.status.idle":"2024-07-07T13:35:41.734655Z","shell.execute_reply.started":"2024-07-07T12:59:09.199664Z","shell.execute_reply":"2024-07-07T13:35:41.733566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_train_dir = \"tmp/small_trn_dir\" #let's refer to the path for the small_trn_dir and small_train_dir","metadata":{"execution":{"iopub.status.busy":"2024-07-07T13:43:54.717399Z","iopub.execute_input":"2024-07-07T13:43:54.717824Z","iopub.status.idle":"2024-07-07T13:43:54.725224Z","shell.execute_reply.started":"2024-07-07T13:43:54.717791Z","shell.execute_reply":"2024-07-07T13:43:54.724430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#well let's see one of the smaller files now\nsmall_train_files = get_image_files(small_train_dir) \nimg = PILImage.create(small_train_files[0]) \nimg","metadata":{"execution":{"iopub.status.busy":"2024-07-07T13:44:07.385432Z","iopub.execute_input":"2024-07-07T13:44:07.385808Z","iopub.status.idle":"2024-07-07T13:44:07.801877Z","shell.execute_reply.started":"2024-07-07T13:44:07.385777Z","shell.execute_reply":"2024-07-07T13:44:07.800954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img.size #is the size correct?","metadata":{"execution":{"iopub.status.busy":"2024-07-07T13:44:23.372183Z","iopub.execute_input":"2024-07-07T13:44:23.373144Z","iopub.status.idle":"2024-07-07T13:44:23.379581Z","shell.execute_reply.started":"2024-07-07T13:44:23.373109Z","shell.execute_reply":"2024-07-07T13:44:23.378419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Important! Removing this .ds_store string which is in one of the entries!!**","metadata":{}},{"cell_type":"code","source":"sorghum_df = pd.read_csv(path +'train_cultivar_mapping.csv')\n#!!!IMPORTANT!! removing this .ds_store string which is in one of the entries\nsorghum_df = sorghum_df[~sorghum_df['image'].str.contains('.DS_Store')]","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:22:34.287681Z","iopub.execute_input":"2024-07-07T16:22:34.288062Z","iopub.status.idle":"2024-07-07T16:22:34.341330Z","shell.execute_reply.started":"2024-07-07T16:22:34.288026Z","shell.execute_reply":"2024-07-07T16:22:34.340381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorghum_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:22:40.171048Z","iopub.execute_input":"2024-07-07T16:22:40.171400Z","iopub.status.idle":"2024-07-07T16:22:40.181802Z","shell.execute_reply.started":"2024-07-07T16:22:40.171374Z","shell.execute_reply":"2024-07-07T16:22:40.180645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Models","metadata":{}},{"cell_type":"markdown","source":"Do we use fine_tune or fit_one_cycle in fastai? https://forums.fast.ai/t/fine-tune-vs-fit-one-cycle/66029/9","metadata":{}},{"cell_type":"code","source":"#let's make a function for training now; using ImageDataLoaders from df #CUT CUT CUT\ndef train(arch, item, batch, epochs=5):\n    dls = ImageDataLoaders.from_df(sorghum_df, train_dir, seed=42, valid_pct=0.2, #so it's either train_dir or small_train_dir\n                                   item_tfms=item, batch_tfms=batch)\n    learn = vision_learner(dls, arch, metrics=error_rate).to_fp16()\n    learn.fine_tune(epochs, 0.01)\n    return learn\n\n#remember the difference for the normal training files i.e. train_dir, and the small ones small_train_dir","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:23:06.680998Z","iopub.execute_input":"2024-07-07T16:23:06.681944Z","iopub.status.idle":"2024-07-07T16:23:06.688675Z","shell.execute_reply.started":"2024-07-07T16:23:06.681906Z","shell.execute_reply":"2024-07-07T16:23:06.687476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arch = 'resnet26d'\nepochs = 2\nitem=Resize(460, method='squish')\nbatch=aug_transforms(size=224, min_scale=0.75)\nepochs=2\ndls = ImageDataLoaders.from_df(sorghum_df, train_dir, seed=42, valid_pct=0.2, #so it's either train_dir or small_train_dir\n                                   item_tfms=item, batch_tfms=batch)\nlearn = vision_learner(dls, arch, metrics=error_rate,  model_dir=\"/tmp/model/\").to_fp16() #model in temp directory","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fine_tune(epochs, 0.01)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:23:35.496606Z","iopub.execute_input":"2024-07-07T16:23:35.497007Z","iopub.status.idle":"2024-07-07T16:49:30.709574Z","shell.execute_reply.started":"2024-07-07T16:23:35.496969Z","shell.execute_reply":"2024-07-07T16:49:30.707559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Finding a learning rate","metadata":{}},{"cell_type":"code","source":"learn.lr_find(suggest_funcs=(minimum, steep, valley, slide))","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:50:58.171739Z","iopub.execute_input":"2024-07-07T16:50:58.172123Z","iopub.status.idle":"2024-07-07T16:50:58.334048Z","shell.execute_reply.started":"2024-07-07T16:50:58.172087Z","shell.execute_reply":"2024-07-07T16:50:58.332564Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_appropriate_lr(model:Learner, lr_diff:int = 15, loss_threshold:float = .05, adjust_value:float = 1, plot:bool = False) -> float:\n    \"\"\"Method that automates the selection of a Learning Rate in Fast.ai\n\n    Parameters:\n    model (Learner): The learner\n    lr_diff (int): The interval distance by units of the “index of LR” (log transform of LRs) between the right and left bound\n    loss_threshold (float): The maximum difference between the left and right bound’s loss values to stop the shift\n    adjust_value (float): A coefficient to the final learning rate for pure manual adjustment\n    plot (bool): A boolean to show two plots\n\n    Returns:\n    float: Best learning rate value to use\n\n   \"\"\"\n    model.lr_find()\n    \n    losses = np.array(model.recorder.losses)\n    assert(lr_diff < len(losses))\n    loss_grad = np.gradient(losses)\n    lrs = model.recorder.lrs\n    \n    r_idx = -1\n    l_idx = r_idx - lr_diff\n    while (l_idx >= -len(losses)) and (abs(loss_grad[r_idx] - loss_grad[l_idx]) > loss_threshold):\n        local_min_lr = lrs[l_idx]\n        r_idx -= 1\n        l_idx -= 1\n\n    lr_to_use = local_min_lr * adjust_value\n    \n    if plot:\n        plt.plot(loss_grad)\n        plt.plot(len(losses)+l_idx, loss_grad[l_idx],markersize=10,marker='o',color='red')\n        plt.ylabel(\"Loss\")\n        plt.xlabel(\"Index of LRs\")\n        plt.show()\n\n        plt.plot(np.log10(lrs), losses)\n        plt.ylabel(\"Loss\")\n        plt.xlabel(\"Log 10 Transform of Learning Rate\")\n        loss_coord = np.interp(np.log10(lr_to_use), np.log10(lrs), losses)\n        plt.plot(np.log10(lr_to_use), loss_coord, markersize=10,marker='o',color='red')\n        plt.show()\n        \n    return lr_to_use","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:49:41.258115Z","iopub.execute_input":"2024-07-07T16:49:41.258472Z","iopub.status.idle":"2024-07-07T16:49:41.271700Z","shell.execute_reply.started":"2024-07-07T16:49:41.258441Z","shell.execute_reply":"2024-07-07T16:49:41.270629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_to_use = find_appropriate_lr(learn)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:49:47.881011Z","iopub.execute_input":"2024-07-07T16:49:47.881373Z","iopub.status.idle":"2024-07-07T16:49:49.082252Z","shell.execute_reply.started":"2024-07-07T16:49:47.881347Z","shell.execute_reply":"2024-07-07T16:49:49.080498Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_to_use","metadata":{"execution":{"iopub.status.busy":"2024-07-07T14:00:07.828172Z","iopub.execute_input":"2024-07-07T14:00:07.828574Z","iopub.status.idle":"2024-07-07T14:00:07.835203Z","shell.execute_reply.started":"2024-07-07T14:00:07.828540Z","shell.execute_reply":"2024-07-07T14:00:07.834120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plain kinda augments, now with the new LR\nitem=Resize(460, method='squish')\nbatch=aug_transforms(size=224, min_scale=0.75)\nepochs=2\ndls = ImageDataLoaders.from_df(sorghum_df, train_dir, seed=42, valid_pct=0.2, #so it's either train_dir or small_train_dir\n                                   item_tfms=item, batch_tfms=batch)\nlearn = vision_learner(dls, arch, metrics=error_rate).to_fp16()","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:52:06.796986Z","iopub.execute_input":"2024-07-07T16:52:06.797697Z","iopub.status.idle":"2024-07-07T16:52:08.964275Z","shell.execute_reply.started":"2024-07-07T16:52:06.797665Z","shell.execute_reply":"2024-07-07T16:52:08.963216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fine_tune(epochs, 0.0229)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T16:52:19.719308Z","iopub.execute_input":"2024-07-07T16:52:19.720121Z","iopub.status.idle":"2024-07-07T17:19:46.489182Z","shell.execute_reply.started":"2024-07-07T16:52:19.720089Z","shell.execute_reply":"2024-07-07T17:19:46.488105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = learn.dls.valid\npreds,targs = learn.get_preds(dl=valid)\nerror_rate(preds, targs) #normal error rate","metadata":{"execution":{"iopub.status.busy":"2024-07-07T17:19:59.357346Z","iopub.execute_input":"2024-07-07T17:19:59.358292Z","iopub.status.idle":"2024-07-07T17:21:36.935622Z","shell.execute_reply.started":"2024-07-07T17:19:59.358241Z","shell.execute_reply":"2024-07-07T17:21:36.934509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tta_preds,_ = learn.tta(dl=valid)\nerror_rate(tta_preds, targs) #TTA error rate","metadata":{"execution":{"iopub.status.busy":"2024-07-07T17:21:52.581926Z","iopub.execute_input":"2024-07-07T17:21:52.583130Z","iopub.status.idle":"2024-07-07T17:30:46.034743Z","shell.execute_reply.started":"2024-07-07T17:21:52.583074Z","shell.execute_reply":"2024-07-07T17:30:46.033685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.train.show_batch(max_n=4, nrows=1, unique=True) #show with augments","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Augmentation with RandomResizedCrop\nitem=[Resize(460, method='squish'), RandomResizedCrop(224, min_scale=0.75, ratio=(1.,1.))]\nbatch=aug_transforms(size=224, min_scale=0.75)\nepochs=2\n\ndls = ImageDataLoaders.from_df(sorghum_df, train_dir, seed=42, valid_pct=0.2, #so it's either train_dir or small_train_dir\n                                   item_tfms=item, batch_tfms=batch)\nlearn = vision_learner(dls, arch, metrics=error_rate).to_fp16()","metadata":{"execution":{"iopub.status.busy":"2024-07-07T17:30:56.897081Z","iopub.execute_input":"2024-07-07T17:30:56.898054Z","iopub.status.idle":"2024-07-07T17:30:59.122364Z","shell.execute_reply.started":"2024-07-07T17:30:56.898014Z","shell.execute_reply":"2024-07-07T17:30:59.121419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fine_tune(epochs, 0.0229)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T17:31:06.778404Z","iopub.execute_input":"2024-07-07T17:31:06.778765Z","iopub.status.idle":"2024-07-07T17:59:48.348441Z","shell.execute_reply.started":"2024-07-07T17:31:06.778735Z","shell.execute_reply":"2024-07-07T17:59:48.346420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = learn.dls.valid\npreds,targs = learn.get_preds(dl=valid)\nerror_rate(preds, targs) #normal error rate","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tta_preds,_ = learn.tta(dl=valid)\nerror_rate(tta_preds, targs) #TTA error rate","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.train.show_batch(max_n=4, nrows=1, unique=True) #show with augments","metadata":{"execution":{"iopub.status.busy":"2024-07-07T18:00:12.050366Z","iopub.execute_input":"2024-07-07T18:00:12.050772Z","iopub.status.idle":"2024-07-07T18:00:15.672367Z","shell.execute_reply.started":"2024-07-07T18:00:12.050738Z","shell.execute_reply":"2024-07-07T18:00:15.671398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#padding augmentation\nitem=Resize((460), method=ResizeMethod.Pad, pad_mode=PadMode.Zeros)\nbatch=aug_transforms(size=224, min_scale=0.75)\nepochs=2\n\ndls = ImageDataLoaders.from_df(sorghum_df, train_dir, seed=42, valid_pct=0.2, #so it's either train_dir or small_train_dir\n                                   item_tfms=item, batch_tfms=batch)\nlearn = vision_learner(dls, arch, metrics=error_rate).to_fp16()","metadata":{"execution":{"iopub.status.busy":"2024-07-07T18:03:19.902650Z","iopub.execute_input":"2024-07-07T18:03:19.903175Z","iopub.status.idle":"2024-07-07T18:03:22.213118Z","shell.execute_reply.started":"2024-07-07T18:03:19.903135Z","shell.execute_reply":"2024-07-07T18:03:22.212080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fine_tune(epochs, 0.0229)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T18:03:30.579941Z","iopub.execute_input":"2024-07-07T18:03:30.580720Z","iopub.status.idle":"2024-07-07T18:30:00.763027Z","shell.execute_reply.started":"2024-07-07T18:03:30.580688Z","shell.execute_reply":"2024-07-07T18:30:00.761917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = learn.dls.valid\npreds,targs = learn.get_preds(dl=valid)\nerror_rate(preds, targs) #normal error rate","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tta_preds,_ = learn.tta(dl=valid)\nerror_rate(tta_preds, targs) #TTA error rate","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.train.show_batch(max_n=4, nrows=1, unique=True) #show with augments","metadata":{"execution":{"iopub.status.busy":"2024-07-07T18:30:36.519883Z","iopub.execute_input":"2024-07-07T18:30:36.520703Z","iopub.status.idle":"2024-07-07T18:30:40.059438Z","shell.execute_reply.started":"2024-07-07T18:30:36.520642Z","shell.execute_reply":"2024-07-07T18:30:40.058400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ok, let's try a different architecture now; with the best augments from before (so let's do padding again)\narch = 'convnext_small'","metadata":{"execution":{"iopub.status.busy":"2024-07-07T18:31:58.052050Z","iopub.execute_input":"2024-07-07T18:31:58.052453Z","iopub.status.idle":"2024-07-07T18:31:58.057772Z","shell.execute_reply.started":"2024-07-07T18:31:58.052424Z","shell.execute_reply":"2024-07-07T18:31:58.056657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#padding augmentation\nitem=Resize((460), method=ResizeMethod.Pad, pad_mode=PadMode.Zeros)\nbatch=aug_transforms(size=224, min_scale=0.75)\nepochs=2\n\ndls = ImageDataLoaders.from_df(sorghum_df, train_dir, seed=42, valid_pct=0.2, #so it's either train_dir or small_train_dir\n                                   item_tfms=item, batch_tfms=batch)\nlearn = vision_learner(dls, arch, metrics=error_rate).to_fp16()","metadata":{"execution":{"iopub.status.busy":"2024-07-07T18:32:36.589430Z","iopub.execute_input":"2024-07-07T18:32:36.590278Z","iopub.status.idle":"2024-07-07T18:32:41.818292Z","shell.execute_reply.started":"2024-07-07T18:32:36.590235Z","shell.execute_reply":"2024-07-07T18:32:41.817415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fine_tune(epochs, 0.0229)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T18:33:05.024198Z","iopub.execute_input":"2024-07-07T18:33:05.024564Z","iopub.status.idle":"2024-07-07T19:10:33.790231Z","shell.execute_reply.started":"2024-07-07T18:33:05.024535Z","shell.execute_reply":"2024-07-07T19:10:33.789269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid = learn.dls.valid\npreds,targs = learn.get_preds(dl=valid)\nerror_rate(preds, targs) #this one seems pretty good!","metadata":{"execution":{"iopub.status.busy":"2024-07-07T19:10:48.962531Z","iopub.execute_input":"2024-07-07T19:10:48.963432Z","iopub.status.idle":"2024-07-07T19:12:10.123587Z","shell.execute_reply.started":"2024-07-07T19:10:48.963391Z","shell.execute_reply":"2024-07-07T19:12:10.122534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tta_preds,_ = learn.tta(dl=valid)\nerror_rate(tta_preds, targs)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T18:04:08.390481Z","iopub.execute_input":"2024-07-06T18:04:08.391356Z","iopub.status.idle":"2024-07-06T18:10:09.994898Z","shell.execute_reply.started":"2024-07-06T18:04:08.391321Z","shell.execute_reply":"2024-07-06T18:10:09.993796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.train.show_batch(max_n=4, nrows=1, unique=True) #show with augments","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Which model should be used!?","metadata":{}},{"cell_type":"markdown","source":"Well, here are some comparisons for those in timm! https://www.kaggle.com/code/jhoward/the-best-vision-models-for-fine-tuning\n","metadata":{}},{"cell_type":"code","source":"arch='convnext_tiny_in22k' #what about this one?","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get testing data","metadata":{}},{"cell_type":"code","source":"test_dl = learn.dls.test_dl(test_files) #test images ","metadata":{"execution":{"iopub.status.busy":"2024-07-07T19:12:21.128442Z","iopub.execute_input":"2024-07-07T19:12:21.128830Z","iopub.status.idle":"2024-07-07T19:12:21.140068Z","shell.execute_reply.started":"2024-07-07T19:12:21.128795Z","shell.execute_reply":"2024-07-07T19:12:21.139159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dl.show_batch()","metadata":{"execution":{"iopub.status.busy":"2024-07-07T19:12:37.748564Z","iopub.execute_input":"2024-07-07T19:12:37.749228Z","iopub.status.idle":"2024-07-07T19:12:43.032109Z","shell.execute_reply.started":"2024-07-07T19:12:37.749197Z","shell.execute_reply":"2024-07-07T19:12:43.030945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TTA Kaggle Submission","metadata":{}},{"cell_type":"code","source":"#submit with TTA\npreds = learn.tta(dl=test_dl)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T19:12:52.798381Z","iopub.execute_input":"2024-07-07T19:12:52.799284Z","iopub.status.idle":"2024-07-07T20:02:44.238192Z","shell.execute_reply.started":"2024-07-07T19:12:52.799249Z","shell.execute_reply":"2024-07-07T20:02:44.237192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2024-07-07T20:02:55.926082Z","iopub.execute_input":"2024-07-07T20:02:55.926471Z","iopub.status.idle":"2024-07-07T20:02:55.936282Z","shell.execute_reply.started":"2024-07-07T20:02:55.926436Z","shell.execute_reply":"2024-07-07T20:02:55.935358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idxs = np.argmax(preds[0], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T20:03:02.044988Z","iopub.execute_input":"2024-07-07T20:03:02.045819Z","iopub.status.idle":"2024-07-07T20:03:02.266345Z","shell.execute_reply.started":"2024-07-07T20:03:02.045789Z","shell.execute_reply":"2024-07-07T20:03:02.265386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idxs","metadata":{"execution":{"iopub.status.busy":"2024-07-07T20:03:10.969428Z","iopub.execute_input":"2024-07-07T20:03:10.970159Z","iopub.status.idle":"2024-07-07T20:03:10.976997Z","shell.execute_reply.started":"2024-07-07T20:03:10.970125Z","shell.execute_reply":"2024-07-07T20:03:10.976107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab = np.array(learn.dls.vocab)\nresults = pd.Series(vocab[idxs], name=\"idxs\")","metadata":{"execution":{"iopub.status.busy":"2024-07-07T20:03:18.304487Z","iopub.execute_input":"2024-07-07T20:03:18.305145Z","iopub.status.idle":"2024-07-07T20:03:18.315901Z","shell.execute_reply.started":"2024-07-07T20:03:18.305113Z","shell.execute_reply":"2024-07-07T20:03:18.314957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({ 'filename': os.listdir(path +'test'), 'cultivar': results })","metadata":{"execution":{"iopub.status.busy":"2024-07-07T20:03:40.888515Z","iopub.execute_input":"2024-07-07T20:03:40.889230Z","iopub.status.idle":"2024-07-07T20:03:41.099600Z","shell.execute_reply.started":"2024-07-07T20:03:40.889197Z","shell.execute_reply":"2024-07-07T20:03:41.098812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-07-07T20:03:49.556814Z","iopub.execute_input":"2024-07-07T20:03:49.557702Z","iopub.status.idle":"2024-07-07T20:03:49.578171Z","shell.execute_reply.started":"2024-07-07T20:03:49.557670Z","shell.execute_reply":"2024-07-07T20:03:49.577301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-07T20:03:56.405641Z","iopub.execute_input":"2024-07-07T20:03:56.406285Z","iopub.status.idle":"2024-07-07T20:03:56.466401Z","shell.execute_reply.started":"2024-07-07T20:03:56.406255Z","shell.execute_reply":"2024-07-07T20:03:56.465350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Normal kaggle submission","metadata":{}},{"cell_type":"code","source":"\nlog_preds_test = learn.get_preds(dl=test_dl)","metadata":{"execution":{"iopub.status.busy":"2024-07-06T20:06:38.825230Z","iopub.execute_input":"2024-07-06T20:06:38.826284Z","iopub.status.idle":"2024-07-06T20:16:37.792058Z","shell.execute_reply.started":"2024-07-06T20:06:38.826247Z","shell.execute_reply":"2024-07-06T20:16:37.791173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_preds_test","metadata":{"execution":{"iopub.status.busy":"2024-07-06T20:16:56.423716Z","iopub.execute_input":"2024-07-06T20:16:56.424370Z","iopub.status.idle":"2024-07-06T20:16:56.433340Z","shell.execute_reply.started":"2024-07-06T20:16:56.424336Z","shell.execute_reply":"2024-07-06T20:16:56.432447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_preds_test = learn.get_preds(dl=test_dl)\nlog_preds_test =  np.argmax(log_preds_test[0], axis=1)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-07-06T20:17:27.204517Z","iopub.execute_input":"2024-07-06T20:17:27.205627Z","iopub.status.idle":"2024-07-06T20:17:27.303397Z","shell.execute_reply.started":"2024-07-06T20:17:27.205586Z","shell.execute_reply":"2024-07-06T20:17:27.302432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_preds_test","metadata":{"execution":{"iopub.status.busy":"2024-07-06T20:17:41.610385Z","iopub.execute_input":"2024-07-06T20:17:41.610750Z","iopub.status.idle":"2024-07-06T20:17:41.617490Z","shell.execute_reply.started":"2024-07-06T20:17:41.610725Z","shell.execute_reply":"2024-07-06T20:17:41.616596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#look at sample\nss = pd.read_csv(path + 'sample_submission.csv')\nss","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:59:54.523210Z","iopub.execute_input":"2024-07-02T19:59:54.524226Z","iopub.status.idle":"2024-07-02T19:59:54.556507Z","shell.execute_reply.started":"2024-07-02T19:59:54.524181Z","shell.execute_reply":"2024-07-02T19:59:54.555320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_classes = [dls.vocab[i] for i in log_preds_test]\nprobs = np.exp(log_preds_test)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:00:15.975388Z","iopub.execute_input":"2024-07-02T20:00:15.975781Z","iopub.status.idle":"2024-07-02T20:00:17.803754Z","shell.execute_reply.started":"2024-07-02T20:00:15.975749Z","shell.execute_reply":"2024-07-02T20:00:17.802804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({ 'filename': os.listdir(path +'test'), 'cultivar': preds_classes })","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:00:39.213526Z","iopub.execute_input":"2024-07-02T20:00:39.214349Z","iopub.status.idle":"2024-07-02T20:00:39.456207Z","shell.execute_reply.started":"2024-07-02T20:00:39.214313Z","shell.execute_reply":"2024-07-02T20:00:39.455129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:00:59.254656Z","iopub.execute_input":"2024-07-02T20:00:59.255073Z","iopub.status.idle":"2024-07-02T20:00:59.267670Z","shell.execute_reply.started":"2024-07-02T20:00:59.255037Z","shell.execute_reply":"2024-07-02T20:00:59.266357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T20:01:30.846871Z","iopub.execute_input":"2024-07-02T20:01:30.847297Z","iopub.status.idle":"2024-07-02T20:01:30.922576Z","shell.execute_reply.started":"2024-07-02T20:01:30.847263Z","shell.execute_reply":"2024-07-02T20:01:30.921183Z"},"trusted":true},"execution_count":null,"outputs":[]}]}