{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n  #  for filename in filenames:\n        # print(os.path.join(dirname, filename))\n   #     pass\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-05T18:22:02.428354Z","iopub.execute_input":"2022-03-05T18:22:02.428995Z","iopub.status.idle":"2022-03-05T18:22:02.456168Z","shell.execute_reply.started":"2022-03-05T18:22:02.42886Z","shell.execute_reply":"2022-03-05T18:22:02.455505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_file_path = '../input/herbarium-2022-fgvc9/sample_submission.csv'\nsample_submission = pd.read_csv(sample_submission_file_path)\nsample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-05T18:22:02.457596Z","iopub.execute_input":"2022-03-05T18:22:02.457902Z","iopub.status.idle":"2022-03-05T18:22:02.543357Z","shell.execute_reply.started":"2022-03-05T18:22:02.457866Z","shell.execute_reply":"2022-03-05T18:22:02.542668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport operator\n\ntrain_metadata_file_path = '../input/herbarium-2022-fgvc9/train_metadata.json'\nf = open(train_metadata_file_path,)\ndata = json.load(f)","metadata":{"execution":{"iopub.status.busy":"2022-03-05T18:22:02.544692Z","iopub.execute_input":"2022-03-05T18:22:02.545168Z","iopub.status.idle":"2022-03-05T18:22:17.757061Z","shell.execute_reply.started":"2022-03-05T18:22:02.545132Z","shell.execute_reply":"2022-03-05T18:22:17.756267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotations = data['annotations']\ndf = pd.DataFrame(annotations)\ndf['category_id'] = df['category_id'].map(str)\n    \ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-05T18:22:17.759159Z","iopub.execute_input":"2022-03-05T18:22:17.759422Z","iopub.status.idle":"2022-03-05T18:22:19.617272Z","shell.execute_reply.started":"2022-03-05T18:22:17.759386Z","shell.execute_reply":"2022-03-05T18:22:19.616566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_file_path(row):\n    image_id = row['image_id']\n    part_1 = '../input/herbarium-2022-fgvc9/train_images'\n    part_2 = image_id[:3]\n    part_3 = image_id[3:5]\n    part_4 = f'{image_id}.jpg'\n    file_path = os.path.join(part_1, part_2, part_3, part_4)\n    \n    return file_path","metadata":{"execution":{"iopub.status.busy":"2022-03-05T18:22:19.618374Z","iopub.execute_input":"2022-03-05T18:22:19.620091Z","iopub.status.idle":"2022-03-05T18:22:19.625469Z","shell.execute_reply.started":"2022-03-05T18:22:19.62005Z","shell.execute_reply":"2022-03-05T18:22:19.624616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['file_path'] = df.apply(lambda row: get_file_path(row), axis=1)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-05T18:22:19.626847Z","iopub.execute_input":"2022-03-05T18:22:19.627125Z","iopub.status.idle":"2022-03-05T18:22:32.696468Z","shell.execute_reply.started":"2022-03-05T18:22:19.627092Z","shell.execute_reply":"2022-03-05T18:22:32.695773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = df.sample(frac=0.8, random_state=200) \nvalidation_df = df.drop(train_df.index)","metadata":{"execution":{"iopub.status.busy":"2022-03-05T18:22:32.697918Z","iopub.execute_input":"2022-03-05T18:22:32.698413Z","iopub.status.idle":"2022-03-05T18:22:33.343747Z","shell.execute_reply.started":"2022-03-05T18:22:32.698374Z","shell.execute_reply":"2022-03-05T18:22:33.342986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras_preprocessing.image import ImageDataGenerator\n\ndatagen = ImageDataGenerator(rescale=1.0/255.0)\ntrain_generator = \\\n    datagen.flow_from_dataframe(dataframe=train_df, x_col='file_path', y_col='category_id', batch_size=32, \n                                seed=42, shuffle=True, class_mode=\"categorical\")","metadata":{"execution":{"iopub.status.busy":"2022-03-05T18:22:33.345163Z","iopub.execute_input":"2022-03-05T18:22:33.345422Z","iopub.status.idle":"2022-03-05T18:56:16.358415Z","shell.execute_reply.started":"2022-03-05T18:22:33.345388Z","shell.execute_reply":"2022-03-05T18:56:16.35767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import Conv2D, Activation, MaxPooling2D, Dropout\nfrom keras.layers import Dense, Flatten\nfrom keras import layers\n\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), padding='same',\n                 input_shape=(256, 256, 3)))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(32, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\nmodel.add(Conv2D(64, (3, 3), padding='same'))\nmodel.add(Activation('relu'))\nmodel.add(Conv2D(64, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\nmodel.add(Flatten())\nmodel.add(Dense(512))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(15330, activation='softmax'))\n\nmodel.compile(optimizer='rmsprop', \n              loss=\"categorical_crossentropy\", \n              metrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2022-03-05T18:56:16.359904Z","iopub.execute_input":"2022-03-05T18:56:16.360165Z","iopub.status.idle":"2022-03-05T18:56:18.90943Z","shell.execute_reply.started":"2022-03-05T18:56:16.360132Z","shell.execute_reply":"2022-03-05T18:56:18.908614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_generator = \\\n    datagen.flow_from_dataframe(dataframe=validation_df, x_col='file_path', y_col='category_id', \n                                batch_size=32, seed=42, shuffle=True, class_mode=\"categorical\")","metadata":{"execution":{"iopub.status.busy":"2022-03-05T18:56:18.912003Z","iopub.execute_input":"2022-03-05T18:56:18.912438Z","iopub.status.idle":"2022-03-05T19:04:16.448356Z","shell.execute_reply.started":"2022-03-05T18:56:18.9124Z","shell.execute_reply":"2022-03-05T19:04:16.447558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN = train_generator.n // train_generator.batch_size\nSTEP_SIZE_VALID = valid_generator.n // valid_generator.batch_size\n\nmodel.fit_generator(generator=train_generator,\n                    steps_per_epoch=STEP_SIZE_TRAIN,\n                    validation_data=valid_generator,\n                    validation_steps=STEP_SIZE_VALID,\n                    epochs=10)","metadata":{"execution":{"iopub.status.busy":"2022-03-05T19:04:16.449874Z","iopub.execute_input":"2022-03-05T19:04:16.450159Z","iopub.status.idle":"2022-03-05T23:11:34.021418Z","shell.execute_reply.started":"2022-03-05T19:04:16.45012Z","shell.execute_reply":"2022-03-05T23:11:34.019833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('model.h5')","metadata":{"execution":{"iopub.status.busy":"2022-03-05T23:11:34.022719Z","iopub.status.idle":"2022-03-05T23:11:34.023381Z","shell.execute_reply.started":"2022-03-05T23:11:34.023144Z","shell.execute_reply":"2022-03-05T23:11:34.023169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TEST = test_generator.n // test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2022-03-05T23:11:34.024667Z","iopub.status.idle":"2022-03-05T23:11:34.025304Z","shell.execute_reply.started":"2022-03-05T23:11:34.025071Z","shell.execute_reply":"2022-03-05T23:11:34.025095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-05T23:11:34.026427Z","iopub.status.idle":"2022-03-05T23:11:34.027037Z","shell.execute_reply.started":"2022-03-05T23:11:34.026798Z","shell.execute_reply":"2022-03-05T23:11:34.02682Z"},"trusted":true},"execution_count":null,"outputs":[]}]}