{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\n","metadata":{"execution":{"iopub.status.busy":"2022-03-09T01:48:46.547211Z","iopub.execute_input":"2022-03-09T01:48:46.547514Z","iopub.status.idle":"2022-03-09T01:48:46.571912Z","shell.execute_reply.started":"2022-03-09T01:48:46.547442Z","shell.execute_reply":"2022-03-09T01:48:46.57123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metadata\n\nLet's look at the metadata:","metadata":{}},{"cell_type":"code","source":"PATH='/kaggle/input/snakeclef2022'\ndf_train=pd.read_csv(os.path.join(PATH,'SnakeCLEF2022-TrainMetadata.csv'))\nos.listdir(PATH)","metadata":{"execution":{"iopub.status.busy":"2022-03-09T01:49:41.975618Z","iopub.execute_input":"2022-03-09T01:49:41.97587Z","iopub.status.idle":"2022-03-09T01:49:42.364927Z","shell.execute_reply.started":"2022-03-09T01:49:41.975845Z","shell.execute_reply":"2022-03-09T01:49:42.363987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distribution of training data per species\n\nNow, let's see the distribution of training data in terms of classses to see how balanced the classes are. ","metadata":{}},{"cell_type":"code","source":"df_hist=df_train.groupby('class_id').count().sort_values('endemic').reset_index()\nimport matplotlib.pyplot as plt\nwidth=20\nheight=10\nplt.figure(figsize=(width,height))\nplt.bar(df_hist.index, df_hist.endemic)\nplt.yscale(\"log\") \nplt.xlabel('Snake species')\nplt.ylabel('# of training images')","metadata":{"execution":{"iopub.status.busy":"2022-03-09T01:48:56.461836Z","iopub.execute_input":"2022-03-09T01:48:56.462075Z","iopub.status.idle":"2022-03-09T01:49:00.07649Z","shell.execute_reply.started":"2022-03-09T01:48:56.462049Z","shell.execute_reply":"2022-03-09T01:49:00.075317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List \n# Create the top 3 most common and top 3 least common species\n# And plot their distribution by converting country to geolocations, and place them in raster map, with frequency.","metadata":{"execution":{"iopub.status.busy":"2022-03-09T00:11:24.949309Z","iopub.execute_input":"2022-03-09T00:11:24.94985Z","iopub.status.idle":"2022-03-09T00:11:25.111191Z","shell.execute_reply.started":"2022-03-09T00:11:24.949813Z","shell.execute_reply":"2022-03-09T00:11:25.110337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataloader\n\nLoad the data with ImageDataGenerator and flow_from_dataframe. Note that this process could be very slow therefore writing custom function would be better in the future. \nIn this example, we use just a subset of the training data to complete the notebook fast. \n\nImage augmentation is notably one of the most important procedure for training CNNs. we use both side of flipping and featurewise normalization techniques. Zoom-in augmentation is neglected because the input resolution is exactly same as the training data resolution in our case. ","metadata":{}},{"cell_type":"code","source":"df_train = df_train.astype({'class_id':str})\ndf_train=df_train.sample(frac=0.05)\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing import image\nBATCH_SIZE=16\nL=240\nbatch_size=BATCH_SIZE\nfPATH='file_path'\nLABEL='class_id'\nFOLDER=\"../input/snakeclef2022/SnakeCLEF2022-small_size/SnakeCLEF2022-small_size\"\n\ntrain_datagen = image.ImageDataGenerator(featurewise_center=True,\n                                 featurewise_std_normalization=True,\n                                 horizontal_flip=True,\n                                 vertical_flip=True,\n                                 #zoom_range=0.3, skip zoom since small image size=240\n                                 validation_split=0.2)\n\n\ntest_datagen = image.ImageDataGenerator(featurewise_center=True,\n                                 featurewise_std_normalization=True,\n                                 validation_split=0.2\n                                 #horizontal_flip=True,\n                                 #vertical_flip=True,\n                                 #zoom_range=0.2\n)\n\n\ntrain_ds = train_datagen.flow_from_dataframe(\n        dataframe = df_train, \n        directory = FOLDER,\n        x_col = fPATH,\n        class_mode=\"categorical\",\n        y_col = LABEL,\n        target_size=(L, L),\n        batch_size=batch_size,\n        shuffle=True,\n        seed=1004,\n        subset='training'\n)\n\nval_ds = train_datagen.flow_from_dataframe(\n        dataframe = df_train, \n        directory = FOLDER,\n        x_col = fPATH,\n        class_mode=\"categorical\",\n        y_col = LABEL,\n        target_size=(L, L),\n        batch_size=batch_size,\n        shuffle=True,\n        seed=1004,\n        subset='validation'\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-09T02:17:29.300032Z","iopub.execute_input":"2022-03-09T02:17:29.300295Z","iopub.status.idle":"2022-03-09T02:17:33.529909Z","shell.execute_reply.started":"2022-03-09T02:17:29.300271Z","shell.execute_reply":"2022-03-09T02:17:33.529403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling with efficientNet\n\nSince the resolution of the small size image data is 240 by 240, we will use EfficientNet B1, developed by google AI, which is meant to work best for image size of 240.\nWe add bottlneck layer to extract minimal dimension for classification.\n","metadata":{}},{"cell_type":"code","source":"N_class=len(set(df_train[LABEL]))\nprint(N_class)\nmodel_c = tf.keras.models.Sequential([\n    tf.keras.layers.InputLayer(input_shape=[L, L, 3]),\n    tf.keras.layers.LayerNormalization(\n    axis=-1,\n    center=True,\n    scale=True,\n    trainable=True,\n    name='input_normalized'),\n    tf.keras.applications.EfficientNetB1(\n    include_top=False, weights='imagenet', input_tensor=None,\n    input_shape=(L,L,3)\n    ),\n    tf.keras.layers.GlobalAveragePooling2D(),\n    tf.keras.layers.Dropout(0.5),\n    tf.keras.layers.Dense(64,activation=\"relu\"),\n    tf.keras.layers.Dropout(0.3),\n    tf.keras.layers.Dense(N_class, activation='softmax'),\n    ])\n\ntf_adam=tf.keras.optimizers.Adam(\n    learning_rate=0.001, beta_1=0.9, beta_2=0.999, epsilon=1e-07, amsgrad=True,\n    name='Adam'\n)\n\nimport tensorflow_addons as tfa\nmodel_c.summary()\nmodel_c.compile(loss=tfa.losses.SigmoidFocalCrossEntropy(), optimizer=tf_adam, metrics=[\"accuracy\"])#,steps_per_execution=16)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-09T02:17:43.717866Z","iopub.execute_input":"2022-03-09T02:17:43.718243Z","iopub.status.idle":"2022-03-09T02:17:47.620875Z","shell.execute_reply.started":"2022-03-09T02:17:43.718216Z","shell.execute_reply":"2022-03-09T02:17:47.619832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_c.fit(train_ds, epochs=1, validation_data=val_ds, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-09T02:17:53.315559Z","iopub.execute_input":"2022-03-09T02:17:53.315972Z","iopub.status.idle":"2022-03-09T02:18:30.835714Z","shell.execute_reply.started":"2022-03-09T02:17:53.315949Z","shell.execute_reply":"2022-03-09T02:18:30.834027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction\n\nTBD!","metadata":{}},{"cell_type":"code","source":"model_c.save(\"SnakeCLEF22-effnetb1.h5\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}