{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\n\nimport numpy as np # linear algebra\nfrom PIL import Image\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport cv2\nimport torchvision\nimport torch\nimport math\nimport json\nfrom zipfile import ZipFile\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-02-28T17:22:45.274302Z","iopub.execute_input":"2023-02-28T17:22:45.274997Z","iopub.status.idle":"2023-02-28T17:22:48.541989Z","shell.execute_reply.started":"2023-02-28T17:22:45.274935Z","shell.execute_reply":"2023-02-28T17:22:48.540938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path='/kaggle/input/iwild19-combine/allofthem'\ndf_2200=pd.read_csv('/kaggle/input/iwild19-combine/df_2200.csv')\ndf_final_class0=pd.read_csv('/kaggle/input/iwild19-combine/df_final_class0.csv',index_col='Unnamed: 0')\ndf_final=pd.concat([df_2200,df_final_class0],ignore_index=True)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:22:48.543836Z","iopub.execute_input":"2023-02-28T17:22:48.544385Z","iopub.status.idle":"2023-02-28T17:22:48.601832Z","shell.execute_reply.started":"2023-02-28T17:22:48.544347Z","shell.execute_reply":"2023-02-28T17:22:48.600918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:22:48.603893Z","iopub.execute_input":"2023-02-28T17:22:48.604665Z","iopub.status.idle":"2023-02-28T17:22:48.618101Z","shell.execute_reply.started":"2023-02-28T17:22:48.604627Z","shell.execute_reply":"2023-02-28T17:22:48.617026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_final.head())\nprint(len(df_final))\n\n# create a DataFrame for category_id=1 and another for all other categories\ncat1_df = df_final[df_final['category_id'] == 0]\nother_df = df_final[df_final['category_id'] != 0]\nother_df['category_id']=1\n# plot histograms of the two DataFrames on the same plot\nplt.hist(other_df['category_id'], alpha=0.5, label='empty')\nplt.hist(cat1_df['category_id'], alpha=0.5, label='animal')\n\n# add title and labels\nplt.title('Category ID Histogram')\nplt.xlabel('Category ID')\nplt.ylabel('Frequency')\nplt.legend()\n\n# show the plot\nplt.show()\nother_df = df_final[df_final['category_id'] != 0]\n\nplt.hist(other_df['category_id'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:22:48.621251Z","iopub.execute_input":"2023-02-28T17:22:48.621619Z","iopub.status.idle":"2023-02-28T17:22:49.087875Z","shell.execute_reply.started":"2023-02-28T17:22:48.621573Z","shell.execute_reply":"2023-02-28T17:22:49.086804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/input/iwild19-combine/allofthem/585daba7-23d2-11e8-a6a3-ec086b02610b.jpg_3.npy','rb') as f:\n    a=np.load(f)\n    plt.imshow(a)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:22:49.089401Z","iopub.execute_input":"2023-02-28T17:22:49.090636Z","iopub.status.idle":"2023-02-28T17:22:49.405159Z","shell.execute_reply.started":"2023-02-28T17:22:49.090591Z","shell.execute_reply":"2023-02-28T17:22:49.404133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:22:49.406289Z","iopub.execute_input":"2023-02-28T17:22:49.406844Z","iopub.status.idle":"2023-02-28T17:22:49.418065Z","shell.execute_reply.started":"2023-02-28T17:22:49.406807Z","shell.execute_reply":"2023-02-28T17:22:49.417042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_final['file_name']=df_final['file_name'].str.replace('/', '-')\ndf_final.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:22:49.419587Z","iopub.execute_input":"2023-02-28T17:22:49.420239Z","iopub.status.idle":"2023-02-28T17:22:49.436564Z","shell.execute_reply.started":"2023-02-28T17:22:49.420193Z","shell.execute_reply":"2023-02-28T17:22:49.435534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize lists to hold the image data and labels\nimages = []\nlabels = []\npath = \"/kaggle/input/iwild19-combine/allofthem/\"\n# Iterate over the rows of the DataFrame\nfor index, row in df_final.iterrows():\n#     print(row['category_id'])\n#     print(row)\n    # Load the image data\n    if row['file_name'].endswith('.jpg'):\n        # If the file is a JPEG image, load it using OpenCV\n        img = np.load(path+row['file_name']+'.npy')\n#         print(img)\n    elif row['file_name'].endswith('.npy'):\n        # If the file is a NumPy array, load it using NumPy\n        img = np.load(path+row['file_name'])\n\n    # Check if the image has a shape of (1, ...)\n    if img.shape[0] == 1:\n        # If the image has a shape of (1, ...), skip it and its corresponding label\n        continue\n\n    # Append the image data and label to the lists\n    images.append(img)\n    labels.append(row['category_id'])\n\n# Convert the lists to NumPy arrays\nimages = np.array(images)\nlabels = np.array(labels)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:22:49.437991Z","iopub.execute_input":"2023-02-28T17:22:49.438721Z","iopub.status.idle":"2023-02-28T17:23:12.031772Z","shell.execute_reply.started":"2023-02-28T17:22:49.438685Z","shell.execute_reply":"2023-02-28T17:23:12.030645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.preprocessing import LabelBinarizer\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.preprocessing import OneHotEncoder\nprint(len(images))\nprint(type(images))\nprint(len(labels))\nprint(labels[800])\n# lb = LabelBinarizer()\n\n# labels = lb.fit_transform(labels)\n# labels = to_categorical(labels)\n# enc = [OneHotEncoder(label) for label in labels]\nprint(labels)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:23:12.033515Z","iopub.execute_input":"2023-02-28T17:23:12.034327Z","iopub.status.idle":"2023-02-28T17:23:20.283690Z","shell.execute_reply.started":"2023-02-28T17:23:12.034282Z","shell.execute_reply":"2023-02-28T17:23:20.282492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# import numpy as np\n# from PIL import Image\n\n# def image_loader(path):\n#     \"\"\"\n#     Load images from the given path. The path can contain JPEG images or NumPy arrays.\n#     \"\"\"\n#     files = os.listdir(path)\n#     images = []\n#     for file in working_df['file_name'].values:\n#             with open(file):\n#                 images.append(np.load(str(path)+str(file)))\n#     return np.array(images)\n# path = \"/kaggle/input/iwild19-combine/allofthem\"\n# images = image_loader(path)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:23:20.287847Z","iopub.execute_input":"2023-02-28T17:23:20.288913Z","iopub.status.idle":"2023-02-28T17:23:20.296170Z","shell.execute_reply.started":"2023-02-28T17:23:20.288879Z","shell.execute_reply":"2023-02-28T17:23:20.295062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resized_images = np.array([cv2.resize(img, (100, 100)) for img in images])","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:23:20.298290Z","iopub.execute_input":"2023-02-28T17:23:20.298753Z","iopub.status.idle":"2023-02-28T17:23:20.580815Z","shell.execute_reply.started":"2023-02-28T17:23:20.298712Z","shell.execute_reply":"2023-02-28T17:23:20.579756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\nfrom keras.utils import to_categorical\n\n\nnum_classes = 23  # number of classes\n\n# Convert labels to one-hot encoded vectors\none_hot_labels = to_categorical(labels, num_classes=num_classes)\n\nprint(one_hot_labels)\n\n# labelencoder = LabelEncoder()\n# enc = OneHotEncoder(sparse=False).fit(df_final['category_id'].unique().reshape(-1,1))\n# encoded = enc.transform(labels)# convert it to a dataframe\n# encoded_df = pd.DataFrame(\n#      encoded, \n#      columns=enc.get_feature_names_out()\n# )\n# encoded_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:23:20.582287Z","iopub.execute_input":"2023-02-28T17:23:20.583237Z","iopub.status.idle":"2023-02-28T17:23:20.592227Z","shell.execute_reply.started":"2023-02-28T17:23:20.583191Z","shell.execute_reply":"2023-02-28T17:23:20.591011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\nfrom sklearn.model_selection import train_test_split\n# Load your training data from numpy array\ntrain_images = resized_images\nX_train, X_valid, y_train, y_valid = train_test_split(resized_images, one_hot_labels, test_size=0.1, shuffle=True)\n\n# train_images = np.expand_dims(train_images, axis=1)\n# train_labels = labels\n\n# Define your ImageDataGenerator\ndatagen = ImageDataGenerator(\n    rotation_range=0,  # Randomly rotate images by up to 10 degrees\n    horizontal_flip=False,  # Randomly flip images horizontally\n    zoom_range=0,  # Randomly zoom images by up to 10%\n    rescale=1/255,\n    fill_mode='nearest',\n    validation_split=0.1\n)\n\n# Define batch size\nbatch_size = 32\n\n# Generate batches of augmented data\ntrain_generator = datagen.flow(\n    x=X_train,\n    y=y_train,\n    batch_size=batch_size,\n    subset='training'\n)\nval_generator = datagen.flow(\n    x=X_valid,\n    y=y_valid,\n    batch_size=batch_size,\n    subset='validation'\n)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T18:10:17.502362Z","iopub.execute_input":"2023-02-28T18:10:17.503493Z","iopub.status.idle":"2023-02-28T18:10:17.651823Z","shell.execute_reply.started":"2023-02-28T18:10:17.503452Z","shell.execute_reply":"2023-02-28T18:10:17.650752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nfrom keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.applications import DenseNet121\nfrom sklearn.preprocessing import LabelEncoder\nfrom keras import backend as K\n\ndef recall_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n    recall = true_positives / (possible_positives + K.epsilon())\n    return recall\n\ndef precision_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    precision = true_positives / (predicted_positives + K.epsilon())\n    return precision\n\ndef f1_m(y_true, y_pred):\n    precision = precision_m(y_true, y_pred)\n    recall = recall_m(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))\n\nmodel = DenseNet121(\n    weights=None, \n    include_top=True, \n    classes=23,\n    input_shape=(100, 100, 3)\n)\n\n# compile the model\nmodel.compile(optimizer='adam',\n              loss='categorical_crossentropy',\n              metrics=['accuracy',f1_m,precision_m, recall_m])\n\n#model.summary()\n\ncheckpoint = ModelCheckpoint(\n    '/kaggle/working/model.h5', \n    monitor='val_accuracy', \n    verbose=0, \n    save_best_only=True, \n    save_weights_only=False,\n    mode='auto'\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T18:10:31.732952Z","iopub.execute_input":"2023-02-28T18:10:31.734170Z","iopub.status.idle":"2023-02-28T18:10:34.206870Z","shell.execute_reply.started":"2023-02-28T18:10:31.734104Z","shell.execute_reply":"2023-02-28T18:10:34.205832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nhistory = model.fit_generator(\n    train_generator,\n    steps_per_epoch= 100, \n    epochs=32,\n    callbacks=[checkpoint],\n    validation_data=val_generator\n#     use_multiprocessing=True,\n#     workers=2, \n#     verbose=1\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T18:10:53.796529Z","iopub.execute_input":"2023-02-28T18:10:53.796907Z","iopub.status.idle":"2023-02-28T18:17:20.717173Z","shell.execute_reply.started":"2023-02-28T18:10:53.796875Z","shell.execute_reply":"2023-02-28T18:17:20.716127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"\n\nwith open('/kaggle/working/history.json', 'w') as f:\n    print(f)\n    json.dump(history.history, f)\n\n\nhistory_df = pd.DataFrame(history.history)\nprint(history_df)\nhistory_df[['loss', 'val_loss']].plot()\nhistory_df[['accuracy', 'val_accuracy']].plot()\nhistory_df[['f1_m', 'val_f1_m']].plot()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T18:18:44.409659Z","iopub.execute_input":"2023-02-28T18:18:44.410355Z","iopub.status.idle":"2023-02-28T18:18:45.016592Z","shell.execute_reply.started":"2023-02-28T18:18:44.410318Z","shell.execute_reply":"2023-02-28T18:18:45.015422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('/kaggle/working/model.h5')\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T18:20:42.671142Z","iopub.execute_input":"2023-02-28T18:20:42.672178Z","iopub.status.idle":"2023-02-28T18:20:43.302353Z","shell.execute_reply.started":"2023-02-28T18:20:42.672139Z","shell.execute_reply":"2023-02-28T18:20:43.301161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, accuracy, f1_score, precision, recall = model.evaluate_generator(\n    generator=val_generator,\n    steps=len(val_generator),\n#     use_multiprocessing=True,\n#     verbose=1,\n#     workers=2\n)\n\n# print('\\nValidation loss:', val_scores[0])\n# print('Validation accuracy:', val_scores[1])\n\n\n# compile the model\n#model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy',f1_m,precision_m, recall_m])\n# load the weigth\n#model.load_weights('/kaggle/working/model.h5')\n# evaluate the model\n#loss, accuracy, f1_score, precision, recall = model.evaluate(X_valid, y_valid, verbose=0)\nprint('\\nValidation loss:', loss)\nprint('Validation accuracy:', accuracy)\nprint('Validation F1 score:', f1_score)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-28T18:20:23.923481Z","iopub.execute_input":"2023-02-28T18:20:23.924487Z","iopub.status.idle":"2023-02-28T18:20:24.069116Z","shell.execute_reply.started":"2023-02-28T18:20:23.924448Z","shell.execute_reply":"2023-02-28T18:20:24.068031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def manual_valid(index):\n    img = X_valid[index]\n    label = y_valid[index]\n    img = img/255\n    img =np.expand_dims(img, axis=0)\n#     print(img.shape)\n#     print(np.where(label==1))\n    predictions = model.predict([img])\n    label_prediction = np.where(predictions==predictions.max())[1]+1\n    print(\"prediction is \" + str(label_prediction))\n    print(\"True label is \"+str(np.where(label==1)[0]+1))\n#     x = np.zeros( (100, 100, 3) )\n    result = img[0,:, :, :]\n    plt.imshow(result)\n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2023-02-28T18:57:26.660320Z","iopub.execute_input":"2023-02-28T18:57:26.660894Z","iopub.status.idle":"2023-02-28T18:57:26.667635Z","shell.execute_reply.started":"2023-02-28T18:57:26.660860Z","shell.execute_reply":"2023-02-28T18:57:26.666512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_valid[0]\n# y_valid[0]\nfor i in range(10,20):\n    manual_valid(i)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T18:57:29.186233Z","iopub.execute_input":"2023-02-28T18:57:29.187168Z","iopub.status.idle":"2023-02-28T18:57:29.470649Z","shell.execute_reply.started":"2023-02-28T18:57:29.187120Z","shell.execute_reply":"2023-02-28T18:57:29.469672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arr=[1,2,3]\narr.index(1)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T18:46:15.450657Z","iopub.execute_input":"2023-02-28T18:46:15.451330Z","iopub.status.idle":"2023-02-28T18:46:15.458494Z","shell.execute_reply.started":"2023-02-28T18:46:15.451293Z","shell.execute_reply":"2023-02-28T18:46:15.457445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import keras\n# from keras.callbacks import ModelCheckpoint\n# from tensorflow.keras.applications import DenseNet121\n# model = DenseNet121(\n#     weights=None, \n#     include_top=True, \n#     classes=23,\n#     input_shape=(100, 100, 3)\n# )\n# from keras.models import Model\n# from keras.layers import Dense\n# from keras.applications.densenet import DenseNet121\n\n# # create a DenseNet121 model\n# base_model = DenseNet121(weights='imagenet', include_top=False)\n\n# # add a new output layer with softmax activation\n# x = base_model.output\n# x = Dense(22, activation='softmax')(x)\n\n# # create a new model with the modified output layer\n# model = Model(inputs=base_model.input, outputs=x)\n\n# # compile the model\n# model.compile(optimizer='adam',\n#               loss='categorical_crossentropy',\n#               metrics=['accuracy'])\n\n\n# model.summary()\n# checkpoint = ModelCheckpoint(\n#     'model.h5', \n#     monitor='val_acc', \n#     verbose=0, \n#     save_best_only=True, \n#     save_weights_only=False,\n#     mode='auto'\n# )\n# history = model.fit(\n#     train_generator,\n#     steps_per_epoch=75000 / 128, \n#     epochs=10,\n#     callbacks=[checkpoint],\n#     validation_data=val_generator\n# #     use_multiprocessing=True,\n# #     workers=2, \n# #     verbose=1\n# )","metadata":{"execution":{"iopub.status.busy":"2023-02-28T17:26:47.877564Z","iopub.execute_input":"2023-02-28T17:26:47.879819Z","iopub.status.idle":"2023-02-28T17:26:47.887687Z","shell.execute_reply.started":"2023-02-28T17:26:47.879778Z","shell.execute_reply":"2023-02-28T17:26:47.886321Z"},"trusted":true},"execution_count":null,"outputs":[]}]}