{"cells":[{"metadata":{},"cell_type":"markdown","source":"This is my first attempt to participiate in Kaggle. I am sure I have made mistakes.\nIt would be nice if you read and let me know my stupidities:D"},{"metadata":{"_uuid":"796838a8-c319-4a06-94d6-3abcd497ddae","_cell_guid":"ecf11bd8-1d9a-4da7-83c0-f4acbc5102d1","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport json,cv2\nfrom glob import glob as gb\nfrom matplotlib import pyplot as plt\nimport tensorflow as tf","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bdf7156f-9b2c-4dd6-bf8a-17e3cf452337","_cell_guid":"0c360c2d-eb27-4d11-87a6-e8098fd98363","trusted":true},"cell_type":"code","source":"#global parameters\nimage_size = (224,224)\nnum_classes = 212\nbatch_size = 64\nnum_samples = 174367\nstp_per_epoch = num_samples//batch_size\nval_num_sample = 43592\nval_stp = val_num_sample//batch_size\ncurrupted_files = ['883572ba-21bc-11ea-a13a-137349068a90.jpg',\n                  '8792549a-21bc-11ea-a13a-137349068a90.jpg',\n                  '99136aa6-21bc-11ea-a13a-137349068a90.jpg',\n                  '87022118-21bc-11ea-a13a-137349068a90.jpg',\n                  '8f17b296-21bc-11ea-a13a-137349068a90.jpg',\n                  '896c1198-21bc-11ea-a13a-137349068a90.jpg']\n\n# for i in enumerate(currupted_files):\n#     currupted_files[i[0]] = '../input/iwildcam-2020-fgvc7/train/'+i[1]\n    \naccelerator = None","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"if accelerator == 'tpu':\n    # detect and init the TPU\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n\n    # instantiate a distribution strategy\n    tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7428cec3-334e-400a-8895-32fff19ea574","_cell_guid":"3f41df75-462e-4391-9173-7e9a6733c3d1","trusted":true},"cell_type":"markdown","source":"**Versions**\n\n0. Loading and preprocessing data\n    1. making a generator\n        1. using keras image utiles, like the document\n        2. building class dataframe from JSON file\n        3. flow from dataframe\n    2. preprocessing as simple as possible\n        1. making the sizes equal(!)\n1. Building the model\n    1. transfer learning\n        1. resnet50\n            1. v1\n            2. v2\n        2. resnet101\n            1. v1\n            2. v2\n        3. resnet152\n            1. v1\n            2. v2\n        4. squeeznet\n        5. alexnet\n        6. inception\n        7. VGG\n        8. ZFNet\n     2. developing my own model\n        1. Create a 50 layer CNN/each layer 500 Units to see how it will work...\n        2. start for reducing"},{"metadata":{"_uuid":"de03a1ab-dcea-4f21-b515-edb4be589217","_cell_guid":"b8792207-4e29-4587-a7b1-d06f7b8f8914","trusted":true},"cell_type":"markdown","source":"<h2>V0.1: making a generator</h2>"},{"metadata":{"_uuid":"34232d6e-0d0f-48cc-bbf9-c77805dadca4","_cell_guid":"343c08b3-0e0c-4739-9f5d-0eb166cba101","trusted":true},"cell_type":"code","source":"#0.11loading the generators\ngen_t = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255)\ngen_v = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a2bbf346-eb83-44ac-bb71-f70f815195d2","_cell_guid":"5ee300ae-cebe-42f6-9e1c-02e038ff44a9","trusted":true},"cell_type":"code","source":"#0.12building the dataframe\nwith open('../input/iwildcam-2020-fgvc7/iwildcam2020_train_annotations.json') as j_data:\n    data = json.load(j_data)\nlabels_dataframe = pd.DataFrame.from_dict(data['annotations'])\n# labels_dataframe.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"13111b08-e812-4fcf-a477-ccfe760a412a","_cell_guid":"6329eaff-88b7-4f84-a605-e2e6d67cfc96","trusted":true},"cell_type":"code","source":"# '''Just a check:\n# 1- make 10 random numbers\n# 2- get the image corresponding to this 10 numbers in the glob list of images\n# 3- show them with their cat ID in the DF\n# '''\nimages = gb('../input/iwildcam-2020-fgvc7/train/*.jpg')\nrandom_numbers = np.random.randint(0,len(images),3)\n\nselected_images = [images[i] for i in random_numbers]\n\nfor i in selected_images:\n    array = cv2.imread(i)\n    category = labels_dataframe.loc[labels_dataframe['image_id'] == i[35:-4]]['category_id'].iloc[0]\n    cv2.cvtColor(array, cv2.COLOR_BGR2RGB)\n    plt.imshow(array)\n    plt.title(category)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cd37189b-27c1-457b-a63d-d033db758ca7","_cell_guid":"62683079-f619-4d45-9a0c-e4f9177a3c9d","trusted":true},"cell_type":"code","source":"#0.13flow from dataframe\n\n#making the dataframe\nlabels_dataframe = labels_dataframe.drop(['id','count'],\n                                        axis = 1)\n#suffle the dataframe\nlabels_dataframe = labels_dataframe.sample(frac=1)\n\n#split the dataframe into validation and training dataset\n\n#finding the number of all the samples\nn_total = len(labels_dataframe.index)\nn_train = round(0.8 * n_total)\nn_valid = n_total - n_train\n\n#defining the dataframes\ndf_train = labels_dataframe.head(n_train)\ndf_valid = labels_dataframe.tail(n_valid)\n\n#changing category IDs to str\ndf_train,df_valid = df_train.astype('str'),df_valid.astype('str')\n\n#adding '.jpg' to all image names\ndf_train['image_id'] = df_train['image_id']+'.jpg'\ndf_valid['image_id'] = df_valid['image_id']+'.jpg'\nfor currupted_image_id in currupted_files:\n    df_train = df_train[df_train.image_id != currupted_image_id]\n    df_valid = df_valid[df_valid.image_id != currupted_image_id]\n\n#show the dataframes\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"23c877b4-d186-48e3-94a5-f25e42c13f1d","_cell_guid":"09287592-78ae-4b2c-bddf-8eb4c7c253d5","trusted":true},"cell_type":"code","source":"df_valid.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#get all the classes\nclass_generator_df = pd.concat([df_valid,df_train])\n\nclasses = list(set(class_generator_df['category_id']))\nnum_classes = len(classes)\nclasses[:20]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3ee3723c-4f9d-4c23-a594-49fa27e5615d","_cell_guid":"b5f112c3-6cca-487a-9d85-b2b270c60911","trusted":true},"cell_type":"code","source":"#making the generator\n\ntrain_generator = gen_t.flow_from_dataframe(df_train,\n                                            x_col='image_id',\n                                            y_col='category_id',\n                                            target_size=image_size,\n                                            class_mode='categorical',\n                                            classes = classes,\n                                           directory = i[:35])\n\nvalid_generator = gen_v.flow_from_dataframe(df_valid,\n                                            x_col='image_id',\n                                            y_col='category_id',\n                                            target_size=image_size,\n                                            class_mode='categorical',\n                                            classes = classes,\n                                           directory = i[:35])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Found 174367 validated image filenames belonging to 212 classes.\nFound 43592 validated image filenames belonging to 45 classes."},{"metadata":{"_uuid":"c86c94fd-2557-4ef1-9139-70be11a8e188","_cell_guid":"08658292-fb3b-46c6-9144-c3c9f5fb41db","trusted":true},"cell_type":"markdown","source":"<h2>V0.12: preprocessing</h2>\n<p>\nAlready done.</p>"},{"metadata":{"_uuid":"9ac308ae-1a91-478a-9317-2dfc523c7ee3","_cell_guid":"e2209463-058d-42eb-b110-a501314d41c6","trusted":true},"cell_type":"markdown","source":"<h2>V1:Build the model</h2>"},{"metadata":{"_uuid":"7f126be9-6778-4d2c-92fc-ce69ead42b7a","_cell_guid":"9c543ffb-3ef9-495b-a8ff-77a47036f1b8","trusted":true},"cell_type":"markdown","source":"<h3>V1.1:transfer learning</h3>"},{"metadata":{"_uuid":"1142fea1-bd6d-49d3-9387-136fb861f32a","_cell_guid":"c1aaeeb8-c4a3-4acd-9383-27b5e9936a49","trusted":true},"cell_type":"markdown","source":"<h4>V1.11:resnet50</h4>"},{"metadata":{"_uuid":"1bcea1bb-ebce-4fc4-9354-31f1895722d7","_cell_guid":"299fcfe2-7eb5-4e42-b71f-89055f240471","trusted":true},"cell_type":"markdown","source":"<h5>V1.111:resnet50v1</h5>"},{"metadata":{"_uuid":"b5490262-53c0-4b51-b893-606ef71dd4cd","_cell_guid":"17d58601-a723-4305-a357-1ac46821ec40","trusted":true},"cell_type":"code","source":"if accelerator == 'tpu':\n    # instantiating the model in the strategy scope creates the model on the TPU\n    with tpu_strategy.scope():\n        model = tf.keras.models.Sequential()\n        model.add(tf.keras.applications.ResNet50(include_top = False, \n                                              pooling = 'avg', \n                                              weights = 'imagenet'))\n        model.add(tf.keras.layers.Dense(num_classes,\n                                     activation='softmax'))\n        model.layers[0].trainable = False\n        model.compile(optimizer = 'adam',\n                     loss = 'categorical_crossentropy',\n                     metrics=['accuracy'])\n        \nelse:\n    model = tf.keras.models.Sequential()\n    model.add(tf.keras.applications.ResNet50(include_top = False, \n                                          pooling = 'avg', \n                                          weights = 'imagenet'))\n    model.add(tf.keras.layers.Dense(num_classes,\n                                 activation='softmax'))\n    model.layers[0].trainable = False\n    model.compile(optimizer = 'adam',\n                 loss = 'categorical_crossentropy',\n                 metrics=['accuracy'])\n\n\n        \nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e874778b-75b3-46cf-a1cf-a2f645f66278","_cell_guid":"c0fb0f5a-8837-4b6b-bf0b-fe9cb5c553cd","trusted":true},"cell_type":"code","source":"#overfit on 1 sample\nov_data,ov_label = next(train_generator)\nov_history = model.fit(ov_data,ov_label,\n                       epochs=50,\n                      verbose = 0)\npd.DataFrame.from_dict(ov_history.history).plot()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"81f056cd-205f-4488-be53-d83e6ab08fd4","_cell_guid":"ed381bfd-cc9e-4083-8654-b2a7f924e4d9","trusted":true},"cell_type":"code","source":"#callbacks\ncsvlogger = tf.keras.callbacks.CSVLogger('./v1_1_1_log.csv',append = True)\nchk_point = tf.keras.callbacks.ModelCheckpoint('./v1_1_1_ckp.h5',save_best_only=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7e3639b0-3a8c-4a3e-97b9-0a66a9e9de4e","_cell_guid":"8c293af1-456a-4a1f-8f47-5bac97c51000","trusted":true},"cell_type":"code","source":"\nfrom PIL import ImageFile\nImageFile.LOAD_TRUNCATED_IMAGES = True\n\nhistory_resnet50v1 = model.fit_generator(train_generator,\n                                       steps_per_epoch=stp_per_epoch,\n                                       validation_data=valid_generator,\n                                       validation_steps=val_stp,\n                                       epochs=5,\n                                       callbacks=[csvlogger,chk_point],\n                                       verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":4}