{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport PIL\nimport tensorflow as tf\n\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.layers import Conv2D,MaxPooling2D,Dense,Dropout,BatchNormalization,Flatten\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.listdir('../input/imet-2020-fgvc7')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data=pd.read_csv('../input/imet-2020-fgvc7/train.csv')\nlabels_data=pd.read_csv('../input/imet-2020-fgvc7/labels.csv')\nsample_submission=pd.read_csv('../input/imet-2020-fgvc7/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_data.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.head","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_submission.head","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_submission.info","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data['id'] += '.png'\nsample_submission['id']+= '.png'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data['attribute_ids']=train_data['attribute_ids'].apply(lambda x: x.split())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.head(5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Image Preprocessing**\nImage Data generator provides the easy way to augment your images"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_datagen=tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255,\n                                 shear_range=0.2,\n                                 zoom_range=0.2,\n                                 horizontal_flip=True,\n                                 validation_split=0.2,                             \n                                 fill_mode='nearest'                             \n                                    )\n\ntest_datagen=tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size=32\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Flow from dataframe is a method in ImageDataGenerator class that allows you to directly augment images by reading its name and target value from dataframe**"},{"metadata":{"trusted":true},"cell_type":"code","source":"train_ds=train_datagen.flow_from_dataframe(dataframe=train_data,\n                                          directory=\"/kaggle/input/imet-2020-fgvc7/train\",\n                                          x_col='id',\n                                          y_col='attribute_ids',\n                                          class_mode='categorical',\n                                          subset='training',\n                                          seed=123,\n                                          shuffle=True,\n                                          batch_size=batch_size,\n                                          target_size=(128,128)\n                                          )\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valid_ds=train_datagen.flow_from_dataframe(dataframe=train_data,\n                                          directory=\"/kaggle/input/imet-2020-fgvc7/train\",\n                                          x_col='id',\n                                          y_col='attribute_ids',\n                                          class_mode='categorical',\n                                          subset='validation',\n                                          seed=123,\n                                          shuflle=True, \n                                          batch_size=batch_size,\n                                          target_size=(128,128)\n                                          )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_ds=test_datagen.flow_from_dataframe(dataframe=sample_submission,\n                                        directory=\"/kaggle/input/imet-2020-fgvc7/test\",\n                                        x_col='id',\n                                        batch_size=batch_size,\n                                        shuffle=False,\n                                        class_mode=None,\n                                        target_size=(128,128))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for image_batch,labels_batch in train_ds:\n    print(image_batch.shape)\n    print(labels_batch.shape)\n    break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"input_shape=(128,128,3)\n\nmodel=Sequential()\n\nmodel.add(Conv2D(16,3,padding='same',input_shape=input_shape,activation='relu'))\nmodel.add(MaxPooling2D())\nmodel.add(Conv2D(32,3 ,padding='same',activation='relu'))\nmodel.add(MaxPooling2D())\nmodel.add(Conv2D(64,3 ,padding='same',activation='relu'))\nmodel.add(MaxPooling2D())\nmodel.add(Dropout(0.2))\nmodel.add(Dense(128,activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\nmodel.add(Flatten())\nmodel.add(Dense(512,activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.2))\nmodel.add(Dense(3471,activation='sigmoid'))\n\nmodel.compile(optimizer='adam',loss='categorical_crossentropy',metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**Training the model**"},{"metadata":{"trusted":true},"cell_type":"code","source":"type(train_ds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"epochs=10\nhistory=model.fit(train_ds,epochs=epochs,steps_per_epoch=200,\n                            validation_data=valid_ds,validation_steps=80,\n                            verbose=1,callbacks=None,\n                           use_multiprocessing=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"accuracy=history.history['accuracy']\nval_accuracy=history.history['val_accuracy']\n\n\nloss=history.history['loss']\nval_loss=history.history['val_loss']\n\nepochs_range=range(epochs)\n\n\nplt.figure(figsize=(8,8))\nplt.subplot(1,2,1)\nplt.plot(epochs_range,accuracy,label='Training Accuracy')\nplt.plot(epochs_range,val_accuracy,label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and validation accuracy')\n\n\n\nplt.subplot(1,2,2)\nplt.plot(epochs_range,loss,label='Training Loss')\nplt.plot(epochs_range,val_loss,label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and validation loss')\n\n\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions=model.predict(test_ds,verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_boolean=(predictions>0.2)\n\nresult=[]\n\nlabels=train_ds.class_indices\n\nlabels=dict((x,y) for y,x in labels.items())\n\nfor i in pred_boolean:\n    list_labels=[]\n    for j,k in enumerate(i):\n        if k:\n            list_labels.append(labels[j])\n    result.append( \" \".join(list_labels))\n\n    \nimagenames=test_ds.filenames\n\nsubmission=pd.DataFrame({\"id\":imagenames,\"attribute_ids\":result})\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission['id']=submission['id'].apply( lambda x: x.split('.')[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}