{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom pathlib import Path\nimport os.path\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport os\nimport cv2\nimport pandas as pd\nimport random\nimport os\nimport PIL\nimport tensorflow as tf\nfrom keras.optimizers import Adam\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '../input/plant-pathology-2021-fgvc8'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(r'../input/plant-pathology-2021-fgvc8/train.csv', index_col='image')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.labels.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,12))\nlabels = sns.barplot(train_df.labels.value_counts().index,train_df.labels.value_counts())\nfor item in labels.get_xticklabels():\n    item.set_rotation(45)\nplt.title('Label Distribution', weight='bold')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img_Path = '../input/plant-pathology-2021-fgvc8/train_images'\ntest_img_Path = '../input/plant-pathology-2021-fgvc8/test_images'\nsample_submission = pd.read_csv(r'../input/plant-pathology-2021-fgvc8/sample_submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Remove duplicates","metadata":{}},{"cell_type":"code","source":"init_len = len(train_df)\n\nwith open('../input/duplicates-image/duplicates.csv', 'r') as file:\n    duplicates = [x.strip().split(',') for x in file.readlines()]\n\nfor row in duplicates:\n   \n    unique_labels = train_df.loc[row].drop_duplicates().values\n    \n    if len(unique_labels) == 1:\n        train_df = train_df.drop(row[1:], axis=0)\n    else:\n        train_df = train_df.drop(row, axis=0)\n        \nprint(f'Dropping {init_len - len(train_df)} duplicate samples.')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['labels'] = train_df['labels'].apply(lambda s: s.split(' '))\ntrain_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfig1 = plt.figure(figsize=(20,10))\n\nfor i in range(1, 10):\n    \n    rand =  random.randrange(1, 18000)\n    sample = os.path.join('../input/plant-pathology-2021-fgvc8/train_images', train_df.index[rand])\n    \n    img = PIL.Image.open(sample)\n    \n    ax = fig1.add_subplot(4,3,i)\n    ax.imshow(img)\n    \n    title = f\"{train_df['labels'][rand]}{img.size}\"\n    plt.title(title)\n    \n    fig1.tight_layout()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = '../input/resized-plant2021/img_sz_256/'\nTEST_PATH = '../input/plant-pathology-2021-fgvc8/test_images/'\n\nIMG_RES = 256","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resized_train = pd.read_csv('../input/plant-pathology-2021-fgvc8/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resized_train.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resized_train['labels'] = resized_train['labels'].apply(lambda s: s.split(' '))\nresized_train.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trans_label = MultiLabelBinarizer().fit(resized_train['labels'])\nlabels = pd.DataFrame(trans_label.transform(resized_train['labels']), columns=trans_label.classes_)\ntrain_df = pd.concat([resized_train['image'], labels], axis=1)\ntrain_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resized_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\ndatagen = keras.preprocessing.image.ImageDataGenerator(rescale=1/255.0,\n                                                        preprocessing_function=None,\n                                                        data_format=None,\n                                                        validation_split= 0.2\n                                                    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = datagen.flow_from_dataframe(\n    resized_train,\n    directory = '../input/resized-plant2021/img_sz_256',\n    x_col = 'image',\n    y_col = 'labels',\n    subset=\"training\",\n    color_mode=\"rgb\",\n    target_size = (224,224),\n    class_mode=\"categorical\",\n    batch_size=32,\n    shuffle=False,\n    seed=40,\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_data = datagen.flow_from_dataframe(\n    resized_train,\n    directory = '../input/resized-plant2021/img_sz_256',\n    x_col = 'image',\n    y_col = 'labels',\n    subset=\"validation\",\n    color_mode=\"rgb\",\n    target_size = (224,224),\n    class_mode=\"categorical\",\n    batch_size=32,\n    shuffle=False,\n    seed=40,\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\nmlb = MultiLabelBinarizer()\nhot_labels = mlb.fit_transform(resized_train['labels'])\nprint(mlb.classes_)\nprint(hot_labels)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_labels = pd.DataFrame(hot_labels,columns=mlb.classes_,index=resized_train.index)\ndf_labels","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,10))\nsns.barplot(x=df_labels.columns,y=df_labels.sum().values)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **DenseNet 169**","metadata":{}},{"cell_type":"code","source":"from keras.applications import InceptionResNetV2\nfrom keras.applications import MobileNetV2\nfrom keras.applications import DenseNet121\nfrom keras.applications import DenseNet169\n\nimport keras\nfrom keras.layers import Dense,Dropout,Flatten\nfrom tensorflow.keras.layers import GlobalAveragePooling2D\nfrom keras.models import Model\nfrom tensorflow.keras.callbacks import EarlyStopping\nimport tensorflow_addons as tfa\n\nweight_path='../input/tf-keras-pretrained-model-weights/No Top/densenet169_weights_tf_dim_ordering_tf_kernels_notop.h5'\nbase_model=DenseNet169(weights=weight_path,include_top=False, input_shape=(224,224,3))\nx=base_model.output\nx=GlobalAveragePooling2D()(x)\nx=Dense(128,activation='relu')(x)\nx=Dropout(0.2)(x)\nx=Dense(64,activation='relu')(x)\npredictions=Dense(6,activation='sigmoid')(x)\n\nmodel=Model(inputs=base_model.input,outputs=predictions)\n\nfor layer in base_model.layers:\n    layer.trainable=False","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1 = tfa.metrics.F1Score(num_classes=6,average='macro')\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy',metrics=['accuracy',f1])\nes=EarlyStopping(patience=4,monitor=f1,mode='max',restore_best_weights=True)\nhist = model.fit_generator(generator=train_data,\n                    validation_data=valid_data,\n                    epochs=20,\n                    steps_per_epoch=train_data.samples//128,\n                    validation_steps=valid_data.samples//128,\n                    callbacks=[es])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.layers[595:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in model.layers[:595]:\n    layer.trainable=False\n\nfor layer in model.layers[143:]:\n    layer.trainable=True\n\nfor layer in model.layers[595:]:\n    layer.trainable=False\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy',metrics=['accuracy',f1])\nhistory = model.fit_generator(generator=train_data,\n                    validation_data=valid_data,\n                    epochs=20,\n                    steps_per_epoch=train_data.samples//128,\n                    validation_steps=valid_data.samples//128,\n                    callbacks=[es])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nepoch_list = list(range(1, len(history.history['accuracy']) + 1))\nplt.plot(epoch_list, history.history['accuracy'],label='accuracy')\nplt.plot(epoch_list, history.history['val_accuracy'],label='val_accuracy')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path=\"../input/plant-pathology-2021-fgvc8/sample_submission.csv\"\ntest = pd.read_csv(test_path)\ntest","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = datagen.flow_from_dataframe(\n    test,\n    directory='../input/plant-pathology-2021-fgvc8/test_images',\n    x_col='image',\n    y_col=None,\n    color_mode='rgb',\n    target_size=(224,224),\n    class_mode=None,\n    shuffle=False\n)\npredictions = model.predict(test_data)\nprint(predictions)\n\nclass_idx=[]\nfor pred in predictions:\n    pred=list(pred)\n    temp=[]\n    for i in pred:\n        if (i>0.4):\n            temp.append(pred.index(i))\n    if (temp!=[]):\n        class_idx.append(temp)\n    else:\n        temp.append(np.argmax(pred))\n        class_idx.append(temp)\nprint(class_idx)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_dict = train_data.class_indices\ndef get_key(val):\n    for key,value in class_dict.items():\n        if (val==value):\n            return key\nprint(class_dict)\n\nsub_pred=[]\nfor img_ in class_idx:\n    img_pred=[]\n    for i in img_:\n        img_pred.append(get_key(i))\n    sub_pred.append( ' '.join(img_pred))\nprint(sub_pred)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = test[['image']]\nsub['labels']=sub_pred\nsub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}