{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')\nimport seaborn as sns\n\nimport shutil\nimport os\n\nfrom sklearn.model_selection import train_test_split\n\n\n\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport numpy as np\nimport pandas as pd\nimport time\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.set_printoptions(suppress=True)\npd.set_option('display.max_columns',8000)\npd.set_option('display.max_rows',7000)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow_addons\n\nimport tensorflow_addons as tfa","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/plant-pathology-2021-fgvc8/train.csv\")\n\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = \"../input/plant-pathology-2021-fgvc8/train_images\"\n\ntrain[\"path\"] = TRAIN_PATH + \"/\" +  train[\"image\"]\n\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_list(string):\n    list = []\n    list.append(string)\n    return list\n\ntrain[\"labels_long\"] = train[\"labels\"].apply(get_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"labels\"] = train[\"labels\"].str.split(\" \")\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n\n\nX_train, X_test, y_train, y_test = train_test_split(train['image'], train['labels'], test_size=1800, random_state = 12,\n                                                      stratify =  train['labels'] )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = train.iloc[y_test.index]\ndata.head(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\nimport matplotlib.pyplot as plt\nimport time\nimport os\nimport numpy as np","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n\ntrain_datagen = ImageDataGenerator(rescale=1./255,rotation_range=45,width_shift_range=0.15,height_shift_range=0.15,\nshear_range=0.2,zoom_range=0.5,horizontal_flip=True,fill_mode='nearest',validation_split=0.33)\n\nval_datagen = ImageDataGenerator(rescale = 1./255,validation_split=0.33 )\n\nIMAGE_RES = 299\nBATCH_SIZE = 128\n\ntraining_set = train_datagen.flow_from_dataframe(dataframe=train,x_col='path',y_col= \"labels\",\ntarget_size = (IMAGE_RES, IMAGE_RES),batch_size = BATCH_SIZE,validate_filenames=False,subset = \"training\",seed = 12)\n\nvalidation_set = val_datagen.flow_from_dataframe(dataframe=train,x_col='path',y_col= \"labels\",target_size = (IMAGE_RES, IMAGE_RES),\nbatch_size = BATCH_SIZE,validate_filenames=False,subset = \"validation\",seed = 12)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = training_set.class_indices\n\nprint(class_names)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_classes = len(class_names)\nprint(num_classes)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = tf.keras.optimizers.Adam(lr=1e-3)\n\nEPOCHS = 2\n\nmetrics = tfa.metrics.F1Score(num_classes = num_classes,average = \"macro\",name = \"f1_score\")\n    \n\n# Create a callback that stops fitting when val loss do not decrease\ncallback = tf.keras.callbacks.EarlyStopping(monitor=\"val_f1_score\", patience=5, mode='max')\n\nreducelr = tf.keras.callbacks.ReduceLROnPlateau( monitor= \"val_f1_score\",mode='max',factor=0.1,patience=4,verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE=[299,299]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Input, Lambda, Dense, Flatten\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.applications.inception_v3 import InceptionV3\n#from keras.applications.vgg16 import VGG16\nfrom tensorflow.keras.applications.inception_v3 import preprocess_input\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator,load_img\nfrom tensorflow.keras.models import Sequential\nimport numpy as np\nfrom glob import glob","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.applications.inception_v3 import InceptionV3\n\n\ninception = InceptionV3(input_shape=IMAGE_SIZE + [3], include_top=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for layer in inception.layers:\n    layer.trainable = False","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = layers.Flatten()(inception.output)\nprediction = layers.Dense  (num_classes, activation='sigmoid')(x)\nmodel = Model( inputs=inception.input, outputs=prediction) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=optimizer,loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\nmetrics=[metrics])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(training_set,epochs=2,validation_data=validation_set,verbose=1,callbacks=[callback,reducelr])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['f1_score']\nval_acc = history.history['val_f1_score']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs_range = range(len(history.history['loss']))\n\nplt.figure(figsize=(14, 8))\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training f1_score')\nplt.plot(epochs_range, val_acc, label='Validation f1_score')\nplt.legend(loc='lower right')\nplt.title('Training and Validation f1_score')\n\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights(checkpoint_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = {v: k for k, v in class_names.items()}\nclass_names","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_sample = train.sample(500)\ndata_sample.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_datagen = ImageDataGenerator(rescale = 1./255)\n\nfull_set = full_datagen.flow_from_dataframe(dataframe=data,\nx_col='path',y_col= None,target_size = (IMAGE_RES, IMAGE_RES),batch_size = BATCH_SIZE,\nshuffle=False,validate_filenames=False,class_mode=None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_labels(prediction,threshold):\n  pred = []\n  idx = np.where(prediction>threshold)[0]\n  for i in idx:\n    pred.append(class_names[i])\n  pred = ' '. join(pred)\n  if len(pred) == 0:\n    pred = []\n    idx = np.argmax(prediction)\n    pred.append(class_names[idx])\n    pred = ' '. join(pred)\n    return pred\n  else :\n    return pred","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicts = model.predict(full_set) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thresholds = [0.1,0.2,0.3,0.4,0.5,0.6,0.7,0.8,0.9]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_macro = []\nfor t in thresholds:\n    labels = []\n    for i in range(len(predicts)):\n        pred = predicts[i]\n        labels.append(get_labels(prediction = pred, threshold = t))\n    data[\"pred\"]= labels","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MultiLabelBinarizer\nmlb = MultiLabelBinarizer()\nmlb.fit(data[\"labels\"])\ny_test = mlb.transform(data[\"labels\"])\ndata[\"pred\"] = data[\"pred\"].str.split(\" \")\ny_score = mlb.transform(data[\"pred\"]) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\nfrom scipy import interp\nfrom itertools import cycle\n\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\nfor i in range(num_classes):\n    fpr[i], tpr[i], _ = roc_curve(y_test[:, i], y_score[:, i])\n    roc_auc[i] = auc(fpr[i], tpr[i])\n\nfpr[\"micro\"], tpr[\"micro\"], _ = roc_curve(y_test.ravel(), y_score.ravel())\nroc_auc[\"micro\"] = auc(fpr[\"micro\"], tpr[\"micro\"])\nall_fpr = np.unique(np.concatenate([fpr[i] for i in range(num_classes)]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_tpr = np.zeros_like(all_fpr)\nfor i in range(num_classes):\n    mean_tpr += interp(all_fpr, fpr[i], tpr[i])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_tpr /= num_classes\nfpr[\"macro\"] = all_fpr\ntpr[\"macro\"] = mean_tpr\nroc_auc[\"macro\"] = auc(fpr[\"macro\"], tpr[\"macro\"])\nroc_auc_macro.append(round(roc_auc[\"macro\"],2))\nprint(\"Threshold:\",t,\"ROC: \",round(roc_auc[\"macro\"],2))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame()\ndf[\"thresholds\"]= thresholds\ndf[\"roc_auc_macro\"]= roc_auc_macro\ndf.sort_values(\"roc_auc_macro\", inplace=True, ascending=False)\ndf.reset_index(inplace=True, drop = True)\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"thresholds\"][0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = df[\"thresholds\"][0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir = '/kaggle/input/plant-pathology-2021-fgvc8/test_images/'\ntest = pd.DataFrame(columns = [\"image\", \"labels\",\"path\"])\ntest['image'] = os.listdir(test_dir)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.sort_values(\"image\", inplace= True)\ntest.reset_index(inplace=True, drop=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TEST_PATH = \"../input/plant-pathology-2021-fgvc8/test_images\"\n\ntest[\"path\"] = TEST_PATH + \"/\" +  test[\"image\"]\n\ntest.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_datagen = ImageDataGenerator(rescale = 1./255)\n\ntest_set = test_datagen.flow_from_dataframe(dataframe=test,\n                                    x_col='path',\n                                    y_col= None,                                    \n                                    target_size = (IMAGE_RES, IMAGE_RES),\n                                    batch_size = BATCH_SIZE,\n                                    shuffle=False,\n                                    class_mode=None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicts = model.predict(test_set)   ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = []\nfor i in range(len(predicts)):\n    pred = predicts[i]\n    labels.append(get_labels(prediction = pred, threshold = threshold))\n    \ntest[\"labels\"] = labels\ntest.head()\nsubmission_1 = test.drop(\"path\", axis = 1)\nsubmission_1.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_1.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}