{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport tensorflow as tf\nfrom tensorflow.keras import models","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_path = \"../input/plant-pathology-2021-fgvc8/train_images/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv = pd.read_csv(\"../input/plant-pathology-2021-fgvc8/train.csv\")\ntrain_csv.tail(7)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(len(train_csv))\n\nprint(len(os.listdir(\"../input/plant-pathology-2021-fgvc8/train_images\")))\nprint(len(os.listdir(\"../input/plant-pathology-2021-fgvc8/test_images\")))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.labels.nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.labels.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, axarr = plt.subplots(2,2)\nidx = [np.random.randint(18632) for _ in range(4)]\naxarr[0,0].imshow(plt.imread(train_path + train_csv.iloc[idx[0],0]))\naxarr[0,1].imshow(plt.imread(train_path + train_csv.iloc[idx[1],0]))\naxarr[1,0].imshow(plt.imread(train_path + train_csv.iloc[idx[2],0]))\naxarr[1,1].imshow(plt.imread(train_path + train_csv.iloc[idx[3],0]))\n\nprint(train_csv.iloc[idx[0],1],\",\", train_csv.iloc[idx[1],1],\",\", train_csv.iloc[idx[2],1],\",\", train_csv.iloc[idx[3],1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in range(50):\n    print(plt.imread(train_path + train_csv.iloc[i,0]).shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_augmentation = tf.keras.Sequential(\n    [\n        tf.keras.layers.experimental.preprocessing.RandomContrast(0.2),\n        #tf.keras.layers.experimental.preprocessing.Resizing(3000, 4000),\n        tf.keras.layers.experimental.preprocessing.RandomFlip(\"horizontal_and_vertical\"),\n        tf.keras.layers.experimental.preprocessing.RandomRotation(0.3),\n        tf.keras.layers.experimental.preprocessing.RandomZoom(0.2),\n    ]\n)\n\n# Add the image to a batch\nimage = tf.expand_dims(plt.imread(train_path + train_csv.iloc[idx[0],0]), 0)\n\nplt.figure(figsize=(10, 10))\nfor i in range(9):\n  augmented_image = data_augmentation(image)\n  ax = plt.subplot(3, 3, i + 1)\n  plt.imshow(augmented_image[0])\n  plt.axis(\"off\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#!cp -r \"../input/plant-pathology-2021-fgvc8/train_images/\" \"./train_images/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!mkdir train_images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tqdm import tqdm\n\nsample = pd.DataFrame(train_csv.labels.value_counts())#.loc[\"scab\"]\ncols = {i for i in train_csv.labels if sample.loc[i][0] < 700}\n\naug = dict()\n\nfor col in cols:\n    aug[col] = []\n    \n    for i in tqdm(range(18632), total=18632):\n        if train_csv.iloc[i, 1] == col:\n            aug[col].append(train_csv.iloc[i, 0])\n            ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def combine(val):\n    return val[0]+\"_\"+val[1]\n    \nfor i in aug:\n    print(combine(i.split(\" \")))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"augmentations = tf.keras.Sequential([\n    \n    tf.keras.layers.experimental.preprocessing.RandomContrast(0.3),\n    tf.keras.layers.experimental.preprocessing.Resizing(3000, 4000),\n    tf.keras.layers.experimental.preprocessing.RandomFlip(mode=\"horizontal_and_vertical\"),\n    tf.keras.layers.experimental.preprocessing.RandomRotation(0.5, fill_mode='reflect', interpolation='bilinear'),\n    tf.keras.layers.experimental.preprocessing.RandomZoom(0.4),\n    #tf.keras.layers.experimental.preprocessing.Rescaling(1./255, offset=0.0),#[0,1]\n    #tf.keras.layers.experimental.preprocessing.Rescaling(1./127.5, offset=-1)#[-1,1]\n    \n])\nout_path = \"./train_images/\"\n\nfor i in aug:\n    rem = 1000 - len(aug[i])\n    for j in tqdm(range(rem), total=rem):\n        image = np.random.choice(aug[i])\n        image = tf.expand_dims(plt.imread(train_path + image), 0)\n        augmented_image = augmentations(image)\n        img_name = combine(i.split(\" \"))\n        img_path = out_path + f\"{img_name}_{j}.jpg\"\n        train_csv = train_csv.append(pd.DataFrame({\"image\":f\"{img_name}_{j}.jpg\", \"labels\":i},index=[0]), ignore_index=True)\n        tf.keras.preprocessing.image.save_img(img_path, np.squeeze(augmented_image), data_format=\"channels_last\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv.to_csv(\"train_aug.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(os.listdir(\"./train_images\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!zip -r aug_images.zip \"./\"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}