{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport os\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!readlink -f ../input/sorghum-id-fgvc-9\n!ls ../input/sorghum-id-fgvc-9","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1>Load Data</h1>","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/sorghum-id-fgvc-9/train_cultivar_mapping.csv\")\nprint(train_df.shape)\ntrain_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check for null values. There is only one image missing the class and we can simply drop it.","metadata":{}},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.dropna()\ntrain_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_IMAGES = \"//kaggle/input/sorghum-id-fgvc-9/train_images\"\nTEST_IMAGES = \"../input/sorghum-id-fgvc-9/test\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1> EDA </h1>  \n<h3>Show some data</h3>","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(20, 20))\ni = 1\n\nfor img in os.listdir(TRAIN_IMAGES):\n    if (i <= 20):\n        image = cv2.imread(os.path.join(TRAIN_IMAGES,img))\n        ax = fig.add_subplot(4, 5, i)\n        plt.imshow(image[:,:,::-1])\n        plt.title(img)\n        # remove labels and ticks x-axis\n        ax.axes.xaxis.set_visible(False)\n        # for y-axis\n        ax.axes.yaxis.set_visible(False)\n        i = i + 1\n    else:\n        break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2> Let's check validity of information in datasets</h2> \nWe should make sure that labels in the train_df match images in the test folder.","metadata":{}},{"cell_type":"code","source":"print('Images in the dataframe: ', train_df.shape[0])\nprint('Cultivars in the dataframe: ', train_df['cultivar'].value_counts().shape[0])\nprint('PNG images in train folder: ',len(os.listdir(TRAIN_IMAGES)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Make sure that all labels are matching.","metadata":{}},{"cell_type":"code","source":"train_df = train_df[train_df['image'].isin(os.listdir(TRAIN_IMAGES))]\ntrain_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check are there any duplicates.","metadata":{}},{"cell_type":"code","source":"train_df.duplicated().any()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>Show amount of photos for different cultivars</h3>\nSome groups have more, and some have less data.","metadata":{}},{"cell_type":"code","source":"f = plt.figure(figsize=(10,17))\nplt.barh(train_df['cultivar'].value_counts().index,train_df['cultivar'].value_counts().values)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categories = train_df['cultivar'].unique()\ncategories","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.chdir(\"/kaggle\")\n# make directory\nos.makedirs('new_input')\nos.chdir(\"new_input\")\n\nfor category in categories:\n    os.makedirs(category)\n!ls","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create new_input folder that will be holding class folders with apropriate images.","metadata":{}},{"cell_type":"code","source":"import shutil\n\ntra = \"//kaggle/input/sorghum-id-fgvc-9/train_images\"\n\nfor img in os.listdir(tra):\n    original_image_path = os.path.join(tra,img)\n    l =train_df[train_df['image'] == img]\n    cultivar = l.iloc[0].cultivar\n    dest_image_path = os.path.join('//kaggle/new_input/', cultivar)\n    try:\n        shutil.copy(original_image_path, dest_image_path) \n    except (shutil.SameFileError):\n        print(\"Source and destination represents the same file.\")\n    # If there is any permission issue\n    except (PermissionError):\n        print(\"Permission denied.\")\n    # For other errors\n    except:\n        print(\"Error occurred while copying file.\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# just checking are images in dirs\nos.chdir(\"PI_152828\")\nos.getcwd()\n!ls","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h2>Explanation</h2>\nI am using keras image_dataset_from_directory to load images into training and validation datasets. We will assign .8 data to training and .2 data to the validation set. Test images are already separated for us.  \n\nhttps://keras.io/api/preprocessing/image/","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers\n\npath = '//kaggle/new_input'\n\ntrain_ds = tf.keras.preprocessing.image_dataset_from_directory(\n  path,\n  labels=\"inferred\",\n  label_mode = 'categorical',\n  color_mode = \"rgb\",\n  batch_size = 32,\n  image_size = (224,224),\n  seed = 1234,\n  validation_split = 0.2, \n  subset = \"training\"\n)\n\nval_ds = tf.keras.preprocessing.image_dataset_from_directory(\n  path,\n  labels=\"inferred\",\n  color_mode = \"rgb\",\n  label_mode = 'categorical',\n  batch_size = 32,\n  image_size = (224, 224),\n  seed = 1234,\n  subset = \"validation\",\n  validation_split = 0.2\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_ds)\nlen(val_ds) ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import datasets, layers, models\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras import activations","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(layers.Conv2D(32, (3, 3), activation = 'relu', input_shape = (224,224,3)))\nmodel.add(layers.MaxPooling2D((2,2)))\nmodel.add(layers.Conv2D(64, (3, 3),activation = 'relu'))\nmodel.add(layers.MaxPooling2D(2))\nmodel.add(layers.Conv2D(64, (3, 3),activation = 'relu'))\nmodel.add(layers.Flatten())\nmodel.add(layers.Dense(512, activation = 'relu'))\nmodel.add(layers.Dense(100, activation = 'softmax'))\n\n\n\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n\nmodel.fit(train_ds, epochs=10, validation_data = val_ds)\n# model.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mse_test = model.evaluate(val_ds)\nmse_test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(training_images, training_labels, epochs = 10, validation_data = (testing_images, testing_labels))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss, accuracy = model.evaluate(testing_images, testing_labels)\nprint('loss: ', loss)\nprint('accuracy: ', accuracy)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}