{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport tensorflow as tf\nimport tensorflow_hub as hub\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_names=os.listdir('../input/cassava-leaf-disease-classification/test_images')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_path='drive/MyDrive/Kaggle/train_images/'\ntrain_path='../input/cassava-leaf-disease-classification/train_images/'\narr=image_names\npath=test_path\nimage_add=['../input/cassava-leaf-disease-classification/test_images/'+fname for fname in arr]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from IPython.display import Image\nImage(image_add[0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"IMG_SIZE = 224\n\ndef process_image(image_path, img_size = IMG_SIZE):\n  \"\"\"\n  Takes an image file path and turns the image into a Tensor. \n  \"\"\"\n  # Read in an image file\n  image = tf.io.read_file(image_path)\n  # Turn the jpeg image into numerical Tensor with 3 colour channels (Red, Green, Blue)\n  image = tf.image.decode_jpeg(image, channels = 3)\n  # Convert the colour channel values from 0-255 to 0-1 values\n  image = tf.image.convert_image_dtype(image, tf.float32)\n  # Resize the image to our desired value (224, 224)\n  image = tf.image.resize(image,size = [img_size, img_size])\n  return image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_image_label(image_path, label):\n  \"\"\"\n  Takes an image file path name and the assosciated label,\n  processes the image and reutrns a typle of (image, label).\n  \"\"\"\n  image = process_image(image_path)\n  return image,label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BATCH_SIZE = 32\n\ndef create_data_batches(X , y = None, batch_size = BATCH_SIZE, valid_data = False, test_data = False):\n  \"\"\"\n  Creates batches of data out of image (X) and label (y) pairs.\n  Shuffles the data if it's training data but doesn't shuffle if it's validation data.\n  Also accepts test data as input (no labels).\n  \"\"\"\n  if test_data:\n    print('Creating test data batches........')\n    data = tf.data.Dataset.from_tensor_slices((tf.constant(X)))\n    data_batch = data.map(process_image).batch(batch_size)\n    return data_batch\n  \n  elif valid_data:\n    print('Creating valid data batches...........')\n    data = tf.data.Dataset.from_tensor_slices((tf.constant(X),tf.constant(y)))\n    data_batch = data.map(get_image_label).batch(batch_size)\n    return data_batch\n  \n  else:\n    print('Creating training data batches...............')\n    data = tf.data.Dataset.from_tensor_slices((tf.constant(X),tf.constant(y)))\n    data = data.shuffle(buffer_size = len(X))\n    data_batch = data.map(get_image_label).batch(batch_size)\n    return data_batch","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df=pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\ndf['image_read']= [train_path+fname for fname in df['image_id']]\nfor i in range(len(df['label'])):\n  if df['label'][i] == 0:\n    df['label'][i] = 'Cassava Bacterial Blight (CBB)'\n  elif df['label'][i] == 1:\n    df['label'][i] = 'Cassava Brown Streak Disease (CBSD)'\n  elif df['label'][i] == 2:\n    df['label'][i] = 'Cassava Green Mottle (CGM)'\n  elif df['label'][i] == 3:\n    df['label'][i] = 'Cassava Mosaic Disease (CMD)'\n  else:\n    df['label'][i] = 'Healthy'\nlabels = df['label'].to_numpy()\nunique_diseases = np.unique(labels)\nboolean_labels = [label == unique_diseases for label in labels]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"NUM_IMAGES = 1000 #@param{type:'slider', min:1000, max:10000, step:1000}\nX = df['image_read']\ny = boolean_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X[:NUM_IMAGES], y[:NUM_IMAGES],\n                                                     test_size = 0.2, random_state = 9)\nlen(X_train), len(X_val), len(y_train), len(y_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = create_data_batches(X_train,y_train)\nvalid_data = create_data_batches(X_val,y_val, valid_data=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"model = create_models()"},{"metadata":{"trusted":true},"cell_type":"code","source":"INPUT_SHAPE = [224,224,3]\n\nOUTPUT_SHAPE = len(y[0])\n\nMODEL_URL = 'https://tfhub.dev/google/imagenet/mobilenet_v2_130_224/classification/4'\ndef create_models(input_shape = INPUT_SHAPE, output_shape = OUTPUT_SHAPE, model_url = MODEL_URL):\n\n  \"\"\"\n  Trains a given model and returns the trained version.\n  \"\"\"\n\n  print('Building a model with :',model_url)\n\n  # Setup the model layers\n  model = tf.keras.models.Sequential([\n    tf.keras.layers.Conv2D(64, (3,3), activation = 'relu', padding = 'Same',input_shape = INPUT_SHAPE),\n    tf.keras.layers.MaxPooling2D(2, 2),\n    tf.keras.layers.Dropout(0.25),\n    tf.keras.layers.Conv2D(128, (3,3), activation = 'relu',padding = 'Same'),\n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Dropout(0.25),\n    tf.keras.layers.Conv2D(128, (3,3), activation = 'relu',padding = 'Same'),\n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Dropout(0.25),\n    tf.keras.layers.Conv2D(128, (3,3), activation = 'relu',padding = 'Same'),\n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Dropout(0.25),\n    tf.keras.layers.Conv2D(256, (3,3), activation = 'relu',padding = 'Same'),\n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Dropout(0.25),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(1024,activation = 'relu'),\n    tf.keras.layers.Dense(5, activation = 'softmax')\n])\n  # Compile the model\n  model.compile(\n      loss = tf.keras.losses.CategoricalCrossentropy(),\n      optimizer = tf.keras.optimizers.Adam(),\n      metrics = ['accuracy']\n  )\n\n  # Build the model\n  model.build(input_shape)\n  \n  return model\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor = 'val_accuracy', patience=3)\ndef train_model():\n  model = create_models()\n\n\n  model.fit(x = train_data, epochs = NUM_EPOCH,\n            validation_data = valid_data, validation_freq = 1,\n            callbacks = early_stopping)\n  \n  return model\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"NUM_EPOCH = 30 #@param {type:'slider', min:10, max:100, step:10}\nmodel = train_model()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"inp=create_data_batches(image_add,test_data=True)\nfinal_pred=np.argmax(p,axis=1)"},{"metadata":{"trusted":true},"cell_type":"code","source":"inp=create_data_batches(image_add,test_data=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"p=model.predict(inp)"},{"metadata":{},"cell_type":"markdown","source":"Xb=create_data_batches(X[:10],test_data=True)\np=model.predict(Xb)"},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_model(model_path):\n  \"\"\"\n  Loads a saved model from a specified path.\n  \"\"\"\n  print(f\"Loading saved model from: {model_path}\")\n  model = tf.keras.models.load_model(model_path, \n                                     custom_objects={'KerasLayer': hub.KerasLayer})\n  return model","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"loaded_full_model = load_model('../input/model-file/20210216-110450-full-image-set-model.h5')\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"pred=model.predict(inp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"final_pred=np.argmax(pred,axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub=pd.read_csv('../input/cassava-leaf-disease-classification/sample_submission.csv')\nOUTPUT_DIR = './'\nsubs=pd.DataFrame(columns=sub.columns)\nsubs['image_id']=arr\nsubs['label']=final_pred\nsubs.to_csv(OUTPUT_DIR+'submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"subs.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}