{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/cassava-leaf-disease-classification/train.csv\")\nimage_count = train_data['image_id'].count()\ntrain_images_dir = \"/kaggle/input/cassava-leaf-disease-classification/train_images\"\ntest_images_dir = \"/kaggle/input/cassava-leaf-disease-classification/test_images\"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**reshuffle_each_itertion must be false in ds.shuffle otherwise anytime u call any operation it will shuffle the dataset. and will create overlapping while splitting the data in train and validation.**\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"list_ds = tf.data.Dataset.list_files(train_images_dir + \"/*.jpg\",shuffle = False)\nlist_ds = list_ds.shuffle(buffer_size = 1000, seed=None, reshuffle_each_iteration=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_size = int(image_count * 0.1)\ntrain_ds = list_ds.skip(val_size)\nval_ds = list_ds.take(val_size)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**many things in tensor flow run in graph mode and not in eager mode. use numpy_function or py_function to make around this.**"},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndef get_label_tensor(path):\n    path_str = str(path)\n    image_id_from_path = path_str.split('/')[-1].split('.')[0]+ \".jpg\" \n    label = train_data[train_data['image_id']== image_id_from_path]['label'].to_numpy()[0]\n    label = tf.convert_to_tensor(label, dtype=tf.int32)\n    return label","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**numpy_function output dtype was declared as [tf.int32] which was creating list of tensors instead of single tensor. replaced that with tf.int32. due to this we were getting issue with loss functions label must be1-d.**"},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_label(path):\n    label = tf.numpy_function(get_label_tensor,[path],tf.int32)\n    return label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_size = [600,600]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**after casting image to float when u plot the image it will not be identical but it will work.**"},{"metadata":{"trusted":true},"cell_type":"code","source":"def decode_img(img):\n    image = tf.image.decode_jpeg(img, channels=3) \n    image = tf.cast(image, tf.float32)/255\n    image = tf.image.resize(image,image_size )\n    return image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def process_path(path):\n    label = get_label(path)\n    img = tf.io.read_file(path)\n    img = decode_img(img)\n    return img,label\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ntrain_ds = train_ds.map(process_path,num_parallel_calls =AUTOTUNE)\nval_ds = val_ds.map(process_path,num_parallel_calls =AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def flip_left_right(image,label):\n    aug_img = tf.image.flip_left_right(image)\n    return aug_img,label\n\ndef flip_top_bottom(image,label):\n    aug_img = tf.image.flip_up_down(image)\n    return aug_img,label\n\ndef rot_90(image,label):\n    aug_img = tf.image.rot90(image)\n    return aug_img,label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_ds_left_right_aug =train_ds.map(flip_left_right,num_parallel_calls =AUTOTUNE)\ntrain_ds_top_bottom_aug =train_ds.map(flip_top_bottom,num_parallel_calls =AUTOTUNE)\ntrain_ds_rot_90_aug =train_ds.map(rot_90,num_parallel_calls =AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_ds_concat_1 = train_ds_left_right_aug.concatenate(train_ds_top_bottom_aug)\ntrain_ds_concat_2 = train_ds_concat_1.concatenate(train_ds_rot_90_aug)\ntrain_ds= train_ds.concatenate(train_ds_concat_2)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"have tried 24 batches taking too much time and overfitting tried 36 but it failed need to look for that\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"def config_for_performance(ds):\n#     ds = ds.cache()\n    ds = ds.batch(24)\n    ds = ds.prefetch(buffer_size=AUTOTUNE)\n    return ds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_ds = config_for_performance(train_ds)\nval_ds = config_for_performance(val_ds)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"****model input for resnet is n_dim =4, means batch size must be given.***"},{"metadata":{},"cell_type":"markdown","source":""},{"metadata":{},"cell_type":"markdown","source":"increasing batches didn't work so now reducing units in fully connected layer from 1000 to 600"},{"metadata":{"trusted":true},"cell_type":"code","source":"reduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2 , patience=2, min_lr=0.000001) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nbase_model = tf.keras.applications.ResNet50(weights=\"../input/keras-pretrained-models/ResNet50_NoTop_ImageNet.h5\", include_top=False)\nbase_model.trainable = False\nimg_adjust_layer = tf.keras.layers.Lambda(tf.keras.applications.resnet50.preprocess_input, input_shape=[*image_size, 3])\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.BatchNormalization(renorm=True),\n    img_adjust_layer,\n    base_model,\n    tf.keras.layers.GlobalAveragePooling2D(),\n    tf.keras.layers.Dropout(0.25),\n   # tf.keras.layers.Dense(600, activation='relu'),\n#     tf.keras.layers.Dense(1000, activation='relu'),\n    tf.keras.layers.Dense(5, activation='softmax')\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(\noptimizer = tf.keras.optimizers.Adam(),\nloss = tf.losses.SparseCategoricalCrossentropy(),\nmetrics = ['sparse_categorical_accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history = model.fit(train_ds,batch_size= 24, validation_data = val_ds, epochs =12, callbacks=[reduce_lr])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_list_ds =  tf.data.Dataset.list_files(test_images_dir + \"/*.jpg\",shuffle = False)\ntest_image_count = len(list(test_list_ds.as_numpy_iterator()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_id_num_tensor(path):\n    path_str = str(path)\n    image_id_from_path = path_str.split('/')[-1].split('.')[0]+ \".jpg\" \n    id_num = tf.convert_to_tensor(image_id_from_path, dtype=tf.string)\n    return id_num","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_id_num(path):\n    id_num = tf.numpy_function(get_id_num_tensor,[path],tf.string)\n    return id_num","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def test_path_process(path):\n    id_num = get_id_num(path)\n    test_img = tf.io.read_file(path)\n    test_img = decode_img(test_img)\n    return test_img,id_num","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_ds = test_list_ds.map(test_path_process,num_parallel_calls =AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_image_ds = test_ds.map(lambda image,id_num : image)\ntest_ids_ds = test_ds.map(lambda image,id_num : id_num)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_image_ds = test_image_ds.batch(24)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = model.predict(test_image_ds)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = np.argmax(predictions,axis=-1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions_output = list(predictions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_ids_list = list(test_ids_ds.as_numpy_iterator())\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_ids_output =[]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in test_ids_list:\n    test_ids_output.append(str(i).split(\"'\")[1])\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.DataFrame(list(zip(test_ids_output,predictions_output)), columns = [\"image_id\",\"label\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv',index =False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}