{"cells":[{"metadata":{},"cell_type":"markdown","source":"This is my first CNN network, so for some of you it can be too simple, but my aim is to explain every step we need to do to create one."},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib.image import imread\n\nimport os\nfrom PIL import Image\n\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Conv2D, MaxPool2D, Dropout, Flatten\nfrom tensorflow.keras.callbacks import EarlyStopping","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tf.test.is_gpu_available(cuda_only=False, min_cuda_compute_capability=None)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"des = pd.read_json('/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json',\n                  lines=True)\nlabels = des.T\nlabels.columns = [\"labels_description\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_train_path = \"/kaggle/input/cassava-leaf-disease-classification/train_images\"\nimg_test_path = \"/kaggle/input/cassava-leaf-disease-classification/test_images\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/cassava-leaf-disease-classification/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Load a image\npath = \"/kaggle/input/cassava-leaf-disease-classification/train_images/1235188286.jpg\"\n\nsingle_picture = tf.keras.preprocessing.image.load_img(path, grayscale=False, color_mode=\"rgb\", target_size=None, interpolation=\"nearest\")\nsingle_picture","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.listdir(img_train_path)[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"picture_path = img_train_path+\"/\"+\"1235188286.jpg\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"picture_path","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# imread(picture_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[train['image_id'] == \"1235188286.jpg\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.title(labels.iloc[2].values[0])\nplt.imshow(imread(picture_path))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(os.listdir(img_train_path))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### CSV train"},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(x=train['label'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train['label'] = train['label'].astype(str)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class_3_picture = img_train_path+'/'+ train[train['label']==3]['image_id'][4]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"picture_class_3 = imread(class_3_picture)\nplt.imshow(picture_class_3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Test picture\ntest_picture_path = img_test_path+\"/\"+ os.listdir(img_test_path)[0]\n\nplt.imshow(imread(test_picture_path))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Shape of our pictures"},{"metadata":{},"cell_type":"markdown","source":"It is very important to check if the pictures have the same size as CNN won't be able to work with different sizes."},{"metadata":{"trusted":true},"cell_type":"code","source":"# dim1 = []\n# dim2 = []\n\n#for image_filename in os.listdir(img_train_path):\n    \n   # img = imread(img_train_path+\"/\"+ image_filename)\n   # d1, d2, colors = img.shape\n   # dim1.append(d1)\n   # dim2.append(d2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#print(f\"Size of all pictures is: ({dim1[0]},{dim2[0]})\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"If the pictures have different size we have to resize it to e.g. mean of those dimentions. Nice way to visualize it is by seaborn jointplot."},{"metadata":{"trusted":true},"cell_type":"code","source":"SHAPE = 456\nIMAGE_SHAPE = (456, 456, 3)\nBATCH_SIZE = 15","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Manipulating images"},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# help(ImageDataGenerator)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# imread(picture_path).max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_gen = ImageDataGenerator(rotation_range=40,\n                               width_shift_range=0.1,\n                               height_shift_range=0.1,\n                               rescale=1/255, # we need to normalize the data\n                               shear_range=0.2,\n                               zoom_range=0.2,\n                               horizontal_flip=True,\n                               fill_mode=\"nearest\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(image_gen.random_transform(picture_class_3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dict_map = {int(i): label for i, label in enumerate(labels['labels_description'].values)}\ntrain[\"class_name\"] = train['label'].map(dict_map)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Split into train/test set\nfrom sklearn.model_selection import train_test_split\n\ntrain, test = train_test_split(train, test_size=0.05, random_state=45, stratify=train['class_name'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_set = image_gen.flow_from_dataframe(  \n                                         train,\n                                        directory=img_train_path,\n                                        seed=42,\n                                        x_col='image_id',\n                                        y_col='class_name',\n                                        target_size = (SHAPE, SHAPE),\n                                        class_mode='categorical',\n                                        interpolation='nearest',\n                                        shuffle = True,\n                                        batch_size = BATCH_SIZE,\n                                    )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Do the same for our test set\ndatagen_val = ImageDataGenerator(\n    preprocessing_function = tf.keras.applications.efficientnet.preprocess_input,\n)\n\ntest_set = datagen_val.flow_from_dataframe(\n    test,\n    directory=img_train_path,\n    seed=42,\n    x_col='image_id',\n    y_col='class_name',\n    target_size = (SHAPE, SHAPE),\n    class_mode='categorical',\n    interpolation='nearest',\n    batch_size=BATCH_SIZE,    \n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Creating CNN model"},{"metadata":{"trusted":true},"cell_type":"code","source":"early_stop = EarlyStopping(monitor='val_loss',mode='min', patience=2, )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"IMAGE_SHAPE","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_model():\n    \n    # Instantiate ann model\n    model = Sequential()\n    \n    # Add Convolution layer first\n    model.add(Conv2D(filters=64, kernel_size=(3,3), input_shape=IMAGE_SHAPE, activation=\"relu\"))\n    model.add(MaxPool2D(pool_size=(2,2)))\n    \n    # Add Convolution layer first\n    model.add(Conv2D(filters=128, kernel_size=(3,3), activation=\"relu\"))\n    model.add(MaxPool2D(pool_size=(2,2)))\n    \n    # Add Convolution layer first\n    model.add(Conv2D(filters=128, kernel_size=(3,3), activation=\"relu\"))\n    model.add(MaxPool2D(pool_size=(2,2)))\n    \n    # Flatten out the model\n    model.add(Flatten())\n    \n    # Add Dense layer now\n    model.add(Dense(128, activation='relu'))\n    model.add(Dropout(0.5))\n    \n    # Add output layer\n    model.add(Dense(5, activation='softmax'))\n    \n    # Compile the model\n    model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ann_model = create_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ann_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_set.image_shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ann_model.fit(train_set, \n              epochs=10, \n              validation_data=test_set, \n              callbacks=[early_stop])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ann_model.save(\"First_CNN_model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# imread(img_test_path+\"/\"+test_images[0])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Submission"},{"metadata":{"trusted":true},"cell_type":"code","source":"final_model = tf.keras.models.load_model(\"First_CNN_model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_images = os.listdir(img_test_path)\n\npredictions = []\n\nfor image in test_images:\n    img = Image.open(img_test_path+\"/\"+ image)\n    img = img.resize((SHAPE, SHAPE))\n    img = np.expand_dims(img, axis=0)\n    predictions.extend(final_model.predict(img).argmax(axis = 1))\n    \n    \nsub = pd.DataFrame({'image_id': test_images, 'label': predictions})\ndisplay(sub)\nsub.to_csv('submission.csv', index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}