{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow.keras as keras\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nimport cv2\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport imagehash\nimport PIL\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-04-26T15:04:50.477705Z","iopub.execute_input":"2022-04-26T15:04:50.478046Z","iopub.status.idle":"2022-04-26T15:04:56.742878Z","shell.execute_reply.started":"2022-04-26T15:04:50.477967Z","shell.execute_reply":"2022-04-26T15:04:56.742102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#config\ntrain_dir = \"../input/plant-pathology-2021-fgvc8/train_images/\"\ntrain_df = pd.read_csv(\"../input/plant-pathology-2021-fgvc8/train.csv\")\n\nclasses = ['scab frog_eye_leaf_spot', 'scab frog_eye_leaf_spot complex', 'complex', 'rust frog_eye_leaf_spot', \n           'powdery_mildew complex', 'powdery_mildew', 'frog_eye_leaf_spot complex', \n           'rust complex', 'rust', 'frog_eye_leaf_spot', 'scab', 'healthy']\n\npaths = train_df['image']\ny_train = train_df['labels']\ny_train = [classes.index(i) for i in y_train]\ny_train = keras.utils.to_categorical(y_train,12)\n\nclass CFG():\n    threshold = .9\n    img_size = 112\n    seed = 42","metadata":{"execution":{"iopub.status.busy":"2022-04-26T15:05:11.958386Z","iopub.execute_input":"2022-04-26T15:05:11.958932Z","iopub.status.idle":"2022-04-26T15:05:12.006034Z","shell.execute_reply.started":"2022-04-26T15:05:11.958895Z","shell.execute_reply":"2022-04-26T15:05:12.005359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#read list of image\n# x_train = []\n# for path in tqdm(paths, total=len(paths)):\n#     image = tf.io.read_file(os.path.join(train_dir, path))\n#     image = tf.image.decode_jpeg(image, channels=3)\n#     image = tf.image.resize(image, [112, 112])\n#     image = tf.cast(image, tf.uint8).numpy()\n#     x_train.append(image)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-26T13:40:07.058222Z","iopub.execute_input":"2022-04-26T13:40:07.058472Z","iopub.status.idle":"2022-04-26T13:41:52.123804Z","shell.execute_reply.started":"2022-04-26T13:40:07.058444Z","shell.execute_reply":"2022-04-26T13:41:52.122035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #reshape\n# x_train = np.array(x_train)\n# x_train = x_train.reshape(-1,112,112,3)\n\n# print(len(x_train))\n# print(len(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-04-26T14:25:33.345228Z","iopub.execute_input":"2022-04-26T14:25:33.34577Z","iopub.status.idle":"2022-04-26T14:25:33.637287Z","shell.execute_reply.started":"2022-04-26T14:25:33.345718Z","shell.execute_reply":"2022-04-26T14:25:33.635882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pickle\n\n# filehandler = open(\"x_train.pkl\",\"wb\") \n# pickle.dump(x_train,filehandler) \n# filehandler.close()","metadata":{"execution":{"iopub.status.busy":"2022-04-26T14:13:50.471335Z","iopub.execute_input":"2022-04-26T14:13:50.471897Z","iopub.status.idle":"2022-04-26T14:13:51.291254Z","shell.execute_reply.started":"2022-04-26T14:13:50.471856Z","shell.execute_reply":"2022-04-26T14:13:51.29028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n\nfilehandler = open(\"../input/resizeddata/x_train2.pkl\",\"rb\") \nx_train = pickle.load(filehandler) \nfilehandler.close()","metadata":{"execution":{"iopub.status.busy":"2022-04-26T15:05:20.555071Z","iopub.execute_input":"2022-04-26T15:05:20.55533Z","iopub.status.idle":"2022-04-26T15:05:26.621446Z","shell.execute_reply.started":"2022-04-26T15:05:20.555302Z","shell.execute_reply":"2022-04-26T15:05:26.620715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train, x_val, y_train, y_val = train_test_split(x_train, y_train, test_size=0.1, random_state=42)\nx_train, x_test, y_train, y_test = train_test_split(x_train, y_train, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-04-26T15:05:28.54028Z","iopub.execute_input":"2022-04-26T15:05:28.540734Z","iopub.status.idle":"2022-04-26T15:05:29.032594Z","shell.execute_reply.started":"2022-04-26T15:05:28.540697Z","shell.execute_reply":"2022-04-26T15:05:29.031676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-26T15:05:31.45016Z","iopub.execute_input":"2022-04-26T15:05:31.450699Z","iopub.status.idle":"2022-04-26T15:05:31.459068Z","shell.execute_reply.started":"2022-04-26T15:05:31.450659Z","shell.execute_reply":"2022-04-26T15:05:31.458259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Model\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import (\n    Dense,\n    Conv2D,\n    MaxPool2D,\n    Flatten,\n    Dropout,\n    BatchNormalization,  \n    RandomFlip,\n    RandomRotation,\n    InputLayer\n)\n\nmodel = Sequential()\nmodel.add(InputLayer(input_shape=(112, 112, 3))),\n\nmodel.add(Conv2D(64, (3, 3), strides=1, padding=\"same\", activation=\"relu\"))\nmodel.add(MaxPool2D((2, 2), strides=2, padding=\"same\"))\nmodel.add(BatchNormalization())\nmodel.add(Conv2D(64, (3, 3), strides=1, padding=\"same\", activation=\"relu\"))\nmodel.add(MaxPool2D((2, 2), strides=2, padding=\"same\"))\nmodel.add(BatchNormalization())\nmodel.add(Conv2D(128, (3, 3), strides=1, padding=\"same\", activation=\"relu\"))\nmodel.add(MaxPool2D((2, 2), strides=2, padding=\"same\"))\nmodel.add(BatchNormalization())\nmodel.add(Conv2D(128, (3, 3), strides=1, padding=\"same\", activation=\"relu\"))\nmodel.add(MaxPool2D((2, 2), strides=2, padding=\"same\"))\nmodel.add(BatchNormalization())\nmodel.add(Conv2D(256, (3, 3), strides=1, padding=\"same\", activation=\"relu\"))\nmodel.add(MaxPool2D((2, 2), strides=2, padding=\"same\"))\nmodel.add(BatchNormalization())\nmodel.add(Conv2D(256, (3, 3), strides=1, padding=\"same\", activation=\"relu\"))\nmodel.add(MaxPool2D((2, 2), strides=2, padding=\"same\"))\nmodel.add(BatchNormalization())\nmodel.add(Conv2D(512, (3, 3), strides=1, padding=\"same\", activation=\"relu\"))\nmodel.add(MaxPool2D((2, 2), strides=2, padding=\"same\"))\nmodel.add(BatchNormalization())\n\nmodel.add(Flatten())\nmodel.add(Dense(units=512, activation=\"relu\"))\nmodel.add(Dense(units=256, activation=\"relu\"))\nmodel.add(Dense(units=12, activation=\"softmax\"))\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-04-26T15:05:34.398792Z","iopub.execute_input":"2022-04-26T15:05:34.39922Z","iopub.status.idle":"2022-04-26T15:05:37.053246Z","shell.execute_reply.started":"2022-04-26T15:05:34.399185Z","shell.execute_reply":"2022-04-26T15:05:37.052589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.image import ImageDataGenerator\n# With data augmentation to prevent overfitting (accuracy 0.99286)\ndatagen = ImageDataGenerator(\n        featurewise_center=False,  # set input mean to 0 over the dataset\n        samplewise_center=False,  # set each sample mean to 0\n        featurewise_std_normalization=False,  # divide inputs by std of the dataset\n        samplewise_std_normalization=False,  # divide each input by its std\n        zca_whitening=True,  # apply ZCA whitening\n        rotation_range=45,  # randomly rotate images in the range (degrees, 0 to 180)\n        zoom_range = 0.1, # Randomly zoom image \n        width_shift_range=0.1,  # randomly shift images horizontally (fraction of total width)\n        height_shift_range=0.1,  # randomly shift images vertically (fraction of total height)\n        horizontal_flip=True,  # randomly flip images\n        vertical_flip=True)  # randomly flip images\n","metadata":{"execution":{"iopub.status.busy":"2022-04-26T15:05:42.621763Z","iopub.execute_input":"2022-04-26T15:05:42.622007Z","iopub.status.idle":"2022-04-26T15:05:42.630964Z","shell.execute_reply.started":"2022-04-26T15:05:42.621979Z","shell.execute_reply":"2022-04-26T15:05:42.629976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Compiling the Model\nopt = keras.optimizers.Adam(learning_rate=0.001)\nmodel.compile(loss='categorical_crossentropy', optimizer=opt, metrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2022-04-26T15:05:50.017959Z","iopub.execute_input":"2022-04-26T15:05:50.018369Z","iopub.status.idle":"2022-04-26T15:05:50.034303Z","shell.execute_reply.started":"2022-04-26T15:05:50.018335Z","shell.execute_reply":"2022-04-26T15:05:50.033487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the model\nhistory = model.fit_generator(datagen.flow(x_train,y_train, batch_size=64),\n                              epochs = 10, validation_data = (x_val,y_val),\n                              verbose = 1, steps_per_epoch=x_train.shape[0] // 128\n                              )","metadata":{"execution":{"iopub.status.busy":"2022-04-26T16:06:08.83193Z","iopub.execute_input":"2022-04-26T16:06:08.83259Z","iopub.status.idle":"2022-04-26T16:06:53.228732Z","shell.execute_reply.started":"2022-04-26T16:06:08.832553Z","shell.execute_reply":"2022-04-26T16:06:53.224498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Evaluate on test data\")\nresults = model.evaluate(x_test, y_test, batch_size=128)\nprint(\"test loss, test acc:\", results)","metadata":{"execution":{"iopub.status.busy":"2022-04-26T16:06:58.904963Z","iopub.execute_input":"2022-04-26T16:06:58.905299Z","iopub.status.idle":"2022-04-26T16:06:59.584842Z","shell.execute_reply.started":"2022-04-26T16:06:58.905268Z","shell.execute_reply":"2022-04-26T16:06:59.584102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n#Create the prediction\npredictions = model.predict(x_test)\npredictions = np.argmax(predictions, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-04-26T16:07:07.46982Z","iopub.execute_input":"2022-04-26T16:07:07.470063Z","iopub.status.idle":"2022-04-26T16:07:08.38835Z","shell.execute_reply.started":"2022-04-26T16:07:07.470036Z","shell.execute_reply":"2022-04-26T16:07:08.387602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"con_mat = tf.math.confusion_matrix(np.argmax(y_test, axis=1), predictions, num_classes=12)\ncon_mat","metadata":{"execution":{"iopub.status.busy":"2022-04-26T16:07:22.902551Z","iopub.execute_input":"2022-04-26T16:07:22.902797Z","iopub.status.idle":"2022-04-26T16:07:22.914993Z","shell.execute_reply.started":"2022-04-26T16:07:22.90277Z","shell.execute_reply":"2022-04-26T16:07:22.914118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nplt.figure(figsize = (13,10))\nsns.heatmap(con_mat, annot=True, cmap='twilight_shifted_r', fmt='g')","metadata":{"execution":{"iopub.status.busy":"2022-04-26T16:07:28.428521Z","iopub.execute_input":"2022-04-26T16:07:28.429154Z","iopub.status.idle":"2022-04-26T16:07:29.254405Z","shell.execute_reply.started":"2022-04-26T16:07:28.429116Z","shell.execute_reply":"2022-04-26T16:07:29.253564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#read test\ntest_df = pd.read_csv(\"../input/digit-recognizer/test.csv\")\n#get value in and store \nx_test = test_df.values\n\n# Normalize our image data\nx_test = x_test / 255\n\n#reshape\nx_test = x_test.reshape(-1,28,28,1)","metadata":{"execution":{"iopub.status.busy":"2022-04-26T16:07:40.918549Z","iopub.execute_input":"2022-04-26T16:07:40.91901Z","iopub.status.idle":"2022-04-26T16:07:40.956276Z","shell.execute_reply.started":"2022-04-26T16:07:40.918974Z","shell.execute_reply":"2022-04-26T16:07:40.955318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n#Create the prediction\nprediction = model.predict(x_test)\nprediction = np.argmax(prediction, axis=1)\n\nprediction","metadata":{"execution":{"iopub.status.busy":"2022-04-26T16:07:45.880883Z","iopub.execute_input":"2022-04-26T16:07:45.881144Z","iopub.status.idle":"2022-04-26T16:07:46.771068Z","shell.execute_reply.started":"2022-04-26T16:07:45.881115Z","shell.execute_reply":"2022-04-26T16:07:46.770212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Submission\nsubmissions = pd.DataFrame({\"ImageId\": list(range(1,len(prediction)+1)),\n                         \"Label\": prediction})\n\nsubmissions.to_csv(\"submissions.csv\", index=False, header=True)\nsubmissions","metadata":{"execution":{"iopub.status.busy":"2022-04-26T16:07:51.855143Z","iopub.execute_input":"2022-04-26T16:07:51.855403Z","iopub.status.idle":"2022-04-26T16:07:51.878285Z","shell.execute_reply.started":"2022-04-26T16:07:51.855361Z","shell.execute_reply":"2022-04-26T16:07:51.877445Z"},"trusted":true},"execution_count":null,"outputs":[]}]}