{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport glob\nimport cv2\nfrom PIL import Image\nfrom PIL import Image\nimport os, os.path\nimport tensorflow as tf\nfrom tensorflow import keras\nimport matplotlib.pyplot as plt\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Conv2D\nfrom tensorflow.keras.layers import MaxPooling2D, AveragePooling2D\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom tensorflow.keras.layers import Flatten\n#plt.style.use('dark_background')","metadata":{"execution":{"iopub.status.busy":"2021-11-17T16:06:36.6119Z","iopub.execute_input":"2021-11-17T16:06:36.612569Z","iopub.status.idle":"2021-11-17T16:06:36.619395Z","shell.execute_reply.started":"2021-11-17T16:06:36.612505Z","shell.execute_reply":"2021-11-17T16:06:36.618604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(r'../input/plant-pathology-2021-fgvc8/train.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2021-11-17T14:23:53.849698Z","iopub.execute_input":"2021-11-17T14:23:53.849953Z","iopub.status.idle":"2021-11-17T14:23:53.903879Z","shell.execute_reply.started":"2021-11-17T14:23:53.849916Z","shell.execute_reply":"2021-11-17T14:23:53.903002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image'][0]","metadata":{"execution":{"iopub.status.busy":"2021-11-17T14:07:18.769811Z","iopub.execute_input":"2021-11-17T14:07:18.770278Z","iopub.status.idle":"2021-11-17T14:07:18.779045Z","shell.execute_reply.started":"2021-11-17T14:07:18.77024Z","shell.execute_reply":"2021-11-17T14:07:18.778318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = list(df.labels)","metadata":{"execution":{"iopub.status.busy":"2021-11-17T15:51:46.065091Z","iopub.execute_input":"2021-11-17T15:51:46.065694Z","iopub.status.idle":"2021-11-17T15:51:46.071469Z","shell.execute_reply.started":"2021-11-17T15:51:46.065654Z","shell.execute_reply":"2021-11-17T15:51:46.07079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = []\npath = '../input/plant-pathology-2021-fgvc8/train_images/'\nfor img in df['image']:\n    path_img = path + img\n    foto = np.array(Image.open(os.path.join(path,img)))\n    foto = cv2.resize(foto, (100,100), interpolation = cv2.INTER_AREA)\n    foto =foto.astype('float32') / 255\n    imgs.append(foto)\n    ","metadata":{"execution":{"iopub.status.busy":"2021-11-17T14:25:39.654904Z","iopub.execute_input":"2021-11-17T14:25:39.655189Z","iopub.status.idle":"2021-11-17T15:47:53.911982Z","shell.execute_reply.started":"2021-11-17T14:25:39.655156Z","shell.execute_reply":"2021-11-17T15:47:53.911133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs[0]","metadata":{"execution":{"iopub.status.busy":"2021-11-17T15:49:43.394453Z","iopub.execute_input":"2021-11-17T15:49:43.395297Z","iopub.status.idle":"2021-11-17T15:49:43.405556Z","shell.execute_reply.started":"2021-11-17T15:49:43.395253Z","shell.execute_reply":"2021-11-17T15:49:43.404576Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Labels a numeros y to_categorical","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder \nencoder = LabelEncoder()\nlabels_enc = encoder.fit_transform(np.array(labels))\n","metadata":{"execution":{"iopub.status.busy":"2021-11-17T15:52:23.822436Z","iopub.execute_input":"2021-11-17T15:52:23.823311Z","iopub.status.idle":"2021-11-17T15:52:24.391461Z","shell.execute_reply.started":"2021-11-17T15:52:23.823272Z","shell.execute_reply":"2021-11-17T15:52:24.390728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_cat = tf.keras.utils.to_categorical(labels_enc)\n","metadata":{"execution":{"iopub.status.busy":"2021-11-17T15:52:34.391306Z","iopub.execute_input":"2021-11-17T15:52:34.391934Z","iopub.status.idle":"2021-11-17T15:52:34.39711Z","shell.execute_reply.started":"2021-11-17T15:52:34.391892Z","shell.execute_reply":"2021-11-17T15:52:34.396222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Separamos en train y val","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(imgs, labels_cat, test_size = 0.3, stratify = labels_cat)","metadata":{"execution":{"iopub.status.busy":"2021-11-17T15:53:07.074271Z","iopub.execute_input":"2021-11-17T15:53:07.074592Z","iopub.status.idle":"2021-11-17T15:53:07.388746Z","shell.execute_reply.started":"2021-11-17T15:53:07.07455Z","shell.execute_reply":"2021-11-17T15:53:07.387917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = np.array(X_train), np.array(X_val), np.array(y_train), np.array(y_val)","metadata":{"execution":{"iopub.status.busy":"2021-11-17T16:26:17.764786Z","iopub.execute_input":"2021-11-17T16:26:17.765557Z","iopub.status.idle":"2021-11-17T16:26:20.281726Z","shell.execute_reply.started":"2021-11-17T16:26:17.765498Z","shell.execute_reply":"2021-11-17T16:26:20.280742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modelo","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import InceptionResNetV2\nbase_model = InceptionResNetV2(weights = \"imagenet\", include_top=False,input_shape=(100,100,3))","metadata":{"execution":{"iopub.status.busy":"2021-11-17T15:55:25.073671Z","iopub.execute_input":"2021-11-17T15:55:25.074342Z","iopub.status.idle":"2021-11-17T15:55:30.504261Z","shell.execute_reply.started":"2021-11-17T15:55:25.074298Z","shell.execute_reply":"2021-11-17T15:55:30.503489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(base_model)\nmodel.add(Flatten())\nmodel.add(Dense(500, activation='sigmoid'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(200, activation='sigmoid'))\nmodel.add(Dense(12, activation='softmax'))","metadata":{"execution":{"iopub.status.busy":"2021-11-17T17:44:39.845753Z","iopub.execute_input":"2021-11-17T17:44:39.846039Z","iopub.status.idle":"2021-11-17T17:44:41.217772Z","shell.execute_reply.started":"2021-11-17T17:44:39.846006Z","shell.execute_reply":"2021-11-17T17:44:41.21699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-11-17T16:51:56.720478Z","iopub.execute_input":"2021-11-17T16:51:56.720791Z","iopub.status.idle":"2021-11-17T16:51:56.769553Z","shell.execute_reply.started":"2021-11-17T16:51:56.720752Z","shell.execute_reply":"2021-11-17T16:51:56.768573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='CategoricalCrossentropy',\n              optimizer='Adam',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-11-17T17:44:42.809839Z","iopub.execute_input":"2021-11-17T17:44:42.810393Z","iopub.status.idle":"2021-11-17T17:44:42.834878Z","shell.execute_reply.started":"2021-11-17T17:44:42.810351Z","shell.execute_reply":"2021-11-17T17:44:42.834019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_filepath = '..input/tmp/checkpoint'\ncheckP = keras.callbacks.ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    save_weights_only=True,\n    monitor='val_accuracy',\n    mode='max',\n    save_best_only=True)\n\nearlyS = keras.callbacks.EarlyStopping(monitor='loss', patience=5)","metadata":{"execution":{"iopub.status.busy":"2021-11-17T15:58:48.303957Z","iopub.execute_input":"2021-11-17T15:58:48.304715Z","iopub.status.idle":"2021-11-17T15:58:48.310006Z","shell.execute_reply.started":"2021-11-17T15:58:48.304673Z","shell.execute_reply":"2021-11-17T15:58:48.30884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, y_train, \n                    epochs=8, \n                    validation_data = (X_val, y_val), \n                    verbose=1, \n                    callbacks=[checkP, earlyS])","metadata":{"execution":{"iopub.status.busy":"2021-11-17T17:44:45.726449Z","iopub.execute_input":"2021-11-17T17:44:45.727113Z","iopub.status.idle":"2021-11-17T17:53:28.106179Z","shell.execute_reply.started":"2021-11-17T17:44:45.727071Z","shell.execute_reply":"2021-11-17T17:53:28.105249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(X_val, y_val)","metadata":{"execution":{"iopub.status.busy":"2021-11-17T17:20:09.329243Z","iopub.execute_input":"2021-11-17T17:20:09.33103Z","iopub.status.idle":"2021-11-17T17:20:20.7963Z","shell.execute_reply.started":"2021-11-17T17:20:09.33098Z","shell.execute_reply":"2021-11-17T17:20:20.795468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot model performance\n\nacc = history.history[\"accuracy\"]\nval_acc = history.history[\"val_accuracy\"]\nloss = history.history[\"loss\"]\nval_loss = history.history[\"val_loss\"]\nepochs_range = range(1, len(history.epoch) + 1)\n\nplt.figure(figsize = (15, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label = \"Train Set\")\nplt.plot(epochs_range, val_acc, label = \"Val Set\")\nplt.legend(loc = \"best\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.title(\"Model Accuracy\")\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label = \"Train Set\")\nplt.plot(epochs_range, val_loss, label = \"Val Set\")\nplt.legend(loc = \"best\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.title(\"Model Loss\")\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-17T17:20:37.904658Z","iopub.execute_input":"2021-11-17T17:20:37.90536Z","iopub.status.idle":"2021-11-17T17:20:38.397076Z","shell.execute_reply.started":"2021-11-17T17:20:37.905317Z","shell.execute_reply":"2021-11-17T17:20:38.396399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediccion","metadata":{"execution":{"iopub.status.busy":"2021-11-17T16:24:32.742138Z","iopub.execute_input":"2021-11-17T16:24:32.742406Z","iopub.status.idle":"2021-11-17T16:24:32.747695Z","shell.execute_reply.started":"2021-11-17T16:24:32.742377Z","shell.execute_reply":"2021-11-17T16:24:32.746909Z"}}},{"cell_type":"code","source":"imgs_out = []\npath = '../input/plant-pathology-2021-fgvc8/test_images'\npics = ['85f8cb619c66b863.jpg', 'ad8770db05586b59.jpg', 'c7b03e718489f3ca.jpg']\n\nfor img in pics:\n    \n    foto = np.array(Image.open(os.path.join(path,img)))\n    foto = cv2.resize(foto, (100,100), interpolation = cv2.INTER_AREA)\n    foto =foto.astype('float32') / 255\n    imgs_out.append(foto)\n    \nimgs_out = np.array(imgs_out)","metadata":{"execution":{"iopub.status.busy":"2021-11-17T18:05:16.107355Z","iopub.execute_input":"2021-11-17T18:05:16.108037Z","iopub.status.idle":"2021-11-17T18:05:16.758688Z","shell.execute_reply.started":"2021-11-17T18:05:16.107985Z","shell.execute_reply":"2021-11-17T18:05:16.757869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(imgs_out)","metadata":{"execution":{"iopub.status.busy":"2021-11-17T18:05:22.820315Z","iopub.execute_input":"2021-11-17T18:05:22.820612Z","iopub.status.idle":"2021-11-17T18:05:26.651429Z","shell.execute_reply.started":"2021-11-17T18:05:22.820578Z","shell.execute_reply":"2021-11-17T18:05:26.650609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clases = []\n\nfor i in pred:\n    clases.append(encoder.inverse_transform(np.array(np.argmax(i)).reshape(-1, 1))[0])\n    \nclases","metadata":{"execution":{"iopub.status.busy":"2021-11-17T18:09:45.267235Z","iopub.execute_input":"2021-11-17T18:09:45.267637Z","iopub.status.idle":"2021-11-17T18:09:45.283228Z","shell.execute_reply.started":"2021-11-17T18:09:45.26759Z","shell.execute_reply":"2021-11-17T18:09:45.282581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_pred = pd.DataFrame([pics,clases]).T\ndf_pred.columns = ['image', 'label']\ndf_pred","metadata":{"execution":{"iopub.status.busy":"2021-11-17T18:13:00.607232Z","iopub.execute_input":"2021-11-17T18:13:00.607512Z","iopub.status.idle":"2021-11-17T18:13:00.625954Z","shell.execute_reply.started":"2021-11-17T18:13:00.607479Z","shell.execute_reply":"2021-11-17T18:13:00.625135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_pred.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-11-17T18:25:19.44575Z","iopub.execute_input":"2021-11-17T18:25:19.446412Z","iopub.status.idle":"2021-11-17T18:25:19.452739Z","shell.execute_reply.started":"2021-11-17T18:25:19.446368Z","shell.execute_reply":"2021-11-17T18:25:19.451872Z"},"trusted":true},"execution_count":null,"outputs":[]}]}