{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## **Petals to the Metal - Flower Classification on TPU - TensorFlow**\n+ Cristhian Ccala Huamani","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport seaborn as sns\nimport re\nimport matplotlib.pyplot as plt \n\nfrom sklearn.metrics import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2024-03-10T20:02:37.270284Z","iopub.execute_input":"2024-03-10T20:02:37.270742Z","iopub.status.idle":"2024-03-10T20:02:51.525650Z","shell.execute_reply.started":"2024-03-10T20:02:37.270711Z","shell.execute_reply":"2024-03-10T20:02:51.524844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Cargamos las direccoines de la carpeta**","metadata":{}},{"cell_type":"code","source":"#dir_img = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-192x192'\n#dir_img = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224'\n#dir_img = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-331x331'\ndir_img = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-512x512'\nimg_size=256","metadata":{"execution":{"iopub.status.busy":"2024-03-10T20:02:51.527197Z","iopub.execute_input":"2024-03-10T20:02:51.527714Z","iopub.status.idle":"2024-03-10T20:02:51.531892Z","shell.execute_reply.started":"2024-03-10T20:02:51.527688Z","shell.execute_reply":"2024-03-10T20:02:51.530971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_train = tf.io.gfile.glob(f\"{dir_img}{'/train/*.tfrec'}\")\nimg_vald = tf.io.gfile.glob(f\"{dir_img}{'/val/*.tfrec'}\")\nimg_test = tf.io.gfile.glob(f\"{dir_img}{'/test/*.tfrec'}\") ","metadata":{"execution":{"iopub.status.busy":"2024-03-10T20:02:51.533133Z","iopub.execute_input":"2024-03-10T20:02:51.533466Z","iopub.status.idle":"2024-03-10T20:02:51.617359Z","shell.execute_reply.started":"2024-03-10T20:02:51.533427Z","shell.execute_reply":"2024-03-10T20:02:51.616709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Procesamos el archivo TFRecord**","metadata":{}},{"cell_type":"code","source":"# procesar ejemplos individuales del conjunto de datos TFRecord\ndef leer_archivo(example):\n    feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'class': tf.io.FixedLenFeature([], tf.int64)\n    }\n    example = tf.io.parse_single_example(example, feature_description)\n    image = tf.io.decode_jpeg(example['image'], channels=3)  # Decodifica la imagen JPEG\n    image = tf.image.resize(image, [img_size, img_size])\n    image = tf.cast(image, tf.float32) / 255.0 \n    label = tf.cast(example['class'], tf.int32)\n    label = tf.one_hot(label, 104)\n    return image, label\n\n# dataset de entrenamiento\nds_train = tf.data.TFRecordDataset(img_train, num_parallel_reads=tf.data.experimental.AUTOTUNE)\nds_train = ds_train.map(leer_archivo, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n#dataset de validadcion\nds_valid = tf.data.TFRecordDataset(img_vald, num_parallel_reads=tf.data.experimental.AUTOTUNE)\nds_valid = ds_valid.map(leer_archivo, num_parallel_calls=tf.data.experimental.AUTOTUNE)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T20:02:51.618217Z","iopub.execute_input":"2024-03-10T20:02:51.618729Z","iopub.status.idle":"2024-03-10T20:02:52.513206Z","shell.execute_reply.started":"2024-03-10T20:02:51.618706Z","shell.execute_reply":"2024-03-10T20:02:52.512273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Visualizamos las imagenes de flores**","metadata":{}},{"cell_type":"code","source":"iterator = iter(ds_train)\nplt.figure(figsize = (11,10))\nfor i in range(18):\n    image, label = next(iterator)\n    plt.subplot(5,6,i+1)\n    plt.imshow(image.numpy())\n    plt.title(label.numpy().argmax())\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2024-03-10T20:02:52.515877Z","iopub.execute_input":"2024-03-10T20:02:52.516169Z","iopub.status.idle":"2024-03-10T20:02:54.094007Z","shell.execute_reply.started":"2024-03-10T20:02:52.516145Z","shell.execute_reply":"2024-03-10T20:02:54.093056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Aumento de datos**","metadata":{}},{"cell_type":"code","source":"data_augmenation = tf.keras.Sequential([\n    tf.keras.layers.RandomFlip(mode = \"horizontal_and_vertical\"),\n    tf.keras.layers.RandomRotation(factor=0.2),\n    tf.keras.layers.RandomTranslation(height_factor=0.1, width_factor=0.1),\n])\nimage, label = next(iter(ds_train))\nimage = tf.expand_dims(image, 0)\nplt.figure(figsize=(10, 10))\nfor i in range(9):\n    augmented_image = data_augmenation(image)\n    ax = plt.subplot(3, 3, i + 1)\n    plt.imshow(augmented_image[0])\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2024-03-10T20:02:54.095088Z","iopub.execute_input":"2024-03-10T20:02:54.095363Z","iopub.status.idle":"2024-03-10T20:02:56.121447Z","shell.execute_reply.started":"2024-03-10T20:02:54.095339Z","shell.execute_reply":"2024-03-10T20:02:56.120535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train = ds_train.map(lambda x, y: (data_augmenation(x, training=True), y), num_parallel_calls=tf.data.AUTOTUNE)\nds_train = ds_train.batch(32)\n\nds_valid = ds_valid.batch(32)\n\nentrenar_ds = ds_train.prefetch(buffer_size=tf.data.AUTOTUNE) \nvalidar_ds = ds_valid.prefetch(buffer_size=tf.data.AUTOTUNE) ","metadata":{"execution":{"iopub.status.busy":"2024-03-10T20:02:56.122575Z","iopub.execute_input":"2024-03-10T20:02:56.122847Z","iopub.status.idle":"2024-03-10T20:02:56.381136Z","shell.execute_reply.started":"2024-03-10T20:02:56.122824Z","shell.execute_reply":"2024-03-10T20:02:56.380242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Modelo Pre-Entrenado xception**","metadata":{}},{"cell_type":"code","source":"xception = tf.keras.applications.xception.Xception(\n    include_top=False,\n    weights=\"imagenet\",\n    input_tensor=None,\n    input_shape=(img_size,img_size,3),\n    pooling='max'\n)\ncapa_salida = (tf.keras.layers.Dense(104, activation='softmax'))(xception.output)\nmodel_xception = tf.keras.models.Model(inputs=xception.input, outputs=capa_salida)\n\nmodel_xception.compile(loss='categorical_crossentropy',\n                     optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n                     metrics=['accuracy'])\n\n#model_xception.summary()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T20:02:56.382308Z","iopub.execute_input":"2024-03-10T20:02:56.382636Z","iopub.status.idle":"2024-03-10T20:03:00.842920Z","shell.execute_reply.started":"2024-03-10T20:02:56.382611Z","shell.execute_reply":"2024-03-10T20:03:00.841977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"historia = model_xception.fit(\n    entrenar_ds,\n    epochs=20,\n    validation_data=validar_ds\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T20:03:00.844038Z","iopub.execute_input":"2024-03-10T20:03:00.844314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Evaluamos el modelo**","metadata":{}},{"cell_type":"code","source":"model_xception.evaluate(validar_ds)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualizamos el diagrama de lineas\npd.DataFrame(historia.history).plot()\nplt.title('Grafico de Lineas')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v = [int(re.compile(r\"-([0-9]*)\\.\").search(file_vald).group(1)) for file_vald in img_vald]\ncount_v=np.sum(v)\nimages_ds = validar_ds.map(lambda image, label: image)\nlabels_ds = validar_ds.map(lambda image, label: label).unbatch()\ny_valid = next(iter(labels_ds.batch(count_v))).numpy()\ny_pred = model_xception.predict(validar_ds)\ny_pred = np.argmax(y_pred, axis=1)\ny_true = np.argmax(y_valid, axis=1)\ny_pred","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualizamos la matriz de confucion\ncm = confusion_matrix(y_true, y_pred)\nplt.figure(figsize = (15,13))\nplt.title('Matrix de Confusion')\nsns.heatmap(pd.DataFrame(cm, index= np.arange(104), columns = np.arange(104)))\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Preparamos datos de Prueba**","metadata":{}},{"cell_type":"code","source":"def leer_archivo_test(example):\n    feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'id': tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, feature_description)\n    image = tf.io.decode_jpeg(example['image'], channels=3)  # Decodifica la imagen\n    image = tf.cast(image, tf.float32) / 255.0 \n    idnum = example['id']\n    return image, idnum\n\n# Abre el archivo TFRecord\nds_test = tf.data.TFRecordDataset(img_test, num_parallel_reads=tf.data.experimental.AUTOTUNE)\nds_test = ds_test.map(leer_archivo_test, num_parallel_calls=tf.data.experimental.AUTOTUNE)\nds_test = ds_test.batch(32)\nds_test = ds_test.prefetch(tf.data.experimental.AUTOTUNE)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = [int(re.compile(r\"-([0-9]*)\\.\").search(file_test).group(1)) for file_test in img_test]\ncount=np.sum(n)\ntest_ids = ds_test.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids.batch(count))).numpy().astype('U')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"etiquetas = np.arange(104)\nprediccion = model_xception.predict(ds_test)\nprediccion = etiquetas[tf.argmax(prediccion, axis=1)]\nprediccion","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Eviamos Resultados**","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame()\nsubmission['id'] = test_ids\nsubmission['label'] = prediccion\nsubmission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}