{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport tensorflow as tf\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tensorflow.keras import optimizers, layers, models, utils\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Conv2D, MaxPooling2D, Flatten, Dropout, Activation\nfrom sklearn.metrics import confusion_matrix, classification_report, accuracy_score, precision_score, recall_score, f1_score\nfrom sklearn.model_selection import train_test_split\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nprint(\"Done\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-16T10:56:14.769801Z","iopub.execute_input":"2022-08-16T10:56:14.770449Z","iopub.status.idle":"2022-08-16T10:56:22.384430Z","shell.execute_reply.started":"2022-08-16T10:56:14.770349Z","shell.execute_reply":"2022-08-16T10:56:22.383270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() \n\nREPLICAS = strategy.num_replicas_in_sync\nprint(\"REPLICAS: \", REPLICAS)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:22.386458Z","iopub.execute_input":"2022-08-16T10:56:22.386915Z","iopub.status.idle":"2022-08-16T10:56:28.831924Z","shell.execute_reply.started":"2022-08-16T10:56:22.386868Z","shell.execute_reply":"2022-08-16T10:56:28.830869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Constants","metadata":{}},{"cell_type":"code","source":"TEST = \"test\"\nTRAIN = \"train\"\nVAL = \"val\"\nTFREC = \"/*.tfrec\"\nIMAGE = \"image\"\nID = \"id\"\nCLASS = \"class\"\nRECORDS_192 = \"../input/tpu-getting-started/tfrecords-jpeg-192x192\" \nRECORDS_224 = \"../input/tpu-getting-started/tfrecords-jpeg-224x224\"\nRECORDS_331 = \"tfrecords-jpeg-331x331\"\nRECORDS_512 = \"../input/tpu-getting-started/tfrecords-jpeg-512x512\"\nALL_RECORDS = [RECORDS_192, RECORDS_224, RECORDS_331,RECORDS_512]\nSAMPLE_SUBMISSION = \"../input/tpu-getting-started/sample_submission.csv\"\nIMAGE_SIZE = [331, 331]\nAUTOTUNE = tf.data.experimental.AUTOTUNE\nSEED = 1111\nBATCH_SIZE = 16*REPLICAS","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:28.833096Z","iopub.execute_input":"2022-08-16T10:56:28.833418Z","iopub.status.idle":"2022-08-16T10:56:28.839637Z","shell.execute_reply.started":"2022-08-16T10:56:28.833389Z","shell.execute_reply":"2022-08-16T10:56:28.838847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions\n**🟦EN** Function that we use later to see the characteristics of the data such as missing values (NaN), all of features and number, records and columns\n\n**🟥ES** Función que usamos más tarde para ver las características de los datos, como valores faltantes (NaN), todas las características y el número, registros y columnas.","metadata":{}},{"cell_type":"code","source":"def data_description(df):\n    print(\"Data description\")\n    print(f\"Total number of records {df.shape[0]}\")\n    print(f'number of features {df.shape[1]}\\n\\n')\n    columns = df.columns\n    data_type = []\n    \n    # Get the datatype of features\n    for col in df.columns:\n        data_type.append(df[col].dtype)\n        \n    n_uni = df.nunique()\n    # Number of NaN values\n    n_miss = df.isna().sum()\n    \n    names = list(zip(columns, data_type, n_uni, n_miss))\n    variable_desc = pd.DataFrame(names, columns=[\"Name\",\"Type\",\"Unique levels\",\"Missing\"])\n\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:28.841557Z","iopub.execute_input":"2022-08-16T10:56:28.841814Z","iopub.status.idle":"2022-08-16T10:56:28.853325Z","shell.execute_reply.started":"2022-08-16T10:56:28.841777Z","shell.execute_reply":"2022-08-16T10:56:28.852261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_grap(data, title):\n    \n    count = count_labels(data) # Get number of labels\n        \n    plt.figure(figsize = (35, 5))\n    sns.countplot(x = count).set_title(title) # Count values all values and create a countplot\n    print(f\"Images: {len(count)}\") # show number of img","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:28.854588Z","iopub.execute_input":"2022-08-16T10:56:28.854921Z","iopub.status.idle":"2022-08-16T10:56:28.870323Z","shell.execute_reply.started":"2022-08-16T10:56:28.854887Z","shell.execute_reply":"2022-08-16T10:56:28.869219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_labels(data):\n    lbl = data.map(lambda img, label: label) # Create a map with img and their label\n    count = list() # Create a list\n    \n    for labels in lbl:\n        count += list(labels.numpy())\n    \n    return count","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:28.872187Z","iopub.execute_input":"2022-08-16T10:56:28.872491Z","iopub.status.idle":"2022-08-16T10:56:28.883464Z","shell.execute_reply.started":"2022-08-16T10:56:28.872458Z","shell.execute_reply":"2022-08-16T10:56:28.882411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def image_decoder(image):\n    img = tf.image.decode_jpeg(image, channels = 3)\n    img = tf.cast(img, tf.float32) / 255  # Normalize the image 255 (0 - Min color palette and 255 max color palette)\n    img = tf.reshape(img, [*IMAGE_SIZE, 3]) # Reshape the tensor, 3 = number of colors (1 to gray palette and 3 to full color)\n    return img\n\n\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:28.885056Z","iopub.execute_input":"2022-08-16T10:56:28.885597Z","iopub.status.idle":"2022-08-16T10:56:28.899283Z","shell.execute_reply.started":"2022-08-16T10:56:28.885556Z","shell.execute_reply":"2022-08-16T10:56:28.897728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dataset(file, labeled = True, ordered = False):\n    ignore = tf.data.Options()\n    \n    if not ordered:\n        ignore.experimental_deterministic = False\n        \n    dataset = tf.data.TFRecordDataset(file, num_parallel_reads = AUTOTUNE) # Read files with TFRecord files\n    dataset = dataset.with_options(ignore) \n    dataset = dataset.map(read_labeled if labeled else read_unlabeled, num_parallel_calls = AUTOTUNE)\n    \n    return dataset\n\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:28.900802Z","iopub.execute_input":"2022-08-16T10:56:28.901677Z","iopub.status.idle":"2022-08-16T10:56:28.912209Z","shell.execute_reply.started":"2022-08-16T10:56:28.901625Z","shell.execute_reply":"2022-08-16T10:56:28.911348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_labeled(exemple):\n    LABELED_TFREC_FORMAT = { # Create a dictionaty with the features\n        IMAGE : tf.io.FixedLenFeature([], tf.string),\n        CLASS : tf.io.FixedLenFeature([], tf.int64),\n    }\n    \n    exemple = tf.io.parse_single_example(exemple, LABELED_TFREC_FORMAT) # Parses a single Example proto.\n    image = image_decoder(exemple[IMAGE]) # Decode the image\n    label = tf.cast(exemple[CLASS], tf.int32) # Casts a tensor to a new type.\n    return image, label\n\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:28.913796Z","iopub.execute_input":"2022-08-16T10:56:28.914103Z","iopub.status.idle":"2022-08-16T10:56:28.923691Z","shell.execute_reply.started":"2022-08-16T10:56:28.914029Z","shell.execute_reply":"2022-08-16T10:56:28.922763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_unlabeled(exemple):\n    UNLABELED_TFREC_FORMAT = {\n        IMAGE : tf.io.FixedLenFeature([], tf.string),\n        ID : tf.io.FixedLenFeature([], tf.string),\n    }\n    exemple = tf.io.parse_single_example(exemple, UNLABELED_TFREC_FORMAT)\n    image = image_decoder(exemple[IMAGE]) # Decode img\n    idn = exemple[ID] # Get id of the img\n    return image, idn\n\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:28.926382Z","iopub.execute_input":"2022-08-16T10:56:28.926873Z","iopub.status.idle":"2022-08-16T10:56:28.945081Z","shell.execute_reply.started":"2022-08-16T10:56:28.926840Z","shell.execute_reply":"2022-08-16T10:56:28.944007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load datasets","metadata":{}},{"cell_type":"code","source":"petals_dir=None\n\nif tpu:\n    from kaggle_datasets import KaggleDatasets\n    img_dir = KaggleDatasets().get_gcs_path('tpu-getting-started')\nelse:\n    img_dir = \".\" \n\nrec = RECORDS_331\nimg_dir_x = os.path.join(img_dir, rec)\n\ntrain_dir = os.path.join(img_dir_x, TRAIN)\ntest_dir = os.path.join(img_dir_x, TEST)\nval_dir = os.path.join(img_dir_x, VAL)\n\ntrain_img = tf.io.gfile.glob(train_dir + TFREC) \ntest_img = tf.io.gfile.glob(test_dir + TFREC)\nval_img = tf.io.gfile.glob(val_dir + TFREC)\n\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:28.948100Z","iopub.execute_input":"2022-08-16T10:56:28.948759Z","iopub.status.idle":"2022-08-16T10:56:29.608977Z","shell.execute_reply.started":"2022-08-16T10:56:28.948720Z","shell.execute_reply":"2022-08-16T10:56:29.608107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:29.610050Z","iopub.execute_input":"2022-08-16T10:56:29.610489Z","iopub.status.idle":"2022-08-16T10:56:29.619958Z","shell.execute_reply.started":"2022-08-16T10:56:29.610446Z","shell.execute_reply":"2022-08-16T10:56:29.619108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = load_dataset(train_img, labeled = True).shuffle(SEED).batch(BATCH_SIZE).prefetch(AUTOTUNE)\ntest_data = load_dataset(test_img, labeled = True).batch(BATCH_SIZE).prefetch(AUTOTUNE)\nval_data = load_dataset(val_img, labeled = False, ordered = True).batch(BATCH_SIZE).cache().prefetch(AUTOTUNE)\n\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:29.621215Z","iopub.execute_input":"2022-08-16T10:56:29.622095Z","iopub.status.idle":"2022-08-16T10:56:30.079743Z","shell.execute_reply.started":"2022-08-16T10:56:29.622050Z","shell.execute_reply":"2022-08-16T10:56:30.078344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data)\ndisplay(test_data)\ndisplay(val_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:30.082750Z","iopub.execute_input":"2022-08-16T10:56:30.083029Z","iopub.status.idle":"2022-08-16T10:56:30.094406Z","shell.execute_reply.started":"2022-08-16T10:56:30.082998Z","shell.execute_reply":"2022-08-16T10:56:30.093181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"itr = train_data.__iter__()\nX,y = itr.next()\n\nprint(X.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:30.095469Z","iopub.execute_input":"2022-08-16T10:56:30.096284Z","iopub.status.idle":"2022-08-16T10:56:32.260523Z","shell.execute_reply.started":"2022-08-16T10:56:30.096247Z","shell.execute_reply":"2022-08-16T10:56:32.259823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 60))\nfor i in range(0,5):\n    plt.subplot(1,5,i+1)\n    plt.imshow(X[i])","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:32.262327Z","iopub.execute_input":"2022-08-16T10:56:32.262567Z","iopub.status.idle":"2022-08-16T10:56:33.317476Z","shell.execute_reply.started":"2022-08-16T10:56:32.262539Z","shell.execute_reply":"2022-08-16T10:56:33.316679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_grap(train_data, \"classes\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:33.318722Z","iopub.execute_input":"2022-08-16T10:56:33.319531Z","iopub.status.idle":"2022-08-16T10:56:46.404651Z","shell.execute_reply.started":"2022-08-16T10:56:33.319451Z","shell.execute_reply":"2022-08-16T10:56:46.403589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_features = 104 #Todo set this with a code :P","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:46.406006Z","iopub.execute_input":"2022-08-16T10:56:46.406328Z","iopub.status.idle":"2022-08-16T10:56:46.410689Z","shell.execute_reply.started":"2022-08-16T10:56:46.406285Z","shell.execute_reply":"2022-08-16T10:56:46.409640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model creation","metadata":{}},{"cell_type":"code","source":"model = Sequential([\n    tf.keras.layers.Conv2D(32, (3,3), activation = \"relu\", input_shape=(331, 331, 3)),\n    tf.keras.layers.Conv2D(64, (3,3), activation = \"relu\"),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.MaxPooling2D(2,2),\n    \n    tf.keras.layers.Conv2D(128, (3,3), activation = \"relu\"),\n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Dropout(0.2),\n    \n    tf.keras.layers.Conv2D(128, (3,3), activation = \"relu\"),\n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Dropout(0.2),\n    \n    tf.keras.layers.Flatten(),\n    \n    tf.keras.layers.Dense(125, activation = \"relu\"),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Dense(250, activation = \"relu\"),\n    tf.keras.layers.Dense(500, activation = \"relu\"),\n    tf.keras.layers.Dropout(0.2),\n    tf.keras.layers.Dense(num_features, activation = \"softmax\"),\n])\n\nprint(\"Model created\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:46.412266Z","iopub.execute_input":"2022-08-16T10:56:46.412608Z","iopub.status.idle":"2022-08-16T10:56:46.626032Z","shell.execute_reply.started":"2022-08-16T10:56:46.412569Z","shell.execute_reply":"2022-08-16T10:56:46.625140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_options=None\nif tpu:\n    checkpoint_options = tf.train.CheckpointOptions(experimental_io_device='/job:localhost')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:46.627364Z","iopub.execute_input":"2022-08-16T10:56:46.628077Z","iopub.status.idle":"2022-08-16T10:56:46.633247Z","shell.execute_reply.started":"2022-08-16T10:56:46.628033Z","shell.execute_reply":"2022-08-16T10:56:46.632070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chp_callback=tf.keras.callbacks.ModelCheckpoint(\n    filepath=\"chp.h5\",\n    monitor='val_loss',\n    verbose=1,\n    save_best_only=True,\n    save_freq='epoch',\n    mode=\"min\",\n    save_weights_only=True,\n    options=checkpoint_options\n)\nmodel.compile(loss = \"sparse_categorical_crossentropy\",\n             optimizer = \"adam\",\n             metrics = [\"sparse_categorical_accuracy\"])\n\nprint(\"Done\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:46.634633Z","iopub.execute_input":"2022-08-16T10:56:46.635740Z","iopub.status.idle":"2022-08-16T10:56:46.661408Z","shell.execute_reply.started":"2022-08-16T10:56:46.635695Z","shell.execute_reply":"2022-08-16T10:56:46.660737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_data, validation_data = val_data,\n                    epochs = 10, callbacks = [chp_callback]) ## Epoch 1/10 21/Unknown - 186s 9s/step - loss: 4.7701 - sparse_categorical_accuracy: 0.0319","metadata":{"execution":{"iopub.status.busy":"2022-08-16T10:56:46.663124Z","iopub.execute_input":"2022-08-16T10:56:46.663741Z","iopub.status.idle":"2022-08-16T11:00:01.782539Z","shell.execute_reply.started":"2022-08-16T10:56:46.663698Z","shell.execute_reply":"2022-08-16T11:00:01.781115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_frame = pd.DataFrame(history.history)\nhistory_frame.loc[:, ['loss', 'val_loss']].plot()\nhistory_frame.loc[:, ['accuracy', 'val_accuracy']].plot()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T11:00:01.783401Z","iopub.status.idle":"2022-08-16T11:00:01.783761Z","shell.execute_reply.started":"2022-08-16T11:00:01.783584Z","shell.execute_reply":"2022-08-16T11:00:01.783601Z"},"trusted":true},"execution_count":null,"outputs":[]}]}