{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.15","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":30777,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, RandomFlip, RandomContrast\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.metrics import f1_score\nimport matplotlib.pyplot as plt\n\nprint(\"TensorFlow version:\", tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T14:38:03.046231Z","iopub.execute_input":"2024-10-03T14:38:03.047083Z","iopub.status.idle":"2024-10-03T14:38:19.885902Z","shell.execute_reply.started":"2024-10-03T14:38:03.047038Z","shell.execute_reply":"2024-10-03T14:38:19.885113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is set. \n    # On Kaggle this is always the case.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print('Running on TPU ', tpu.master())\nexcept (ValueError, TypeError) as e:\n    # If there is an error, it usually means TPU is not available.\n    print(\"Can't connect to a TPU: {}\".format(e))\n    # Fallback to a different strategy or handling\n    strategy = tf.distribute.get_strategy() # Default strategy that works on CPU and GPU","metadata":{"execution":{"iopub.status.busy":"2024-10-03T14:38:19.887305Z","iopub.execute_input":"2024-10-03T14:38:19.887759Z","iopub.status.idle":"2024-10-03T14:38:29.524158Z","shell.execute_reply.started":"2024-10-03T14:38:19.887731Z","shell.execute_reply":"2024-10-03T14:38:29.523373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\ndata_dir = \"/kaggle/input/tpu-getting-started/tfrecords-jpeg-512x512\"","metadata":{"execution":{"iopub.status.busy":"2024-10-03T14:38:29.525054Z","iopub.execute_input":"2024-10-03T14:38:29.525302Z","iopub.status.idle":"2024-10-03T14:38:29.529418Z","shell.execute_reply.started":"2024-10-03T14:38:29.525276Z","shell.execute_reply":"2024-10-03T14:38:29.528645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_height, img_width = 512, 512\nbatch_size = 128","metadata":{"execution":{"iopub.status.busy":"2024-10-03T14:38:29.531031Z","iopub.execute_input":"2024-10-03T14:38:29.531353Z","iopub.status.idle":"2024-10-03T14:38:29.548523Z","shell.execute_reply.started":"2024-10-03T14:38:29.531323Z","shell.execute_reply":"2024-10-03T14:38:29.547819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(image):\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.cast(image, tf.float32)\n    image = keras.applications.densenet.preprocess_input(image)\n    image = image / 255.0\n    image = tf.reshape(image, [img_height, img_width, 3])\n    return image","metadata":{"execution":{"iopub.status.busy":"2024-10-03T14:38:29.549355Z","iopub.execute_input":"2024-10-03T14:38:29.549596Z","iopub.status.idle":"2024-10-03T14:38:29.558797Z","shell.execute_reply.started":"2024-10-03T14:38:29.549572Z","shell.execute_reply":"2024-10-03T14:38:29.558121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_tfrecord(example, labeled=True):\n    tfrecord_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    } if labeled else {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = decode_image(example['image'])\n    if labeled:\n        label = tf.one_hot(example['class'], 104)\n        return image, label\n    return image, example['id']\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    dataset = tf.data.TFRecordDataset(filenames)\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False  # disable order, increase speed\n    dataset = dataset.with_options(ignore_order)\n    return dataset.map(lambda x: read_tfrecord(x, labeled), num_parallel_calls=tf.data.experimental.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T14:38:29.559646Z","iopub.execute_input":"2024-10-03T14:38:29.559875Z","iopub.status.idle":"2024-10-03T14:38:29.569783Z","shell.execute_reply.started":"2024-10-03T14:38:29.559851Z","shell.execute_reply":"2024-10-03T14:38:29.569259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files = tf.io.gfile.glob(data_dir + '/train/*.tfrec')\nval_files = tf.io.gfile.glob(data_dir + '/val/*.tfrec')\ntest_files = tf.io.gfile.glob(data_dir + '/test/*.tfrec')","metadata":{"execution":{"iopub.status.busy":"2024-10-03T14:38:29.570529Z","iopub.execute_input":"2024-10-03T14:38:29.570751Z","iopub.status.idle":"2024-10-03T14:38:29.610905Z","shell.execute_reply.started":"2024-10-03T14:38:29.570727Z","shell.execute_reply":"2024-10-03T14:38:29.610229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_train_examples = 12753  # example value\nnum_val_examples = 3480     # example value\nnum_test_files = len(test_files)\nnum_test_examples = num_test_files * 462  # Assuming batch size for test dataset is 2048\nTRAIN_STEPS = num_train_examples // batch_size\nVAL_STEPS = num_val_examples // batch_size","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-10-03T14:38:40.900011Z","iopub.execute_input":"2024-10-03T14:38:40.900369Z","iopub.status.idle":"2024-10-03T14:38:40.904508Z","shell.execute_reply.started":"2024-10-03T14:38:40.900338Z","shell.execute_reply":"2024-10-03T14:38:40.903663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = load_dataset(train_files).repeat().shuffle(462).batch(batch_size).prefetch(tf.data.experimental.AUTOTUNE)\nval_dataset = load_dataset(val_files).batch(batch_size).cache().prefetch(tf.data.experimental.AUTOTUNE)\ntest_dataset = load_dataset(test_files, labeled=False, ordered=True).batch(batch_size)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T14:41:06.374148Z","iopub.execute_input":"2024-10-03T14:41:06.374611Z","iopub.status.idle":"2024-10-03T14:41:06.645055Z","shell.execute_reply.started":"2024-10-03T14:41:06.374574Z","shell.execute_reply":"2024-10-03T14:41:06.643777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n#     pretrained_model = tf.keras.applications.DenseNet201(\n#         weights='imagenet',\n#         include_top=False,\n#         input_shape=(512, 512, 3)\n#     )\n\n#     pretrained_model.trainable = True\n\n#     model = keras.Sequential([\n#         pretrained_model,\n#         tf.keras.layers.GlobalAveragePooling2D(),\n#         tf.keras.layers.Dropout(0.4),\n#         tf.keras.layers.Dense(480),\n#         tf.keras.layers.LeakyReLU(alpha=0.1),\n#         tf.keras.layers.BatchNormalization(),\n#         tf.keras.layers.Dense(360, activation='relu'),\n#         tf.keras.layers.Dropout(0.3),\n#         tf.keras.layers.BatchNormalization(),\n#         tf.keras.layers.Dense(240, activation='relu'),\n#         tf.keras.layers.Dropout(0.2),\n#         tf.keras.layers.BatchNormalization(),\n#         tf.keras.layers.Dense(104, activation='softmax')\n#     ])\n    \n    model = tf.keras.models.load_model('/kaggle/working/flower-densenet201.keras')\n        \n    model.compile(\n    optimizer= tf.keras.optimizers.Adam(learning_rate=0.00005),\n    loss = 'categorical_crossentropy',\n    metrics=['accuracy'],\n    )\n    early_stopping = EarlyStopping(\n      monitor='val_loss',\n  \t  min_delta=0.01, # minimium amount of change to count as an improvement\n   \t  patience=2, # how many epochs to wait before stopping\n   \t  restore_best_weights=True,\n      verbose=1\n    )\n    \n    lr_schedule = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.9, patience=2, min_lr=1e-6)\n    \n    history=model.fit(\n    train_dataset,\n    validation_data=(val_dataset),\n    epochs=15,\n    callbacks=[early_stopping, lr_schedule],\n    steps_per_epoch=TRAIN_STEPS, \n    validation_steps=VAL_STEPS,\n    )\n    \n    model.save('flower-densenet201.keras')","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:44:13.330014Z","iopub.execute_input":"2024-10-03T17:44:13.330934Z","iopub.status.idle":"2024-10-03T17:56:34.462294Z","shell.execute_reply.started":"2024-10-03T17:44:13.330891Z","shell.execute_reply":"2024-10-03T17:56:34.461271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = test_dataset\n\nprint('Computing predictions...')\ntest_images_ds = test_ds.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\nprint(predictions)\n\nprint('Generating submission.csv file...')","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:56:38.349535Z","iopub.execute_input":"2024-10-03T17:56:38.349918Z","iopub.status.idle":"2024-10-03T17:58:36.726741Z","shell.execute_reply.started":"2024-10-03T17:56:38.349883Z","shell.execute_reply":"2024-10-03T17:58:36.725572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(7392))).numpy().astype('U')\n\n# Write the submission file\nnp.savetxt(\n    'submission.csv',\n    np.rec.fromarrays([test_ids, predictions]),\n    fmt=['%s', '%d'],\n    delimiter=',',\n    header='id,label',\n    comments='',\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:58:36.728570Z","iopub.execute_input":"2024-10-03T17:58:36.729013Z","iopub.status.idle":"2024-10-03T17:58:42.465566Z","shell.execute_reply.started":"2024-10-03T17:58:36.728968Z","shell.execute_reply":"2024-10-03T17:58:42.464361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = pd.read_csv('submission.csv')\npred.head(pred.shape[0])","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:58:42.466878Z","iopub.execute_input":"2024-10-03T17:58:42.467323Z","iopub.status.idle":"2024-10-03T17:58:42.485139Z","shell.execute_reply.started":"2024-10-03T17:58:42.467273Z","shell.execute_reply":"2024-10-03T17:58:42.483956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}