{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":30734,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow.keras.backend as K\nfrom kaggle_datasets import KaggleDatasets\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Dropout, Input, Conv2D, MaxPooling2D\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom sklearn.metrics import f1_score\nfrom tensorflow.keras.models import Model\nimport os\n\n# Detect TPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\n# Connect to TPU\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)  # Updated to non-experimental TPUStrategy\nelse:\n    strategy = tf.distribute.get_strategy()  # Default strategy for CPU and single GPU","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-09T09:12:14.784026Z","iopub.execute_input":"2024-07-09T09:12:14.784298Z","iopub.status.idle":"2024-07-09T09:12:40.710373Z","shell.execute_reply.started":"2024-07-09T09:12:14.784260Z","shell.execute_reply":"2024-07-09T09:12:40.709467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, TensorBoard, ReduceLROnPlateau\nimport datetime\n\ncheckpoint_path = \"/kaggle/working/model_checkpoint.keras\"\ncheckpoint_callback = ModelCheckpoint(filepath=checkpoint_path, save_best_only=True, verbose=1)\nearly_stopping_callback = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True, verbose=1)\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, min_lr=0.0001)\nlog_dir = \"/kaggle/working/logs/\" + datetime.datetime.now().strftime(\"%Y%m%d-%H%M%S\")\ntensorboard_callback = TensorBoard(log_dir=log_dir, histogram_freq=1)\n\ncallbacks = [checkpoint_callback, early_stopping_callback, tensorboard_callback, reduce_lr]","metadata":{"execution":{"iopub.status.busy":"2024-07-09T09:13:33.324900Z","iopub.execute_input":"2024-07-09T09:13:33.325674Z","iopub.status.idle":"2024-07-09T09:13:33.333419Z","shell.execute_reply.started":"2024-07-09T09:13:33.325633Z","shell.execute_reply":"2024-07-09T09:13:33.332516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Configuration\nAUTOTUNE = tf.data.experimental.AUTOTUNE\nIMAGE_SIZE = [224, 224]  # Specify the image size\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\n# Data access\nGCS_PATH = '/kaggle/input/flower-classification-with-tpus'\ntrain_dir = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/train'\nval_dir = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/val'\ntest_dir = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-224x224/test'\n\nprint(\"Training files:\", os.listdir(train_dir)[:5])  # List first 5 training files\nprint(\"Validation files:\", os.listdir(val_dir)[:5])  # List first 5 validation files\nprint(\"Test files:\", os.listdir(test_dir)[:5])  # List first 5 test files\n\n# Load the data\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.image.resize(image, IMAGE_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\ndef read_tfrecord(example, labeled=True):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64)\n    }\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT if labeled else UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    if labeled:\n        label = tf.cast(example['class'], tf.int32)\n        return image, label\n    idnum = example['id']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True):\n    dataset = tf.data.TFRecordDataset(filenames)\n    dataset = dataset.map(lambda x: read_tfrecord(x, labeled), num_parallel_calls=AUTOTUNE)\n    return dataset\n\n# Define file paths\nTRAINING_FILENAMES = tf.io.gfile.glob(train_dir + '/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(val_dir + '/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(test_dir + '/*.tfrec')\n\n# Verify data loading\ntrain_dataset = load_dataset(TRAINING_FILENAMES, labeled=True).take(1)\nvalid_dataset = load_dataset(VALIDATION_FILENAMES, labeled=True).take(1)\ntest_dataset = load_dataset(TEST_FILENAMES, labeled=False).take(1)\n\nfor image, label in train_dataset:\n    print(\"Train image shape:\", image.numpy().shape)\n    print(\"Train label:\", label.numpy())\n\nfor image, label in valid_dataset:\n    print(\"Validation image shape:\", image.numpy().shape)\n    print(\"Validation label:\", label.numpy())\n\nfor image, idnum in test_dataset:\n    print(\"Test image shape:\", image.numpy().shape)\n    print(\"Test id:\", idnum.numpy().decode('utf-8'))\n\n# Data pipeline\ndef get_dataset(filenames, shuffle=False, repeat=False, labeled=True, batch_size=BATCH_SIZE):\n    dataset = load_dataset(filenames, labeled)\n    if shuffle:\n        dataset = dataset.shuffle(2048)\n    if repeat:\n        dataset = dataset.repeat()\n    dataset = dataset.batch(batch_size)\n    dataset = dataset.prefetch(buffer_size=AUTOTUNE)\n    return dataset\n\n# Load datasets\ntrain_dataset = get_dataset(TRAINING_FILENAMES, shuffle=True, repeat=True)\nvalid_dataset = get_dataset(VALIDATION_FILENAMES, repeat=True)\ntest_dataset = get_dataset(TEST_FILENAMES, repeat=False, labeled=False)\n\n# Calculate number of steps per epoch\ndef count_data_items(filenames):\n    return np.sum([tf.data.TFRecordDataset(f).reduce(0, lambda x, _: x + 1).numpy() for f in filenames])\n\nnum_train_samples = count_data_items(TRAINING_FILENAMES)\nnum_val_samples = count_data_items(VALIDATION_FILENAMES)\nsteps_per_epoch = num_train_samples // BATCH_SIZE\nvalidation_steps = num_val_samples // BATCH_SIZE\n\nprint(f\"Number of training samples: {num_train_samples}\")\nprint(f\"Number of validation samples: {num_val_samples}\")\nprint(f\"Steps per epoch: {steps_per_epoch}\")\nprint(f\"Validation steps: {validation_steps}\")\n\n# Build the model\nwith strategy.scope():\n    model1 = Sequential([\n        EfficientNetB0(input_shape=(*IMAGE_SIZE, 3), include_top=False, weights='imagenet'),\n        Flatten(),\n        Dense(256, activation='relu'),\n        Dropout(0.3),\n        Dense(104, activation='softmax')\n    ])\n    model1.compile(\n        optimizer='adam',\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n\n    # Train the model\n    EPOCHS = 50\n    history = model1.fit(\n        train_dataset,\n        validation_data=valid_dataset,\n        epochs=EPOCHS,\n        steps_per_epoch=steps_per_epoch,\n        validation_steps=validation_steps, shuffle=True, callbacks=callbacks\n    )","metadata":{"execution":{"iopub.status.busy":"2024-07-09T09:13:37.710276Z","iopub.execute_input":"2024-07-09T09:13:37.710596Z","iopub.status.idle":"2024-07-09T09:28:36.997387Z","shell.execute_reply.started":"2024-07-09T09:13:37.710568Z","shell.execute_reply":"2024-07-09T09:28:36.996118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = model1","metadata":{"execution":{"iopub.status.busy":"2024-07-09T09:29:47.828704Z","iopub.execute_input":"2024-07-09T09:29:47.829910Z","iopub.status.idle":"2024-07-09T09:29:47.833717Z","shell.execute_reply.started":"2024-07-09T09:29:47.829864Z","shell.execute_reply":"2024-07-09T09:29:47.832731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions\ntest_images = test_dataset.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images)\npredictions = np.argmax(probabilities, axis=-1)\n\n# Prepare the submission file\ntest_ids = [idnum.numpy().decode('utf-8') for image, idnum in test_dataset.unbatch()]\nsubmission_df = pd.DataFrame({'id': test_ids, 'label': predictions})\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-09T09:29:49.960097Z","iopub.execute_input":"2024-07-09T09:29:49.960521Z","iopub.status.idle":"2024-07-09T09:30:16.988707Z","shell.execute_reply.started":"2024-07-09T09:29:49.960487Z","shell.execute_reply":"2024-07-09T09:30:16.987332Z"},"trusted":true},"execution_count":null,"outputs":[]}]}