{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-12T19:08:16.131591Z","iopub.execute_input":"2023-09-12T19:08:16.132215Z","iopub.status.idle":"2023-09-12T19:08:16.722558Z","shell.execute_reply.started":"2023-09-12T19:08:16.132178Z","shell.execute_reply":"2023-09-12T19:08:16.721682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\nimport matplotlib.pyplot as plt\n\n# Define data preprocessing steps (e.g., resizing, normalization)\nimage_size = (128, 128)  # Change this to your desired image size\nbatch_size = 32\n\ndatagen = keras.preprocessing.image.ImageDataGenerator(\n    rescale=1.0 / 255.0,\n    validation_split=0.2,\n)\n\n# Create a function to parse TFRecords\ndef parse_tfrecord_fn(example):\n    feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'class': tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, feature_description)\n    image = tf.image.decode_jpeg(example['image'], channels=3)\n    image = tf.image.resize(image, image_size)  # Use the image_size you want\n    image = tf.cast(image, tf.float32) / 255.0  # Normalize to [0, 1]\n    label = example['class']\n    return image, label\n\n# Create a dataset from TFRecords\ndef load_dataset(tfrecords_dir):\n    tfrecords_pattern = tfrecords_dir + '/*.tfrec'\n    tfrecords = tf.io.gfile.glob(tfrecords_pattern)\n    autotune = tf.data.AUTOTUNE\n\n    dataset = tf.data.TFRecordDataset(tfrecords)\n    dataset = dataset.map(parse_tfrecord_fn, num_parallel_calls=autotune)\n    \n    return dataset\n\n# Load the datasets\ntfrecords_dir = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-192x192'  # Update this path\nfull_dataset = load_dataset(tfrecords_dir)\n\n# Calculate the total number of samples\nnum_samples = sum(1 for _ in full_dataset)\n\n# Split the dataset into training and validation sets\ntrain_size = int(0.8 * num_samples)\nval_size = num_samples - train_size\n\nif train_size > 0:\n    train_dataset = full_dataset.take(train_size)\n    val_dataset = full_dataset.skip(train_size)\n\n    # Shuffle and batch the datasets\n    train_dataset = train_dataset.shuffle(10000).batch(batch_size)\n    val_dataset = val_dataset.batch(batch_size)\n\n    # Build a simple CNN model\n    model = keras.Sequential([\n        Conv2D(32, (3, 3), activation='relu', input_shape=(128, 128, 3)),\n        MaxPooling2D((2, 2)),\n        Conv2D(64, (3, 3), activation='relu'),\n        MaxPooling2D((2, 2)),\n        Flatten(),\n        Dense(128, activation='relu'),\n        Dense(104, activation='softmax')  # 104 classes for simplicity\n    ])\n\n    # Compile the model\n    model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n\n    # Custom training loop\n    epochs = 10  # Adjust as needed\n\n    train_losses, train_accuracies = [], []\n    val_losses, val_accuracies = [], []\n\n    for epoch in range(epochs):\n        print(f\"Epoch {epoch + 1}/{epochs}\")\n\n        # Training loop\n        train_loss = 0.0\n        train_correct = 0\n        train_total = 0\n\n        for images, labels in train_dataset:\n            with tf.GradientTape() as tape:\n                predictions = model(images)\n                loss = model.compiled_loss(labels, predictions)\n\n            grads = tape.gradient(loss, model.trainable_variables)\n            model.optimizer.apply_gradients(zip(grads, model.trainable_variables))\n\n            train_loss += loss\n            train_correct += tf.reduce_sum(tf.cast(tf.argmax(predictions, axis=1) == labels, tf.int32))\n            train_total += labels.shape[0]\n\n        if train_total > 0:\n            train_losses.append(train_loss / train_total)\n            train_accuracy = train_correct / train_total\n            train_accuracies.append(train_accuracy)\n\n            # Validation loop\n            val_loss = 0.0\n            val_correct = 0\n            val_total = 0\n\n            for images, labels in val_dataset:\n                predictions = model(images)\n                loss = model.compiled_loss(labels, predictions)\n\n                val_loss += loss\n                val_correct += tf.reduce_sum(tf.cast(tf.argmax(predictions, axis=1) == labels, tf.int32))\n                val_total += labels.shape[0]\n\n            val_losses.append(val_loss / val_total)\n            val_accuracy = val_correct / val_total\n            val_accuracies.append(val_accuracy)\n\n            print(f\"Train Loss: {train_loss / train_total:.4f}, Train Accuracy: {train_accuracy:.4f}\")\n            print(f\"Val Loss: {val_loss / val_total:.4f}, Val Accuracy: {val_accuracy:.4f}\")\n        else:\n            print(\"Skipping epoch due to empty training dataset.\")\n\n    # Visualize training history (loss and accuracy)\n    plt.figure(figsize=(12, 4))\n    plt.subplot(1, 2, 1)\n    plt.plot(train_losses, label='Train Loss')\n    plt.plot(val_losses, label='Validation Loss')\n    plt.legend()\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.title('Loss vs. Epochs')\n\n    plt.subplot(1, 2, 2)\n    plt.plot(train_accuracies, label='Train Accuracy')\n    plt.plot(val_accuracies, label='Validation Accuracy')\n    plt.legend()\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.title('Accuracy vs. Epochs')\n\n    plt.show()\nelse:\n    print(\"No samples in the training dataset. Please check your data.\")\n","metadata":{"execution":{"iopub.status.busy":"2023-09-12T19:08:16.723951Z","iopub.execute_input":"2023-09-12T19:08:16.724569Z","iopub.status.idle":"2023-09-12T19:08:27.547012Z","shell.execute_reply.started":"2023-09-12T19:08:16.724534Z","shell.execute_reply":"2023-09-12T19:08:27.545737Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}