{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":19991,"databundleVersionId":1117522,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!apt-get update && apt-get install -y python3-opencv\n!pip install opencv-python\n\nimport os\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom typing import List, Dict\nfrom tensorflow import keras\nfrom keras import models, layers\nfrom sklearn.model_selection import train_test_split\nfrom kaggle_datasets import KaggleDatasets\n\n!pip install lorem-text \nfrom lorem_text import lorem\n\n!pip install stegano \nfrom stegano import lsb\n\n!pip install efficientnet \nimport efficientnet.tfkeras as efn","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-06T06:27:19.896457Z","iopub.execute_input":"2024-09-06T06:27:19.896800Z","iopub.status.idle":"2024-09-06T06:27:38.211487Z","shell.execute_reply.started":"2024-09-06T06:27:19.896773Z","shell.execute_reply":"2024-09-06T06:27:38.210652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NEW on TPU in TensorFlow 24: shorter cross-compatible TPU/GPU/multi-GPU/cluster-GPU detection code\n\n# Detect TPU, return appropriate distribution strategy\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() \n    \nprint(\"Number of accelerators: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:27:45.770215Z","iopub.execute_input":"2024-09-06T06:27:45.771696Z","iopub.status.idle":"2024-09-06T06:27:53.139114Z","shell.execute_reply.started":"2024-09-06T06:27:45.771657Z","shell.execute_reply":"2024-09-06T06:27:53.138354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For tf.dataset\nAUTO = tf.data.experimental.AUTOTUNE\n\n# Data access\nGCS_DS_PATH = KaggleDatasets().get_gcs_path()\n\n# Configuration\nEPOCHS = 60\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:28:01.141016Z","iopub.execute_input":"2024-09-06T06:28:01.141336Z","iopub.status.idle":"2024-09-06T06:28:01.146566Z","shell.execute_reply.started":"2024-09-06T06:28:01.141309Z","shell.execute_reply":"2024-09-06T06:28:01.145767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:28:30.235171Z","iopub.execute_input":"2024-09-06T06:28:30.235522Z","iopub.status.idle":"2024-09-06T06:28:30.239656Z","shell.execute_reply.started":"2024-09-06T06:28:30.235492Z","shell.execute_reply":"2024-09-06T06:28:30.238951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SHAPE = (512, 512, 3)\n#SHAPE = (128, 128, 3)\n\n\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        efn.EfficientNetB3(\n            input_shape=SHAPE,\n            weights='imagenet',\n            include_top=False\n        ),\n        layers.GlobalAveragePooling2D(),\n        layers.Dense(1, activation='sigmoid')\n    ])\n        \n    model.compile(\n        optimizer='adam',\n        loss = 'binary_crossentropy',\n        metrics=['accuracy']\n    )\n    model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:28:35.088176Z","iopub.execute_input":"2024-09-06T06:28:35.088527Z","iopub.status.idle":"2024-09-06T06:28:59.190287Z","shell.execute_reply.started":"2024-09-06T06:28:35.088496Z","shell.execute_reply":"2024-09-06T06:28:59.189274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_paths(directory):\n    image_paths = []\n    for filename in os.listdir(directory):\n        image_path = os.path.join(directory, filename)\n        image_paths.append(image_path)\n    return image_paths\n\nSAMPLE_SIZE = 5000\ncover_paths = np.array(get_paths('/kaggle/input/alaska2-image-steganalysis/Cover')[:SAMPLE_SIZE])\ncover_labels = np.array([0 for path in cover_paths])\nstego_paths = np.array(get_paths('/kaggle/input/alaska2-image-steganalysis/JUNIWARD')[:SAMPLE_SIZE])\nstego_labels = np.array([1 for path in stego_paths])","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:29:06.292430Z","iopub.execute_input":"2024-09-06T06:29:06.292836Z","iopub.status.idle":"2024-09-06T06:29:07.406641Z","shell.execute_reply.started":"2024-09-06T06:29:06.292802Z","shell.execute_reply":"2024-09-06T06:29:07.405401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\n    cover_paths.shape,\n    cover_labels.shape,\n    stego_paths.shape,\n    stego_labels.shape\n)","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:29:14.475325Z","iopub.execute_input":"2024-09-06T06:29:14.475794Z","iopub.status.idle":"2024-09-06T06:29:14.481058Z","shell.execute_reply.started":"2024-09-06T06:29:14.475759Z","shell.execute_reply":"2024-09-06T06:29:14.480075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_paths = np.concatenate((cover_paths, stego_paths), axis=None)\nimg_labels = np.concatenate((cover_labels, stego_labels), axis=None)\n\nX_train_paths, X_test_paths, y_train, y_test = train_test_split(img_paths, img_labels, test_size=0.15, shuffle=True)\nX_train_paths, X_val_paths, y_train, y_val = train_test_split(X_train_paths, y_train, test_size=0.15, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:29:18.492094Z","iopub.execute_input":"2024-09-06T06:29:18.492522Z","iopub.status.idle":"2024-09-06T06:29:18.501717Z","shell.execute_reply.started":"2024-09-06T06:29:18.492487Z","shell.execute_reply":"2024-09-06T06:29:18.500708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msg = \"Tamaños set {}. X: {}, y: {}\"\nprint(msg.format(\"train\", len(X_train_paths), len(y_train)))\nprint(msg.format(\"test\", len(X_test_paths), len(y_test)))\nprint(msg.format(\"validation\", len(X_val_paths), len(y_val)))","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:29:23.367351Z","iopub.execute_input":"2024-09-06T06:29:23.367757Z","iopub.status.idle":"2024-09-06T06:29:23.373412Z","shell.execute_reply.started":"2024-09-06T06:29:23.367726Z","shell.execute_reply":"2024-09-06T06:29:23.372417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X_train_paths[:5])\nprint(y_train[:5])","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:29:27.546162Z","iopub.execute_input":"2024-09-06T06:29:27.546563Z","iopub.status.idle":"2024-09-06T06:29:27.551848Z","shell.execute_reply.started":"2024-09-06T06:29:27.546514Z","shell.execute_reply":"2024-09-06T06:29:27.550892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_augment(image, label):\n    # data augmentation. Thanks to the dataset.prefetch(AUTO) statement in the next function (below),\n    # this happens essentially for free on TPU. Data pipeline code is executed on the \"CPU\" part\n    # of the TPU while the TPU itself is computing gradients.\n    image = tf.image.random_flip_left_right(image)\n    return image, label  \n\ndef decode_img(path, label):\n    bits = tf.io.read_file(path)\n    image = tf.image.decode_jpeg(bits, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    #image = tf.image.resize(image, (SHAPE[0], SHAPE[1]))\n    return image, label\n\ndef build_dataset(X, y):\n    return tf.data.Dataset.from_tensor_slices((X, y)) \\\n        .map(decode_img, num_parallel_calls=AUTO) \\\n        .map(data_augment, num_parallel_calls=AUTO) \\\n        .batch(BATCH_SIZE) \\\n        .prefetch(AUTO)\n    \ntrain_ds = build_dataset(X_train_paths, y_train)\ntest_ds = build_dataset(X_test_paths, y_test)\nval_ds = build_dataset(X_val_paths, y_val)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:29:31.352897Z","iopub.execute_input":"2024-09-06T06:29:31.353313Z","iopub.status.idle":"2024-09-06T06:29:31.564462Z","shell.execute_reply.started":"2024-09-06T06:29:31.353280Z","shell.execute_reply":"2024-09-06T06:29:31.563437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:29:39.242905Z","iopub.execute_input":"2024-09-06T06:29:39.243322Z","iopub.status.idle":"2024-09-06T06:29:39.249367Z","shell.execute_reply.started":"2024-09-06T06:29:39.243286Z","shell.execute_reply":"2024-09-06T06:29:39.248534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_ds, \n    epochs=EPOCHS,\n    batch_size=BATCH_SIZE,\n    validation_data=val_ds\n)","metadata":{"execution":{"iopub.status.busy":"2024-09-06T06:29:45.853257Z","iopub.execute_input":"2024-09-06T06:29:45.853693Z","iopub.status.idle":"2024-09-06T07:27:05.993986Z","shell.execute_reply.started":"2024-09-06T06:29:45.853658Z","shell.execute_reply":"2024-09-06T07:27:05.992920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find the best epoch and accuracy\nbest_epoch = np.argmax(history.history['accuracy']) + 1\nbest_accuracy = max(history.history['accuracy'])*100\n\n# Find the best epoch and val_accuracy\nbest_epoch_val = np.argmax(history.history['val_accuracy']) + 1\nbest_accuracy_val = max(history.history['val_accuracy'])*100\n\nprint(f\"Best Value Accuracy: {best_accuracy_val:.4f} at Epoch {best_epoch_val}\")\n\nprint(f\"Best Accuracy: {best_accuracy:.4f} at Epoch {best_epoch}\")","metadata":{"execution":{"iopub.status.busy":"2024-09-06T07:40:51.862897Z","iopub.execute_input":"2024-09-06T07:40:51.863280Z","iopub.status.idle":"2024-09-06T07:40:51.869462Z","shell.execute_reply.started":"2024-09-06T07:40:51.863249Z","shell.execute_reply":"2024-09-06T07:40:51.868600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submission\n#submission_test_paths = np.array(get_paths('/kaggle/input/alaska2-image-steganalysis/Test')[:SAMPLE_SIZE])\n#submission_test_paths","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}