{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"\nimport math, re, os\n\nimport numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nimport tensorflow as tf\nimport tensorflow.keras.layers as L\nfrom sklearn import metrics\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# For tf.dataset\nAUTO = tf.data.experimental.AUTOTUNE\n\n# Data access\nGCS_DS_PATH = KaggleDatasets().get_gcs_path()\n\n# Configuration\nEPOCHS = 1\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndef append_path(pre):\n    return np.vectorize(lambda file: os.path.join(GCS_DS_PATH,pre,file))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/alaska2-image-steganalysis/sample_submission.csv')\ntrain_filenames=np.array(os.listdir('/kaggle/input/alaska2-image-steganalysis/Cover/'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.random.seed(0)\npositives = train_filenames.copy()\nnegatives = train_filenames.copy()\n\nnp.random.shuffle(positives)\nnp.random.shuffle(negatives)\n\njmipod=append_path('JMiPOD')(positives[:10000])\njuniward=append_path('JUNIWARD')(positives[10000:])\nuerd=append_path('UERD')(positives[20000:30000])\n\npos_paths=np.concatenate([jmipod, juniward,uerd])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_paths=append_path('Test')(sub.Id.values)\nneg_paths =append_path('Cover')(negatives[:30000])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_paths=np.concatenate([pos_paths,neg_paths])\ntrain_labels= np.array([1] * len(pos_paths)+[0] * len(neg_paths))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_paths, valid_paths, train_labels, valid_labels=train_test_split(train_paths,train_labels,test_size=0.2,random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def decode_image(filename, label=None, image_size=(512,512)):\n    bits=tf.io.read_file(filename)\n    image=tf.image.decode_jpeg(bits,channels=3)\n    image=tf.cast(image,tf.float32)#image to tf.float32 data type\n    image=tf.image.resize(image,image_size)\n    \n    if label is None:\n        return image\n    else:\n        return image,label\n\ndef data_augment(image,label=None):\n    image=tf.image.random_flip_left_right(image)\n    image=tf.image.random_flip_up_down(image)\n    \n    if label is None:\n        return image\n    else:\n        return image,label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dataset = (tf.data.Dataset\n                 .from_tensor_slices((train_paths,train_labels))\n                 .map(decode_image, num_parallel_calls=AUTO)\n                 .cache()\n                 .repeat()\n                 .shuffle(1024)\n                 .batch(BATCH_SIZE)\n                 .prefetch(AUTO)\n                )\nvalid_dataset= (tf.data.Dataset\n                .from_tensor_slices((valid_paths,valid_labels))\n                .map(decode_image, num_parallel_calls=AUTO)\n                .batch(BATCH_SIZE)\n                .prefetch(AUTO)\n               )\ntest_dataset= (tf.data.Dataset\n               .from_tensor_slices(test_paths)\n               .map(decode_image, num_parallel_calls=AUTO)\n               .batch(BATCH_SIZE)\n              )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_lrfn(lr_start=0.0001,lr_max=0.000075,\n              lr_min=0.000001, lr_rampup_epochs=20,\n              lr_sustain_epochs=0,lr_exp_decay=0.8):\n    lr_max = lr_max * strategy.num_replicas_in_sync\n    \n    def lrfn(epoch):\n        if epoch < lr_rampup_epochs:\n            lr = (lr_max- lr_start)/lr_rampup * epoch + lr_start\n        elif epoch < lr_rampup_epochs + lr_sustain_epochs:\n            lr = lr_max\n        else:\n            lr = (lr_max - lr_min) * lr_exp_decay ** (epoch - lr_rampup_epochs - lr_sustain_epochs) + lr_min\n            return lr\n        \n        return lrfn","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.applications.resnet50 import ResNet50","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with strategy.scope():\n    model = tf.keras.Sequential([\n        ResNet50(\n        input_shape=(512,512,3),\n        weights='imagenet',\n        include_top=False),\n        L.GlobalAveragePooling2D(),\n        L.Dense(1, activation='sigmoid')\n    ])\n    \n    model.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['accuracy']\n    )\n    model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"STEPS_PER_EPOCH=train_labels.shape[0] // BATCH_SIZE\n\nhistory= model.fit(\ntrain_dataset,\nepochs=EPOCHS,\nsteps_per_epoch=STEPS_PER_EPOCH,\nvalidation_data=valid_dataset)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-output":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}