{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Alaska2,Try to predict by EfficientNet","metadata":{"trusted":true}},{"cell_type":"markdown","source":"As my first try for the ALSKA2 competition, I made predictions using tensorflow efficient net B7 model.<br>","metadata":{}},{"cell_type":"code","source":"! pip install -q efficientnet","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:29:21.326475Z","iopub.execute_input":"2022-03-22T18:29:21.326938Z","iopub.status.idle":"2022-03-22T18:29:31.855665Z","shell.execute_reply.started":"2022-03-22T18:29:21.326814Z","shell.execute_reply":"2022-03-22T18:29:31.854872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Basic library\nimport numpy as np \nimport pandas as pd \nimport os\n\n# Data preprocessing\nfrom sklearn.model_selection import train_test_split\n\n# Visualization\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nplt.style.use('fivethirtyeight')\n\n# tensorflow\nimport tensorflow as tf\nimport tensorflow.keras.layers as l\nimport efficientnet.tfkeras as efn\n\n# data set\nfrom kaggle_datasets import KaggleDatasets","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2022-03-22T18:30:19.031183Z","iopub.execute_input":"2022-03-22T18:30:19.031557Z","iopub.status.idle":"2022-03-22T18:30:19.039339Z","shell.execute_reply.started":"2022-03-22T18:30:19.031525Z","shell.execute_reply":"2022-03-22T18:30:19.038398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"TPU Setting","metadata":{}},{"cell_type":"code","source":"# TPU setting\n# detect and init the TPU\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(tpu)\ntf.tpu.experimental.initialize_tpu_system(tpu)\n\n# instantiate a distribution strategy\ntpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:30:26.189820Z","iopub.execute_input":"2022-03-22T18:30:26.190544Z","iopub.status.idle":"2022-03-22T18:30:32.090536Z","shell.execute_reply.started":"2022-03-22T18:30:26.190497Z","shell.execute_reply":"2022-03-22T18:30:32.089543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tpu.master())\nprint(tpu_strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:30:47.442073Z","iopub.execute_input":"2022-03-22T18:30:47.442748Z","iopub.status.idle":"2022-03-22T18:30:47.449328Z","shell.execute_reply.started":"2022-03-22T18:30:47.442701Z","shell.execute_reply":"2022-03-22T18:30:47.448239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For tensorflow dataset\nAUTO = tf.data.experimental.AUTOTUNE\nignore_order = tf.data.Options()\nignore_order.experimental_deterministic = False\n\n# Pass\ngcs_path = KaggleDatasets().get_gcs_path()","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:30:50.782573Z","iopub.execute_input":"2022-03-22T18:30:50.783463Z","iopub.status.idle":"2022-03-22T18:30:51.205611Z","shell.execute_reply.started":"2022-03-22T18:30:50.783423Z","shell.execute_reply":"2022-03-22T18:30:51.204530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data set loadng","metadata":{}},{"cell_type":"code","source":"# Sample dataframe\nsample = pd.read_csv(\"/kaggle/input/alaska2-image-steganalysis/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:30:58.084019Z","iopub.execute_input":"2022-03-22T18:30:58.084629Z","iopub.status.idle":"2022-03-22T18:30:58.095619Z","shell.execute_reply.started":"2022-03-22T18:30:58.084592Z","shell.execute_reply":"2022-03-22T18:30:58.094731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# batch size in tpu\nBATCH_SIZE = 32 * tpu_strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:31:05.011682Z","iopub.execute_input":"2022-03-22T18:31:05.012213Z","iopub.status.idle":"2022-03-22T18:31:05.016067Z","shell.execute_reply.started":"2022-03-22T18:31:05.012176Z","shell.execute_reply":"2022-03-22T18:31:05.015376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Directrory and file name\n# Drop csv from dir_name\ndir_name = ['Test', 'JUNIWARD', 'JMiPOD', 'Cover', 'UERD']\ndf = pd.DataFrame({})\n\n# Create empty dataframe and list\nlists = []\ncate = []\n\n# get the filenames\nfor dir_ in dir_name:\n    # file name\n    list_ = os.listdir(\"/kaggle/input/alaska2-image-steganalysis/\"+dir_+\"/\")\n    lists = lists+list_\n    # category name\n    cate_ = np.tile(dir_,len(list_))\n    cate = np.concatenate([cate,cate_])\n    \n# insert dataframe\ndf[\"cate\"] = cate\ndf[\"name\"] = lists","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:31:10.095876Z","iopub.execute_input":"2022-03-22T18:31:10.096337Z","iopub.status.idle":"2022-03-22T18:31:16.953035Z","shell.execute_reply.started":"2022-03-22T18:31:10.096278Z","shell.execute_reply":"2022-03-22T18:31:16.951902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data and path preprocessing","metadata":{}},{"cell_type":"code","source":"# path line\ndf[\"path\"] = [str(os.path.join(gcs_path,cate,name)) for cate, name in zip(df[\"cate\"], df[\"name\"])]","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:31:26.415892Z","iopub.execute_input":"2022-03-22T18:31:26.417006Z","iopub.status.idle":"2022-03-22T18:31:27.451582Z","shell.execute_reply.started":"2022-03-22T18:31:26.416936Z","shell.execute_reply":"2022-03-22T18:31:27.450216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Labeling positive and negative\ndef cate_label(x):\n    if x[\"cate\"] == \"Cover\":\n        res = 0\n    else:\n        res = 1\n    return res\n\n# Test dataframe and Train dataframe\nTest_df = df.query(\"cate=='Test'\").sort_values(by=\"name\")\nTrain_df = df.query(\"cate!='Test'\")\n# Apply the function\nTrain_df[\"flg\"] = df.apply(cate_label, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:31:31.035502Z","iopub.execute_input":"2022-03-22T18:31:31.035869Z","iopub.status.idle":"2022-03-22T18:31:35.414578Z","shell.execute_reply.started":"2022-03-22T18:31:31.035828Z","shell.execute_reply":"2022-03-22T18:31:35.413753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label_counts\nTrain_df[\"cate\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:31:41.778182Z","iopub.execute_input":"2022-03-22T18:31:41.779104Z","iopub.status.idle":"2022-03-22T18:31:41.840851Z","shell.execute_reply.started":"2022-03-22T18:31:41.779062Z","shell.execute_reply":"2022-03-22T18:31:41.839877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since I need to keeping memory and running time over, the number of samples was smalled 60000 data wset.","metadata":{}},{"cell_type":"code","source":"Train_df = Train_df.sample(60000)\n# label_counts\nTrain_df[\"cate\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:32:38.595950Z","iopub.execute_input":"2022-03-22T18:32:38.596763Z","iopub.status.idle":"2022-03-22T18:32:38.647632Z","shell.execute_reply.started":"2022-03-22T18:32:38.596709Z","shell.execute_reply":"2022-03-22T18:32:38.646456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create train data and val data\nX = Train_df[\"path\"]\ny = Train_df[\"flg\"]\n\n# split train and val data\nX_train, X_val, y_train, y_val = train_test_split(X,y, test_size=0.2, random_state=10)","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:32:43.945762Z","iopub.execute_input":"2022-03-22T18:32:43.946095Z","iopub.status.idle":"2022-03-22T18:32:43.967395Z","shell.execute_reply.started":"2022-03-22T18:32:43.946062Z","shell.execute_reply":"2022-03-22T18:32:43.966345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Change to numpy array\nX_train, X_val, y_train, y_val = np.array(X_train), np.array(X_val), np.array(y_train), np.array(y_val)","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:32:51.602033Z","iopub.execute_input":"2022-03-22T18:32:51.602462Z","iopub.status.idle":"2022-03-22T18:32:51.611518Z","shell.execute_reply.started":"2022-03-22T18:32:51.602420Z","shell.execute_reply":"2022-03-22T18:32:51.610499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create test data\nX_test = np.array(Test_df[\"path\"])","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:32:58.370111Z","iopub.execute_input":"2022-03-22T18:32:58.371022Z","iopub.status.idle":"2022-03-22T18:32:58.376096Z","shell.execute_reply.started":"2022-03-22T18:32:58.370965Z","shell.execute_reply":"2022-03-22T18:32:58.375395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(filename, label=None, image_size=(512,512)):\n    bits = tf.io.read_file(filename)\n    image = tf.image.decode_jpeg(bits, channels=3)\n    image = tf.cast(image, tf.float32)/255.0\n    image = tf.image.resize(image, image_size)\n    \n    if label is None:\n        return image\n    else:\n        return image, label","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:33:01.578419Z","iopub.execute_input":"2022-03-22T18:33:01.579111Z","iopub.status.idle":"2022-03-22T18:33:01.586851Z","shell.execute_reply.started":"2022-03-22T18:33:01.579051Z","shell.execute_reply":"2022-03-22T18:33:01.586092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nignore_order = tf.data.Options()\nignore_order.experimental_deterministic = False\n\n# Build input pipeline\ntrain_dataset = (tf.data.Dataset.from_tensor_slices((X_train, y_train)).prefetch(AUTO).with_options(ignore_order)\n                 .map(decode_image, num_parallel_calls=AUTO).shuffle(512).batch(BATCH_SIZE).repeat())\n\nvalid_dataset = (tf.data.Dataset.from_tensor_slices((X_val, y_val)).map(decode_image, num_parallel_calls=AUTO)\n                    .cache().batch(BATCH_SIZE).prefetch(AUTO))\n\ntest_dataset = (tf.data.Dataset.from_tensor_slices((X_test)).map(decode_image, num_parallel_calls=AUTO)\n                    .batch(BATCH_SIZE))","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:33:04.406975Z","iopub.execute_input":"2022-03-22T18:33:04.407865Z","iopub.status.idle":"2022-03-22T18:33:04.901758Z","shell.execute_reply.started":"2022-03-22T18:33:04.407821Z","shell.execute_reply":"2022-03-22T18:33:04.900604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"markdown","source":"Model : EfficientNetB7<br>","metadata":{}},{"cell_type":"code","source":"with tpu_strategy.scope():\n    model_b7 = tf.keras.Sequential([\n        efn.EfficientNetB7(input_shape=(512,512,3),weights='imagenet',include_top=False),\n        l.GlobalAveragePooling2D(),\n        l.Dense(1, activation=\"sigmoid\")\n    ])\n    \n    model_b7.compile(optimizer=\"adam\", loss=\"binary_crossentropy\", metrics=[\"accuracy\"])\n    model_b7.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:33:11.079512Z","iopub.execute_input":"2022-03-22T18:33:11.080361Z","iopub.status.idle":"2022-03-22T18:33:57.082331Z","shell.execute_reply.started":"2022-03-22T18:33:11.080295Z","shell.execute_reply":"2022-03-22T18:33:57.081361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Calculation","metadata":{}},{"cell_type":"code","source":"STEPS_PER_EPOCH = X_train.shape[0] // BATCH_SIZE\ncallbacks = [tf.keras.callbacks.EarlyStopping(patience=3, restore_best_weights=True)]\n\nEPOCHS = 5\nhist_b7 = model_b7.fit(train_dataset, epochs=EPOCHS,\n                   steps_per_epoch=STEPS_PER_EPOCH, validation_data=valid_dataset, callbacks=callbacks, workers=4, use_multiprocessing=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-22T18:34:34.930395Z","iopub.execute_input":"2022-03-22T18:34:34.931185Z","iopub.status.idle":"2022-03-22T20:38:40.703495Z","shell.execute_reply.started":"2022-03-22T18:34:34.931115Z","shell.execute_reply":"2022-03-22T20:38:40.701493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prediction\npred_b7 = model_b7.predict(test_dataset, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-22T20:38:57.957226Z","iopub.execute_input":"2022-03-22T20:38:57.957620Z","iopub.status.idle":"2022-03-22T20:42:10.318571Z","shell.execute_reply.started":"2022-03-22T20:38:57.957586Z","shell.execute_reply":"2022-03-22T20:42:10.317309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training history\ntrain_loss = hist_b7.history[\"loss\"]\nval_loss = hist_b7.history[\"val_loss\"]\ntrain_acc = hist_b7.history[\"accuracy\"]\nval_acc = hist_b7.history[\"val_accuracy\"]\n\nfig, ax = plt.subplots(1,2,figsize=(10,6))\nax[0].plot(range(len(train_loss)), train_loss, label=\"train_loss\")\nax[0].plot(range(len(val_loss)), val_loss, label=\"val_loss\")\nax[0].set_xlabel(\"epochs\")\nax[0].set_ylabel(\"loss\")\nax[0].set_title(\"EfficientNetB7 loss\")\nax[0].legend()\n\nax[1].plot(range(len(train_acc)), train_acc, label=\"train_accuracy\")\nax[1].plot(range(len(val_acc)), val_acc, label=\"val_accuracy\")\nax[1].set_xlabel(\"epochs\")\nax[1].set_ylabel(\"accuracy\")\nax[1].set_title(\"EfficientNetB7 accurary\")\nax[1].legend()","metadata":{"execution":{"iopub.status.busy":"2022-03-22T20:42:38.956027Z","iopub.execute_input":"2022-03-22T20:42:38.956472Z","iopub.status.idle":"2022-03-22T20:42:39.548456Z","shell.execute_reply.started":"2022-03-22T20:42:38.956415Z","shell.execute_reply":"2022-03-22T20:42:39.547558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"# EfficientNetB7\nsample_7 = sample.copy()\nsample_7[\"Label\"] = pred_b7\nsample_7.to_csv(\"submission_B7.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-22T20:43:47.834810Z","iopub.execute_input":"2022-03-22T20:43:47.835215Z","iopub.status.idle":"2022-03-22T20:43:47.860791Z","shell.execute_reply.started":"2022-03-22T20:43:47.835157Z","shell.execute_reply":"2022-03-22T20:43:47.859074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_7[\"Label\"].describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-22T20:44:31.321752Z","iopub.execute_input":"2022-03-22T20:44:31.322444Z","iopub.status.idle":"2022-03-22T20:44:31.332907Z","shell.execute_reply.started":"2022-03-22T20:44:31.322387Z","shell.execute_reply":"2022-03-22T20:44:31.332196Z"},"trusted":true},"execution_count":null,"outputs":[]}]}