{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# !pip install --upgrade tensorflow\n# !pip install --upgrade transformers\n\nfrom transformers import CLIPProcessor, TFCLIPVisionModel, CLIPFeatureExtractor\nimport tensorflow as tf; print(tf.__version__)\nimport tensorflow_addons as tfa; print(tfa.__version__)\nimport matplotlib.pyplot as plt\nfrom zipfile import ZipFile\nimport tensorflow as tf\nfrom PIL import Image\nimport pandas as pd\nimport numpy as np\nimport requests\nimport os\n\n\n# model = TFCLIPVisionModel.from_pretrained(\"openai/clip-vit-base-patch32\")\n# extractor = CLIPFeatureExtractor.from_pretrained(\"openai/clip-vit-base-patch32\")\n\nmodel = TFCLIPVisionModel.from_pretrained(\"openai/clip-vit-large-patch14\")\n# extractor = CLIPFeatureExtractor.from_pretrained(\"openai/clip-vit-large-patch14\")\n\nurl = \"http://images.cocodataset.org/val2017/000000039769.jpg\"\nimage = Image.open(requests.get(url, stream=True).raw)\ntf_img = tf.expand_dims(np.asarray(image), axis=0)\n\nimage","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T12:47:09.478156Z","iopub.execute_input":"2022-07-28T12:47:09.479254Z","iopub.status.idle":"2022-07-28T12:47:10.294233Z","shell.execute_reply.started":"2022-07-28T12:47:09.479205Z","shell.execute_reply":"2022-07-28T12:47:10.291953Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def hf_preprocess_symbolic_input(x, data_format=\"channels_last\", mode=\"torch\"):\n  \"\"\"Preprocesses a tensor encoding a batch of images.\n  Args:\n    x: Input tensor, 3D or 4D.\n    data_format: Data format of the image tensor.\n    mode: One of \"caffe\", \"tf\" or \"torch\".\n      - caffe: will convert the images from RGB to BGR,\n          then will zero-center each color channel with\n          respect to the ImageNet dataset,\n          without scaling.\n      - tf: will scale pixels between -1 and 1,\n          sample-wise.\n      - torch: will scale pixels between 0 and 1 and then\n          will normalize each channel with respect to the\n          ImageNet dataset.\n  Returns:\n      Preprocessed tensor.\n  \"\"\"\n  if mode == 'tf':\n    x /= 127.5\n    x -= 1.\n    return x\n  elif mode == 'torch':\n    x /= 255.\n    mean = [0.48145466, 0.4578275, 0.40821073]\n    std = [0.26862954, 0.26130258, 0.27577711]\n  else:\n    if data_format == 'channels_first':\n      # 'RGB'->'BGR'\n      if tf.keras.backend.ndim(x) == 3:\n        x = x[::-1, ...]\n      else:\n        x = x[:, ::-1, ...]\n    else:\n      # 'RGB'->'BGR'\n      x = x[..., ::-1]\n    mean = [103.939, 116.779, 123.68]\n    std = None\n\n  mean_tensor = tf.keras.backend.constant(-np.array(mean))\n\n  # Zero-center by mean pixel\n  if tf.keras.backend.dtype(x) != tf.keras.backend.dtype(mean_tensor):\n    x = tf.keras.backend.bias_add(\n        x, tf.keras.backend.cast(mean_tensor, tf.keras.backend.dtype(x)), data_format=data_format)\n  else:\n    x = tf.keras.backend.bias_add(x, mean_tensor, data_format)\n  if std is not None:\n    std_tensor = tf.keras.backend.constant(np.array(std), dtype=tf.keras.backend.dtype(x))\n    if data_format == 'channels_first':\n      std_tensor = tf.keras.backend.reshape(std_tensor, (-1, 1, 1))\n    x /= std_tensor\n  return x","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Contrast\ntf_img_batch = tf.repeat(tf_img, 4, axis=0)\ntf_img_batch_1 = tf.keras.layers.RandomContrast(factor=0.025)(tf_img_batch, training=True)\n\n# 2. Flip\ntf_img_batch_2 = tf.keras.layers.RandomFlip()(tf_img_batch_1, training=True)\n\n# 3. Zoom\ntf_img_batch_3 = tf.keras.layers.RandomZoom((-0.025, 0.025))(tf_img_batch_2, training=True)\n\n# 4. Rotation\ntf_img_batch_4 = tf.keras.layers.RandomRotation(0.025, fill_mode=\"reflect\")(tf_img_batch_3, training=True)\n\n# 4. Translation\ntf_img_batch_5 = tf.keras.layers.RandomTranslation(height_factor=(-0.025, 0.025), width_factor=(-0.025, 0.025))(tf_img_batch_4, training=True)\n\nplt.figure(figsize=(5,3))\nplt.imshow(tf_img_batch[0])\nplt.title(\"ORIGINAL\")\nplt.axis(False)\nplt.show()\n\nprint(\"\\n\\n... CONTRAST ...\\n\")\nplt.figure(figsize=(20,8))\nfor i in range(4):\n    plt.subplot(1,4,i+1)\n    plt.imshow(tf_img_batch_1[i])\n    plt.axis(False)\nplt.tight_layout()\nplt.show()\n\nprint(\"\\n\\n... FLIP ...\\n\")\nplt.figure(figsize=(20,8))\nfor i in range(4):\n    plt.subplot(1,4,i+1)\n    plt.imshow(tf_img_batch_2[i])\n    plt.axis(False)\nplt.tight_layout()\nplt.show()\n\nprint(\"\\n\\n... ZOOM ...\\n\")\nplt.figure(figsize=(20,8))\nfor i in range(4):\n    plt.subplot(1,4,i+1)\n    plt.imshow(tf_img_batch_3[i])\n    plt.axis(False)\nplt.tight_layout()\nplt.show()\n\nprint(\"\\n\\n... ROTATION ...\\n\")\nplt.figure(figsize=(20,8))\nfor i in range(4):\n    plt.subplot(1,4,i+1)\n    plt.imshow(tf_img_batch_4[i])\n    plt.axis(False)\nplt.tight_layout()\nplt.show()\n\nprint(\"\\n\\n... TRANSLATION ...\\n\")\nplt.figure(figsize=(20,8))\nfor i in range(4):\n    plt.subplot(1,4,i+1)\n    plt.imshow(tf_img_batch_5[i])\n    plt.axis(False)\nplt.tight_layout()\nplt.show()\n\n# def build_clip_guie_model(clip_model, _h=224, _w=224, tta_n=2):\n    \n#     # Inputs\n#     _inputs = tf.keras.layers.Input(shape=(None,None,3), batch_size=1, dtype=tf.uint8, name=\"inputs\")\n\n#     # TTA\n#     x1 = tf.keras.layers.Lambda(lambda _x: tf.repeat(_x, repeats=tta_n, axis=0))(_inputs)\n\n#     # 1. Contrast\n#     x2 = tf.keras.layers.RandomContrast(factor=0.025)(x1, training=True)\n\n#     # 2. Flip\n#     x2 = tf.keras.layers.RandomFlip(mode=\"horizontal\")(x2, training=True)\n\n#     # 3. Zoom\n#     x2 = tf.keras.layers.RandomZoom((-0.025, 0.025))(x2, training=True)\n\n#     # 4. Rotation\n#     x2 = tf.keras.layers.RandomRotation(0.025, fill_mode=\"reflect\")(x2, training=True)\n\n#     # 4. Translation\n#     x2 = tf.keras.layers.RandomTranslation(height_factor=(-0.025, 0.025), width_factor=(-0.025, 0.025))(x2, training=True)\n    \n#     x = tf.keras.layers.Concatenate(axis=0)([x1,x2])\n    \n#     # Resize, normalize pixel values & move channels-last to channels-first\n#     x = tf.keras.layers.Resizing(_h, _w, crop_to_aspect_ratio=True)(tf.cast(x, tf.float32))\n#     x = hf_preprocess_symbolic_input(x)\n#     x = tf.keras.layers.Permute((3,1,2))(x) #bhwc --> bchw\n    \n#     # Inference\n#     x = clip_model({'pixel_values':x}).pooler_output\n    \n#     # Adaptive Pooling\n#     x = tf.keras.layers.Reshape((-1,1))(x)\n#     x = tfa.layers.AdaptiveAveragePooling1D(output_size=64)(x)\n#     x = tf.keras.layers.Lambda(lambda _x: tf.math.reduce_mean(_x, axis=0, keepdims=True))(x)\n    \n#     # Embedding and embedding norm\n#     _output_1 = tf.keras.layers.Reshape((-1,), name=\"embedding\")(x)\n#     _output_2 = tf.keras.layers.Lambda(lambda x: tf.nn.l2_normalize(x), name=\"embedding_norm\")(_output_1)    \n    \n#     # Return model\n#     return tf.keras.Model(inputs=_inputs, outputs=[_output_1, _output_2])\n\n# guie_model = build_clip_guie_model(model)\n# guie_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T12:52:18.190392Z","iopub.execute_input":"2022-07-28T12:52:18.190754Z","iopub.status.idle":"2022-07-28T12:52:25.087995Z","shell.execute_reply.started":"2022-07-28T12:52:18.190722Z","shell.execute_reply":"2022-07-28T12:52:25.086984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_clip_guie_model(clip_model, _h=224, _w=224, output_dim=64, _inter=\"bicubic\", dim_redux=\"aap\"):\n    \n    # Inputs\n    _inputs = tf.keras.layers.Input(shape=(None,None,3), batch_size=1, dtype=tf.uint8, name=\"inputs\")\n\n    # Resize, normalize pixel values & move channels-last to channels-first\n    x = tf.keras.layers.Resizing(_h, _w, crop_to_aspect_ratio=True)(tf.cast(_inputs, tf.float32))\n    x = hf_preprocess_symbolic_input(x)\n    x = tf.keras.layers.Permute((3,1,2))(x) #bhwc --> bchw\n    \n    # Inference\n    x = clip_model({'pixel_values':x}).pooler_output\n    \n    # Adaptive Pooling\n    x = tf.keras.layers.Reshape((-1,1))(x)\n    \n    if dim_redux==\"aap\":\n        x = tf.keras.layers.Reshape((-1,1))(x)\n        x = tfa.layers.AdaptiveAveragePooling1D(output_size=output_dim)(x)\n        _output_1 = tf.keras.layers.Reshape((-1,), name=\"embedding\")(x)\n    elif dim_redux==\"amp\":\n        x = tf.keras.layers.Reshape((-1,1))(x)\n        x = tfa.layers.AdaptiveMaxPooling1D(output_size=output_dim)(x)\n        _output_1 = tf.keras.layers.Reshape((-1,), name=\"embedding\")(x)\n    elif dim_redux==\"rand\":\n        x = tf.keras.layers.Reshape((-1,))(x)\n        _output_1 = tf.keras.layers.Dense(output_dim, use_bias=False, name=\"embedding\")(x)\n    \n    # embedding normalization\n    _output_2 = tf.keras.layers.Lambda(lambda x: tf.nn.l2_normalize(x), name=\"embedding_norm\")(_output_1)    \n    \n    # Return model\n    return tf.keras.Model(inputs=_inputs, outputs=[_output_1, _output_2])\n\nREDUX_METHOD=\"aap\"\nguie_model = build_clip_guie_model(model, dim_redux=REDUX_METHOD)\nguie_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T12:56:58.958860Z","iopub.execute_input":"2022-07-28T12:56:58.959295Z","iopub.status.idle":"2022-07-28T12:57:01.934323Z","shell.execute_reply.started":"2022-07-28T12:56:58.959260Z","shell.execute_reply":"2022-07-28T12:57:01.933276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_submission_zip(model_dir, output_dir=\".\"):\n    with ZipFile(os.path.join(output_dir, 'submission.zip'),'w') as zip:           \n        zip.write(os.path.join(model_dir, 'saved_model.pb'), arcname='saved_model.pb') \n        zip.write(os.path.join(model_dir, 'variables', 'variables.data-00000-of-00001'), arcname='variables/variables.data-00000-of-00001') \n        zip.write(os.path.join(model_dir, 'variables', 'variables.index'), arcname='variables/variables.index') \n\n# Save fresh model to directory\n!rm -rf ./models\nos.makedirs(\"./models\", exist_ok=True)\nguie_model.save(\"./models\")\n\n# Show unzipped contents\nprint(os.listdir(\"./models\"))\n\nmake_submission_zip(\"./models\", output_dir=\".\")","metadata":{"execution":{"iopub.status.busy":"2022-07-28T12:47:10.302853Z","iopub.status.idle":"2022-07-28T12:47:10.304488Z","shell.execute_reply.started":"2022-07-28T12:47:10.304190Z","shell.execute_reply":"2022-07-28T12:47:10.304217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf ./model*\n!rm -rf ./tmp*\n!ls .","metadata":{"execution":{"iopub.status.busy":"2022-07-28T12:47:10.305990Z","iopub.status.idle":"2022-07-28T12:47:10.306766Z","shell.execute_reply.started":"2022-07-28T12:47:10.306513Z","shell.execute_reply":"2022-07-28T12:47:10.306536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}