{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Wikipedia Image-Caption Competition\n\n### Import libraries","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = \"2\"\n\nimport base64\nimport io\nimport json\nimport numpy as np\nimport tensorflow as tf\nimport time\nimport IPython\nimport PIL\n\nfrom kaggle_secrets import UserSecretsClient\n\n\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print(\"Device:\", tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    strategy = tf.distribute.get_strategy()\n\nAUTOTUNE = tf.data.experimental.AUTOTUNE\nprint(\"Number of replicas:\", strategy.num_replicas_in_sync)\nprint(tf.__version__)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-09T23:17:14.617157Z","iopub.execute_input":"2021-11-09T23:17:14.618335Z","iopub.status.idle":"2021-11-09T23:17:14.628671Z","shell.execute_reply.started":"2021-11-09T23:17:14.618292Z","shell.execute_reply":"2021-11-09T23:17:14.628045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_secrets = UserSecretsClient()\nuser_credential = user_secrets.get_gcloud_credential()\nuser_secrets.set_tensorflow_credential(user_credential)","metadata":{"execution":{"iopub.status.busy":"2021-11-09T23:17:14.630142Z","iopub.execute_input":"2021-11-09T23:17:14.631139Z","iopub.status.idle":"2021-11-09T23:17:16.409196Z","shell.execute_reply.started":"2021-11-09T23:17:14.631095Z","shell.execute_reply":"2021-11-09T23:17:16.407978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Import data\nData consists of first 10 joined datasets (00000-00009) from the archive here: https://analytics.wikimedia.org/published/datasets/one-off/caption_competition/training/joined/","metadata":{}},{"cell_type":"code","source":"DS_PATH = \"../input/wikipedia-train-0\"\n\nDESC_COLUMN = \"caption_title_and_reference_description\"\nIMG_COLUMN = \"b64_bytes\"\nFEAT_COLUMN = \"wit_features\"\nURL_COLUMN = \"image_url\"\n\nfilenames = sorted(os.listdir(DS_PATH))\njson_content = []\n\nstart_time = time.time()\nstep_time = start_time\nfor file in filenames:\n    filename = os.path.join(DS_PATH, file)\n    with open(filename, \"rb\") as fr:\n        for line in fr:\n            if line:\n                obj = json.loads(line)                \n                content = {}\n                content[URL_COLUMN] = obj[URL_COLUMN]\n                content[DESC_COLUMN] = []\n                content[IMG_COLUMN] = obj[IMG_COLUMN]\n                \n                for element in obj[FEAT_COLUMN]:\n                    if element.get(DESC_COLUMN):\n                        content[DESC_COLUMN].append(element[DESC_COLUMN])\n                \n                # Only keep content if both description and image are not empty\n                if content[URL_COLUMN] != \"\" and content[IMG_COLUMN] != \"\" and len(content[DESC_COLUMN]) > 0:\n                    json_content.append(content)\n    cur_time = time.time()\n    print(f\"Import took {(cur_time - step_time):.3f} seconds from file: {file}\")\n    step_time = cur_time\n\nend_time = time.time()\nprint(f\"Time to read {len(filenames)} files: {(end_time - start_time):.3f} seconds\")","metadata":{"execution":{"iopub.status.busy":"2021-11-09T23:17:16.415415Z","iopub.execute_input":"2021-11-09T23:17:16.415996Z","iopub.status.idle":"2021-11-09T23:19:53.812397Z","shell.execute_reply.started":"2021-11-09T23:17:16.415957Z","shell.execute_reply":"2021-11-09T23:19:53.811511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(json_content[0].keys())\nprint(f\"Total items: {len(json_content)}\")","metadata":{"execution":{"iopub.status.busy":"2021-11-09T23:19:53.813516Z","iopub.execute_input":"2021-11-09T23:19:53.813729Z","iopub.status.idle":"2021-11-09T23:19:53.819098Z","shell.execute_reply.started":"2021-11-09T23:19:53.813703Z","shell.execute_reply":"2021-11-09T23:19:53.818478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualize data","metadata":{}},{"cell_type":"code","source":"def display_from_json(content):\n    decoded = base64.b64decode(content[IMG_COLUMN])\n    image = PIL.Image.open(io.BytesIO(decoded)).convert(\"RGB\")\n    print(f\"\\nImage URL: {content[URL_COLUMN]}\")\n    print(f\"Description: {content[DESC_COLUMN]}\\n\")\n    IPython.display.display(image)\n\n    \nfor _ in range(3):\n    rand_index = np.random.randint(0, len(json_content))\n    display_from_json(json_content[rand_index])","metadata":{"execution":{"iopub.status.busy":"2021-11-09T23:23:18.777189Z","iopub.execute_input":"2021-11-09T23:23:18.778410Z","iopub.status.idle":"2021-11-09T23:23:18.875612Z","shell.execute_reply.started":"2021-11-09T23:23:18.778352Z","shell.execute_reply":"2021-11-09T23:23:18.874666Z"},"trusted":true},"execution_count":null,"outputs":[]}]}