{"cells":[{"metadata":{"_uuid":"902a6598fce0e425315de411150bab718e5fbc7b"},"cell_type":"markdown","source":"This kernel will take the Google tsv files that were designed for downloading the full dataset and remove all the lines that are not in the training set. Make sure you run it with \"internet connected\"\n\nDownload the trimmed tsv from the output tab of the kernel"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"430a35352bc7cbca4f8ee275c61fcb23477d6d48","collapsed":true},"cell_type":"code","source":"!wget https://storage.googleapis.com/openimages/2018_04/train/train-images-boxable-with-rotation.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d05800641f6f9d94812e1f8b3125d7a64722b6c2","collapsed":true},"cell_type":"code","source":"#load the image list\ntrain_images = {}\nwith open(\"train-images-boxable-with-rotation.csv\", \"r\") as f:\n    for l in f:\n        r = l.split(\",\")\n        if r[0] == 'ImageID':\n            continue\n        train_images[r[2]] = r[0]\n            \nprint(\"loaded {} image ids\".format(len(train_images)))\n    ","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"collapsed":true},"cell_type":"code","source":"!wget https://storage.googleapis.com/cvdf-datasets/oid/open-images-dataset-train0.tsv\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"88c86c63b44080bc6e057e7b6d917065886ecfc5","collapsed":true},"cell_type":"code","source":"#screen the images to include only the ones in the training set\ninfn = \"open-images-dataset-train0.tsv\"\ncount = 0\noutf = open(\"trimmed_\"+infn, \"w+\")\n\nwith open(infn, \"r\") as f:\n    for l in f:\n        r = l[:-1].split(\"\\t\")\n        if len(r) == 1: #header\n            outf.write(l)\n            continue\n        try:\n            id = train_images[r[0]]\n            outf.write(l)\n            count += 1\n        except:\n            continue\n#        print(r[2])\noutf.close()\nprint(count)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a666682e1c4aa240593a885bd296fe403b7e3e45","collapsed":true},"cell_type":"code","source":"!head trimmed*","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}