{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"from __future__ import print_function\nfrom imageio import imread\n#from scipy import ndimage\nfrom collections import Counter\nfrom core.vggnet import Vgg19\nfrom core.utils import *\n\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport hickle\nimport os\nimport json\n\n\ndef _process_caption_data(caption_file, image_dir, max_length):\n    with open(caption_file) as f:\n        caption_data = json.load(f)\n\n    # id_to_filename is a dictionary such as {image_id: filename]} \n    id_to_filename = {image['id']: image['file_name'] for image in caption_data['images']}\n\n    # data is a list of dictionary which contains 'captions', 'file_name' and 'image_id' as key.\n    data = []\n    for annotation in caption_data['annotations']:\n        image_id = annotation['image_id']\n        annotation['file_name'] = os.path.join(image_dir, id_to_filename[image_id])\n        data += [annotation]\n    \n    # convert to pandas dataframe (for later visualization or debugging)\n    caption_data = pd.DataFrame.from_dict(data)\n    del caption_data['id']\n    caption_data.sort_values(by='image_id', inplace=True)\n    caption_data = caption_data.reset_index(drop=True)\n    \n    del_idx = []\n    for i, caption in enumerate(caption_data['caption']):\n        caption = caption.replace('.','').replace(',','').replace(\"'\",\"\").replace('\"','')\n        caption = caption.replace('&','and').replace('(','').replace(\")\",\"\").replace('-',' ')\n        caption = \" \".join(caption.split())  # replace multiple spaces\n        \n        caption_data.set_value(i, 'caption', caption.lower())\n        if len(caption.split(\" \")) > max_length:\n            del_idx.append(i)\n    \n    # delete captions if size is larger than max_length\n    print \"The number of captions before deletion: %d\" %len(caption_data)\n    caption_data = caption_data.drop(caption_data.index[del_idx])\n    caption_data = caption_data.reset_index(drop=True)\n    print \"The number of captions after deletion: %d\" %len(caption_data)\n    return caption_data\n\n\ndef _build_vocab(annotations, threshold=1):\n    counter = Counter()\n    max_len = 0\n    for i, caption in enumerate(annotations['caption']):\n        words = caption.split(' ') # caption contrains only lower-case words\n        for w in words:\n            counter[w] +=1\n        \n        if len(caption.split(\" \")) > max_len:\n            max_len = len(caption.split(\" \"))\n\n    vocab = [word for word in counter if counter[word] >= threshold]\n    print ('Filtered %d words to %d words with word count threshold %d.' % (len(counter), len(vocab), threshold))\n\n    word_to_idx = {u'<NULL>': 0, u'<START>': 1, u'<END>': 2}\n    idx = 3\n    for word in vocab:\n        word_to_idx[word] = idx\n        idx += 1\n    print \"Max length of caption: \", max_len\n    return word_to_idx\n\n\ndef _build_caption_vector(annotations, word_to_idx, max_length=15):\n    n_examples = len(annotations)\n    captions = np.ndarray((n_examples,max_length+2)).astype(np.int32)   \n\n    for i, caption in enumerate(annotations['caption']):\n        words = caption.split(\" \") # caption contrains only lower-case words\n        cap_vec = []\n        cap_vec.append(word_to_idx['<START>'])\n        for word in words:\n            if word in word_to_idx:\n                cap_vec.append(word_to_idx[word])\n        cap_vec.append(word_to_idx['<END>'])\n        \n        # pad short caption with the special null token '<NULL>' to make it fixed-size vector\n        if len(cap_vec) < (max_length + 2):\n            for j in range(max_length + 2 - len(cap_vec)):\n                cap_vec.append(word_to_idx['<NULL>']) \n        \n        captions[i, :] = np.asarray(cap_vec)\n    print \"Finished building caption vectors\"\n    return captions\n\n\ndef _build_file_names(annotations):\n    image_file_names = []\n    id_to_idx = {}\n    idx = 0\n    image_ids = annotations['image_id']\n    file_names = annotations['file_name']\n    for image_id, file_name in zip(image_ids, file_names):\n        if not image_id in id_to_idx:\n            id_to_idx[image_id] = idx\n            image_file_names.append(file_name)\n            idx += 1\n\n    file_names = np.asarray(image_file_names)\n    return file_names, id_to_idx\n\n\ndef _build_image_idxs(annotations, id_to_idx):\n    image_idxs = np.ndarray(len(annotations), dtype=np.int32)\n    image_ids = annotations['image_id']\n    for i, image_id in enumerate(image_ids):\n        image_idxs[i] = id_to_idx[image_id]\n    return image_idxs\n\n\ndef main():\n    # batch size for extracting feature vectors from vggnet.\n    batch_size = 100\n    # maximum length of caption(number of word). if caption is longer than max_length, deleted.  \n    max_length = 15\n    # if word occurs less than word_count_threshold in training dataset, the word index is special unknown token.\n    word_count_threshold = 1\n    # vgg model path \n    vgg_model_path = './data/imagenet-vgg-verydeep-19.mat'\n\n    caption_file = 'data/annotations/captions_train2014.json'\n    image_dir = 'image/%2014_resized/'\n\n    # about 80000 images and 400000 captions for train dataset\n    train_dataset = _process_caption_data(caption_file='data/annotations/captions_train2014.json',\n                                          image_dir='image/train2014_resized/',\n                                          max_length=max_length)\n\n    # about 40000 images and 200000 captions\n    val_dataset = _process_caption_data(caption_file='data/annotations/captions_val2014.json',\n                                        image_dir='image/val2014_resized/',\n                                        max_length=max_length)\n\n    # about 4000 images and 20000 captions for val / test dataset\n    val_cutoff = int(0.1 * len(val_dataset))\n    test_cutoff = int(0.2 * len(val_dataset))\n    print 'Finished processing caption data'\n\n    save_pickle(train_dataset, 'data/train/train.annotations.pkl')\n    save_pickle(val_dataset[:val_cutoff], 'data/val/val.annotations.pkl')\n    save_pickle(val_dataset[val_cutoff:test_cutoff].reset_index(drop=True), 'data/test/test.annotations.pkl')\n\n    for split in ['train', 'val', 'test']:\n        annotations = load_pickle('./data/%s/%s.annotations.pkl' % (split, split))\n\n        if split == 'train':\n            word_to_idx = _build_vocab(annotations=annotations, threshold=word_count_threshold)\n            save_pickle(word_to_idx, './data/%s/word_to_idx.pkl' % split)\n        \n        captions = _build_caption_vector(annotations=annotations, word_to_idx=word_to_idx, max_length=max_length)\n        save_pickle(captions, './data/%s/%s.captions.pkl' % (split, split))\n\n        file_names, id_to_idx = _build_file_names(annotations)\n        save_pickle(file_names, './data/%s/%s.file.names.pkl' % (split, split))\n\n        image_idxs = _build_image_idxs(annotations, id_to_idx)\n        save_pickle(image_idxs, './data/%s/%s.image.idxs.pkl' % (split, split))\n\n        # prepare reference captions to compute bleu scores later\n        image_ids = {}\n        feature_to_captions = {}\n        i = -1\n        for caption, image_id in zip(annotations['caption'], annotations['image_id']):\n            if not image_id in image_ids:\n                image_ids[image_id] = 0\n                i += 1\n                feature_to_captions[i] = []\n            feature_to_captions[i].append(caption.lower() + ' .')\n        save_pickle(feature_to_captions, './data/%s/%s.references.pkl' % (split, split))\n        print \"Finished building %s caption dataset\" %split\n\n    # extract conv5_3 feature vectors\n    vggnet = Vgg19(vgg_model_path)\n    vggnet.build()\n    with tf.Session() as sess:\n        tf.initialize_all_variables().run()\n        for split in ['train', 'val', 'test']:\n            anno_path = './data/%s/%s.annotations.pkl' % (split, split)\n            save_path = './data/%s/%s.features.hkl' % (split, split)\n            annotations = load_pickle(anno_path)\n            image_path = list(annotations['file_name'].unique())\n            n_examples = len(image_path)\n\n            all_feats = np.ndarray([n_examples, 196, 512], dtype=np.float32)\n\n            for start, end in zip(range(0, n_examples, batch_size),\n                                  range(batch_size, n_examples + batch_size, batch_size)):\n                image_batch_file = image_path[start:end]\n                image_batch = np.array(map(lambda x: imread(x, mode='RGB'), image_batch_file)).astype(\n                    np.float32)\n                feats = sess.run(vggnet.features, feed_dict={vggnet.images: image_batch})\n                all_feats[start:end, :] = feats\n                print (\"Processed %d %s features..\" % (end, split))\n\n            # use hickle to save huge feature vectors\n            hickle.dump(all_feats, save_path)\n            print (\"Saved %s..\" % (save_path))\n\n\nif __name__ == \"__main__\":\n    main()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}