{"cells":[{"metadata":{"trusted":false,"_uuid":"608f7b1b3c94db86451ca1f4f8d8fb15a7c256d7"},"cell_type":"code","source":"from random import shuffle\nimport glob\nimport pandas as pd\nimport cv2\n\n\ndf = pd.read_csv(\"train.csv\")\nshuffle_data = False  # shuffle the addresses before saving\nhdf5_path = 'dataset.hdf5'  # address to where you want to save the hdf5 file\n# read addresses and labels from the 'train' folder\naddrs = df.Image.apply(lambda x: \"train/\" + x).values\nlabels = df.Id.values  # 0 = Cat, 1 = Dog\nimg_names = df.Image.values\n# to shuffle data\nif shuffle_data:\n    c = list(zip(addrs, labels,img_names))\n    shuffle(c)\n    train_addrs, train_labels, train_img_name = zip(*c)\nelse:\n    train_addrs = list(addrs)\n    train_labels = list(labels)\n    train_img_name = list(img_names)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"020518012a119953acd01d34406d08e9726c6d8c"},"cell_type":"code","source":"import numpy as np\nimport tables\n\ndata_order = 'tf'  # 'th' for Theano, 'tf' for Tensorflow\nimg_dtype = tables.UInt8Atom()  # dtype in which the images will be saved\n# check the order of data and chose proper data shape to save images\nif data_order == 'th':\n    data_shape = (0, 3, 224, 224)\nelif data_order == 'tf':\n    data_shape = (0, 128,256, 3)\n# open a hdf5 file and create earrays\nhdf5_file = tables.open_file(hdf5_path, mode='w')\ntrain_storage = hdf5_file.create_earray(hdf5_file.root, 'train_img', img_dtype, shape=data_shape)\nmean_storage = hdf5_file.create_earray(hdf5_file.root, 'train_mean', img_dtype, shape=data_shape)\n# create the label arrays and copy the labels data in them\nhdf5_file.create_array(hdf5_file.root, 'train_labels', train_labels)\nhdf5_file.create_array(hdf5_file.root, 'train_img_name', train_img_name)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c488268de31cc82c11043e9c23e60621546ed5ac"},"cell_type":"code","source":"# a numpy array to save the mean of the images\nmean = np.zeros(data_shape[1:], np.float32)\n# loop over train addresses\nfor i in range(len(train_addrs)):\n    # print how many images are saved every 1000 images\n    if i % 1000 == 0 and i > 1:\n        print('Train data: ' + str(i) + \"/\" + str(len(train_addrs)))\n    # read an image and resize to (224, 224)\n    # cv2 load images as BGR, convert it to RGB\n    addr = train_addrs[i]\n    img = cv2.imread(addr)\n    img = cv2.resize(img, (256, 128))\n    #img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    # add any image pre-processing here\n    # if the data order is Theano, axis orders should change\n    if data_order == 'th':\n        img = np.rollaxis(img, 2)\n    # save the image and calculate the mean so far\n    train_storage.append(img[None])\n    mean += img / float(len(train_labels))\n# save the mean and close the hdf5 file\nmean_storage.append(mean[None])\nhdf5_file.close()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"93caf56417d4aaa60233ccad76bfda888c3a69f3"},"cell_type":"code","source":"hdf5_path = 'dataset.hdf5'\nsubtract_mean = False\n# open the hdf5 file\nhdf5_file = tables.open_file(hdf5_path, mode='r')\n# subtract the training mean\nif subtract_mean:\n    mm = hdf5_file.root.train_mean[0]\n    mm = mm[np.newaxis, ...]\n# Total number of samples\ndata_num = hdf5_file.root.train_img.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"9c475eba4fb9ba2552974ae54990a94f5425b462"},"cell_type":"code","source":"hdf5_file.root.train_img[:10].shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"9ad29b5fc25c47151e25af0289db4c08cf149ea8"},"cell_type":"code","source":"hdf5_file.root.train_img_name[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"74eb86dd7788ffe49d025c78be74141aa97af284"},"cell_type":"code","source":"hdf5_file.root.train_labels[:10]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"53a29bedf686a5c447ad5c4f38619fb69010fdb9"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}