{"cells":[{"metadata":{"_uuid":"4e2b135cd5cf3ee82f0023f41b1eeb3b3982a1fd"},"cell_type":"markdown","source":"The goal of this notebook is to precompute the batches using multiprocessing CPU and store these batches as savez_compressed numpy arrays as described in the discussion https://www.kaggle.com/c/human-protein-atlas-image-classification/discussion/68118\n\nDue to the limited storage on Kaggle kernel, we process only part of the training and test set (first 64 batches). The results are stored to the output directory, so that other kernels can use the computation results as input."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom tqdm import tqdm\nfrom PIL import Image\nfrom multiprocessing import Pool\nimport multiprocessing as mp\n#from dataProcessing import loadImage\nimport matplotlib.pyplot as plt\nimport time","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe2e7bb79f90b1c0a22460be36d182fcb1b2ebbb"},"cell_type":"code","source":"BATCH_SIZE = 16\nTEST_BATCH_SIZE = 16\nSEED = 777\nSHAPE = (512, 512, 4)\nCORES = mp.cpu_count() #4\nDIR = '../input'\nOUTPUT_DIR = '.'\nDEBUG = True","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9f7e2c15d3bffd788c34da8a70d90d100b2cdf54"},"cell_type":"code","source":"def getTrainDataset():\n    \n    path_to_train = DIR + '/train/'\n    data = pd.read_csv(DIR + '/train.csv')\n\n    paths = []\n    labels = []\n    \n    for name, lbl in zip(data['Id'], data['Target'].str.split(' ')):\n        y = np.zeros(28)\n        for key in lbl:\n            y[int(key)] = 1\n        paths.append(os.path.join(path_to_train, name))\n        labels.append(y)\n\n    return np.array(paths), np.array(labels)\n\n\ndef getTestDataset():\n    \n    path_to_test = DIR + '/test/'\n    data = pd.read_csv(DIR + '/sample_submission.csv')\n\n    paths = []\n    labels = []\n    \n    for name in data['Id']:\n        y = np.ones(28)\n        paths.append(os.path.join(path_to_test, name))\n        labels.append(y)\n\n    return np.array(paths), np.array(labels)\n\n\ndef prepareData(paths, labels, shuffle = True, shape = SHAPE, seed = SEED, batch_size = BATCH_SIZE, debug = False):\n    \n    keys = np.arange(paths.shape[0], dtype=np.int)\n    if(shuffle):\n        np.random.seed(seed)\n        np.random.shuffle(keys)\n\n    if(paths.shape[0] % batch_size != 0):\n        remaining = (paths.shape[0] // batch_size + 1) * batch_size - paths.shape[0]\n        keys = np.append(keys, np.zeros(remaining, dtype=np.int32))\n        \n    keys = keys.reshape(-1,batch_size)\n    \n    if debug == True:\n        keys = keys[0:8]\n    \n    paths = paths[keys]\n    labels = labels[keys]\n    \n    processImages(paths, labels, shape)\n    \n    return paths, labels\n\n\ndef loadImage(path):\n\n        R = Image.open(path + '_red.png')\n        G = Image.open(path + '_green.png')\n        B = Image.open(path + '_blue.png')\n        Y = Image.open(path + '_yellow.png')\n\n        im = np.stack((\n            np.array(R), \n            np.array(G), \n            np.array(B),\n            np.array(Y)), -1)\n        \n        im = np.divide(im, 255)\n        return im\n\n    \ndef processImages(paths, labels, shape):\n    \n    p = Pool(CORES)\n    \n    for batch in tqdm(range(paths.shape[0])):\n\n        batch_size = paths[batch].shape[0]\n        n_labels = labels[batch].shape[1]\n        \n        batch_labels = np.zeros((batch_size, n_labels))\n        \n        batch_images = np.array(p.map(loadImage, paths[batch]))\n        batch_labels = labels[batch]\n        np.savez(os.path.dirname(paths[batch][0]).replace(DIR, OUTPUT_DIR) + '-memory'+str(batch), images=batch_images, labels=batch_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9113091d65c928b4a435bda9896485da10332e9c"},"cell_type":"code","source":"paths, labels = getTrainDataset()\npathsTrain, labelsTrain = prepareData(paths, labels, debug = DEBUG)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c34697a512f1cc3b2349e63f535dda63dcfd8ef7"},"cell_type":"code","source":"paths, labels = getTestDataset()\npathsTest, labelsTest = prepareData(paths, labels, shuffle=False, batch_size = TEST_BATCH_SIZE, debug = DEBUG)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9132f5b10133fc671d67caeb7547b856778f422d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}