{"cells":[{"metadata":{"_uuid":"a264b4edc2f09f4b0460560fa80e335c1820dab1"},"cell_type":"markdown","source":"# Notebok Info\nHi all! This notebook creates a lighter dataset for the Human Protein Atlas Image Classification. In particular the code below reads all the images and saves them into RGBY format. This is done for a faster reading of the data in the training and testing phases.\n\nIn this example I resize the images from $[512,512,4]$ to $[128,128,4]$ and the output dataset sizes are: 1.055 GB  for the train dataset and 0.356 GB for test dataset.\nIf you don't want to resize the images you have to run this kernel in you PC locally because in kaggle kernels you do not have much disk memory."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# Import Packages\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport os\nfrom tqdm import tqdm_notebook\nfrom IPython.display import clear_output\nfrom contextlib import closing\nfrom zipfile import ZipFile, ZIP_DEFLATED\n\n# Variables\nDATA_DIR = '../input/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a928a40fa8141d28333c0d4bc19bc872c8b5cdf2"},"cell_type":"code","source":"# Read Single Image\ndef read_img(img_id, mode='train', img_size=512):\n    img_dir = 'train/'\n    if mode=='test':\n        img_dir = 'test/'\n    \n    channels = ['red','green','blue','yellow']\n    img = []\n    for ch in channels:\n        img_ch = cv2.imread(DATA_DIR+img_dir+img_id+'_{}.png'.format(ch), cv2.IMREAD_GRAYSCALE)\n        \n        # Resize\n        if img_size!=512:\n            img_ch = cv2.resize(img_ch, (img_size, img_size))\n        img.append(img_ch)\n    return np.stack(img, axis=-1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f4b5e3852625cfe146116e421c7ed9a564ff61f"},"cell_type":"code","source":"# Compress A Folder\ndef zipdir(basedir, archivename):\n    assert os.path.isdir(basedir)\n    with closing(ZipFile(archivename, \"w\", ZIP_DEFLATED)) as z:\n        for root, dirs, files in os.walk(basedir):\n            #NOTE: ignore empty directories\n            for fn in files:\n                absfn = os.path.join(root, fn)\n                zfn = absfn[len(basedir)+len(os.sep):] #XXX: relative path\n                z.write(absfn, zfn)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"46eee6deaa75ed87e197f607b44974d68be98b76"},"cell_type":"markdown","source":"# Load and Save Train Images"},{"metadata":{"trusted":true,"_uuid":"714d3b97cd53c17b9f18671a5984b9088208ba7b"},"cell_type":"code","source":"df_tr = pd.read_csv(DATA_DIR+'train.csv')\ntrain_generator = ([img_id, read_img(img_id, mode='train', img_size=128)] for img_id in df_tr['Id'])\n\n# Save Train Images\nos.makedirs('train/') if not os.path.exists('train/') else None\nfor img_id,img in tqdm_notebook(train_generator, total=df_tr.shape[0]):\n    cv2.imwrite('train/{}.png'.format(img_id), img)\n    \n# Compress Data\nzipdir('/kaggle/working/train', 'train.zip')\n!rm train/*\n!rmdir train","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8a925320e2a30129e9ee48fc105055839ef23e7e"},"cell_type":"markdown","source":"# Load and Save Test Images"},{"metadata":{"trusted":true,"_uuid":"8212a0200a06cf2bdaf64abe816935b74be4c571"},"cell_type":"code","source":"df_te = pd.read_csv(DATA_DIR+'sample_submission.csv')\ntest_generator = ([img_id, read_img(img_id, mode='test', img_size=128)] for img_id in df_te['Id'])\n\n# Save Train Images\nos.makedirs('test/') if not os.path.exists('test/') else None\nfor img_id,img in tqdm_notebook(test_generator, total=df_te.shape[0]):\n    cv2.imwrite('test/{}.png'.format(img_id), img)\n    \n# Compress Data\nzipdir('/kaggle/working/test', 'test.zip')\n!rm test/*\n!rmdir test","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3a829fc9e781cbbcab2b6d84e7cacd6f2a2b2539"},"cell_type":"markdown","source":"# Check Dataset Size"},{"metadata":{"trusted":true,"_uuid":"2d47538912fb82feb47f16f9505625154f64b69b"},"cell_type":"code","source":"!ls -l ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7f18770c3f69dc3ef0532ab6f2f63d7c686d2a60"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}