{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Methods for extracting a sample were sourced from https://www.kaggle.com/thedrcat/hpa-cell-tiles-sample-balanced by @thedrcat","metadata":{}},{"cell_type":"code","source":"from fastai.vision.all import *\nimport pandas as pd\nimport numpy as np\nimport os\nfrom tqdm import tqdm\nimport cv2\nfrom shutil import copyfile","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path('../input/hpa-single-cell-image-classification')\ndf = pd.read_csv(path/'train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = '../input/hpa-single-cell-image-classification/'\ntrain_or_test = 'train'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = df.sample(n=1500, replace=False, random_state=58).reset_index(drop=True)\ndfs_train = dfs[0:1000]\ndfs_val = dfs[1000:].reset_index(drop=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_img(image_id, color, train_or_test='train', image_size=None):\n    filename = f'{ROOT}/{train_or_test}/{image_id}_{color}.png'\n    assert os.path.exists(filename), f'not found {filename}'\n    img = cv2.imread(filename, cv2.IMREAD_UNCHANGED)\n    if image_size is not None:\n        img = cv2.resize(img, (image_size, image_size))\n    if img.max() > 255:\n        img_max = img.max()\n        img = (img/255).astype('uint8')\n    return img","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_files = len(dfs)\ncell_mask_dir = '../input/hpa-mask/hpa_cell_mask'\n\nif not os.path.exists('masks'):\n    os.makedirs('masks')\n\nchannels = [\"red\", \"green\", \"blue\", \"yellow\"]\nwith zipfile.ZipFile('hpa_sample.zip', 'w') as img_out:\n\n    for idx in tqdm(range(num_files)):\n        image_id = dfs.iloc[idx].ID\n        labels = dfs.iloc[idx].Label\n        #cell_mask = np.load(f'{cell_mask_dir}/{image_id}.npz')\n        #red = read_img(image_id, \"red\", train_or_test, None)\n        fname_mask = f'masks/{image_id}.npz'\n        copyfile(f'{cell_mask_dir}/{image_id}.npz', fname_mask)\n        \n        for chan in channels:\n            curr_img = read_img(image_id, chan, train_or_test, None)\n            fname = f'{image_id}_{chan}.jpg'\n\n            im = cv2.imencode('.jpg', curr_img)[1]\n            img_out.writestr(fname, im)\n                    \n\n\ndfs_train.to_csv('train_sample.csv', index=False)\ndfs_val.to_csv('val_sample.csv', index=False)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sourced from https://stackoverflow.com/questions/1855095/how-to-create-a-zip-archive-of-a-directory-in-python\ndef zipdir(path, ziph):\n    # ziph is zipfile handle\n    for root, dirs, files in os.walk(path):\n        for file in files:\n            ziph.write(os.path.join(root, file), \n                       os.path.relpath(os.path.join(root, file), \n                                       os.path.join(path, '..')))\n            \nzipf = zipfile.ZipFile('hpa_sample_masks.zip', 'w', zipfile.ZIP_DEFLATED)\nzipdir('masks/', zipf)\nzipf.close()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}