{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"from tqdm import tqdm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os, cv2 \nfrom multiprocessing import Pool\nimport threading","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# This notebook creates a new dataset by resizing each single image"},{"metadata":{},"cell_type":"markdown","source":"Initialize the Kaggle API with my Auth token and the dataset metadata"},{"metadata":{"trusted":true},"cell_type":"code","source":"! mkdir -p /root/.kaggle/\n! cp ../input/api-token/kaggle.json /root/.kaggle/kaggle.json\n! mkdir -p /kaggle/tmp/hpa_512x512_dataset\n! kaggle datasets init -p /kaggle/tmp/hpa_512x512_dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%bash\necho \"{\n  \\\"title\\\": \\\"HPA: 512x512 dataset\\\",\n  \\\"id\\\": \\\"tchaye59/HPA512x512DATASET\\\",\n  \\\"licenses\\\": [\n    {\n      \\\"name\\\": \\\"CC0-1.0\\\"\n    }\n  ]\n}\" > /kaggle/tmp/hpa_512x512_dataset/dataset-metadata.json","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"DATA_DIR = \"../input/hpa-single-cell-image-classification/\"\nTRAIN_DIR = os.path.join(DATA_DIR,\"train/\")\nTEST_DIR = os.path.join(DATA_DIR,\"test/\")\n\nDS_DIR = '/kaggle/tmp/hpa_512x512_dataset/'\ntrain_df = pd.read_csv('../input/hpa-single-cell-image-classification/train.csv')\ntest_df = pd.read_csv('../input/hpa-single-cell-image-classification/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Runs a process in a thread\nclass Worker(threading.Thread):\n    \n    def __init__(self, process,args,pbar=None):\n        super().__init__()\n        self.pbar = pbar\n        self.process = process\n        self.args = args\n\n    def run(self):\n        res = self.process(self.args)\n        if self.pbar:\n            self.pbar.update(1)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Build dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"def rgb_worker_fn(args):\n    id_,src_path,dest_path,rgb_dest_dir = args\n    dim = (512, 512)\n    \n    red_image = cv2.imread(src_path+id_+\"_red.png\", cv2.IMREAD_UNCHANGED)\n    red_image = cv2.resize(red_image, dim, interpolation = cv2.INTER_AREA)\n    cv2.imwrite(dest_path+id_+\"_red.png\" , red_image)\n    \n    green_image = cv2.imread(src_path+id_+\"_green.png\", cv2.IMREAD_UNCHANGED)\n    green_image = cv2.resize(green_image, dim, interpolation = cv2.INTER_AREA)\n    cv2.imwrite(dest_path+id_+\"_green.png\" , green_image)\n    \n    blue_image = cv2.imread(src_path+id_+\"_blue.png\", cv2.IMREAD_UNCHANGED)\n    blue_image = cv2.resize(blue_image, dim, interpolation = cv2.INTER_AREA)\n    cv2.imwrite(dest_path+id_+\"_blue.png\" , blue_image)\n    \n    yellow_image = cv2.imread(src_path+id_+\"_yellow.png\", cv2.IMREAD_UNCHANGED)\n    yellow_image = cv2.resize(yellow_image, dim, interpolation = cv2.INTER_AREA)\n    cv2.imwrite(dest_path+id_+\"_yellow.png\" , yellow_image)\n        \n    image = np.array([red_image, green_image, blue_image,])\n    image = np.transpose(image, (1,2,0))\n    #image=image.astype(np.uint8) \n    cv2.imwrite(rgb_dest_dir+id_+\".png\" , image)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Training images"},{"metadata":{"trusted":true},"cell_type":"code","source":"ids = train_df.ID.values\nsrc_dir  =  TRAIN_DIR\ndest_dir = DS_DIR+'train/'\nrgb_dest_dir =  DS_DIR+'rgb_train/'\nos.makedirs(dest_dir,exist_ok=True)\nos.makedirs(rgb_dest_dir,exist_ok=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"size = len(ids)\npbar = tqdm(total=size)\nn = 1000\nfor i in range(0,size,n):\n    workers = []\n    for id_ in ids[i:min(size,i+n)]:\n        worker =  Worker(rgb_worker_fn,(id_,src_dir,dest_dir,rgb_dest_dir),pbar=pbar)\n        worker.start()\n        workers.append(worker)\n    for worker in workers:\n        worker.join()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Testing images"},{"metadata":{"trusted":true},"cell_type":"code","source":"ids = test_df.ID.values\nsrc_dir  =  TEST_DIR\ndest_dir = DS_DIR+'test/'\nrgb_dest_dir =  DS_DIR+'rgb_test/'\nos.makedirs(dest_dir,exist_ok=True)\nos.makedirs(rgb_dest_dir,exist_ok=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"size = len(ids)\npbar = tqdm(total=size)\nn = 1000\nfor i in range(0,size,n):\n    workers = []\n    for id_ in ids[i:min(size,i+n)]:\n        worker =  Worker(rgb_worker_fn,(id_,src_dir,dest_dir,rgb_dest_dir),pbar=pbar)\n        worker.start()\n        workers.append(worker)\n    for worker in workers:\n        worker.join()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_kg_hide-output":true},"cell_type":"code","source":"! kaggle datasets version -p /kaggle/tmp/hpa_512x512_dataset -m \"add rgb images\"  --dir-mode tar\n#! kaggle datasets create -p /kaggle/tmp/hpa_512x512_dataset/ -u --dir-mode tar","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"! rm -rf /root/.kaggle/kaggle.json","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Dataset link : https://www.kaggle.com/tchaye59/hpa512x512dataset"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}