{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nimport gc\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport sys\nfrom PIL import Image\nfrom joblib import Parallel, delayed","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"! mkdir ../working/images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class Config:\n    \"\"\"\n        Here we save the configuration for the experiments\n    \"\"\"\n    # dirs\n    base_dir = os.path.abspath('../')\n    data_dir = os.path.join(base_dir, 'input/bengaliai-cv19')\n    working_dir = os.path.join(base_dir, 'working')\n    images_dir = os.path.join(working_dir, 'images')\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_parquet_lists():\n    \"\"\"\n    Load all .parquet files and get train and test splits\n    \"\"\"\n    \n    parquet_files = [f for f in os.listdir(Config.data_dir) if f.endswith(\".parquet\")]\n    train_files = [f for f in parquet_files if 'train' in f]\n    test_files = [f for f in parquet_files if 'test' in f]\n\n    return train_files, test_files\n\n\ndef convert_files(file:str):\n    try:\n        print(f'[INFO] Loading {file}')\n\n        df = pd.read_parquet(os.path.join(Config.data_dir, file), engine='pyarrow')\n\n\n        images_ids = df.image_id.values\n        images = df.drop(\"image_id\", axis=1)\n\n        print(f'[INFO] Working on {file}')\n\n        for index, img_id in enumerate(tqdm(images_ids, desc=f\"Converting files in {file} to .npy format\")):\n            path = os.path.join(Config.working_dir, \"images\", f'{img_id}.npy')\n            # print(path)\n            #break\n            try:        \n                np.save(path, images.iloc[index].values.reshape(137, 236))\n\n            except Exception as ex:\n                print(ex)\n\n        del df \n        del images\n        del images_ids\n    except:\n        pass","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_files, test_files = get_parquet_lists()\n\ntrain_files+test_files","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for file in train_files+test_files:\n    convert_files(file=file)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"! zip -q -r images.zip ../working/images/","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}