{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport PIL.Image","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":false},"cell_type":"markdown","source":"Parquet files can be read directly into pandas"},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/bengaliai-ocr-2019/train_image_data_1.parquet')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"There's one row per image, plus an image_id column"},{"metadata":{"trusted":true},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"flattened_image = df.iloc[123].drop('image_id').values.astype(np.uint8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unpacked_image = PIL.Image.fromarray(flattened_image.reshape(137, 236))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"unpacked_image","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":""}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}