{"cells":[{"metadata":{},"cell_type":"markdown","source":"#### Right after joining the competion, I've checked that loading the dataset is too (x100) slow, so I decided to convert these into memory saving and faster form. This might potentially lose some feature information. if you don't mind anyways, Feel free to use it if you think it is useful. ;-)"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.\nimport cv2\nimport gc\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm_notebook as tqdm\nimport time","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Resize Train set (64x64) and Convert feather format"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"start_time = time.time()\ndata0 = pd.read_parquet('/kaggle/input/bengaliai-cv19/train_image_data_0.parquet')\ndata1 = pd.read_parquet('/kaggle/input/bengaliai-cv19/train_image_data_1.parquet')\ndata2 = pd.read_parquet('/kaggle/input/bengaliai-cv19/train_image_data_2.parquet')\ndata3 = pd.read_parquet('/kaggle/input/bengaliai-cv19/train_image_data_3.parquet')\nprint(\"--- %s seconds ---\" % (time.time() - start_time))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Resize(df,size=64):\n    resized = {} \n    df = df.set_index('image_id')\n    for i in tqdm(range(df.shape[0])):\n        image = cv2.resize(df.loc[df.index[i]].values.reshape(137,236),(size,size))\n        resized[df.index[i]] = image.reshape(-1)\n    resized = pd.DataFrame(resized).T.reset_index()\n    resized.columns = resized.columns.astype(str)\n    resized.rename(columns={'index':'image_id'},inplace=True)\n    return resized","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data0 = Resize(data0)\ndata0.to_feather('train_data_0.feather')\ndel data0\ndata1 = Resize(data1)\ndata1.to_feather('train_data_1.feather')\ndel data1\ndata2 = Resize(data2)\ndata2.to_feather('train_data_2.feather')\ndel data2\ndata3 = Resize(data3)\ndata3.to_feather('train_data_3.feather')\ndel data3","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Resize test set (64x64) and Convert feather format"},{"metadata":{"trusted":true},"cell_type":"code","source":"start_time = time.time()\ndata0 = pd.read_parquet('/kaggle/input/bengaliai-cv19/test_image_data_0.parquet')\ndata1 = pd.read_parquet('/kaggle/input/bengaliai-cv19/test_image_data_1.parquet')\ndata2 = pd.read_parquet('/kaggle/input/bengaliai-cv19/test_image_data_2.parquet')\ndata3 = pd.read_parquet('/kaggle/input/bengaliai-cv19/test_image_data_3.parquet')\nprint(\"--- %s seconds ---\" % (time.time() - start_time))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data0 = Resize(data0)\ndata0.to_feather('test_data_0.feather')\ndel data0\ndata1 = Resize(data1)\ndata1.to_feather('test_data_1.feather')\ndel data1\ndata2 = Resize(data2)\ndata2.to_feather('test_data_2.feather')\ndel data2\ndata3 = Resize(data3)\ndata3.to_feather('test_data_3.feather')\ndel data3","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Reload trainset and Check the images"},{"metadata":{"trusted":true},"cell_type":"code","source":"start_time = time.time()\ndata0 = pd.read_feather('train_data_0.feather')\ndata1 = pd.read_feather('train_data_1.feather')\ndata2 = pd.read_feather('train_data_2.feather')\ndata3 = pd.read_feather('train_data_3.feather')\nprint(\"--- %s seconds ---\" % (time.time() - start_time))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def Grapheme_plot(df):\n    df_sample = df.sample(15)\n    im_id, img = df_sample.iloc[:,0].values,df_sample.iloc[:,1:].values.astype(np.float)\n    \n    fig,ax = plt.subplots(3,5,figsize=(20,20))\n    for i in range(15):\n        j=i%3\n        k=i//3\n        ax[j,k].imshow(img[i].reshape(64,64), cmap='gray')\n        ax[j,k].set_title(im_id[i],fontsize=20)\n    plt.tight_layout()\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Grapheme_plot(data0)\nGrapheme_plot(data1)\nGrapheme_plot(data2)\nGrapheme_plot(data3)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}