{"cells":[{"metadata":{},"cell_type":"markdown","source":"### Bengali language\n\nBengali (/bɛŋˈɡɔːli/),also known by its endonym Bangla (বাংলা [ˈbaŋla]), is an Indo-Aryan language primarily spoken by the Bengalis in South Asia. It is the official and most widely spoken language of Bangladesh and second most widely spoken of the 22 scheduled languages of India, behind Hindi. With approximately 228 million native speakers and another 37 million as second language speakers, Bengali is the fifth most-spoken native language and the seventh most spoken language by total number of speakers in the world.\n\n![](https://github.com/seriousran/img_link/blob/master/kg/bengali.png?raw=true)\n\nref: https://en.wikipedia.org/wiki/Bengali_language"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\nfrom matplotlib import pyplot as plt\nimport cv2\n\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# File Analysis\n\n## train\n1. dataframe\n    1. train.csv\n    2. test.csv\n    3. class_map.csv\n    4. sample_submission\n2. image data\n    1. train_image_data_*.parquet\n    2. test_image_data_*.parquet"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/bengaliai-cv19/train.csv')\ndf_test = pd.read_csv('/kaggle/input/bengaliai-cv19/test.csv')\ndf_class = pd.read_csv('/kaggle/input/bengaliai-cv19/class_map.csv')\ndf_submission = pd.read_csv('/kaggle/input/bengaliai-cv19/sample_submission.csv')\n\ndf_train_img_0 = pd.read_parquet('/kaggle/input/bengaliai-cv19/train_image_data_0.parquet')\ndf_train_img_1 = pd.read_parquet('/kaggle/input/bengaliai-cv19/train_image_data_1.parquet')\ndf_train_img_2 = pd.read_parquet('/kaggle/input/bengaliai-cv19/train_image_data_2.parquet')\ndf_train_img_3 = pd.read_parquet('/kaggle/input/bengaliai-cv19/train_image_data_3.parquet')\ndf_test_img_0 = pd.read_parquet('/kaggle/input/bengaliai-cv19/test_image_data_0.parquet')\ndf_test_img_1 = pd.read_parquet('/kaggle/input/bengaliai-cv19/test_image_data_1.parquet')\ndf_test_img_2 = pd.read_parquet('/kaggle/input/bengaliai-cv19/test_image_data_2.parquet')\ndf_test_img_3 = pd.read_parquet('/kaggle/input/bengaliai-cv19/test_image_data_3.parquet')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('shape of df_train:', df_train.shape)\nprint('shape of df_test:', df_test.shape)\nprint('shape of df_shape:', df_class.shape)\nprint('shape of df_submission:', df_submission.shape)\n\nprint('shape of df_train_img_0:', df_train_img_0.shape)\nprint('shape of df_train_img_1:', df_train_img_1.shape)\nprint('shape of df_train_img_2:', df_train_img_2.shape)\nprint('shape of df_train_img_3:', df_train_img_3.shape)\nprint('shape of df_test_img_0:', df_test_img_0.shape)\nprint('shape of df_test_img_1:', df_test_img_1.shape)\nprint('shape of df_test_img_2:', df_test_img_2.shape)\nprint('shape of df_test_img_3:', df_test_img_3.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train_imgs = [df_train_img_0, df_train_img_1, df_train_img_2, df_train_img_3]\ndf_train_img = pd.concat(df_train_imgs)\n\ndf_test_imgs = [df_test_img_0, df_test_img_1, df_test_img_2, df_test_img_3]\ndf_test_img = pd.concat(df_test_imgs)\n\nprint('shape of df_train_img:', df_train_img.shape)\nprint('shape of df_test_img:', df_test_img.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# memory release\n\ndel df_train_img_0\ndel df_train_img_1\ndel df_train_img_2\ndel df_train_img_3\ndel df_test_img_0\ndel df_test_img_1\ndel df_test_img_2\ndel df_test_img_3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_class.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train_img.iloc[0].values[1:].astype(np.uint8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in range(10):\n    plt.imshow(df_train_img.iloc[i].values[1:].astype(np.uint8).reshape(137,236), cmap='gray')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in range(3):\n    plt.imshow(df_test_img.iloc[i].values[1:].astype(np.uint8).reshape(137,236), cmap='gray')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_submission['target'] = 4\ndf_submission.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}