{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Just a Quick Look on Dataset...\n"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nfrom glob import glob\nimport seaborn as sns\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images_path='/kaggle/input/bms-molecular-translation/train'\ntrain_df=pd.read_csv('/kaggle/input/bms-molecular-translation/train_labels.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Shape of Dataset: ',train_df.shape)\nprint('Number of Unique IDs: ',train_df['image_id'].nunique())\ntrain_df.head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Lets plot first 10 images\nimg_ids=[glob(os.path.join(images_path,'*','*','*',i+'.png')) for i in train_df['image_id'].values[:10]]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"r,c=5,2\nfig=plt.figure(figsize=(20,20))\nfor i in range(1,r*c+1):\n    img=cv2.imread(img_ids[i-1][0])\n    lbl=train_df.loc[i-1,'InChI']\n    fig.add_subplot(r,c,i)\n    plt.imshow(img,cmap='gray')\n    plt.title(lbl[:20])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"**String length Distribution**"},{"metadata":{"trusted":true},"cell_type":"code","source":"string_len=[len(i) for i in train_df['InChI']]\nsns.displot(string_len,kde=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_path(id_):\n    return os.path.join(id_[0],id_[1],id_[2],id_+'.png')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Lets take a look at image shape distribution of 10k images\nimg_h,img_w=[],[]\nfor i in tqdm(train_df['image_id'].values[:10000]):\n    img=cv2.imread(os.path.join(images_path,get_path(i)))\n    h,w=img.shape[0],img.shape[1]\n    img_h.append(h)\n    img_w.append(w)\n\nsns.displot((img_h,img_w))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}