{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"markdown","source":"**Hello everyone. This is a simple data visualization notebook for this competition. If you somehow find it useful, please upvote. **\n\nI will keep updating this notebook gradually.. ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"#Setting up directories\nDATA_DIR = '/kaggle/input/imet-2020-fgvc7/'\nTRAIN_DIR = DATA_DIR + 'train/'\nTEST_DIR = DATA_DIR + 'test/'\nLABELS = DATA_DIR + '/labels.csv'\nSAMPLE_SUB = DATA_DIR + '/sample_submission.csv'\nTRAIN_LABELS = DATA_DIR + '/train.csv'\n\n#Loading CSV files. \nlabels = pd.read_csv(LABELS) \ntrain_labels = pd.read_csv(TRAIN_LABELS)\nsub = pd.read_csv(SAMPLE_SUB)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_labels.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_train, _ = train_labels.shape\nn_labels, _ = labels.shape\n\n# Adding another column as the number of attributes per image\nattrib_freq = pd.DataFrame(train_labels['id'])\nattrib_freq['no_of_attribute'] = np.nan\nfor i in tqdm (range(n_train)):\n    attrib_freq.iloc[i, 1] = int(train_labels['attribute_ids'][i].count(' ') + 1)\n\n# Plotting no_of_attribue for the images\nplt.figure()\nplt.hist(attrib_freq['no_of_attribute'], color = 'blue', edgecolor = 'black',\n         bins = int(attrib_freq['no_of_attribute'].max()))\nplt.xlabel('no_of_attribute')\nplt.ylabel('images_count')\nplt.title('Count of no_of_attribute')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true},"cell_type":"code","source":"labels_custom = labels\nlabels_custom[['attribute_type','attribute_info']] = labels_custom.attribute_name.str.split('::',expand=True,)\nlabels_custom.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_custom['attribute_type'].unique()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.array(train_labels['attribute_ids'][0].split(' '), dtype = np.int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels.iloc[np.array(train_labels['attribute_ids'][1].split(' '), dtype = np.int), 1].tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"index = 2\nprint(labels.iloc[np.array(train_labels['attribute_ids'][index].split(' '), dtype = np.int), 1].tolist())\nimg = Image.open(TRAIN_DIR + train_labels['id'][index] + '.png')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}