{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data exploration\nChecking what data do we have available and what is the labels distribution...","metadata":{}},{"cell_type":"code","source":"# jsu to see what is the data location\n! ls /kaggle/input -l\n! ls /kaggle/input/imet-2021-fgvc8 -l","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:05:29.458969Z","iopub.execute_input":"2021-06-10T15:05:29.459352Z","iopub.status.idle":"2021-06-10T15:05:30.923784Z","shell.execute_reply.started":"2021-06-10T15:05:29.459266Z","shell.execute_reply":"2021-06-10T15:05:30.922727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:05:30.925359Z","iopub.execute_input":"2021-06-10T15:05:30.925831Z","iopub.status.idle":"2021-06-10T15:05:30.933193Z","shell.execute_reply.started":"2021-06-10T15:05:30.925781Z","shell.execute_reply":"2021-06-10T15:05:30.932433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# View the labels details\n\nCheckout the label mapping , there seems to be some topics\nnote that country has operand inside AND and OR which would be better encoded with just primitives","metadata":{}},{"cell_type":"code","source":"df_label_map = pd.read_csv(\"/kaggle/input/imet-2021-fgvc8/label_map.csv\", index_col=\"attribute_id\")\nprint(f\"len: {len(df_label_map)}\")\nprint(df_label_map.head())","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:13:20.088984Z","iopub.execute_input":"2021-06-10T15:13:20.089378Z","iopub.status.idle":"2021-06-10T15:13:20.105952Z","shell.execute_reply.started":"2021-06-10T15:13:20.089344Z","shell.execute_reply":"2021-06-10T15:13:20.104940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_label_map[\"is_OR\"] = ['or' in name for name in df_label_map[\"attribute_name\"]]\ndf_label_map[\"is_AND\"] = ['and' in name for name in df_label_map[\"attribute_name\"]]\ndf_label_map[\"topic\"] = [name.split(\"::\")[0] if \"::\" in name else \".\" for name in df_label_map[\"attribute_name\"]]\ntopics = set(df_label_map['topic'])\n\nprint(f\"include OR: {sum(df_label_map['is_OR'])}\")\nprint(f\"include AND: {sum(df_label_map['is_AND'])}\")\n\ndf_label_map['topic'].hist()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:13:22.516240Z","iopub.execute_input":"2021-06-10T15:13:22.516609Z","iopub.status.idle":"2021-06-10T15:13:22.709535Z","shell.execute_reply.started":"2021-06-10T15:13:22.516578Z","shell.execute_reply":"2021-06-10T15:13:22.708451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Annotations - labed data","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:11:22.279912Z","iopub.execute_input":"2021-06-10T15:11:22.280525Z","iopub.status.idle":"2021-06-10T15:11:22.283824Z","shell.execute_reply.started":"2021-06-10T15:11:22.280473Z","shell.execute_reply":"2021-06-10T15:11:22.283067Z"}}},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/imet-2021-fgvc8/train-from-kaggle.csv\")\nprint(f\"len: {len(df_train)}\")\nprint(df_train.head())","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:14:25.442353Z","iopub.execute_input":"2021-06-10T15:14:25.442749Z","iopub.status.idle":"2021-06-10T15:14:25.640833Z","shell.execute_reply.started":"2021-06-10T15:14:25.442714Z","shell.execute_reply":"2021-06-10T15:14:25.639737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import itertools\n\nlabel_map = dict(zip(df_label_map.index, df_label_map['attribute_name']))\nCOUNT_THR = 1000\n\nlabels_all = list(itertools.chain(*[[int(lb) for lb in lbs.split(\" \")] for lbs in df_train['attribute_ids']]))\nlb_hist = dict(zip(range(max(labels_all) + 1), np.bincount(labels_all)))\ndf_hist = pd.DataFrame([dict(lb=label_map[lb], count=count) for lb, count in lb_hist.items() if count > COUNT_THR]).set_index(\"lb\")\n\nax = df_hist.plot(kind=\"bar\", grid=True, title=\"ids counts\", figsize=(18, 4))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:16:04.293795Z","iopub.execute_input":"2021-06-10T15:16:04.294171Z","iopub.status.idle":"2021-06-10T15:16:07.088381Z","shell.execute_reply.started":"2021-06-10T15:16:04.294140Z","shell.execute_reply":"2021-06-10T15:16:07.087265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_counts = np.bincount(labels_all)\n[[ax]] = pd.DataFrame(labels_counts).hist(bins=500)\nax.set_yscale('log')\nax.set_xscale('log')\nax.set_ylabel('nb labes with this nymber of samples')\nax.set_xlabel('nb of sampls per label')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:43:13.020139Z","iopub.execute_input":"2021-06-10T15:43:13.020510Z","iopub.status.idle":"2021-06-10T15:43:15.151733Z","shell.execute_reply.started":"2021-06-10T15:43:13.020478Z","shell.execute_reply":"2021-06-10T15:43:15.150819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Frequency - labels per sample\n\nAlso labeling per image seems to be one to many with multiple same topis per image...","metadata":{}},{"cell_type":"code","source":"df_train['nb_classes'] = [len(lbs.split(\" \")) for lbs in df_train['attribute_ids']]\nlb_hist = dict(zip(range(25), np.bincount(df_train['nb_classes'])))\ndf_hist = pd.DataFrame([dict(lb=lb, count=count) for lb, count in lb_hist.items()]).set_index(\"lb\")\n\nprint(f\"max lbs: {max(df_train['nb_classes'])}\")\ndf_hist.plot(kind=\"bar\", grid=True, title=\"ids histogram\", figsize=(10, 4))\nprint(lb_hist)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:09:18.459034Z","iopub.execute_input":"2021-06-10T15:09:18.459441Z","iopub.status.idle":"2021-06-10T15:09:19.340469Z","shell.execute_reply.started":"2021-06-10T15:09:18.459392Z","shell.execute_reply":"2021-06-10T15:09:19.339484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Frequency - topic per sample\n\nExploring the cumulation of same topics per image, for some it is atural for soem it can be confusing...","metadata":{}},{"cell_type":"code","source":"import tqdm\nfrom itertools import groupby\n\nls_tops = []\nfor idx, row in tqdm.tqdm(df_train.iterrows(), total=len(df_train)):\n    ids = row[\"attribute_ids\"].split()\n    tops = [df_label_map.loc[int(i), 'topic'] for i in ids]\n    t_hist = {k: len(list(g)) for k, g in groupby(tops)}\n    # print((ids, tops)\n    ls_tops.append(dict(id=row['id'], **t_hist))\n\ndf_train = pd.merge(df_train, pd.DataFrame(ls_tops), on=\"id\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:10:14.285731Z","iopub.execute_input":"2021-06-10T15:10:14.286132Z","iopub.status.idle":"2021-06-10T15:10:38.953590Z","shell.execute_reply.started":"2021-06-10T15:10:14.286094Z","shell.execute_reply":"2021-06-10T15:10:38.952738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[list(topics)].plot(kind=\"hist\", grid=True, alpha=0.5, bins=50)\ndf_train[list(topics)].max()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T15:10:38.955325Z","iopub.execute_input":"2021-06-10T15:10:38.955707Z","iopub.status.idle":"2021-06-10T15:10:39.894882Z","shell.execute_reply.started":"2021-06-10T15:10:38.955673Z","shell.execute_reply":"2021-06-10T15:10:39.893703Z"},"trusted":true},"execution_count":null,"outputs":[]}]}