{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport gc\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nfrom PIL import Image\n%matplotlib inline\n\ndata_dir = \"../input/imet-2021-fgvc8/\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(data_dir + \"train-from-kaggle.csv\")\ntrain[\"attribute_ids_lst\"] = train.attribute_ids.apply(lambda x: x.split(\" \"))\n\nlabel_map = pd.read_csv(data_dir + \"label_map.csv\")\nlabel_map[\"category\"] = label_map.attribute_name.apply(lambda x: x.split(\"::\")[0])\nlabel_map[\"category_value\"] = label_map.attribute_name.apply(lambda x: x.split(\"::\")[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check the label_map","metadata":{}},{"cell_type":"code","source":"sns.countplot(label_map.category)\nlabel_map[\"category\"].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check how many attribute_ids per id","metadata":{}},{"cell_type":"code","source":"ax = sns.countplot(train.attribute_ids_lst.apply(lambda x: len(x)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check the distribution of category over ids","metadata":{}},{"cell_type":"code","source":"train_attr = (train.explode(\"attribute_ids_lst\")\n .rename({\"attribute_ids_lst\": \"attribute_id\"}, axis=1)\n .astype({\"attribute_id\": \"int64\"})\n .join(label_map.set_index(\"attribute_id\"), on=\"attribute_id\"))\nprint(f\"total ids: {train.shape[0]}\")\ncategory = train_attr.groupby(\"category\").agg({\"id\": [\"count\", \"nunique\"]}).reset_index()\ncategory.columns = [category.columns[0][0], *list(map('_'.join, category.columns.values[1:]))]\ncategory[\"id_count_pct\"] = 1.0 * category[\"id_count\"] / train.shape[0]\ncategory[\"id_nunique_pct\"] = 1.0 * category[\"id_nunique\"] / train.shape[0]\ncategory","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, only dimension is unique per id","metadata":{}},{"cell_type":"markdown","source":"Continuous...","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}