{"cells":[{"metadata":{},"cell_type":"markdown","source":"![](https://storage.googleapis.com/kaggle-competitions/kaggle/13251/logos/header.png?t=2019-03-14-16-01-52)\n\n<br>\n\nJust simple visualization and quick insights ...\n\nBased on last year: https://www.kaggle.com/ttahara/eda-compare-number-of-culture-and-tag-attributes\n\n**I will hash the images and detect the repeated images between 2019 and 2020 data so we can identify the new images :)**\n\n```\nIn this dataset, you are presented with a large number of artwork images and associated attributes of the art. \nThe dataset has been expanded from the 2019 edition of this competition. \nMultiple modalities can be expected and the camera sources are unknown. \nThe photographs are often centered for objects, and in the case where the museum artifact is an entire room, the images are scenic in nature.\n```"},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport glob\nimport json\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nfrom collections import Counter\nimport gc\nfrom PIL import Image\n\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls /kaggle/input/imet-2020-fgvc7/","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/imet-2020-fgvc7/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/imet-2020-fgvc7/sample_submission.csv\")\nlabels_df = pd.read_csv(\"/kaggle/input/imet-2020-fgvc7/labels.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"print(\"Train\", train_df.shape)\ntrain_df.sample(10).head()","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"print(\"Test\", test_df.shape)\ntest_df.sample(10).head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Labels"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"labels_df.sample(10).head(10)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"labels_df[\"attribute_type\"] = labels_df.attribute_name.apply(lambda x: x.split(\"::\")[0])\nprint(labels_df[\"attribute_type\"].value_counts())\nsns.countplot(labels_df.attribute_type)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_df.attribute_id.nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"1920 + 768 + 681 + 100 + 5 # number of attributes = N_CLASSES","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_df.attribute_type.unique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels_df[labels_df.attribute_type == \"tags\"]","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/ttahara/eda-compare-number-of-culture-and-tag-attributes\ntrain_attr_ohot = np.zeros((len(train_df), len(labels_df)), dtype=int)\n\nfor idx, attr_arr in enumerate(train_df.attribute_ids.str.split(\" \").apply(lambda l: list(map(int, l))).values):\n    train_attr_ohot[idx, attr_arr] = 1\n    \nnames_arr = labels_df.attribute_name.values\ntrain_df[\"attribute_names\"] = [\", \".join(names_arr[arr == 1]) for arr in train_attr_ohot]\n\ntrain_df[\"attr_num\"] = train_attr_ohot.sum(axis=1)\ntrain_df[\"culture_attr_num\"] = train_attr_ohot[:, :398].sum(axis=1)\ntrain_df[\"tag_attr_num\"] = train_attr_ohot[:, 398:].sum(axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/ttahara/eda-compare-number-of-culture-and-tag-attributes\nfig = plt.figure(figsize=(5 * 5, 5 * 6))\nfig.subplots_adjust(wspace=0.5, hspace=0.5)\nfor i, (art_id, attr_names) in enumerate(train_df.sort_values(by=\"culture_attr_num\", ascending=False)[[\"id\", \"attribute_names\"]].values[:15]):\n    ax = fig.add_subplot(5, 3, i // 3 * 3 + i % 3 + 1)\n    im = Image.open(\"/kaggle/input/imet-2020-fgvc7/train/{}.png\".format(art_id))\n    ax.imshow(im)\n    im.close()\n    attr_split = attr_names.split(\", \")\n    attr_culture = list(map(lambda x: x.split(\"::\")[-1], filter(lambda x: x[:7] == \"culture\", attr_split)))\n    attr_tag = list(map(lambda x: x.split(\"::\")[-1], filter(lambda x: x[:3] == \"tag\", attr_split)))\n    ax.set_title(\"art id: {}\\nculture: {}\\ntag: {}\".format(art_id, attr_culture, attr_tag))","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/ttahara/eda-compare-number-of-culture-and-tag-attributes\nfig = plt.figure(figsize=(5 * 6, 5 * 5))\nfig.subplots_adjust(wspace=0.6, hspace=0.6)\nfor i, (art_id, attr_names) in enumerate(train_df.sort_values(by=\"tag_attr_num\", ascending=False)[[\"id\", \"attribute_names\"]].values[:12]):\n    ax = fig.add_subplot(4, 3, i // 3 * 3 + i % 3 + 1)\n    im = Image.open(\"/kaggle/input/imet-2020-fgvc7/train/{}.png\".format(art_id))\n    ax.imshow(im)\n    im.close()\n    attr_split = attr_names.split(\", \")\n    attr_culture = list(map(lambda x: x.split(\"::\")[-1], filter(lambda x: x[:6] == \"culture\", attr_split)))\n    attr_tag = list(map(lambda x: x.split(\"::\")[-1], filter(lambda x: x[:3] == \"tag\", attr_split)))\n    ax.set_title(\"art id: {}\\nculture: {}\\ntag: {}\".format(art_id, attr_culture, attr_tag))","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"# https://www.kaggle.com/ttahara/eda-compare-number-of-culture-and-tag-attributes\nfig = plt.figure(figsize=(5 * 8, 5 * 7))\nfig.subplots_adjust(wspace=0.6, hspace=0.6)\nfor i, (art_id, attr_names) in enumerate(train_df[train_df.tag_attr_num == 1][[\"id\", \"attribute_names\"]].values[:49]):\n    ax = fig.add_subplot(7, 7, i // 7 * 7 + i % 7 + 1)\n    im = Image.open(\"/kaggle/input/imet-2020-fgvc7/train/{}.png\".format(art_id))\n    ax.imshow(im)\n    im.close()\n    attr_split = attr_names.split(\", \")\n    attr_culture = list(map(lambda x: x.split(\"::\")[-1], filter(lambda x: x[:7] == \"culture\", attr_split)))\n    attr_tag = list(map(lambda x: x.split(\"::\")[-1], filter(lambda x: x[:3] == \"tag\", attr_split)))\n    ax.set_title(\"art id: {}\\nculture: {}\\ntag: {}\".format(art_id, attr_culture, attr_tag))","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(5 * 8, 5 * 7))\nfig.subplots_adjust(wspace=0.6, hspace=0.6)\nfor i, (art_id, attr_names) in enumerate(train_df[train_df.tag_attr_num == 2][[\"id\", \"attribute_names\"]].values[:49]):\n    ax = fig.add_subplot(7, 7, i // 7 * 7 + i % 7 + 1)\n    im = Image.open(\"/kaggle/input/imet-2020-fgvc7/train/{}.png\".format(art_id))\n    ax.imshow(im)\n    im.close()\n    attr_split = attr_names.split(\", \")\n    attr_culture = list(map(lambda x: x.split(\"::\")[-1], filter(lambda x: x[:7] == \"culture\", attr_split)))\n    attr_tag = list(map(lambda x: x.split(\"::\")[-1], filter(lambda x: x[:3] == \"tag\", attr_split)))\n    ax.set_title(\"art id: {}\\nculture: {}\\ntag: {}\".format(art_id, attr_culture, attr_tag))","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-input":true,"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(5 * 8, 5 * 7))\nfig.subplots_adjust(wspace=0.6, hspace=0.6)\nfor i, (art_id, attr_names) in enumerate(train_df[train_df.tag_attr_num == 3][[\"id\", \"attribute_names\"]].values[:49]):\n    ax = fig.add_subplot(7, 7, i // 7 * 7 + i % 7 + 1)\n    im = Image.open(\"/kaggle/input/imet-2020-fgvc7/train/{}.png\".format(art_id))\n    ax.imshow(im)\n    im.close()\n    attr_split = attr_names.split(\", \")\n    attr_culture = list(map(lambda x: x.split(\"::\")[-1], filter(lambda x: x[:7] == \"culture\", attr_split)))\n    attr_tag = list(map(lambda x: x.split(\"::\")[-1], filter(lambda x: x[:3] == \"tag\", attr_split)))\n    ax.set_title(\"art id: {}\\nculture: {}\\ntag: {}\".format(art_id, attr_culture, attr_tag))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2020 vs 2019 Dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"train20 = pd.read_csv(\"/kaggle/input/imet-2020-fgvc7/train.csv\")\ntest20 = pd.read_csv(\"/kaggle/input/imet-2020-fgvc7/sample_submission.csv\")\nlabels20 = pd.read_csv(\"/kaggle/input/imet-2020-fgvc7/labels.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train19 = pd.read_csv(\"/kaggle/input/imet-2019-fgvc6/train.csv\")\ntest19 = pd.read_csv(\"/kaggle/input/imet-2019-fgvc6/sample_submission.csv\")\nlabels19 = pd.read_csv(\"/kaggle/input/imet-2019-fgvc6/labels.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train20.shape, train19.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.intersect1d(train20.id.values , train19.id.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train19.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels19.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels19[\"attribute_type\"] = labels19.attribute_name.apply(lambda x: x.split(\"::\")[0])\nprint(labels19[\"attribute_type\"].value_counts())\nsns.countplot(labels19.attribute_type)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print (labels20.attribute_name.nunique())\nprint (labels19.attribute_name.nunique())\nprint (np.intersect1d(labels20.attribute_name.values , labels19.attribute_name.values).shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.4"}},"nbformat":4,"nbformat_minor":4}