{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input/googlenewsvectorsnegative300\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"pd.options.display.max_rows = 64\npd.options.display.max_columns = 512","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label = pd.read_csv('../input/imet-2019-fgvc6/labels.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label_list = list(label.attribute_name.str.split(pat='::').map(lambda x: x[1]))\nlabel_dict = dict(zip(label_list, list(label.attribute_id)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data = pd.read_csv('../input/imet-2019-fgvc6/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from gensim import models","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"w2v = models.KeyedVectors.load_word2vec_format('../input/googlenewsvectorsnegative300/GoogleNews-vectors-negative300.bin.gz', binary=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"map_label = {}\nexceptions = []\nfor item in label_list:\n    try:\n        map_label[item] = w2v[item]\n    except BaseException:\n        try:\n            item_C = item[0].upper() + item[1:]\n            map_label[item] = w2v[item_C]\n        except BaseException:\n            try:\n                item__ = item.replace(' ','_')\n                map_label[item] = w2v[item__]\n            except BaseException:\n                exceptions.append(item)\nprint(len(exceptions),len(map_label))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"exceptions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.zeros(2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"no_embed = np.zeros(len(exceptions))\nno_embeds = []\nfor item in exceptions:\n    no_embeds.append(label_dict[item])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for item in data.attribute_ids:\n    for num in item.split(' '):\n        num = int(num)\n        for i in range(len(exceptions)):\n            if num == no_embeds[i]:\n                no_embed[i] += 1\nno_embed.astype(np.int32)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"count = 0\nfor item in data.attribute_ids:\n    for num in item.split(' '):\n        num = int(num)\n        count += 1\ncount","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_labels = pd.DataFrame(map_label).T\ndf_labels.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.cluster import KMeans\nn_clusters = 12\nkm = KMeans(n_clusters=n_clusters, random_state=42, n_jobs=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"kmeans = km.fit(df_labels)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"centers = kmeans.cluster_centers_\ncenters","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.concatenate((df_labels.values,centers),axis=0).shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.manifold import TSNE\n\nlabel_embedded = TSNE(n_components=2,perplexity=100).fit_transform(np.concatenate((centers,df_labels.values),axis=0))\nlabel_embedded","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.cm as cm\nimport matplotlib.pyplot as plt\n\ncolors = cm.gist_rainbow(np.linspace(0, 1, n_clusters))\nplt.style.use('ggplot')\n\nX = label_embedded[:,0]\nY = label_embedded[:,1]\nc = kmeans.labels_\n\nplt.figure(figsize=[10,10])\nfor i in range(len(X)):\n    if i >= n_clusters:\n        plt.scatter(X[i],Y[i],color=colors[c[i-n_clusters]])\n    else:\n        plt.scatter(X[i],Y[i],color=colors[i],s=500,label=str(i))\nplt.legend()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_labels['C'] = kmeans.labels_\ndf_labels.C.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}