{"cells":[{"metadata":{},"cell_type":"markdown","source":"* This is a derivative of the face clustering notebook created by Henriqe Mendoca! [Link to Notebook](https://www.kaggle.com/hmendonca/proper-clustering-with-facenet-embeddings-eda/)\n\n- Embeddings of the first frame of each video in the training dataset are stored in a pickle file\n- The following nb is used to calculate the clusters of each face and store it in the working directory \n"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom matplotlib import pyplot as plt\n\nfrom sklearn.decomposition import PCA","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"os.listdir('/kaggle/input/sample-face-crop')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"def scatter_thumbnails(data, images, zoom=0.12, colors=None):\n    assert len(data) == len(images)\n\n    # reduce embedding dimentions to 2\n    x = PCA(n_components=2).fit_transform(data) if len(data[0]) > 2 else data\n\n    # create a scatter plot.\n    f = plt.figure(figsize=(22, 15))\n    ax = plt.subplot(aspect='equal')\n    sc = ax.scatter(x[:,0], x[:,1], s=4)\n    _ = ax.axis('off')\n    _ = ax.axis('tight')\n\n    # add thumbnails :)\n#     from matplotlib.offsetbox import OffsetImage, AnnotationBbox\n#     for i in range(len(images)):\n#         image = plt.imread(images[i])\n#         im = OffsetImage(image, zoom=zoom)\n#         bboxprops = dict(edgecolor=colors[i]) if colors is not None else None\n#         ab = AnnotationBbox(im, x[i], xycoords='data',\n#                             frameon=(bboxprops is not None),\n#                             pad=0.02,\n#                             bboxprops=bboxprops)\n#         ax.add_artist(ab)\n    return ax","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pickle\n\nembeddings = pd.read_pickle('/kaggle/input/sample-face-crop/embeddings_face_clusters.pkl')\nprint(embeddings.shape)\nembeddings.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# _ = scatter_thumbnails(embeddings.embedding.tolist(), embeddings.faceFile.tolist())\n# plt.title('Facial Embeddings - Principal Component Analysis')\n# plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\nfrom sklearn.manifold import TSNE","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n# PCA first to speed it up\nx = PCA(n_components=50).fit_transform(embeddings['embedding'].tolist())\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\nx = TSNE(perplexity=50,\n         n_components=3).fit_transform(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# _ = scatter_thumbnails(x, embeddings.faceFile.tolist(), zoom=0.06)\n# plt.title('3D t-Distributed Stochastic Neighbor Embedding')\n# plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install -q hdbscan\nimport hdbscan","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import sklearn.cluster as cluster","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndef plot_clusters(data, algorithm, *args, **kwds):\n    labels = algorithm(*args, **kwds).fit_predict(data)\n    #palette = sns.color_palette('deep', np.max(labels) + 1)\n    #colors = [palette[x] if x >= 0 else (0,0,0) for x in labels]\n    #ax = scatter_thumbnails(x, df.face.tolist(), 0.06, colors)\n    #plt.title(f'Clusters found by {algorithm.__name__}')\n    return labels\n\n# clusters = plot_clusters(x, hdbscan.HDBSCAN, alpha=1.0, min_cluster_size=2, min_samples=1)\nclusters = plot_clusters(x, cluster.DBSCAN, n_jobs=-1, eps=1.0, min_samples=1)\nembeddings['cluster'] = clusters","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(type(x))\nprint(x.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embeddings['TSNE'] = x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embeddings.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embeddings.to_pickle('/kaggle/working/embedding_clusters.pkl')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}