{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport cv2\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0\nimport matplotlib.pyplot as plt\nfrom cuml import TruncatedSVD","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-29T01:36:30.397699Z","iopub.execute_input":"2022-08-29T01:36:30.398446Z","iopub.status.idle":"2022-08-29T01:36:43.396131Z","shell.execute_reply.started":"2022-08-29T01:36:30.398399Z","shell.execute_reply":"2022-08-29T01:36:43.394278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_article_images_df(path= '../input/h-and-m-personalized-fashion-recommendations/images'):\n    article_ids = []\n    image_paths = []\n    for dirname, _, filenames in os.walk(path):\n        for filename in filenames:\n            fullpath = os.path.join(dirname, filename)\n            image_path = fullpath\n            article_id = fullpath.split('/')[-1].replace('.jpg', '')\n            article_ids.append(article_id)\n            image_paths.append(fullpath)\n    return pd.DataFrame({'article_id': article_ids, 'image': image_paths})","metadata":{"execution":{"iopub.status.busy":"2022-08-29T00:01:50.754448Z","iopub.execute_input":"2022-08-29T00:01:50.75693Z","iopub.status.idle":"2022-08-29T00:01:50.765288Z","shell.execute_reply.started":"2022-08-29T00:01:50.756898Z","shell.execute_reply":"2022-08-29T00:01:50.7636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = get_article_images_df()","metadata":{"execution":{"iopub.status.busy":"2022-08-29T00:01:50.766885Z","iopub.execute_input":"2022-08-29T00:01:50.767293Z","iopub.status.idle":"2022-08-29T00:03:48.471588Z","shell.execute_reply.started":"2022-08-29T00:01:50.767251Z","shell.execute_reply":"2022-08-29T00:03:48.470467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataGenerator(tf.keras.utils.Sequence):\n    'Generates data for Keras'\n    def __init__(self, df, img_size=512, batch_size=32):\n        self.df = df\n        self.img_size = img_size\n        self.batch_size = batch_size\n        self.indexes = np.arange(len(self.df))\n\n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        ct = len(self.df) // self.batch_size\n        ct += int(((len(self.df)) % self.batch_size) != 0)\n        return ct\n    \n    def __getitem__(self, index):\n        'Generate one batch of data'\n        indexes = self.indexes[index * self.batch_size:(index + 1) * self.batch_size]\n        X = self.__data_generation(indexes)\n        return X\n\n    def __data_generation(self, indexes):\n        'Generates data containing batch_size samples'\n        X = np.zeros((len(indexes), self.img_size, self.img_size, 3), dtype='float32')\n        df = self.df.iloc[indexes]\n        for i, (index, row) in enumerate(df.iterrows()):\n            img = cv2.imread(row.image)\n            X[i,] = cv2.resize(img, (self.img_size, self.img_size))  # /128.0 - 1.0\n            \n        return X","metadata":{"execution":{"iopub.status.busy":"2022-08-29T00:03:48.474504Z","iopub.execute_input":"2022-08-29T00:03:48.475241Z","iopub.status.idle":"2022-08-29T00:03:48.484869Z","shell.execute_reply.started":"2022-08-29T00:03:48.475199Z","shell.execute_reply":"2022-08-29T00:03:48.483952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = EfficientNetB0(weights='imagenet', include_top=False, pooling='avg', input_shape=None)\ntrain_gen = DataGenerator(df, batch_size=32)\nimage_embeddings = model.predict(train_gen, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-29T00:03:48.4866Z","iopub.execute_input":"2022-08-29T00:03:48.486959Z","iopub.status.idle":"2022-08-29T01:17:49.635409Z","shell.execute_reply.started":"2022-08-29T00:03:48.486924Z","shell.execute_reply":"2022-08-29T01:17:49.634196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('hm_embeddings_effb0.npy', 'wb') as f:\n    np.save(f, image_embeddings)","metadata":{"execution":{"iopub.status.busy":"2022-08-29T01:23:18.028058Z","iopub.execute_input":"2022-08-29T01:23:18.028794Z","iopub.status.idle":"2022-08-29T01:23:18.46781Z","shell.execute_reply.started":"2022-08-29T01:23:18.028756Z","shell.execute_reply":"2022-08-29T01:23:18.466812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_embeddings = np.load(\"./hm_embeddings_effb0.npy\")\nplt.hist(np.var(image_embeddings, axis=0), bins=100)\nplt.title('Variance on the Embedding Coordinates')\nplt.show()\n\n\nsvd = TruncatedSVD(n_components=1280, random_state=0)\nsvd.fit(image_embeddings)\nprint(\"Explained variance ratio:\", svd.explained_variance_ratio_.sum().item())\nimage_embeddings=svd.transform(image_embeddings)","metadata":{"execution":{"iopub.status.busy":"2022-08-29T01:36:46.320441Z","iopub.execute_input":"2022-08-29T01:36:46.321218Z","iopub.status.idle":"2022-08-29T01:36:46.885068Z","shell.execute_reply.started":"2022-08-29T01:36:46.321184Z","shell.execute_reply":"2022-08-29T01:36:46.882577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}