{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Read Article Data","metadata":{}},{"cell_type":"code","source":"import cudf\n\ndf = cudf.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\nprint(df.shape)\ndf.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-16T23:24:30.315454Z","iopub.execute_input":"2022-09-16T23:24:30.315808Z","iopub.status.idle":"2022-09-16T23:24:37.359664Z","shell.execute_reply.started":"2022-09-16T23:24:30.315722Z","shell.execute_reply":"2022-09-16T23:24:37.358507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find Categorical Columns to Represent The Data","metadata":{}},{"cell_type":"code","source":"ohe_columns = []\ntotal = 0\n\nfor col in df.columns:\n    if df[col].dtype == \"int64\" and len(df[col].unique()) <= 500:\n        ohe_columns.append(col)\n        total += len(df[col].unique())\n    \n    print(col, df[col].dtype, len(df[col].unique()))\n    \n    \nprint(\"Columns to use:\", ohe_columns)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# One-Hot-Encoding","metadata":{}},{"cell_type":"code","source":"V = cudf.get_dummies(df[ohe_columns], columns=ohe_columns).values\nV.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TFIDF for Article Description","metadata":{}},{"cell_type":"code","source":"from cuml.feature_extraction.text import TfidfVectorizer\n\ntfidf = TfidfVectorizer(min_df=3)\nV_desc = tfidf.fit_transform(df[\"detail_desc\"].fillna(\"nodesc\"))\nV_desc.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Represent The Articles as Vectors of size 512","metadata":{}},{"cell_type":"code","source":"from cuml import TruncatedSVD\nimport cupy\n\n\nEMB_SIZE = 512\n\nV = cupy.hstack([V.astype(\"float32\"), V_desc.todense()])\n\nsvd = TruncatedSVD(n_components=EMB_SIZE, random_state=0)\nsvd.fit(V)\nprint(\"Explained variance ratio:\", svd.explained_variance_ratio_.sum().item())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save The Article Embeddings","metadata":{}},{"cell_type":"code","source":"V = svd.transform(V)\nprint(V.shape)\n\ncupy.save(\"articles.npy\", V)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T06:51:53.220651Z","iopub.execute_input":"2022-05-06T06:51:53.220955Z","iopub.status.idle":"2022-05-06T06:51:53.828631Z","shell.execute_reply.started":"2022-05-06T06:51:53.220915Z","shell.execute_reply":"2022-05-06T06:51:53.827573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get similar article examples","metadata":{}},{"cell_type":"code","source":"from cuml.neighbors import NearestNeighbors\n\n\nmatcher = NearestNeighbors(n_neighbors=2, metric=\"cosine\")\nmatcher.fit(V)\n\n\ndistances, indices = matcher.kneighbors(V)\n\nd, idx = distances[:, 1], indices[:, 1]  # exclude self-match, only get the best match","metadata":{"execution":{"iopub.status.busy":"2022-05-06T06:51:53.830427Z","iopub.execute_input":"2022-05-06T06:51:53.830803Z","iopub.status.idle":"2022-05-06T06:51:55.768719Z","shell.execute_reply.started":"2022-05-06T06:51:53.830762Z","shell.execute_reply":"2022-05-06T06:51:55.767667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_examples = cupy.argsort(d)\n\ndef get_example(i):\n    index1 = sorted_examples[i]\n    index2 = idx[index1]\n    \n    print(\"Match score:\", cupy.round(1 - d[index1], 2))\n    \n    return df.iloc[[index1, index2]].to_pandas().T","metadata":{"execution":{"iopub.status.busy":"2022-05-06T06:53:15.390068Z","iopub.execute_input":"2022-05-06T06:53:15.390716Z","iopub.status.idle":"2022-05-06T06:53:15.398975Z","shell.execute_reply.started":"2022-05-06T06:53:15.390682Z","shell.execute_reply":"2022-05-06T06:53:15.397867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### A perfect match (probably a duplicate article)","metadata":{}},{"cell_type":"code","source":"get_example(0)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T06:53:16.283236Z","iopub.execute_input":"2022-05-06T06:53:16.283848Z","iopub.status.idle":"2022-05-06T06:53:16.675422Z","shell.execute_reply.started":"2022-05-06T06:53:16.283814Z","shell.execute_reply":"2022-05-06T06:53:16.674132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### An average match","metadata":{}},{"cell_type":"code","source":"get_example(df.shape[0]//2)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T06:53:17.676636Z","iopub.execute_input":"2022-05-06T06:53:17.676958Z","iopub.status.idle":"2022-05-06T06:53:17.715325Z","shell.execute_reply.started":"2022-05-06T06:53:17.676912Z","shell.execute_reply":"2022-05-06T06:53:17.714143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}