{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"},{"sourceId":3365816,"sourceType":"datasetVersion","datasetId":2029985},{"sourceId":3638692,"sourceType":"datasetVersion","datasetId":2179351},{"sourceId":3643237,"sourceType":"datasetVersion","datasetId":2181695},{"sourceId":3666299,"sourceType":"datasetVersion","datasetId":2192715}],"dockerImageVersionId":30176,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Building recommender system using collaborative filtering approach**\n\nThe following code:\n\n* Transforms data from transactions.csv into a (user, article, numbers_of_pusrchases) cllection which is the requirded form to feed to ALS model in PySpark.\n\n* Fits the model on the data to produce 10d feature vectors \n\n* Produces and plots top k recommendations\n\nInput data limited to 100000 transactions ","metadata":{}},{"cell_type":"code","source":"!pip install pyspark","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:01:40.965951Z","iopub.execute_input":"2023-09-12T08:01:40.966400Z","iopub.status.idle":"2023-09-12T08:02:24.697512Z","shell.execute_reply.started":"2023-09-12T08:01:40.966289Z","shell.execute_reply":"2023-09-12T08:02:24.696347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.ml.evaluation import RegressionEvaluator\nfrom pyspark.ml.recommendation import ALS\nfrom pyspark.sql import Row\nfrom pyspark.sql import SparkSession\nfrom pyspark.sql import Row\nimport plotly.express as px\nimport pickle\nimport pandas as pd\nfrom sklearn.neighbors import KNeighborsClassifier as KNN\nfrom sklearn.preprocessing import MinMaxScaler\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport sklearn\nspark = SparkSession.builder.appName('Recommendations').getOrCreate()\nimport networkx as nx\nfrom gensim.models import Word2Vec\nfrom sklearn.metrics.pairwise import cosine_similarity\nimport matplotlib.image as mpimg\nimport random\nfrom lightgbm.sklearn import LGBMRanker\nfrom datetime import timedelta\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nfrom tqdm import tqdm\n\nlines = spark.read.options(header=True).csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\ndf = lines.drop('sales_channel_id').drop('price').drop('t_dat').limit(100000)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:24.699638Z","iopub.execute_input":"2023-09-12T08:02:24.699962Z","iopub.status.idle":"2023-09-12T08:02:39.023951Z","shell.execute_reply.started":"2023-09-12T08:02:24.699923Z","shell.execute_reply":"2023-09-12T08:02:39.023263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get unique customer and article ids so you can map them to integers as per pyspark requirements for ALS model\nunique_customers = df.select('customer_id').distinct()\nunique_articles = df.select('article_id').distinct()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:39.025025Z","iopub.execute_input":"2023-09-12T08:02:39.025233Z","iopub.status.idle":"2023-09-12T08:02:39.098233Z","shell.execute_reply.started":"2023-09-12T08:02:39.025204Z","shell.execute_reply":"2023-09-12T08:02:39.097691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create a list of Row objects that map each custmoer id to a unique integer\ncustomer_id_mapping = []\nfor i, c in enumerate(unique_customers.collect()):\n    customer_id_mapping.append(Row(c['customer_id'], i))","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:39.099668Z","iopub.execute_input":"2023-09-12T08:02:39.099840Z","iopub.status.idle":"2023-09-12T08:02:44.201301Z","shell.execute_reply.started":"2023-09-12T08:02:39.099816Z","shell.execute_reply":"2023-09-12T08:02:44.200766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create a list of Row objects that map each article id to a unique integer\n\narticle_id_mapping = []\nfor i, c in enumerate(unique_articles.collect()):\n    article_id_mapping.append(Row(c['article_id'], i))","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:44.202465Z","iopub.execute_input":"2023-09-12T08:02:44.202864Z","iopub.status.idle":"2023-09-12T08:02:47.240329Z","shell.execute_reply.started":"2023-09-12T08:02:44.202834Z","shell.execute_reply":"2023-09-12T08:02:47.239685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customer_map = spark.createDataFrame(customer_id_mapping, ['customer_id', 'int_customer_id'])","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:47.241543Z","iopub.execute_input":"2023-09-12T08:02:47.241899Z","iopub.status.idle":"2023-09-12T08:02:48.099807Z","shell.execute_reply.started":"2023-09-12T08:02:47.241866Z","shell.execute_reply":"2023-09-12T08:02:48.098891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_map = spark.createDataFrame(article_id_mapping, ['article_id', 'int_article_id'])","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:48.101135Z","iopub.execute_input":"2023-09-12T08:02:48.101366Z","iopub.status.idle":"2023-09-12T08:02:48.559309Z","shell.execute_reply.started":"2023-09-12T08:02:48.101336Z","shell.execute_reply":"2023-09-12T08:02:48.558677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"map_df = df.join(customer_map, 'customer_id').join(article_map, on='article_id')","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:48.560375Z","iopub.execute_input":"2023-09-12T08:02:48.560887Z","iopub.status.idle":"2023-09-12T08:02:48.603294Z","shell.execute_reply.started":"2023-09-12T08:02:48.560859Z","shell.execute_reply":"2023-09-12T08:02:48.602554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"map_df.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:48.604677Z","iopub.execute_input":"2023-09-12T08:02:48.605093Z","iopub.status.idle":"2023-09-12T08:02:53.070829Z","shell.execute_reply.started":"2023-09-12T08:02:48.605054Z","shell.execute_reply":"2023-09-12T08:02:53.070091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get number of purchases of each item by each users \n#which will be treated as a rating of customers liking of this specific article\n\nuser_item = map_df.groupby(['int_customer_id', 'int_article_id']).count()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:53.073346Z","iopub.execute_input":"2023-09-12T08:02:53.073600Z","iopub.status.idle":"2023-09-12T08:02:53.120660Z","shell.execute_reply.started":"2023-09-12T08:02:53.073570Z","shell.execute_reply":"2023-09-12T08:02:53.119928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_item.show(5)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:53.121654Z","iopub.execute_input":"2023-09-12T08:02:53.121888Z","iopub.status.idle":"2023-09-12T08:02:57.898232Z","shell.execute_reply.started":"2023-09-12T08:02:53.121856Z","shell.execute_reply":"2023-09-12T08:02:57.897606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# user_item.write.parquet(\"./user_item_matrix.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:57.899075Z","iopub.execute_input":"2023-09-12T08:02:57.899257Z","iopub.status.idle":"2023-09-12T08:02:57.909970Z","shell.execute_reply.started":"2023-09-12T08:02:57.899231Z","shell.execute_reply":"2023-09-12T08:02:57.908761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(training, test) = user_item.randomSplit([0.8, 0.2])","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:57.911166Z","iopub.execute_input":"2023-09-12T08:02:57.911818Z","iopub.status.idle":"2023-09-12T08:02:57.977227Z","shell.execute_reply.started":"2023-09-12T08:02:57.911781Z","shell.execute_reply":"2023-09-12T08:02:57.976532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build the recommendation model using ALS on the training data\n# Note we set cold start strategy to 'drop' to ensure we don't get NaN evaluation metrics\nals = ALS(maxIter=5, regParam=0.01, userCol=\"int_customer_id\", itemCol=\"int_article_id\", ratingCol=\"count\",\n          coldStartStrategy=\"drop\")\nmodel = als.fit(training)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:02:57.978157Z","iopub.execute_input":"2023-09-12T08:02:57.978367Z","iopub.status.idle":"2023-09-12T08:03:09.991371Z","shell.execute_reply.started":"2023-09-12T08:02:57.978336Z","shell.execute_reply":"2023-09-12T08:03:09.990754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model by computing the RMSE on the test data\npredictions = model.transform(test)\nevaluator = RegressionEvaluator(metricName=\"rmse\", labelCol=\"count\",\n                                predictionCol=\"prediction\")\nrmse = evaluator.evaluate(predictions)\nprint(\"Root-mean-square error = \" + str(rmse))","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:09.992345Z","iopub.execute_input":"2023-09-12T08:03:09.992931Z","iopub.status.idle":"2023-09-12T08:03:15.369212Z","shell.execute_reply.started":"2023-09-12T08:03:09.992901Z","shell.execute_reply":"2023-09-12T08:03:15.368654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = open('../input/hm-product-image-embeddings/embeds.pickle', 'rb')\npaths = open('../input/hm-product-image-embeddings/paths.pickle', 'rb')\nlabels = open('../input/hm-product-image-embeddings/labels.pickle', 'rb')\npaths = pickle.load(paths)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:15.370167Z","iopub.execute_input":"2023-09-12T08:03:15.370393Z","iopub.status.idle":"2023-09-12T08:03:15.457311Z","shell.execute_reply.started":"2023-09-12T08:03:15.370359Z","shell.execute_reply":"2023-09-12T08:03:15.456591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = np.array([path[-14:-4] for path in paths])","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:15.461959Z","iopub.execute_input":"2023-09-12T08:03:15.463984Z","iopub.status.idle":"2023-09-12T08:03:15.510834Z","shell.execute_reply.started":"2023-09-12T08:03:15.463936Z","shell.execute_reply":"2023-09-12T08:03:15.509988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = np.array([pickle.load(labels) for i in range(105100)])","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:15.514901Z","iopub.execute_input":"2023-09-12T08:03:15.516646Z","iopub.status.idle":"2023-09-12T08:03:17.204860Z","shell.execute_reply.started":"2023-09-12T08:03:15.516585Z","shell.execute_reply":"2023-09-12T08:03:17.204046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeds = np.array([pickle.load(f)[0] for i, path in zip(range(105100), paths)])","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:17.205834Z","iopub.execute_input":"2023-09-12T08:03:17.206029Z","iopub.status.idle":"2023-09-12T08:03:24.162216Z","shell.execute_reply.started":"2023-09-12T08:03:17.206005Z","shell.execute_reply":"2023-09-12T08:03:24.161599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(embeds)\ndf['article_id'] = ids\ndf['label'] = labels","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:24.163412Z","iopub.execute_input":"2023-09-12T08:03:24.163637Z","iopub.status.idle":"2023-09-12T08:03:24.202044Z","shell.execute_reply.started":"2023-09-12T08:03:24.163592Z","shell.execute_reply":"2023-09-12T08:03:24.200700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[['article_id', 'label']].to_csv('cnn_labels.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:24.203429Z","iopub.execute_input":"2023-09-12T08:03:24.203661Z","iopub.status.idle":"2023-09-12T08:03:24.330984Z","shell.execute_reply.started":"2023-09-12T08:03:24.203610Z","shell.execute_reply":"2023-09-12T08:03:24.329650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = df.drop(['article_id', 'label'], axis=1)\ny = df.article_id","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:24.332155Z","iopub.execute_input":"2023-09-12T08:03:24.332394Z","iopub.status.idle":"2023-09-12T08:03:24.765367Z","shell.execute_reply.started":"2023-09-12T08:03:24.332364Z","shell.execute_reply":"2023-09-12T08:03:24.764094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn = KNN(20, metric='cosine')","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:24.767529Z","iopub.execute_input":"2023-09-12T08:03:24.767875Z","iopub.status.idle":"2023-09-12T08:03:24.780407Z","shell.execute_reply.started":"2023-09-12T08:03:24.767808Z","shell.execute_reply":"2023-09-12T08:03:24.778848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn.fit(x,y)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:24.781852Z","iopub.execute_input":"2023-09-12T08:03:24.782138Z","iopub.status.idle":"2023-09-12T08:03:25.004965Z","shell.execute_reply.started":"2023-09-12T08:03:24.782104Z","shell.execute_reply":"2023-09-12T08:03:25.004130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scores = knn.kneighbors()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:25.006101Z","iopub.execute_input":"2023-09-12T08:03:25.006956Z","iopub.status.idle":"2023-09-12T08:03:25.010517Z","shell.execute_reply.started":"2023-09-12T08:03:25.006917Z","shell.execute_reply":"2023-09-12T08:03:25.009790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_items(items, scores, labels):\n    path = \"../input/h-and-m-personalized-fashion-recommendations/images\"\n\n    k = len(items)\n    fig = plt.figure(figsize=(3*k, 10))\n    for item, i, score, label in zip(items, range(1, k+1), scores, labels):\n        sub = item[:3]\n        image = path + \"/\"+ sub + \"/\"+ item +\".jpg\"\n        image = plt.imread(image)\n        fig.add_subplot(1, k, i)\n        plt.axis(False)\n        plt.title(str(score)+' - '+(label))\n        plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:25.011657Z","iopub.execute_input":"2023-09-12T08:03:25.012073Z","iopub.status.idle":"2023-09-12T08:03:25.028115Z","shell.execute_reply.started":"2023-09-12T08:03:25.012005Z","shell.execute_reply":"2023-09-12T08:03:25.026968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nrand = x.sample(950)\nsample = knn.kneighbors(rand, 20)\nrcmnds = ids[sample[1][0]]\nscores = np.round(1- sample[0][0], 2)\nsample_labels = df[df.article_id.isin(rcmnds)].label\nprint(scores.mean())\nplot_items(rcmnds, scores, labels)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:25.029363Z","iopub.execute_input":"2023-09-12T08:03:25.029660Z","iopub.status.idle":"2023-09-12T08:03:34.772522Z","shell.execute_reply.started":"2023-09-12T08:03:25.029630Z","shell.execute_reply":"2023-09-12T08:03:34.771797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"everthing = pd.read_csv('../input/hm-embeddings-4-different-approaches/everthing.csv')\ncustomer_all = pd.read_csv('../input/hm-embeddings-4-different-approaches/customer_all.csv')\ncustomers_history = pd.read_csv('../input/hm-data-transformation/customer_sequence.csv').set_index('Unnamed: 0')","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:03:34.778533Z","iopub.execute_input":"2023-09-12T08:03:34.779392Z","iopub.status.idle":"2023-09-12T08:04:44.296059Z","shell.execute_reply.started":"2023-09-12T08:03:34.779339Z","shell.execute_reply":"2023-09-12T08:04:44.294498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfrs_knn = pickle.load(open('../input/hm-embeddings-4-different-approaches/tfrs_knn.pickle', 'rb' ))\nimage_knn = pickle.load(open('../input/hm-embeddings-4-different-approaches/image_knn.pickle', 'rb'))\ntext_knn = pickle.load(open('../input/hm-embeddings-4-different-approaches/text_knn.pickle', 'rb'))\nfeature_knn = pickle.load(open('../input/hm-embeddings-4-different-approaches/feature_knn.pickle', 'rb'))\nall_knn = pickle.load(open('../input/hm-embeddings-4-different-approaches/all_knn.pickle', 'rb'))","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:04:44.298172Z","iopub.execute_input":"2023-09-12T08:04:44.298399Z","iopub.status.idle":"2023-09-12T08:05:04.828446Z","shell.execute_reply.started":"2023-09-12T08:04:44.298375Z","shell.execute_reply":"2023-09-12T08:05:04.826989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef plot_items(items, scores=['None']):\n    path = \"../input/h-and-m-personalized-fashion-recommendations/images\"\n\n    k = len(items)\n    fig = plt.figure(figsize=(3*k, 10))\n    for item, i, score in zip(items, range(1, k+1), scores):\n        item = '0'+ str(item)\n        sub = item[:3]\n        image = path + \"/\"+ sub + \"/\"+ item +\".jpg\"\n        image = plt.imread(image)\n        fig.add_subplot(1, k, i)\n        plt.axis('off')\n        plt.title(score)\n        plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:04.830693Z","iopub.execute_input":"2023-09-12T08:05:04.831023Z","iopub.status.idle":"2023-09-12T08:05:04.838522Z","shell.execute_reply.started":"2023-09-12T08:05:04.830990Z","shell.execute_reply":"2023-09-12T08:05:04.837392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_rand_article(article_id):\n    \n    article_mask = everthing.article_id == article_id\n    \n    combined_article = everthing[article_mask].values[0][1:]\n    article_tfrs = everthing[article_mask].filter(regex='^tfrs',axis=1).values\n    article_image = everthing[article_mask].filter(regex='^image',axis=1).values\n    article_text = everthing[article_mask].filter(regex='^text',axis=1).values\n    article_feature = everthing[article_mask].filter(regex='^feature',axis=1).values\n\n    return combined_article, article_tfrs, article_image, article_text, article_feature","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:04.840307Z","iopub.execute_input":"2023-09-12T08:05:04.840678Z","iopub.status.idle":"2023-09-12T08:05:04.865134Z","shell.execute_reply.started":"2023-09-12T08:05:04.840608Z","shell.execute_reply":"2023-09-12T08:05:04.864235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_rand_customer(customer):\n        \n    customer = customer_all[customer_all.customer_id == customer].drop('customer_id', axis=1)\n    \n    customer_tfrs = customer.filter(regex='^tfrs',axis=1)\n    customer_image = customer.filter(regex='^image',axis=1)\n    customer_text = customer.filter(regex='^text',axis=1)\n    customer_feature = customer.filter(regex='^feature',axis=1)\n    \n    return customer.values[0], customer_tfrs, customer_image, customer_text, customer_feature","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:04.866401Z","iopub.execute_input":"2023-09-12T08:05:04.866980Z","iopub.status.idle":"2023-09-12T08:05:04.882754Z","shell.execute_reply.started":"2023-09-12T08:05:04.866940Z","shell.execute_reply":"2023-09-12T08:05:04.881818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_new_customer(new_customer_history):\n    \n    new_customer_embeddings = pd.DataFrame(everthing[everthing.article_id.isin(new_customer_history)].mean()).T\n    \n    new_tfrs = new_customer_embeddings.filter(regex='^tfrs')\n    new_image = new_customer_embeddings.filter(regex='^image')\n    new_text = new_customer_embeddings.filter(regex='^text')\n    new_feature = new_customer_embeddings.filter(regex='^feature')\n    \n    return new_customer_embeddings.values[0], new_tfrs, new_image, new_text, new_feature","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:04.883982Z","iopub.execute_input":"2023-09-12T08:05:04.884208Z","iopub.status.idle":"2023-09-12T08:05:04.900405Z","shell.execute_reply.started":"2023-09-12T08:05:04.884181Z","shell.execute_reply":"2023-09-12T08:05:04.899057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_rcmnds(combined, tfrs, image, text, feature ,k=20):\n    \n    combined = all_knn.kneighbors([combined], k)\n    combined_rcmnds, combined_scores = everthing.iloc[combined[1][0]].article_id.values, np.round(1- combined[0][0], 2)\n    \n    tfrs = tfrs_knn.kneighbors(tfrs, k)\n    tfrs_rcmnds, tfrs_scores = everthing.iloc[tfrs[1][0]].article_id.values, np.round(1- tfrs[0][0], 2)\n\n    image = image_knn.kneighbors(image, k)\n    image_rcmnds, image_scores = everthing.iloc[image[1][0]].article_id.values, np.round(1- image[0][0], 2)\n\n    text = text_knn.kneighbors(text, k)\n    text_rcmnds, text_scores = everthing.iloc[text[1][0]].article_id.values, np.round(1- text[0][0], 2)\n\n    feature = feature_knn.kneighbors(feature, k)\n    feature_rcmnds, feature_scores = everthing.iloc[feature[1][0]].article_id.values, np.round(1- feature[0][0], 2)\n    \n    \n    return (combined_rcmnds, combined_scores), (tfrs_rcmnds, tfrs_scores), (image_rcmnds, image_scores), (text_rcmnds, text_scores), (feature_rcmnds, feature_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:04.903612Z","iopub.execute_input":"2023-09-12T08:05:04.903995Z","iopub.status.idle":"2023-09-12T08:05:04.922471Z","shell.execute_reply.started":"2023-09-12T08:05:04.903956Z","shell.execute_reply":"2023-09-12T08:05:04.920767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Finding Item to Similar Items given from user","metadata":{}},{"cell_type":"code","source":"article_id = everthing.sample(255).article_id.values[0]\n\ncombined_article, article_tfrs, article_image, article_text, article_feature = get_rand_article(article_id)\n\n(combined_rcmnds, combined_scores), (image_rcmnds, image_scores), (tfrs_rcmnds, tfrs_scores), (text_rcmnds, text_scores), (feature_rcmnds, feature_scores) = get_rcmnds(combined_article, article_tfrs, article_image, article_text, article_feature)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:04.924052Z","iopub.execute_input":"2023-09-12T08:05:04.924318Z","iopub.status.idle":"2023-09-12T08:05:06.905977Z","shell.execute_reply.started":"2023-09-12T08:05:04.924288Z","shell.execute_reply":"2023-09-12T08:05:06.905116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items([article_id], ['Item'])","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:06.907120Z","iopub.execute_input":"2023-09-12T08:05:06.908327Z","iopub.status.idle":"2023-09-12T08:05:07.309125Z","shell.execute_reply.started":"2023-09-12T08:05:06.908293Z","shell.execute_reply":"2023-09-12T08:05:07.307650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(combined_rcmnds, combined_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:07.310590Z","iopub.execute_input":"2023-09-12T08:05:07.310880Z","iopub.status.idle":"2023-09-12T08:05:14.062475Z","shell.execute_reply.started":"2023-09-12T08:05:07.310848Z","shell.execute_reply":"2023-09-12T08:05:14.059010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(image_rcmnds, image_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:14.063914Z","iopub.execute_input":"2023-09-12T08:05:14.064656Z","iopub.status.idle":"2023-09-12T08:05:20.540027Z","shell.execute_reply.started":"2023-09-12T08:05:14.064590Z","shell.execute_reply":"2023-09-12T08:05:20.538610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(tfrs_rcmnds, tfrs_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:20.541375Z","iopub.execute_input":"2023-09-12T08:05:20.541673Z","iopub.status.idle":"2023-09-12T08:05:27.510817Z","shell.execute_reply.started":"2023-09-12T08:05:20.541638Z","shell.execute_reply":"2023-09-12T08:05:27.509783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(text_rcmnds, text_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:27.511898Z","iopub.execute_input":"2023-09-12T08:05:27.512106Z","iopub.status.idle":"2023-09-12T08:05:34.038326Z","shell.execute_reply.started":"2023-09-12T08:05:27.512079Z","shell.execute_reply":"2023-09-12T08:05:34.036949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(feature_rcmnds, feature_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:34.040424Z","iopub.execute_input":"2023-09-12T08:05:34.041061Z","iopub.status.idle":"2023-09-12T08:05:40.778741Z","shell.execute_reply.started":"2023-09-12T08:05:34.041018Z","shell.execute_reply":"2023-09-12T08:05:40.777805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Finding item based items from user history ","metadata":{}},{"cell_type":"code","source":"customer = customer_all.sample(11).customer_id.values[0]\ncustomer_history = eval(customers_history[customers_history.customer == customer].sequence.values[0])\ncustomer_history = [int(i) for i in customer_history]\n\ncombined_customer, customer_tfrs, customer_image, customer_text, customer_feature = get_rand_customer(customer)\n(combined_rcmnds, combined_scores), (image_rcmnds, image_scores), (tfrs_rcmnds, tfrs_scores), (text_rcmnds, text_scores), (feature_rcmnds, feature_scores) = get_rcmnds(combined_customer, customer_tfrs, customer_image, customer_text, customer_feature)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:40.780140Z","iopub.execute_input":"2023-09-12T08:05:40.780516Z","iopub.status.idle":"2023-09-12T08:05:42.816292Z","shell.execute_reply.started":"2023-09-12T08:05:40.780473Z","shell.execute_reply":"2023-09-12T08:05:42.815647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(customer_history[:9], range(9))","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:42.817543Z","iopub.execute_input":"2023-09-12T08:05:42.817942Z","iopub.status.idle":"2023-09-12T08:05:45.753527Z","shell.execute_reply.started":"2023-09-12T08:05:42.817907Z","shell.execute_reply":"2023-09-12T08:05:45.752697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(combined_rcmnds, combined_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:45.754698Z","iopub.execute_input":"2023-09-12T08:05:45.754894Z","iopub.status.idle":"2023-09-12T08:05:52.259277Z","shell.execute_reply.started":"2023-09-12T08:05:45.754870Z","shell.execute_reply":"2023-09-12T08:05:52.258786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(image_rcmnds,image_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:52.260253Z","iopub.execute_input":"2023-09-12T08:05:52.260487Z","iopub.status.idle":"2023-09-12T08:05:59.560452Z","shell.execute_reply.started":"2023-09-12T08:05:52.260465Z","shell.execute_reply":"2023-09-12T08:05:59.559113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(tfrs_rcmnds, text_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:05:59.562812Z","iopub.execute_input":"2023-09-12T08:05:59.563315Z","iopub.status.idle":"2023-09-12T08:06:05.959423Z","shell.execute_reply.started":"2023-09-12T08:05:59.563276Z","shell.execute_reply":"2023-09-12T08:06:05.958319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(text_rcmnds, text_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:06:05.960789Z","iopub.execute_input":"2023-09-12T08:06:05.961024Z","iopub.status.idle":"2023-09-12T08:06:12.639110Z","shell.execute_reply.started":"2023-09-12T08:06:05.960992Z","shell.execute_reply":"2023-09-12T08:06:12.637990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(feature_rcmnds, feature_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:06:12.640234Z","iopub.execute_input":"2023-09-12T08:06:12.640427Z","iopub.status.idle":"2023-09-12T08:06:19.039041Z","shell.execute_reply.started":"2023-09-12T08:06:12.640394Z","shell.execute_reply":"2023-09-12T08:06:19.038125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Recommending Items for newly generated Users","metadata":{}},{"cell_type":"code","source":"new_customer_history = np.random.choice(everthing.article_id.values, size=6, replace=False)\n\ncombined_customer, customer_tfrs, customer_image, customer_text, customer = get_new_customer(new_customer_history)\n(combined_rcmnds, combined_scores), (image_rcmnds, image_scores), (tfrs_rcmnds, tfrs_scores), (text_rcmnds, text_scores), (feature_rcmnds, feature_scores) = get_rcmnds(combined_customer[1:], customer_tfrs, customer_image, customer_text, customer_feature)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:06:19.040297Z","iopub.execute_input":"2023-09-12T08:06:19.041444Z","iopub.status.idle":"2023-09-12T08:06:21.121702Z","shell.execute_reply.started":"2023-09-12T08:06:19.041384Z","shell.execute_reply":"2023-09-12T08:06:21.120796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(new_customer_history[:9], range(9))","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:06:21.122973Z","iopub.execute_input":"2023-09-12T08:06:21.123199Z","iopub.status.idle":"2023-09-12T08:06:23.075925Z","shell.execute_reply.started":"2023-09-12T08:06:21.123171Z","shell.execute_reply":"2023-09-12T08:06:23.074999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(combined_rcmnds, combined_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:06:23.077342Z","iopub.execute_input":"2023-09-12T08:06:23.077612Z","iopub.status.idle":"2023-09-12T08:06:29.778489Z","shell.execute_reply.started":"2023-09-12T08:06:23.077575Z","shell.execute_reply":"2023-09-12T08:06:29.777323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(image_rcmnds,image_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:06:29.780228Z","iopub.execute_input":"2023-09-12T08:06:29.780531Z","iopub.status.idle":"2023-09-12T08:06:36.682331Z","shell.execute_reply.started":"2023-09-12T08:06:29.780498Z","shell.execute_reply":"2023-09-12T08:06:36.681029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(tfrs_rcmnds, text_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:06:36.683612Z","iopub.execute_input":"2023-09-12T08:06:36.683883Z","iopub.status.idle":"2023-09-12T08:06:43.137450Z","shell.execute_reply.started":"2023-09-12T08:06:36.683852Z","shell.execute_reply":"2023-09-12T08:06:43.136716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(text_rcmnds, text_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:06:43.138715Z","iopub.execute_input":"2023-09-12T08:06:43.139093Z","iopub.status.idle":"2023-09-12T08:06:50.570664Z","shell.execute_reply.started":"2023-09-12T08:06:43.139061Z","shell.execute_reply":"2023-09-12T08:06:50.569532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_items(feature_rcmnds, feature_scores)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:06:50.572199Z","iopub.execute_input":"2023-09-12T08:06:50.572421Z","iopub.status.idle":"2023-09-12T08:06:56.356836Z","shell.execute_reply.started":"2023-09-12T08:06:50.572394Z","shell.execute_reply":"2023-09-12T08:06:56.355468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\n# customers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:06:56.358257Z","iopub.execute_input":"2023-09-12T08:06:56.358479Z","iopub.status.idle":"2023-09-12T08:07:47.547449Z","shell.execute_reply.started":"2023-09-12T08:06:56.358452Z","shell.execute_reply":"2023-09-12T08:07:47.546046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In order to depict this dataset as a Graph, we need to retain only the meaningful rows. The goal is to reduce sparsity as much as possible for better performance.\n\nSo, the first thing to do is to calculate the frequency of 'article_id' (product ID) and 'customer_id' (user ID), and consider filtering only users who have many purchase records & products with many purchase records.","metadata":{}},{"cell_type":"code","source":"item_freq = transactions.groupby('article_id')['customer_id'].nunique()\nuser_freq = transactions.groupby('customer_id')['article_id'].nunique()\n\nitems = item_freq[item_freq >= 100].index\nusers = user_freq[user_freq >= 100].index\n\nfiltered_df = transactions[transactions['article_id'].isin(items) & transactions['customer_id'].isin(users)]","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:07:47.549491Z","iopub.execute_input":"2023-09-12T08:07:47.550321Z","iopub.status.idle":"2023-09-12T08:08:34.763806Z","shell.execute_reply.started":"2023-09-12T08:07:47.550282Z","shell.execute_reply":"2023-09-12T08:08:34.762797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, let's aggregate to reflect the weight we learned in the previous tutorial on the edges. This is a more in-depth application of what we've learned on the journey so far. The weight given to the edge can contain various information values. By assigning appropriate values, we can also reflect in the graph how much the user prefers the item.\n\nLet's take an example. Consider user 0, item A, and item B as nodes. If user 0 purchased item A 10 times and item B once... We could say the preference for item A is higher, right?\n\nThere are two edges connecting these three nodes. The edges that stretch from the user 0 node to each item node. What relation might these edges imply? It's the purchase history! Therefore, we can assign a weight of 10 to the edge connected to item A node and a weight of 1 to the edge connected to item B.","metadata":{}},{"cell_type":"code","source":"freq = filtered_df.groupby(['customer_id', 'article_id']).size().reset_index(name='frequency')\n\nGraphTravel_HM = filtered_df.merge(freq, on=['customer_id', 'article_id'], how='left')\n\nGraphTravel_HM = GraphTravel_HM[GraphTravel_HM['frequency'] >= 10]","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:08:34.765228Z","iopub.execute_input":"2023-09-12T08:08:34.765459Z","iopub.status.idle":"2023-09-12T08:08:43.337604Z","shell.execute_reply.started":"2023-09-12T08:08:34.765430Z","shell.execute_reply":"2023-09-12T08:08:43.336185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we have information that allows us to understand a specific user's preference (repeat purchase frequency) for an item. It seems we can assign weight information to the edge using the 'frequency' column!\n\nIs it okay to use weight and random walk search bias together?\n\nYes! Weights and search bias can be used simultaneously and do not conflict with each other. Each controls a different aspect of the random walk, and using them together allows for more detailed control over how the random walk navigates the graph. To explain each concept a little more,\n\nWeight is used to indicate the importance of an edge and is used to determine the probability that the random walk algorithm will follow a particular edge. Edges with high weights have a higher probability of being chosen by the random walk than edges with low weights.\n\nSearch bias is controlled by the two parameters p and q in Node2Vec. These parameters control the degree to which the random walk prefers to revisit nodes it has visited before or visit new nodes it has not yet visited.\n\nAnd lastly, in order to reflect only meaningful information in the Graph, the dataset was filtered toleave only rows of products purchased more than 10 times by a user.\n\nNow that some preprocessing seems to have been done, shall we take a moment to check the state of the dataset?","metadata":{}},{"cell_type":"code","source":"display(GraphTravel_HM)\n\nprint(\"unique customer_id\" , GraphTravel_HM.customer_id.nunique())\nprint(\"unique article_id\" , GraphTravel_HM.article_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:08:43.339217Z","iopub.execute_input":"2023-09-12T08:08:43.339494Z","iopub.status.idle":"2023-09-12T08:08:43.379174Z","shell.execute_reply.started":"2023-09-12T08:08:43.339459Z","shell.execute_reply":"2023-09-12T08:08:43.377879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#unique customer_id 1013\n#unique article_id 922","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:08:43.381044Z","iopub.execute_input":"2023-09-12T08:08:43.381330Z","iopub.status.idle":"2023-09-12T08:08:43.385060Z","shell.execute_reply.started":"2023-09-12T08:08:43.381303Z","shell.execute_reply":"2023-09-12T08:08:43.384332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nsns.distplot(GraphTravel_HM['frequency'], kde=True, bins=30)\n\nplt.title('Distribution of frequency')\nplt.xlabel('Frequency')\nplt.ylabel('Density')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:08:43.386424Z","iopub.execute_input":"2023-09-12T08:08:43.387211Z","iopub.status.idle":"2023-09-12T08:08:43.849735Z","shell.execute_reply.started":"2023-09-12T08:08:43.387164Z","shell.execute_reply":"2023-09-12T08:08:43.848759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_customer_ids = GraphTravel_HM['customer_id'].unique()\ncustomer_id_mapping = {id: i for i, id in enumerate(unique_customer_ids)}\nGraphTravel_HM['customer_id'] = GraphTravel_HM['customer_id'].map(customer_id_mapping)\n\nitem_name_mapping = dict(zip(articles['article_id'], articles['prod_name'])) # prod_name","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:08:43.851056Z","iopub.execute_input":"2023-09-12T08:08:43.851306Z","iopub.status.idle":"2023-09-12T08:08:43.904449Z","shell.execute_reply.started":"2023-09-12T08:08:43.851274Z","shell.execute_reply":"2023-09-12T08:08:43.903209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"G = nx.Graph()\n\nfor index, row in GraphTravel_HM.iterrows():\n    G.add_node(row['customer_id'], type='user')\n    G.add_node(row['article_id'], type='item')\n    G.add_edge(row['customer_id'], row['article_id'], weight=row['frequency'])","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:08:43.905955Z","iopub.execute_input":"2023-09-12T08:08:43.906190Z","iopub.status.idle":"2023-09-12T08:08:45.287083Z","shell.execute_reply.started":"2023-09-12T08:08:43.906161Z","shell.execute_reply":"2023-09-12T08:08:45.286143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# biased random walk  \ndef biased_random_walk(G, start_node, walk_length, p=1, q=1):\n    walk = [start_node]\n\n    while len(walk) < walk_length:\n        cur_node = walk[-1]\n        cur_neighbors = list(G.neighbors(cur_node))\n\n        if len(cur_neighbors) > 0:\n            if len(walk) == 1:\n                walk.append(random.choice(cur_neighbors))\n            else:\n                prev_node = walk[-2]\n\n                probability = []\n                for neighbor in cur_neighbors:\n                    if neighbor == prev_node:\n                        # Return parameter \n                        probability.append(1/p)\n                    elif G.has_edge(neighbor, prev_node):\n                        # Stay parameter \n                        probability.append(1)\n                    else:\n                        # In-out parameter \n                        probability.append(1/q)\n\n                probability = np.array(probability)\n                probability = probability / probability.sum()  # normalize\n\n                next_node = np.random.choice(cur_neighbors, p=probability)\n                walk.append(next_node)\n        else:\n            break\n\n    return walk","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:08:45.288306Z","iopub.execute_input":"2023-09-12T08:08:45.288553Z","iopub.status.idle":"2023-09-12T08:08:45.296610Z","shell.execute_reply.started":"2023-09-12T08:08:45.288522Z","shell.execute_reply":"2023-09-12T08:08:45.296007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_walks(G, num_walks, walk_length, p=1, q=1):\n    walks = []\n    nodes = list(G.nodes())\n    for _ in range(num_walks):\n        random.shuffle(nodes)  # to ensure randomness\n        for node in nodes:\n            walk_from_node = biased_random_walk(G, node, walk_length, p, q)\n            walks.append(walk_from_node)\n    return walks","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:08:45.299077Z","iopub.execute_input":"2023-09-12T08:08:45.299433Z","iopub.status.idle":"2023-09-12T08:08:45.316482Z","shell.execute_reply.started":"2023-09-12T08:08:45.299380Z","shell.execute_reply":"2023-09-12T08:08:45.315568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Random Walk \nwalks = generate_walks(G, num_walks=10, walk_length=20, p=9, q=1)\nfiltered_walks = [walk for walk in walks if len(walk) >= 5]\n\n# to String  (for Word2Vec input)\nwalks = [[str(node) for node in walk] for walk in walks]\n\n# Word2Vec train\nmodel = Word2Vec(walks, vector_size=128, window=5, min_count=0,  hs=1, sg=1, workers=4, epochs=10)\n\n# node embedding extract\nembeddings = {node_id: model.wv[node_id] for node_id in model.wv.index_to_key}","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:08:45.318067Z","iopub.execute_input":"2023-09-12T08:08:45.318782Z","iopub.status.idle":"2023-09-12T08:09:09.428003Z","shell.execute_reply.started":"2023-09-12T08:08:45.318740Z","shell.execute_reply":"2023-09-12T08:09:09.427015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_user_embedding(user_id, embeddings):\n    return embeddings[str(user_id)]","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:09:09.429174Z","iopub.execute_input":"2023-09-12T08:09:09.429399Z","iopub.status.idle":"2023-09-12T08:09:09.434100Z","shell.execute_reply.started":"2023-09-12T08:09:09.429372Z","shell.execute_reply":"2023-09-12T08:09:09.433244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_rated_items(user_id, df):\n    return set(df[df['customer_id'] == user_id]['article_id'])","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:09:09.435197Z","iopub.execute_input":"2023-09-12T08:09:09.436065Z","iopub.status.idle":"2023-09-12T08:09:09.455043Z","shell.execute_reply.started":"2023-09-12T08:09:09.436000Z","shell.execute_reply":"2023-09-12T08:09:09.453813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_similarities(user_id, df, embeddings):\n    rated_items = get_rated_items(user_id, df)\n    user_embedding = get_user_embedding(user_id, embeddings)\n\n    item_similarities = []\n    for item_id in set(df['article_id']):\n        if item_id not in rated_items:  \n            item_embedding = embeddings[str(item_id)]\n            similarity = cosine_similarity([user_embedding], [item_embedding])[0][0]\n            item_similarities.append((item_id, similarity))\n\n    return item_similarities","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:09:09.456354Z","iopub.execute_input":"2023-09-12T08:09:09.456558Z","iopub.status.idle":"2023-09-12T08:09:09.470282Z","shell.execute_reply.started":"2023-09-12T08:09:09.456532Z","shell.execute_reply":"2023-09-12T08:09:09.469294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_images(items, item_name_mapping, num_items, show_similarity=False):\n    f, ax = plt.subplots(1, num_items, figsize=(20,10))\n    if num_items == 1:\n        ax = [ax]\n    for i, item in enumerate(items):\n        item_id, similarity = item\n        print(f\"- Item {item_id}: {item_name_mapping[item_id]}\", end='')\n        if show_similarity:\n            print(f\" with similarity score: {similarity}\")\n        else:\n            print()\n        img_path = f\"../input/h-and-m-personalized-fashion-recommendations/images/0{str(item_id)[:2]}/0{int(item_id)}.jpg\"\n        try:\n            img = mpimg.imread(img_path)\n            ax[i].imshow(img)\n            ax[i].set_title(f'Item {item_id}')\n            ax[i].set_xticks([], [])\n            ax[i].set_yticks([], [])\n            ax[i].grid(False)\n        except FileNotFoundError:\n            print(f\"Image for item {item_id} not found.\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:09:09.471460Z","iopub.execute_input":"2023-09-12T08:09:09.472840Z","iopub.status.idle":"2023-09-12T08:09:09.486635Z","shell.execute_reply.started":"2023-09-12T08:09:09.472749Z","shell.execute_reply":"2023-09-12T08:09:09.485772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def recommend_items(user_id, df, embeddings, item_name_mapping, num_items=5):\n    rated_items = get_rated_items(user_id, df)\n    \n    print(f\"User {user_id} has purchased:\")\n    show_images([(item_id, 0) for item_id in list(rated_items)[:5]], item_name_mapping, min(len(rated_items), 5))\n    \n    item_similarities = calculate_similarities(user_id, df, embeddings)\n\n    recommended_items = sorted(item_similarities, key=lambda x: x[1], reverse=True)[:num_items]\n\n    print(f\"\\nRecommended items for user {user_id}:\")\n    show_images(recommended_items, item_name_mapping, num_items, show_similarity=True)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:09:09.488082Z","iopub.execute_input":"2023-09-12T08:09:09.488457Z","iopub.status.idle":"2023-09-12T08:09:09.501054Z","shell.execute_reply.started":"2023-09-12T08:09:09.488423Z","shell.execute_reply":"2023-09-12T08:09:09.500170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# costomer 45's top 5 \nrecommend_items(45, GraphTravel_HM, embeddings, item_name_mapping, num_items=5)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:09:09.502369Z","iopub.execute_input":"2023-09-12T08:09:09.502754Z","iopub.status.idle":"2023-09-12T08:09:12.385153Z","shell.execute_reply.started":"2023-09-12T08:09:09.502722Z","shell.execute_reply":"2023-09-12T08:09:12.383895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. Do customers buy the same product multiple times?¶\n","metadata":{}},{"cell_type":"code","source":"df_trans = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv',dtype={'article_id': str})\ndf_trans['t_dat'] = pd.to_datetime(df_trans['t_dat'])\ndf_trans = df_trans[df_trans['t_dat'] >= pd.to_datetime('2020-07-01')]\ndf_article = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv',dtype={'article_id': str})\n\ndf_article['idxgrp_idx_prdtyp'] = df_article['index_group_name'] + '_' + df_article['index_name'] + '_' + df_article['product_type_name']\n\ndf = pd.merge(\n    df_trans,\n    df_article,\n    on='article_id',\n    how='left'\n)\ndf['product_code'] = df['product_code'].astype(str)\ndf['num_week'] = df['t_dat'].dt.isocalendar().week\ndf['product_code'] = df['product_code'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:09:12.386782Z","iopub.execute_input":"2023-09-12T08:09:12.387032Z","iopub.status.idle":"2023-09-12T08:10:03.807835Z","shell.execute_reply.started":"2023-09-12T08:09:12.387002Z","shell.execute_reply":"2023-09-12T08:10:03.806916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compare a set of products purchased in any given week with a set of products purchased 1, 2, or 3 weeks ago to see if they contain the SAME products.","metadata":{}},{"cell_type":"code","source":"def do_customers_purchase_same_AGGKEY(df, agg_key):\n    dfagg = df.groupby(['num_week','customer_id'])[[agg_key]].agg({\n            agg_key: lambda x: ','.join(x)\n    }).reset_index().rename(columns={agg_key: 'purchased_set'})\n    dfagg['num_2wk_before'] = dfagg['num_week'] + 2\n    dfagg = pd.merge(\n        dfagg[['num_week','customer_id','purchased_set']],\n        dfagg.rename(columns={'purchased_set': '2wk_before_purchased_set'})[['num_2wk_before','customer_id','2wk_before_purchased_set']],\n        left_on=['num_week', 'customer_id'],\n        right_on=['num_2wk_before', 'customer_id'],\n        how='left'\n    )\n    dfagg['num_1wk_before'] = dfagg['num_week'] + 1\n    dfagg = pd.merge(\n        dfagg,\n        dfagg.rename(columns={'purchased_set': '1wk_before_purchased_set'})[['num_1wk_before','customer_id','1wk_before_purchased_set']],\n        left_on=['num_week', 'customer_id'],\n        right_on=['num_1wk_before', 'customer_id'],\n        how='left'\n    )\n    dfagg['num_3wk_before'] = dfagg['num_week'] + 3\n    dfagg = pd.merge(\n        dfagg,\n        dfagg.rename(columns={'purchased_set': '3wk_before_purchased_set'})[['num_3wk_before','customer_id','3wk_before_purchased_set']],\n        left_on=['num_week', 'customer_id'],\n        right_on=['num_3wk_before', 'customer_id'],\n        how='left'\n\n    )\n    dfagg = dfagg[['num_week','customer_id','purchased_set','1wk_before_purchased_set','2wk_before_purchased_set','3wk_before_purchased_set']]\n    for col in ['purchased_set','1wk_before_purchased_set', '2wk_before_purchased_set', '3wk_before_purchased_set']:\n        dfagg[col] = dfagg[col].fillna('')\n        dfagg[col] = dfagg[col].str.split(',')\n    dfagg['2wk_before_purchased_set'] = dfagg['2wk_before_purchased_set'] + dfagg['1wk_before_purchased_set']\n    dfagg['3wk_before_purchased_set'] = dfagg['3wk_before_purchased_set'] + dfagg['2wk_before_purchased_set']\n    for col in ['purchased_set','1wk_before_purchased_set', '2wk_before_purchased_set', '3wk_before_purchased_set']:\n        dfagg[col] = dfagg[col].map(set)\n\n    dfagg['is_purchased_same_within_1wk'] = (dfagg['purchased_set'] & dfagg['1wk_before_purchased_set']).astype(int)\n    dfagg['is_purchased_same_within_2wk'] = (dfagg['purchased_set'] & dfagg['2wk_before_purchased_set']).astype(int)\n    dfagg['is_purchased_same_within_3wk'] = (dfagg['purchased_set'] & dfagg['3wk_before_purchased_set']).astype(int)\n    print(len(dfagg[dfagg['is_purchased_same_within_3wk'] == 1]['customer_id'].unique()) / len(dfagg['customer_id'].unique()) * 100,\n        len(dfagg[dfagg['is_purchased_same_within_2wk'] == 1]['customer_id'].unique()) / len(dfagg['customer_id'].unique()) * 100,\n        len(dfagg[dfagg['is_purchased_same_within_1wk'] == 1]['customer_id'].unique()) / len(dfagg['customer_id'].unique()) * 100\n    )\n    df_vis = pd.DataFrame({\n        'Pediod': ['Within_1wk', 'Within_2wk', 'Within_3wk'],\n        'Ratio': [len(dfagg[dfagg['is_purchased_same_within_1wk'] == 1]['customer_id'].unique()) / len(dfagg['customer_id'].unique()) * 100,\n                  len(dfagg[dfagg['is_purchased_same_within_2wk'] == 1]['customer_id'].unique()) / len(dfagg['customer_id'].unique()) * 100,\n                  len(dfagg[dfagg['is_purchased_same_within_3wk'] == 1]['customer_id'].unique()) / len(dfagg['customer_id'].unique()) * 100]\n    })\n    fig = px.bar(df_vis, x='Pediod', y='Ratio')\n    fig.show()\n    return dfagg","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:10:03.808925Z","iopub.execute_input":"2023-09-12T08:10:03.809113Z","iopub.status.idle":"2023-09-12T08:10:03.827199Z","shell.execute_reply.started":"2023-09-12T08:10:03.809090Z","shell.execute_reply":"2023-09-12T08:10:03.825759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfagg_article = do_customers_purchase_same_AGGKEY(df, 'article_id')\n","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:10:03.828905Z","iopub.execute_input":"2023-09-12T08:10:03.829254Z","iopub.status.idle":"2023-09-12T08:10:38.047141Z","shell.execute_reply.started":"2023-09-12T08:10:03.829221Z","shell.execute_reply":"2023-09-12T08:10:38.046270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. Conclusion - Do customers buy the same product multiple times?\n5.1% of customers will buy the same product in one week\n6.2% will buy the same product within two weeks\n6.6% will buy the same product within three weeks\nIn other words, most customers who purchase a product again purchase the same product within two weeks.\n\n2. Do customers buy different colors and sizes of the same product?\nIn this analysis I aggregate data at 'product code'-granularity instead of 'article_id'-granularity.","metadata":{}},{"cell_type":"code","source":"dfagg_prdcd = do_customers_purchase_same_AGGKEY(df, 'product_code')\n","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:10:38.048785Z","iopub.execute_input":"2023-09-12T08:10:38.049078Z","iopub.status.idle":"2023-09-12T08:11:10.269738Z","shell.execute_reply.started":"2023-09-12T08:10:38.049040Z","shell.execute_reply":"2023-09-12T08:11:10.268412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2. Conclusion - Do customers buy different colors and sizes of the same product?\n6.8% of customers will buy the same product code item in one week\n8.5% will buy the same product code item within two weeks\n9.4% will buy the same product code item within three weeks\nCustomers seem to need a little more time when buying products with the same product code but different colors and sizes.\n\n3. Do customers buy the same product type?\nIn this analysis I aggregate data at 'idxgrp_idx_prdtyp'-granularity instead of 'product code'-granularity.\n\n*idxgrp_idx_prdtyp explanation\nhttps://www.kaggle.com/lichtlab/h-m-data-deep-dive-chap-1-understand-article\n\nResult","metadata":{}},{"cell_type":"code","source":"dfagg_idxgrp_idx_prdtyp = do_customers_purchase_same_AGGKEY(df, 'idxgrp_idx_prdtyp')\n","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:11:10.271262Z","iopub.execute_input":"2023-09-12T08:11:10.271584Z","iopub.status.idle":"2023-09-12T08:11:45.610395Z","shell.execute_reply.started":"2023-09-12T08:11:10.271541Z","shell.execute_reply":"2023-09-12T08:11:45.609346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"3. Conclusion - Do customers buy the same product type?¶\n10.8% of customers will buy the same idxgrp_idx_prdtyp item in one week\n14.3% will buy the same idxgrp_idx_prdtyp item within two weeks\n16.5% will buy the same idxgrp_idx_prdtyp item within three weeks\n16.5% is huge! If we observe the purchasing behavior of customers over a short period of time, they seem to buy multiple similar products.","metadata":{}},{"cell_type":"markdown","source":"4. What kind of customer feature or product feature lead to 'one more time purchase'?\n      4-1 Product feature\n","metadata":{}},{"cell_type":"markdown","source":"Data prep¶\n\nFor each customer, set a binary flag to indicate whether the customer has purchased a product with the same product_code in the last three weeks.\n\n","metadata":{}},{"cell_type":"code","source":"dfagg = df.sort_values(\"t_dat\")\\\n        .set_index('t_dat')\\\n        .groupby(['customer_id','product_code'])\\\n        .rolling(\"21d\")[[\"price\"]]\\\n        .count()\\\n        .reset_index()\\\n        .rename(columns={'price': 'num_purchased_same_article'})\ndfagg['is_purchased_same_prdcd_within_3wk'] = (dfagg['num_purchased_same_article'] > 1).astype(int)\ndfagg = dfagg[dfagg['t_dat'] > '2020-09-01']\ndfagg = pd.merge(\n    dfagg,\n    df.groupby(['idxgrp_idx_prdtyp','product_code'])[[]].count().reset_index(),\n    on='product_code'\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:11:45.612515Z","iopub.execute_input":"2023-09-12T08:11:45.612968Z","iopub.status.idle":"2023-09-12T08:12:51.387649Z","shell.execute_reply.started":"2023-09-12T08:11:45.612930Z","shell.execute_reply":"2023-09-12T08:12:51.386802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model design\nCreate a logistic regression model with the idxgrp_idx_prdtyp as the explanatory variable to see if it can explain the binary flags I have created earlier.","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\ndftrain = pd.get_dummies(dfagg[['idxgrp_idx_prdtyp']], drop_first=True)\ndftrain['is_purchased_same_prdcd_within_3wk'] = dfagg['is_purchased_same_prdcd_within_3wk']\ndftrain_partial = dftrain.sample(frac=0.01, random_state=0)\ndftrain_partial = dftrain_partial.dropna()\nfeature_cols = [col for col in dftrain.columns if col not in ['is_purchased_same_prdcd_within_3wk']]\nmodel = LogisticRegression(C=5.0, penalty=\"l1\", tol=0.01, solver=\"saga\")\nmodel.fit(dftrain_partial[feature_cols], dftrain_partial['is_purchased_same_prdcd_within_3wk'])\ndf_coef = pd.DataFrame(\n        model.coef_,\n        index=['coefficient'],\n        columns=feature_cols).T.reset_index()\ndf_coef['index'] = df_coef['index'].str.replace('idxgrp_idx_prdtyp_', '')","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:12:51.389758Z","iopub.execute_input":"2023-09-12T08:12:51.390055Z","iopub.status.idle":"2023-09-12T08:12:54.739323Z","shell.execute_reply.started":"2023-09-12T08:12:51.390017Z","shell.execute_reply":"2023-09-12T08:12:54.738084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Types of products that are easy to purchase multiple times¶\n","metadata":{}},{"cell_type":"code","source":"px.bar(df_coef.sort_values(by='coefficient', ascending=False)[:15], x='index', y='coefficient')","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:12:54.740699Z","iopub.execute_input":"2023-09-12T08:12:54.740948Z","iopub.status.idle":"2023-09-12T08:12:54.804520Z","shell.execute_reply.started":"2023-09-12T08:12:54.740918Z","shell.execute_reply":"2023-09-12T08:12:54.803139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Types of products that are not purchased multiple times¶\n","metadata":{}},{"cell_type":"code","source":"px.bar(df_coef.sort_values(by='coefficient', ascending=False)[-15:], x='index', y='coefficient')","metadata":{"execution":{"iopub.status.busy":"2023-09-12T08:12:54.806180Z","iopub.execute_input":"2023-09-12T08:12:54.806472Z","iopub.status.idle":"2023-09-12T08:12:54.866950Z","shell.execute_reply.started":"2023-09-12T08:12:54.806439Z","shell.execute_reply":"2023-09-12T08:12:54.865686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}