{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-25T14:01:23.916264Z","iopub.execute_input":"2022-03-25T14:01:23.916545Z","iopub.status.idle":"2022-03-25T14:01:23.920686Z","shell.execute_reply.started":"2022-03-25T14:01:23.916515Z","shell.execute_reply":"2022-03-25T14:01:23.919743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### If you found this helpful Do UpVote / Comment please.","metadata":{}},{"cell_type":"markdown","source":"# Combined Customer Data with Articles data basesd on article_id","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv', chunksize=100000)\narticles = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/articles.csv')\nusers = next(df)\ndf = users.merge(articles, on='article_id')","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:37:47.240999Z","iopub.execute_input":"2022-03-25T12:37:47.241295Z","iopub.status.idle":"2022-03-25T12:37:48.287058Z","shell.execute_reply.started":"2022-03-25T12:37:47.241262Z","shell.execute_reply":"2022-03-25T12:37:48.286216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['t_dat'] = pd.to_datetime(df['t_dat'], format=\"%Y-%m-%d\")","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:37:48.288489Z","iopub.execute_input":"2022-03-25T12:37:48.288701Z","iopub.status.idle":"2022-03-25T12:37:48.309428Z","shell.execute_reply.started":"2022-03-25T12:37:48.288677Z","shell.execute_reply":"2022-03-25T12:37:48.308493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\ndf['timestamp'] = df.t_dat.values.astype(np.int64) // 10 ** 9\n","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:37:48.403394Z","iopub.execute_input":"2022-03-25T12:37:48.403979Z","iopub.status.idle":"2022-03-25T12:37:48.410182Z","shell.execute_reply.started":"2022-03-25T12:37:48.403942Z","shell.execute_reply":"2022-03-25T12:37:48.409332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:37:48.811745Z","iopub.execute_input":"2022-03-25T12:37:48.812408Z","iopub.status.idle":"2022-03-25T12:37:48.896832Z","shell.execute_reply.started":"2022-03-25T12:37:48.812368Z","shell.execute_reply":"2022-03-25T12:37:48.895786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ar = pd.read_csv(r\"/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\", dtype={'article_id': 'str'})\ndf_ar.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:37:51.369593Z","iopub.execute_input":"2022-03-25T12:37:51.370365Z","iopub.status.idle":"2022-03-25T12:37:52.0669Z","shell.execute_reply.started":"2022-03-25T12:37:51.37032Z","shell.execute_reply":"2022-03-25T12:37:52.066001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Total Articles","metadata":{}},{"cell_type":"code","source":"for col in df_ar.columns:\n    print(col,len(pd.unique(df_ar[col])))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:37:53.280197Z","iopub.execute_input":"2022-03-25T12:37:53.280479Z","iopub.status.idle":"2022-03-25T12:37:53.433505Z","shell.execute_reply.started":"2022-03-25T12:37:53.280449Z","shell.execute_reply":"2022-03-25T12:37:53.432596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Unique Customers & Transactions from combied DataFrame ( ChunkSize 100000 )","metadata":{}},{"cell_type":"code","source":"for col in df.columns:\n    print(col,len(pd.unique(df[col])))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:37:54.997347Z","iopub.execute_input":"2022-03-25T12:37:54.997622Z","iopub.status.idle":"2022-03-25T12:37:55.135397Z","shell.execute_reply.started":"2022-03-25T12:37:54.997593Z","shell.execute_reply":"2022-03-25T12:37:55.134551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reduced multicollinearity from articles data\nSeveral columns are [category_text, encoded_value]. Thus, in order to eliminate multicollinearity, only one column will be included in each couple\nFollowing are a couple of columns in item features, and we will keep one of them:\n\n* use product_type_no - skip product_type_name\n* use graphical_appearance_no - skip graphical_appearance_name\n* use colour_group_code - skip colour_group_name\n* use perceived_colour_value_id - skip perceived_colour_value_name\n* use perceived_colour_master_id - skip perceived_colour_master_name\n* use index_code - skip index_name\n* use index_group_no - skip index_group_name\n* use section_no - skip section_name\n* use garment_group_no - skip garment_group_name\n* use product_code, skip product_name\n* use department_no, skip department_name","metadata":{}},{"cell_type":"code","source":"df_ar = df_ar.drop(columns = ['product_type_name', 'graphical_appearance_name', 'colour_group_name', 'perceived_colour_value_name',\n                        'perceived_colour_master_name', 'index_name', 'index_group_name', 'section_name', \n                        'garment_group_name', 'prod_name', 'department_name', 'detail_desc'])\ndf_ar.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:37:56.606689Z","iopub.execute_input":"2022-03-25T12:37:56.606994Z","iopub.status.idle":"2022-03-25T12:37:56.625524Z","shell.execute_reply.started":"2022-03-25T12:37:56.60694Z","shell.execute_reply":"2022-03-25T12:37:56.624641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /kaggle/working/recbox_data\ndf_ar.to_csv(r'/kaggle/working/recbox_data/recbox_data.item', index=False, sep='\\t')","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:38:03.037352Z","iopub.execute_input":"2022-03-25T12:38:03.037613Z","iopub.status.idle":"2022-03-25T12:38:04.353467Z","shell.execute_reply.started":"2022-03-25T12:38:03.037586Z","shell.execute_reply":"2022-03-25T12:38:04.352475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = df[df['timestamp'] < 1585620000][['customer_id', 'article_id', 'timestamp']].rename(\n    columns={'customer_id': 'user_id:token', 'article_id': 'item_id:token', 'timestamp': 'timestamp:float'})\ntemp","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:38:04.35531Z","iopub.execute_input":"2022-03-25T12:38:04.355674Z","iopub.status.idle":"2022-03-25T12:38:04.457578Z","shell.execute_reply.started":"2022-03-25T12:38:04.355626Z","shell.execute_reply":"2022-03-25T12:38:04.456665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('/kaggle/working/recbox_data/recbox_data.inter', index=False, sep='\\t')\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:38:04.529818Z","iopub.execute_input":"2022-03-25T12:38:04.530298Z","iopub.status.idle":"2022-03-25T12:38:06.615611Z","shell.execute_reply.started":"2022-03-25T12:38:04.530251Z","shell.execute_reply":"2022-03-25T12:38:06.614815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:38:06.617439Z","iopub.execute_input":"2022-03-25T12:38:06.618178Z","iopub.status.idle":"2022-03-25T12:38:06.621815Z","shell.execute_reply.started":"2022-03-25T12:38:06.618144Z","shell.execute_reply":"2022-03-25T12:38:06.621215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub0 = pd.read_csv('../input/hm-pre-recommendation/submissio_byfone_chris.csv').sort_values('customer_id').reset_index(drop=True)\nsub1 = pd.read_csv('../input/hm-pre-recommendation/submission_trending.csv').sort_values('customer_id').reset_index(drop=True)\nsub2 = pd.read_csv('../input/hm-pre-recommendation/submission_exponential_decay.csv').sort_values('customer_id').reset_index(drop=True)\n\nsub0.shape, sub1.shape, sub2.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:38:06.623345Z","iopub.execute_input":"2022-03-25T12:38:06.623642Z","iopub.status.idle":"2022-03-25T12:38:37.30979Z","shell.execute_reply.started":"2022-03-25T12:38:06.623605Z","shell.execute_reply":"2022-03-25T12:38:37.304244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring PySpark","metadata":{}},{"cell_type":"code","source":"!pip install pyspark","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:01:54.40237Z","iopub.execute_input":"2022-03-25T14:01:54.403012Z","iopub.status.idle":"2022-03-25T14:02:39.926412Z","shell.execute_reply.started":"2022-03-25T14:01:54.402965Z","shell.execute_reply":"2022-03-25T14:02:39.925635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyspark\nfrom pyspark.sql import SparkSession\nfrom pyspark.sql.types import StructType,StructField, StringType, IntegerType \nfrom pyspark.sql.types import ArrayType, DoubleType, BooleanType\nfrom pyspark.sql.functions import col,array_contains\nfrom pyspark.sql import SQLContext \nfrom pyspark.ml.recommendation import ALS\nfrom pyspark.sql.functions import udf,col,when\nfrom pyspark.sql.functions import to_timestamp,date_format\nimport numpy as np\nimport pandas as pd\nfrom pyspark.sql.types import *\nfrom pyspark.sql.functions import *\nfrom pyspark.sql.window import *\n\nsc = SparkSession.builder.appName(\"Recommendations\").config(\"spark.sql.files.maxPartitionBytes\", 5000000).getOrCreate()\nspark = SparkSession(sc)","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:02:39.928321Z","iopub.execute_input":"2022-03-25T14:02:39.928581Z","iopub.status.idle":"2022-03-25T14:02:46.631643Z","shell.execute_reply.started":"2022-03-25T14:02:39.92855Z","shell.execute_reply":"2022-03-25T14:02:46.630796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction = spark.read.option(\"header\",True) \\\n              .csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\ntransaction.printSchema()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:02:46.633165Z","iopub.execute_input":"2022-03-25T14:02:46.634125Z","iopub.status.idle":"2022-03-25T14:02:51.556383Z","shell.execute_reply.started":"2022-03-25T14:02:46.634076Z","shell.execute_reply":"2022-03-25T14:02:51.555529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article = spark.read.option(\"header\",True) \\\n              .csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\narticle.printSchema()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:02:51.55865Z","iopub.execute_input":"2022-03-25T14:02:51.558948Z","iopub.status.idle":"2022-03-25T14:02:51.981215Z","shell.execute_reply.started":"2022-03-25T14:02:51.558907Z","shell.execute_reply":"2022-03-25T14:02:51.980397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Duration of transaction","metadata":{}},{"cell_type":"code","source":"from pyspark.sql.functions import min, max\nfrom pyspark.sql.functions import unix_timestamp, lit\nmin_date, max_date = transaction.select(min(\"t_dat\"), max(\"t_dat\")).first()\nmin_date, max_date","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:02:51.982394Z","iopub.execute_input":"2022-03-25T14:02:51.983389Z","iopub.status.idle":"2022-03-25T14:03:29.927061Z","shell.execute_reply.started":"2022-03-25T14:02:51.983337Z","shell.execute_reply":"2022-03-25T14:03:29.926207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this transaction dataset we have 31,788,324 rows and 5 columns.Let's capture first what are the most recently bought articles.For recommendation I am selecting only date 2020-09-22 which is the last transaction date.","metadata":{}},{"cell_type":"code","source":"hm =  transaction.withColumn('t_dat', transaction['t_dat'].cast('string'))\nhm = hm.withColumn('date', from_unixtime(unix_timestamp('t_dat', 'yyyy-MM-dd')))\nhm = hm.withColumn('year', year(col('date')))\nhm = hm.withColumn('month', month(col('date')))\nhm = hm.withColumn('day', date_format(col('date'), \"d\"))\n\nhm = hm[hm['year'] == 2020]\nhm = hm[hm['month'] == 9]\nhm = hm[hm['day'] == 22]\ntransaction.unpersist()\n\n# Prepare the dataset\nhm = hm.groupby('customer_id', 'article_id').count()\nhm.show(5)","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:03:29.931116Z","iopub.execute_input":"2022-03-25T14:03:29.933478Z","iopub.status.idle":"2022-03-25T14:04:41.458421Z","shell.execute_reply.started":"2022-03-25T14:03:29.933423Z","shell.execute_reply":"2022-03-25T14:04:41.457554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hm =  transaction.withColumn('t_dat', transaction['t_dat'].cast('string'))\nhm = hm.withColumn('date', from_unixtime(unix_timestamp('t_dat', 'yyyy-MM-dd')))\nhm = hm.withColumn('year', year(col('date')))\nhm = hm.withColumn('month', month(col('date')))\nhm = hm.withColumn('day', date_format(col('date'), \"d\"))\n\ntransaction.unpersist()\n\n# Prepare the dataset\nhm = hm.groupby('customer_id', 'article_id').count()\nhm.show(5)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print((hm.count(), len(hm.columns)))","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:04:41.459917Z","iopub.execute_input":"2022-03-25T14:04:41.460217Z","iopub.status.idle":"2022-03-25T14:05:48.032422Z","shell.execute_reply.started":"2022-03-25T14:04:41.460177Z","shell.execute_reply":"2022-03-25T14:05:48.0315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count the total number of article count in the dataset\nnumerator = hm.select(\"count\").count()\n\n# Count the number of distinct customerid and distinct articleid\nnum_users = hm.select(\"customer_id\").distinct().count()\nnum_articles = hm.select(\"article_id\").distinct().count()\n\n# Set the denominator equal to the number of customer multiplied by the number of articles\ndenominator = num_users * num_articles\n\n# Divide the numerator by the denominator\nsparsity = (1.0 - (numerator *1.0)/denominator)*100\nprint(\"Sparsity: \", \"%.2f\" % sparsity + \"%.\")","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:05:48.033707Z","iopub.execute_input":"2022-03-25T14:05:48.033995Z","iopub.status.idle":"2022-03-25T14:09:02.641185Z","shell.execute_reply.started":"2022-03-25T14:05:48.033955Z","shell.execute_reply":"2022-03-25T14:09:02.640171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"userId_count = hm.groupBy(\"customer_id\").count().orderBy('count', ascending=False)\nuserId_count.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:09:02.643481Z","iopub.execute_input":"2022-03-25T14:09:02.643779Z","iopub.status.idle":"2022-03-25T14:10:06.53904Z","shell.execute_reply.started":"2022-03-25T14:09:02.643739Z","shell.execute_reply":"2022-03-25T14:10:06.538082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articleId_count = hm.groupBy(\"article_id\").count().orderBy('count', ascending=False)\narticleId_count.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:10:06.543666Z","iopub.execute_input":"2022-03-25T14:10:06.544335Z","iopub.status.idle":"2022-03-25T14:11:12.508273Z","shell.execute_reply.started":"2022-03-25T14:10:06.544284Z","shell.execute_reply":"2022-03-25T14:11:12.50732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.ml.evaluation import RegressionEvaluator\nfrom pyspark.ml.recommendation import ALS","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:11:12.509794Z","iopub.execute_input":"2022-03-25T14:11:12.510106Z","iopub.status.idle":"2022-03-25T14:11:12.516654Z","shell.execute_reply.started":"2022-03-25T14:11:12.510065Z","shell.execute_reply":"2022-03-25T14:11:12.515777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.ml.feature import StringIndexer\nfrom pyspark.ml import Pipeline\nfrom pyspark.sql.functions import col\nindexer = [StringIndexer(inputCol=column, outputCol=column+\"_index\") for column in list(set(hm.columns)-set(['count'])) ]\npipeline = Pipeline(stages=indexer)\ntransformed = pipeline.fit(hm).transform(hm)\ntransformed.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:11:12.518426Z","iopub.execute_input":"2022-03-25T14:11:12.518931Z","iopub.status.idle":"2022-03-25T14:14:29.012748Z","shell.execute_reply.started":"2022-03-25T14:11:12.5189Z","shell.execute_reply":"2022-03-25T14:14:29.010899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(training,test)=transformed.randomSplit([0.8, 0.2])","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:14:29.01506Z","iopub.execute_input":"2022-03-25T14:14:29.015373Z","iopub.status.idle":"2022-03-25T14:14:29.059336Z","shell.execute_reply.started":"2022-03-25T14:14:29.015334Z","shell.execute_reply":"2022-03-25T14:14:29.058421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To build the model explicitly specify the columns. Set nonnegative as ‘True’, since we are looking count greater than 0. The model also gives an option to select implicit ratings. Since we are working with explicit, set it to ‘False’ or by default it takes explicit.\n\nWhen using simple random splits as in Spark’s CrossValidator or TrainValidationSplit, it is actually very common to encounter users and/or items in the evaluation set that are not in the training set. By default, Spark assigns NaN predictions during ALSModel.transform when a user and/or item factor is not present in the model.We set cold start strategy to ‘drop’ to ensure we don’t get NaN evaluation metrics.","metadata":{}},{"cell_type":"code","source":"als=ALS(maxIter=5,regParam=0.09,rank=25,userCol=\"customer_id_index\",itemCol=\"article_id_index\",ratingCol=\"count\",coldStartStrategy=\"drop\",nonnegative=True)\nmodel=als.fit(training)","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:14:29.060516Z","iopub.execute_input":"2022-03-25T14:14:29.065639Z","iopub.status.idle":"2022-03-25T14:16:45.417822Z","shell.execute_reply.started":"2022-03-25T14:14:29.065591Z","shell.execute_reply":"2022-03-25T14:16:45.416859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluator=RegressionEvaluator(metricName=\"rmse\",labelCol=\"count\",predictionCol=\"prediction\")\npredictions=model.transform(test)\nrmse=evaluator.evaluate(predictions)","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:16:45.419539Z","iopub.execute_input":"2022-03-25T14:16:45.420411Z","iopub.status.idle":"2022-03-25T14:17:48.985074Z","shell.execute_reply.started":"2022-03-25T14:16:45.420365Z","shell.execute_reply":"2022-03-25T14:17:48.984209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"RMSE=\"+str(rmse))","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:17:48.986292Z","iopub.execute_input":"2022-03-25T14:17:48.986593Z","iopub.status.idle":"2022-03-25T14:17:48.994163Z","shell.execute_reply.started":"2022-03-25T14:17:48.986556Z","shell.execute_reply":"2022-03-25T14:17:48.993129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:17:48.996056Z","iopub.execute_input":"2022-03-25T14:17:48.996621Z","iopub.status.idle":"2022-03-25T14:18:52.437351Z","shell.execute_reply.started":"2022-03-25T14:17:48.996579Z","shell.execute_reply":"2022-03-25T14:18:52.436493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_recs=model.recommendForAllItems(10).show(10)","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:18:52.438742Z","iopub.execute_input":"2022-03-25T14:18:52.439063Z","iopub.status.idle":"2022-03-25T14:19:01.964592Z","shell.execute_reply.started":"2022-03-25T14:18:52.439022Z","shell.execute_reply":"2022-03-25T14:19:01.963774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_recs=model.recommendForAllUsers(10).show(10)","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:19:01.965771Z","iopub.execute_input":"2022-03-25T14:19:01.966104Z","iopub.status.idle":"2022-03-25T14:19:08.383259Z","shell.execute_reply.started":"2022-03-25T14:19:01.966062Z","shell.execute_reply":"2022-03-25T14:19:08.382417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nuserRecsDf = model.recommendForAllUsers(10).cache()\nuserRecsDf.count()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:58:08.525007Z","iopub.execute_input":"2022-03-25T12:58:08.525275Z","iopub.status.idle":"2022-03-25T12:58:19.338767Z","shell.execute_reply.started":"2022-03-25T12:58:08.525247Z","shell.execute_reply":"2022-03-25T12:58:19.337939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"userRecsDf.printSchema()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:58:19.340309Z","iopub.execute_input":"2022-03-25T12:58:19.340585Z","iopub.status.idle":"2022-03-25T12:58:19.348457Z","shell.execute_reply.started":"2022-03-25T12:58:19.340546Z","shell.execute_reply":"2022-03-25T12:58:19.347855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"userRecsDf.select(\"customer_id_index\",\"recommendations.article_id_index\").show(10,False)","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:58:19.349669Z","iopub.execute_input":"2022-03-25T12:58:19.350114Z","iopub.status.idle":"2022-03-25T12:58:19.514933Z","shell.execute_reply.started":"2022-03-25T12:58:19.350073Z","shell.execute_reply":"2022-03-25T12:58:19.514117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc #This is to free up the memory\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T12:58:19.516156Z","iopub.execute_input":"2022-03-25T12:58:19.516429Z","iopub.status.idle":"2022-03-25T12:58:19.697879Z","shell.execute_reply.started":"2022-03-25T12:58:19.51639Z","shell.execute_reply":"2022-03-25T12:58:19.697073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As seen in above image the results are in integer form we need to convert it back to its original name.The code is little bit longer given so many conversions","metadata":{}},{"cell_type":"code","source":"from pyspark.ml.evaluation import RegressionEvaluator\nfrom pyspark.ml.recommendation import ALS\nfrom pyspark.sql import Row\nimport pandas as pd\nrecs=model.recommendForAllUsers(10).toPandas()\nnrecs=recs.recommendations.apply(pd.Series) \\\n            .merge(recs, right_index = True, left_index = True) \\\n            .drop([\"recommendations\"], axis = 1) \\\n            .melt(id_vars = ['customer_id_index'], value_name = \"recommendations\") \\\n            .drop(\"variable\", axis = 1) \\\n            .dropna() \nnrecs=nrecs.sort_values('customer_id_index')\nnrecs=pd.concat([nrecs['recommendations'].apply(pd.Series), nrecs['customer_id_index']], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:20:23.815517Z","iopub.execute_input":"2022-03-25T14:20:23.815855Z","iopub.status.idle":"2022-03-25T14:20:56.009155Z","shell.execute_reply.started":"2022-03-25T14:20:23.815819Z","shell.execute_reply":"2022-03-25T14:20:56.008494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nrecs.columns = ['ArticleID_index','count','UserID_index']\nmd=transformed.select(transformed['article_id'],transformed['article_id_index'],transformed['customer_id'],transformed['customer_id_index'])\nmd=md.toPandas()","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:22:09.391167Z","iopub.execute_input":"2022-03-25T14:22:09.392401Z","iopub.status.idle":"2022-03-25T14:23:13.770233Z","shell.execute_reply.started":"2022-03-25T14:22:09.392343Z","shell.execute_reply":"2022-03-25T14:23:13.769407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict1 =dict(zip(md['article_id_index'],md['article_id']))\ndict2=dict(zip(md['customer_id_index'],md['customer_id']))\nnrecs['article_id']=nrecs['ArticleID_index'].map(dict1)\nnrecs['customer_id']=nrecs['UserID_index'].map(dict2)","metadata":{"execution":{"iopub.status.busy":"2022-03-25T14:23:30.15588Z","iopub.execute_input":"2022-03-25T14:23:30.156186Z","iopub.status.idle":"2022-03-25T14:23:30.211257Z","shell.execute_reply.started":"2022-03-25T14:23:30.156152Z","shell.execute_reply":"2022-03-25T14:23:30.210611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}