{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-25T19:23:06.673740Z","iopub.execute_input":"2024-08-25T19:23:06.674141Z","iopub.status.idle":"2024-08-25T19:25:25.166835Z","shell.execute_reply.started":"2024-08-25T19:23:06.674108Z","shell.execute_reply":"2024-08-25T19:25:25.165348Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyspark","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:27:21.608916Z","iopub.execute_input":"2024-08-25T19:27:21.609863Z","iopub.status.idle":"2024-08-25T19:28:16.875816Z","shell.execute_reply.started":"2024-08-25T19:27:21.609825Z","shell.execute_reply":"2024-08-25T19:28:16.874219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyspark\nfrom pyspark.sql import SparkSession\nfrom pyspark.sql import SQLContext \nfrom pyspark.ml.recommendation import ALS\nfrom pyspark.ml.evaluation import RegressionEvaluator\nfrom pyspark.ml.tuning import CrossValidator, ParamGridBuilder\nfrom pyspark.ml.feature import StringIndexer\nfrom pyspark.ml import Pipeline\n\nfrom pyspark.sql.types import *\nfrom pyspark.sql.functions import *\nfrom pyspark.sql.window import *\n\nsc = SparkSession.builder.appName(\"Recommendations\").config(\"spark.sql.files.maxPartitionBytes\", 5000000).getOrCreate()\nspark = SparkSession(sc)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:28:16.878458Z","iopub.execute_input":"2024-08-25T19:28:16.878839Z","iopub.status.idle":"2024-08-25T19:28:23.410064Z","shell.execute_reply.started":"2024-08-25T19:28:16.878802Z","shell.execute_reply":"2024-08-25T19:28:23.408663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:28:38.936389Z","iopub.execute_input":"2024-08-25T19:28:38.938827Z","iopub.status.idle":"2024-08-25T19:28:38.950795Z","shell.execute_reply.started":"2024-08-25T19:28:38.938753Z","shell.execute_reply":"2024-08-25T19:28:38.948163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import dataset\narticles=pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers=pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv\")\n#sample_submission=pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\",nrows=1000)\ntransactions_train=pd.read_csv(\"/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:28:41.961670Z","iopub.execute_input":"2024-08-25T19:28:41.962680Z","iopub.status.idle":"2024-08-25T19:30:26.032820Z","shell.execute_reply.started":"2024-08-25T19:28:41.962641Z","shell.execute_reply":"2024-08-25T19:30:26.031688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.info()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:39:11.554102Z","iopub.execute_input":"2024-08-25T19:39:11.556321Z","iopub.status.idle":"2024-08-25T19:39:11.749148Z","shell.execute_reply.started":"2024-08-25T19:39:11.556244Z","shell.execute_reply":"2024-08-25T19:39:11.747992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:39:14.415790Z","iopub.execute_input":"2024-08-25T19:39:14.416612Z","iopub.status.idle":"2024-08-25T19:39:14.454172Z","shell.execute_reply.started":"2024-08-25T19:39:14.416576Z","shell.execute_reply":"2024-08-25T19:39:14.452927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:39:24.662402Z","iopub.execute_input":"2024-08-25T19:39:24.663486Z","iopub.status.idle":"2024-08-25T19:39:24.674638Z","shell.execute_reply.started":"2024-08-25T19:39:24.663450Z","shell.execute_reply":"2024-08-25T19:39:24.673246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:39:27.805589Z","iopub.execute_input":"2024-08-25T19:39:27.806700Z","iopub.status.idle":"2024-08-25T19:39:27.822184Z","shell.execute_reply.started":"2024-08-25T19:39:27.806653Z","shell.execute_reply":"2024-08-25T19:39:27.820820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transaction = spark.read.option(\"header\",True) \\\n              .csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\ntransaction.printSchema()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:39:37.022796Z","iopub.execute_input":"2024-08-25T19:39:37.023239Z","iopub.status.idle":"2024-08-25T19:39:45.401617Z","shell.execute_reply.started":"2024-08-25T19:39:37.023206Z","shell.execute_reply":"2024-08-25T19:39:45.400397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min_date, max_date = transaction.select(min(\"t_dat\"), max(\"t_dat\")).first()\nmin_date, max_date","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:39:47.732640Z","iopub.execute_input":"2024-08-25T19:39:47.733383Z","iopub.status.idle":"2024-08-25T19:40:25.879326Z","shell.execute_reply.started":"2024-08-25T19:39:47.733351Z","shell.execute_reply":"2024-08-25T19:40:25.878063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# There is total 31,788,324 transactions. Selecting only the latest date purchases as the training data. This reduces to 29,486 transactions.","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hm =  transaction.withColumn('t_dat', transaction['t_dat'].cast('string'))\nhm = hm.withColumn('date', from_unixtime(unix_timestamp('t_dat', 'yyyy-MM-dd')))\nhm = hm.withColumn('year', year(col('date')))\nhm = hm.withColumn('month', month(col('date')))\nhm = hm.withColumn('day', date_format(col('date'), \"d\"))\n\nhm = hm[hm['year'] == 2020]\nhm = hm[hm['month'] == 9]\nhm = hm[hm['day'] == 22]\ntransaction.unpersist()\n\n# count of purchase per customer and per article\nhm = hm.groupby('customer_id', 'article_id').count()\nhm.show(5)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:40:28.825760Z","iopub.execute_input":"2024-08-25T19:40:28.826187Z","iopub.status.idle":"2024-08-25T19:41:48.858197Z","shell.execute_reply.started":"2024-08-25T19:40:28.826154Z","shell.execute_reply":"2024-08-25T19:41:48.856880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count the total number of article in the dataset\nnumerator = hm.select(\"count\").count()\n\n# Count the number of distinct customerid and distinct articleid\nnum_users = hm.select(\"customer_id\").distinct().count()\nnum_articles = hm.select(\"article_id\").distinct().count()\n\n# Set the denominator equal to the number of customer multiplied by the number of articles\ndenominator = num_users * num_articles\n\n# Divide the numerator by the denominator\nsparsity = (1.0 - (numerator *1.0)/denominator)*100\nprint(\"Sparsity: \", \"%.2f\" % sparsity + \"%.\")","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:41:48.861078Z","iopub.execute_input":"2024-08-25T19:41:48.861526Z","iopub.status.idle":"2024-08-25T19:45:35.377496Z","shell.execute_reply.started":"2024-08-25T19:41:48.861486Z","shell.execute_reply":"2024-08-25T19:45:35.376242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count of transactions per customers\nuserId_count = hm.groupBy(\"customer_id\").count().orderBy('count', ascending=False)\nuserId_count.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:45:35.378889Z","iopub.execute_input":"2024-08-25T19:45:35.379359Z","iopub.status.idle":"2024-08-25T19:46:51.896563Z","shell.execute_reply.started":"2024-08-25T19:45:35.379318Z","shell.execute_reply":"2024-08-25T19:46:51.895378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articleId_count = hm.groupBy(\"article_id\").count().orderBy('count', ascending=False)\narticleId_count.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:46:51.899768Z","iopub.execute_input":"2024-08-25T19:46:51.900221Z","iopub.status.idle":"2024-08-25T19:48:08.014248Z","shell.execute_reply.started":"2024-08-25T19:46:51.900180Z","shell.execute_reply":"2024-08-25T19:48:08.012901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ALS only accepts integer values as paratmers. Indexing customer_id and article_id\nindexer = [StringIndexer(inputCol=column, outputCol=column+\"_index\") for column in list(set(hm.columns)-set(['count'])) ]\npipeline = Pipeline(stages=indexer)\ntransformed = pipeline.fit(hm).transform(hm)\ntransformed.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:48:08.015638Z","iopub.execute_input":"2024-08-25T19:48:08.016127Z","iopub.status.idle":"2024-08-25T19:52:01.403206Z","shell.execute_reply.started":"2024-08-25T19:48:08.016083Z","shell.execute_reply":"2024-08-25T19:52:01.402010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(training,test)=transformed.randomSplit([0.8, 0.2])","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:52:01.404579Z","iopub.execute_input":"2024-08-25T19:52:01.404997Z","iopub.status.idle":"2024-08-25T19:52:01.445457Z","shell.execute_reply.started":"2024-08-25T19:52:01.404959Z","shell.execute_reply":"2024-08-25T19:52:01.444149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create ALS model\nals=ALS(userCol=\"customer_id_index\",itemCol=\"article_id_index\",ratingCol=\"count\",coldStartStrategy=\"drop\",nonnegative=True)\n\n#tune model using ParamGridBuilder\nparam_grid = ParamGridBuilder()\\\n            .addGrid(als.rank, [15,20,25])\\\n            .addGrid(als.maxIter,[5,10,15])\\\n            .addGrid(als.regParam,[0.09,0.14,0.19])\\\n            .build()\n#define evaluator as RMSE\nevaluator = RegressionEvaluator(metricName = \"rmse\",labelCol = 'count', predictionCol = 'prediction')\n\n#Build cross validation using CrossValidator\ncv = CrossValidator(estimator=als,estimatorParamMaps=param_grid, evaluator=evaluator,numFolds=3)\n\n\n#Fit ALS model to training data\nmodel = cv.fit(training)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T19:52:33.409271Z","iopub.execute_input":"2024-08-25T19:52:33.409900Z","iopub.status.idle":"2024-08-25T20:23:31.180731Z","shell.execute_reply.started":"2024-08-25T19:52:33.409859Z","shell.execute_reply":"2024-08-25T20:23:31.179475Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Extract best model from the tuning exercise using ParamGridBuilder\nbest_model = model.bestModel\n\n#Generate predictions and evaluate using RMSE\npredictions = best_model.transform(test)\nrmse = evaluator.evaluate(predictions)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T20:23:31.183060Z","iopub.execute_input":"2024-08-25T20:23:31.184048Z","iopub.status.idle":"2024-08-25T20:24:46.559571Z","shell.execute_reply.started":"2024-08-25T20:23:31.184011Z","shell.execute_reply":"2024-08-25T20:24:46.558333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print evaluation metrics and model parameters\nprint(\"RMSE =\" + str(rmse))\nprint(\"**Best Model**\")\nprint(\"Rank : {}\".format(best_model.rank))\nprint(\"MaxIter: {}\".format(best_model._java_obj.parent().getMaxIter()))\nprint(\"RegParam: {}\".format(best_model._java_obj.parent().getRegParam()))","metadata":{"execution":{"iopub.status.busy":"2024-08-25T20:29:13.613832Z","iopub.execute_input":"2024-08-25T20:29:13.614417Z","iopub.status.idle":"2024-08-25T20:29:13.630519Z","shell.execute_reply.started":"2024-08-25T20:29:13.614378Z","shell.execute_reply":"2024-08-25T20:29:13.628644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# providing top 10 user recommendation by article id\nuser_recs=best_model.recommendForAllItems(10)\nuser_recs.show(10)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T20:29:28.426525Z","iopub.execute_input":"2024-08-25T20:29:28.427121Z","iopub.status.idle":"2024-08-25T20:29:38.155934Z","shell.execute_reply.started":"2024-08-25T20:29:28.427081Z","shell.execute_reply":"2024-08-25T20:29:38.154457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# providing top 10 item recommendations by customer id\ndf_recom = best_model.recommendForAllUsers(10)\ndf_recom.show(10)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T20:31:20.183842Z","iopub.execute_input":"2024-08-25T20:31:20.184476Z","iopub.status.idle":"2024-08-25T20:31:26.601065Z","shell.execute_reply.started":"2024-08-25T20:31:20.184433Z","shell.execute_reply":"2024-08-25T20:31:26.599843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recs_for_user = df_recom.where(df_recom.customer_id_index == 1).take(1)","metadata":{"execution":{"iopub.status.busy":"2024-08-25T20:39:08.076126Z","iopub.execute_input":"2024-08-25T20:39:08.076683Z","iopub.status.idle":"2024-08-25T20:39:16.916930Z","shell.execute_reply.started":"2024-08-25T20:39:08.076644Z","shell.execute_reply":"2024-08-25T20:39:16.915708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#def name_retriever(movie_id, movie_title_df):\n#    return movie_title_df.where(movie_title_df.movieId == movie_id).take(1)[0]['title']","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for ranking, (user_id, rating) in enumerate(recs_for_user[0]['recommendations']):\n    #movie_string = name_retriever(movie_id, movie_title_df)\n    print('Recommendation {}: {} | predicted score: {}'.format(ranking+1, 'n/a', rating))","metadata":{"execution":{"iopub.status.busy":"2024-08-25T20:39:54.737010Z","iopub.execute_input":"2024-08-25T20:39:54.737538Z","iopub.status.idle":"2024-08-25T20:39:54.745396Z","shell.execute_reply.started":"2024-08-25T20:39:54.737502Z","shell.execute_reply":"2024-08-25T20:39:54.743976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}