{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# H&M RECOMMENDATION SYSTEM","metadata":{}},{"cell_type":"code","source":"# Install Pyspark\n!pip install pyspark","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:28:36.332120Z","iopub.execute_input":"2022-04-28T19:28:36.333077Z","iopub.status.idle":"2022-04-28T19:29:25.793251Z","shell.execute_reply.started":"2022-04-28T19:28:36.332884Z","shell.execute_reply":"2022-04-28T19:29:25.791964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import packages\n\nimport pyspark\nfrom pyspark.sql import SparkSession\nfrom pyspark.sql.types import StructType,StructField, StringType, IntegerType \nfrom pyspark.sql.types import ArrayType, DoubleType, BooleanType\nfrom pyspark.sql.functions import col,array_contains\nfrom pyspark.sql import SQLContext \nfrom pyspark.ml.recommendation import ALS\nfrom pyspark.sql.functions import udf,col,when\nfrom pyspark.sql.functions import to_timestamp,date_format\nfrom pyspark.sql.functions import weekofyear\nimport numpy as np\nimport pandas as pd\nfrom pyspark.sql.types import *\nfrom pyspark.sql.functions import *\nfrom pyspark.sql.window import *","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:29:25.797323Z","iopub.execute_input":"2022-04-28T19:29:25.798022Z","iopub.status.idle":"2022-04-28T19:29:26.258217Z","shell.execute_reply.started":"2022-04-28T19:29:25.797955Z","shell.execute_reply":"2022-04-28T19:29:26.257042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc = SparkSession.builder \\\n    .appName(\"Recommendations\") \\\n    .config(\"spark.sql.files.maxPartitionBytes\", 5000000) \\\n    .getOrCreate()\n\nspark = SparkSession(sc)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:29:26.259631Z","iopub.execute_input":"2022-04-28T19:29:26.259883Z","iopub.status.idle":"2022-04-28T19:29:33.256101Z","shell.execute_reply.started":"2022-04-28T19:29:26.259854Z","shell.execute_reply":"2022-04-28T19:29:33.255225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.sql import Row\nfrom pyspark.sql.types import *\n\nschema = StructType([\n    StructField(\"t_dat\", DateType()),\n    StructField(\"customer_id\", StringType()),\n    StructField(\"article_id\", IntegerType()),\n    StructField(\"price\", DoubleType()),\n    StructField(\"sales_channel_id\", IntegerType())\n])\n\ndataset = spark.read.option(\"header\", True) \\\n    .csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\",\n        schema = schema)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:29:33.259230Z","iopub.execute_input":"2022-04-28T19:29:33.259621Z","iopub.status.idle":"2022-04-28T19:29:36.215970Z","shell.execute_reply.started":"2022-04-28T19:29:33.259569Z","shell.execute_reply":"2022-04-28T19:29:36.215162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.show(5)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:29:36.217282Z","iopub.execute_input":"2022-04-28T19:29:36.217652Z","iopub.status.idle":"2022-04-28T19:29:39.229644Z","shell.execute_reply.started":"2022-04-28T19:29:36.217604Z","shell.execute_reply":"2022-04-28T19:29:39.228552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.printSchema()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:29:39.231101Z","iopub.execute_input":"2022-04-28T19:29:39.231437Z","iopub.status.idle":"2022-04-28T19:29:39.247313Z","shell.execute_reply.started":"2022-04-28T19:29:39.231388Z","shell.execute_reply":"2022-04-28T19:29:39.246081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.sql.functions import min, max\nfrom pyspark.sql.functions import unix_timestamp, lit\nmin_date, max_date = dataset.select(min(\"t_dat\"), max(\"t_dat\")).first()\nmin_date, max_date","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:29:39.249563Z","iopub.execute_input":"2022-04-28T19:29:39.249908Z","iopub.status.idle":"2022-04-28T19:30:22.457355Z","shell.execute_reply.started":"2022-04-28T19:29:39.249863Z","shell.execute_reply":"2022-04-28T19:30:22.456094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just something to note here... There seems to be at least two years worth of data. Maybe we could do TopPop on the same week in 2018, 2019 and 2020?\n\nSomeone mentioned that we should weigh towards 2020?","metadata":{}},{"cell_type":"code","source":"# Create Calendar Weeks\ndataset = dataset.withColumn('week_of_year',weekofyear(dataset.t_dat))\ndataset.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:30:22.459575Z","iopub.execute_input":"2022-04-28T19:30:22.459987Z","iopub.status.idle":"2022-04-28T19:30:22.751473Z","shell.execute_reply.started":"2022-04-28T19:30:22.459895Z","shell.execute_reply":"2022-04-28T19:30:22.750143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select CW 38 (Arbitrary at this point)\nfrom pyspark.sql.functions import col\nfrom pyspark.sql import functions as F\n\nrecommend = dataset \\\n    .filter((F.col('week_of_year') == F.lit('38'))) \\\n    .groupby('article_id').count()\n\nrecommend = recommend \\\n    .withColumn('count', col('count')/3) \\\n    .sort(\"count\", ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:30:22.753043Z","iopub.execute_input":"2022-04-28T19:30:22.753385Z","iopub.status.idle":"2022-04-28T19:30:23.102324Z","shell.execute_reply.started":"2022-04-28T19:30:22.753343Z","shell.execute_reply":"2022-04-28T19:30:23.101542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommend.show(12)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:30:23.106067Z","iopub.execute_input":"2022-04-28T19:30:23.106517Z","iopub.status.idle":"2022-04-28T19:31:15.878020Z","shell.execute_reply.started":"2022-04-28T19:30:23.106341Z","shell.execute_reply":"2022-04-28T19:31:15.876910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommend_df = recommend.drop(col('count')) \\\n    .limit(12) \\\n    .toPandas()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:31:15.879432Z","iopub.execute_input":"2022-04-28T19:31:15.879805Z","iopub.status.idle":"2022-04-28T19:31:59.513221Z","shell.execute_reply.started":"2022-04-28T19:31:15.879757Z","shell.execute_reply":"2022-04-28T19:31:59.512029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommend_df","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:31:59.515268Z","iopub.execute_input":"2022-04-28T19:31:59.515823Z","iopub.status.idle":"2022-04-28T19:31:59.539903Z","shell.execute_reply.started":"2022-04-28T19:31:59.515772Z","shell.execute_reply":"2022-04-28T19:31:59.538546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recommend_array = recommend_df['article_id'] \\\n    .astype(str) \\\n    .astype(int) \\\n    .to_numpy()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:31:59.541966Z","iopub.execute_input":"2022-04-28T19:31:59.542505Z","iopub.status.idle":"2022-04-28T19:31:59.549751Z","shell.execute_reply.started":"2022-04-28T19:31:59.542452Z","shell.execute_reply":"2022-04-28T19:31:59.548503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/customers.csv')","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:31:59.552284Z","iopub.execute_input":"2022-04-28T19:31:59.552680Z","iopub.status.idle":"2022-04-28T19:32:06.341117Z","shell.execute_reply.started":"2022-04-28T19:31:59.552629Z","shell.execute_reply":"2022-04-28T19:32:06.339979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:32:06.342768Z","iopub.execute_input":"2022-04-28T19:32:06.343160Z","iopub.status.idle":"2022-04-28T19:32:06.376511Z","shell.execute_reply.started":"2022-04-28T19:32:06.343030Z","shell.execute_reply":"2022-04-28T19:32:06.375582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = customers[['customer_id']].copy()\nsubmission['y_score'] = submission.apply(lambda x: recommend_array, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:32:06.378420Z","iopub.execute_input":"2022-04-28T19:32:06.378995Z","iopub.status.idle":"2022-04-28T19:32:16.715169Z","shell.execute_reply.started":"2022-04-28T19:32:06.378941Z","shell.execute_reply":"2022-04-28T19:32:16.713986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:32:16.716645Z","iopub.execute_input":"2022-04-28T19:32:16.717219Z","iopub.status.idle":"2022-04-28T19:32:16.737788Z","shell.execute_reply.started":"2022-04-28T19:32:16.717168Z","shell.execute_reply":"2022-04-28T19:32:16.736937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = submission.rename(columns={'customer_id': 'customer_id', 'y_score': 'prediction'})\nsubmission['prediction'] = submission.prediction.apply(lambda x: ' '.join([f'{e:010d}' for e in x]))\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:32:16.739069Z","iopub.execute_input":"2022-04-28T19:32:16.739774Z","iopub.status.idle":"2022-04-28T19:32:31.479575Z","shell.execute_reply.started":"2022-04-28T19:32:16.739734Z","shell.execute_reply":"2022-04-28T19:32:31.478562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:32:31.481299Z","iopub.execute_input":"2022-04-28T19:32:31.482522Z","iopub.status.idle":"2022-04-28T19:32:45.089524Z","shell.execute_reply.started":"2022-04-28T19:32:31.482459Z","shell.execute_reply":"2022-04-28T19:32:45.088498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers = spark.read.option(\"header\", True) \\\n    .csv('../input/h-and-m-personalized-fashion-recommendations/customers.csv')\n\ndataset.createOrReplaceTempView(\"transaction\") # Create temp view\ncustomers.createOrReplaceTempView(\"customer\")\n\ndf_bracketed_customers = spark.sql(\"\"\"\n                                   with age_bracketed_customers as(\n                                   select customer_id,\n                                    CASE\n                                    WHEN age < 20 then 'Under 20'\n                                    WHEN age between 20 and 30 then '20-30'\n                                    WHEN age between 31 and 40 then '30-40'\n                                    WHEN age between 41 and 50 then '40-50'\n                                    WHEN age between 51 and 60 then '50-60'\n                                    ELSE '60+'\n                                    END AS `age bracket`\n                                    from customer\n                                    ),\n                                    recs as (\n                                    select\n                                    article_id\n                                    , `age bracket` as `purchaser_age_bracket`\n                                    , row_number() over (partition by `age bracket` order by count(*) desc) as rank_within_age_bracket\n                                    , count(*) as `purchase count`\n                                    from transaction t\n                                    join age_bracketed_customers a on a.customer_id = t.customer_id\n                                    group by article_id, `age bracket`\n                                    )\n                                    select * from recs\n                                    where rank_within_age_bracket <= 12\n                                    order by purchaser_age_bracket, rank_within_age_bracket asc\n                                   \"\"\")","metadata":{"execution":{"iopub.status.busy":"2022-04-28T19:32:45.091392Z","iopub.execute_input":"2022-04-28T19:32:45.092231Z","iopub.status.idle":"2022-04-28T19:32:46.062767Z","shell.execute_reply.started":"2022-04-28T19:32:45.092186Z","shell.execute_reply":"2022-04-28T19:32:46.061705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd_df_bracket_cust = df_bracketed_customers.toPandas()","metadata":{"execution":{"iopub.status.busy":"2022-04-28T20:06:24.132349Z","iopub.execute_input":"2022-04-28T20:06:24.132794Z","iopub.status.idle":"2022-04-28T20:08:54.021337Z","shell.execute_reply.started":"2022-04-28T20:06:24.132747Z","shell.execute_reply":"2022-04-28T20:08:54.020369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd_df_bracket_cust","metadata":{"execution":{"iopub.status.busy":"2022-04-28T20:23:29.103641Z","iopub.execute_input":"2022-04-28T20:23:29.104389Z","iopub.status.idle":"2022-04-28T20:23:29.125066Z","shell.execute_reply.started":"2022-04-28T20:23:29.104319Z","shell.execute_reply":"2022-04-28T20:23:29.123495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"groups = pd_df_bracket_cust['purchaser_age_bracket'].unique().tolist()\ngroups","metadata":{"execution":{"iopub.status.busy":"2022-04-28T20:11:57.387813Z","iopub.execute_input":"2022-04-28T20:11:57.388216Z","iopub.status.idle":"2022-04-28T20:11:57.395829Z","shell.execute_reply.started":"2022-04-28T20:11:57.388175Z","shell.execute_reply":"2022-04-28T20:11:57.394892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"group_variables = ['twenty_to_thirty', \n                   'thirty_to_forty', \n                   'forty_to_fifty', \n                   'fifty_to_sixty', \n                   'over_sixty', \n                   'under_twenty']","metadata":{"execution":{"iopub.status.busy":"2022-04-28T20:21:00.023606Z","iopub.execute_input":"2022-04-28T20:21:00.024310Z","iopub.status.idle":"2022-04-28T20:21:00.029196Z","shell.execute_reply.started":"2022-04-28T20:21:00.024253Z","shell.execute_reply":"2022-04-28T20:21:00.028124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def array_maker(source_df, targeted_filter):\n    filtered_df = source_df[source_df['purchaser_age_bracket'] == targeted_filter]\n    \n    df_array = filtered_df['article_id'] \\\n    .astype(str) \\\n    .astype(int) \\\n    .to_numpy()\n    \n    return df_array","metadata":{"execution":{"iopub.status.busy":"2022-04-28T20:19:27.108435Z","iopub.execute_input":"2022-04-28T20:19:27.108791Z","iopub.status.idle":"2022-04-28T20:19:27.115282Z","shell.execute_reply.started":"2022-04-28T20:19:27.108759Z","shell.execute_reply":"2022-04-28T20:19:27.113991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = {}\nfor variable in group_variables:\n    for g in groups:\n        variable = array_maker(pd_df_bracket_cust, g)\n        d.update({g : variable})","metadata":{"execution":{"iopub.status.busy":"2022-04-28T20:37:06.855091Z","iopub.execute_input":"2022-04-28T20:37:06.855622Z","iopub.status.idle":"2022-04-28T20:37:06.890725Z","shell.execute_reply.started":"2022-04-28T20:37:06.855567Z","shell.execute_reply":"2022-04-28T20:37:06.889892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d","metadata":{"execution":{"iopub.status.busy":"2022-04-28T20:37:08.934944Z","iopub.execute_input":"2022-04-28T20:37:08.935303Z","iopub.status.idle":"2022-04-28T20:37:08.943019Z","shell.execute_reply.started":"2022-04-28T20:37:08.935266Z","shell.execute_reply":"2022-04-28T20:37:08.942117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}