{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Building recommender system using content based filtering approach**\n\nThe following code:\n\n* Builds weighted one-hot endcoded item embeddings in the feature space\n* Projecting customers into the embeddings space\n* Performing dimensionality reduction usng PCA and picking the first 150 principal component\n* Finds N similar items using ApproximateNearestKneighbor from spark MLLib\n\nCold start approach: recommend most frequent items.\n\nInput data limited to 10000 transactions due to memory constraints","metadata":{}},{"cell_type":"code","source":"!pip install pyspark","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:47:26.31038Z","iopub.execute_input":"2022-05-06T13:47:26.311397Z","iopub.status.idle":"2022-05-06T13:48:27.76427Z","shell.execute_reply.started":"2022-05-06T13:47:26.311338Z","shell.execute_reply":"2022-05-06T13:48:27.763149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.ml.evaluation import RegressionEvaluator\nfrom pyspark.ml.recommendation import ALS\nfrom pyspark.sql import Row\nfrom pyspark.sql import SparkSession\nfrom pyspark.sql import Row\nfrom pyspark.sql.functions import col, lit, lower\nfrom pyspark.ml.feature import BucketedRandomProjectionLSH\n\nfeatures = ['article_id', 'prod_name', 'product_type_name',\n       'product_group_name', \n       'graphical_appearance_name', 'colour_group_name',\n       'perceived_colour_value_name',\n       'perceived_colour_master_name',\n       'department_name', 'index_name',\n       'index_group_name', 'section_name',\n       'garment_group_name', 'detail_desc']\n\npivot_cols = ['product_group_name', \n       'graphical_appearance_name', 'colour_group_name',\n       'perceived_colour_value_name',\n       'perceived_colour_master_name',\n       'department_name', 'index_name',\n       'index_group_name', 'section_name',\n       'garment_group_name']\n\nspark = SparkSession.builder.appName('Recommendations').getOrCreate()\n\ntransactions = spark.read.options(header=True).csv(\n    \"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\").drop(\n    'sales_channel_id').drop('price').limit(10000)\n    \n\nitems = spark.read.options(header=True).csv(\n    \"../input/h-and-m-personalized-fashion-recommendations/articles.csv\").select(features)\n\nrcmnds = spark.read.options(header=True).csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv'\n                       ).select('customer_id')","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:27.768184Z","iopub.execute_input":"2022-05-06T13:48:27.768721Z","iopub.status.idle":"2022-05-06T13:48:42.148402Z","shell.execute_reply.started":"2022-05-06T13:48:27.768661Z","shell.execute_reply":"2022-05-06T13:48:42.147381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:42.150035Z","iopub.execute_input":"2022-05-06T13:48:42.150331Z","iopub.status.idle":"2022-05-06T13:48:42.201699Z","shell.execute_reply.started":"2022-05-06T13:48:42.150287Z","shell.execute_reply":"2022-05-06T13:48:42.200712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"items","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:42.205015Z","iopub.execute_input":"2022-05-06T13:48:42.206559Z","iopub.status.idle":"2022-05-06T13:48:42.261488Z","shell.execute_reply.started":"2022-05-06T13:48:42.206479Z","shell.execute_reply":"2022-05-06T13:48:42.259939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rcmnds","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:42.267892Z","iopub.execute_input":"2022-05-06T13:48:42.269859Z","iopub.status.idle":"2022-05-06T13:48:42.287901Z","shell.execute_reply.started":"2022-05-06T13:48:42.269763Z","shell.execute_reply":"2022-05-06T13:48:42.287192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_lower(items):\n    for c in pivot_cols:\n        items = items.withColumn(c, lower(col(c)))\n    \n    return items","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:42.292334Z","iopub.execute_input":"2022-05-06T13:48:42.293781Z","iopub.status.idle":"2022-05-06T13:48:42.299411Z","shell.execute_reply.started":"2022-05-06T13:48:42.293723Z","shell.execute_reply":"2022-05-06T13:48:42.298597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ohe(items):\n    keys = ['article_id']\n    def join_all(dfs,keys):\n        if len(dfs) > 1:\n            return dfs[0].join(join_all(dfs[1:],keys), on = keys, how = 'inner')\n        else:\n            return dfs[0]\n\n    dfs = []\n    combined = []\n    for pivot_col in pivot_cols:\n        pivotDF = items.groupBy(keys).pivot(pivot_col).count()\n        new_names = pivotDF.columns[:len(keys)] +  [\"e_{0}_{1}\".format(pivot_col, i) for i, c in enumerate(pivotDF.columns[len(keys):])]        \n        newdf = pivotDF.toDF(*new_names).fillna(0)    \n        combined.append(newdf)\n\n    item_feature = join_all(combined,keys)\n    \n    return item_feature","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:42.301595Z","iopub.execute_input":"2022-05-06T13:48:42.302659Z","iopub.status.idle":"2022-05-06T13:48:42.314758Z","shell.execute_reply.started":"2022-05-06T13:48:42.302609Z","shell.execute_reply":"2022-05-06T13:48:42.313939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"items = to_lower(items)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:42.316329Z","iopub.execute_input":"2022-05-06T13:48:42.317319Z","iopub.status.idle":"2022-05-06T13:48:42.605135Z","shell.execute_reply.started":"2022-05-06T13:48:42.317276Z","shell.execute_reply":"2022-05-06T13:48:42.60404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_feature = ohe(items)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:42.60674Z","iopub.execute_input":"2022-05-06T13:48:42.607067Z","iopub.status.idle":"2022-05-06T13:48:56.597239Z","shell.execute_reply.started":"2022-05-06T13:48:42.607022Z","shell.execute_reply":"2022-05-06T13:48:56.59658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions = transactions.join(item_feature, on='article_id', how='left').sort('t_dat').drop(*features[1:])","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:56.599617Z","iopub.execute_input":"2022-05-06T13:48:56.599933Z","iopub.status.idle":"2022-05-06T13:48:56.829619Z","shell.execute_reply.started":"2022-05-06T13:48:56.599893Z","shell.execute_reply":"2022-05-06T13:48:56.828721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummy_features = transactions.columns[3:]","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:56.830803Z","iopub.execute_input":"2022-05-06T13:48:56.83109Z","iopub.status.idle":"2022-05-06T13:48:57.857122Z","shell.execute_reply.started":"2022-05-06T13:48:56.831053Z","shell.execute_reply":"2022-05-06T13:48:57.856432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_feature = transactions.groupBy('customer_id').sum(*dummy_features)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:57.858807Z","iopub.execute_input":"2022-05-06T13:48:57.859468Z","iopub.status.idle":"2022-05-06T13:48:58.230571Z","shell.execute_reply.started":"2022-05-06T13:48:57.859419Z","shell.execute_reply":"2022-05-06T13:48:58.22962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from pyspark.ml.feature import VectorAssembler, StandardScaler, PCA\n\n# def get_pca(df, col):\n    \n    \n#     assembler = VectorAssembler(inputCols=df.columns[1:], outputCol=\"sparse_features\")\n    \n#     feature_vectors = assembler.transform(df).select(*(col, \"sparse_features\"))\n\n\n#     scaler = StandardScaler(inputCol=\"sparse_features\", outputCol=\"scaled_features\")\n#     scalerModel = scaler.fit(feature_vectors)\n\n#     scaled_feature_vectors = scalerModel.transform(feature_vectors).select(*(col, \"scaled_features\"))\n\n#     pca = PCA(k=100, inputCol=\"scaled_features\", outputCol=\"pca\")\n#     pcaModel = pca.fit(scaled_feature_vectors)\n#     x = pcaModel.transform(scaled_feature_vectors).select(*(col, \"pca\"))\n    \n#     return x\n\n\n# user_feature_pca = get_pca(weighted_user_feature, 'customer_id')\n# item_feature_pca = get_pca(item_feature, 'article_id')","metadata":{"execution":{"iopub.status.busy":"2022-05-02T15:25:33.234624Z","iopub.execute_input":"2022-05-02T15:25:33.235273Z","iopub.status.idle":"2022-05-02T15:25:33.240053Z","shell.execute_reply.started":"2022-05-02T15:25:33.235227Z","shell.execute_reply":"2022-05-02T15:25:33.239342Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.ml.feature import VectorAssembler, StandardScaler, PCA\nfrom pyspark.ml import Pipeline\n\ndef scale(df, col):\n    \n    assembler = VectorAssembler(inputCols=df.columns[1:], outputCol=\"sparse_features\")\n    feature_vectors = assembler.transform(df).select(*(col, \"sparse_features\"))\n\n    scaler = StandardScaler(inputCol=\"sparse_features\", outputCol=\"scaled_features\")\n    scalerModel = scaler.fit(feature_vectors)\n    \n    scaled_feature_vectors = scalerModel.transform(feature_vectors).select(*(col, \"scaled_features\"))\n    \n    return scaled_feature_vectors\n\n\ndef get_pca(df, col):\n    \n    pca = PCA(k=100, inputCol=\"scaled_features\", outputCol=\"pca\")\n    pcaModel = pca.fit(df)\n    \n    return pcaModel","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:58.300112Z","iopub.execute_input":"2022-05-06T13:48:58.300451Z","iopub.status.idle":"2022-05-06T13:48:58.309004Z","shell.execute_reply.started":"2022-05-06T13:48:58.300406Z","shell.execute_reply":"2022-05-06T13:48:58.307989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaled_user_feature = scale(user_feature, 'customer_id')\nscaled_item_feature = scale(item_feature, 'article_id')","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:48:58.311094Z","iopub.execute_input":"2022-05-06T13:48:58.311771Z","iopub.status.idle":"2022-05-06T13:51:27.337483Z","shell.execute_reply.started":"2022-05-06T13:48:58.31172Z","shell.execute_reply":"2022-05-06T13:51:27.336572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca_model = get_pca(scaled_user_feature, 'customer_id')","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:52:58.514632Z","iopub.execute_input":"2022-05-06T13:52:58.514995Z","iopub.status.idle":"2022-05-06T13:54:22.049328Z","shell.execute_reply.started":"2022-05-06T13:52:58.514951Z","shell.execute_reply":"2022-05-06T13:54:22.048048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_feature_pca = pca_model.transform(scaled_user_feature)\nitem_feature_pca = pca_model.transform(scaled_item_feature)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:54:22.051603Z","iopub.execute_input":"2022-05-06T13:54:22.051946Z","iopub.status.idle":"2022-05-06T13:54:22.25436Z","shell.execute_reply.started":"2022-05-06T13:54:22.051904Z","shell.execute_reply":"2022-05-06T13:54:22.253413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_feature_pca","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:54:22.255655Z","iopub.execute_input":"2022-05-06T13:54:22.256601Z","iopub.status.idle":"2022-05-06T13:54:22.269282Z","shell.execute_reply.started":"2022-05-06T13:54:22.256546Z","shell.execute_reply":"2022-05-06T13:54:22.26843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_feature_pca","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:54:22.270768Z","iopub.execute_input":"2022-05-06T13:54:22.271375Z","iopub.status.idle":"2022-05-06T13:54:22.281689Z","shell.execute_reply.started":"2022-05-06T13:54:22.271344Z","shell.execute_reply":"2022-05-06T13:54:22.28068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# user_feature_pca.write.parquet(\"./user_feature_pca.parquet\")\n# item_feature_pca.write.parquet(\"./item_feature_pca.parquet\")","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:54:22.283471Z","iopub.execute_input":"2022-05-06T13:54:22.284236Z","iopub.status.idle":"2022-05-06T13:54:22.290157Z","shell.execute_reply.started":"2022-05-06T13:54:22.284187Z","shell.execute_reply":"2022-05-06T13:54:22.289353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.ml.feature import BucketedRandomProjectionLSH\nfrom pyspark.sql.functions import col, udf\nimport pyspark.sql.functions as F\n\ndef get_rcmnds(customer, k=12):\n    brp = BucketedRandomProjectionLSH(inputCol=\"pca\", outputCol=\"hashes\", seed=12345, bucketLength=1.0)\n    model = brp.fit(user_feature_pca)\n    temp = model.approxNearestNeighbors(item_feature_pca, customer.pca, k).select('article_id').collect()\n    return temp","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:54:22.291882Z","iopub.execute_input":"2022-05-06T13:54:22.292447Z","iopub.status.idle":"2022-05-06T13:54:22.305073Z","shell.execute_reply.started":"2022-05-06T13:54:22.292405Z","shell.execute_reply":"2022-05-06T13:54:22.304193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"flagged = rcmnds.join(user_feature_pca.withColumn('flag', F.lit(True)), 'customer_id', 'left').fillna(False)\n\ncold_start = flagged.where('!flag').drop('flag')\nwith_history = flagged.where('flag').drop('flag')","metadata":{"execution":{"iopub.status.busy":"2022-05-06T14:03:41.091455Z","iopub.execute_input":"2022-05-06T14:03:41.091818Z","iopub.status.idle":"2022-05-06T14:03:41.435776Z","shell.execute_reply.started":"2022-05-06T14:03:41.091785Z","shell.execute_reply":"2022-05-06T14:03:41.434617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rows = with_history.collect()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T14:10:22.537347Z","iopub.execute_input":"2022-05-06T14:10:22.537696Z","iopub.status.idle":"2022-05-06T14:11:51.184684Z","shell.execute_reply.started":"2022-05-06T14:10:22.537661Z","shell.execute_reply":"2022-05-06T14:11:51.183756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rows[0]","metadata":{"execution":{"iopub.status.busy":"2022-05-06T14:11:51.189588Z","iopub.execute_input":"2022-05-06T14:11:51.19071Z","iopub.status.idle":"2022-05-06T14:11:51.212671Z","shell.execute_reply.started":"2022-05-06T14:11:51.190655Z","shell.execute_reply":"2022-05-06T14:11:51.211595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers = []\nitems = []\nfor row in rows:\n    temp = get_rcmnds(row)\n    customers.append(row[0])\n    items.append(' '.join([i[0] for i in temp]))","metadata":{"execution":{"iopub.status.busy":"2022-05-06T14:23:35.213615Z","iopub.execute_input":"2022-05-06T14:23:35.213921Z","iopub.status.idle":"2022-05-06T14:24:21.694957Z","shell.execute_reply.started":"2022-05-06T14:23:35.213891Z","shell.execute_reply":"2022-05-06T14:24:21.69373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nwith_history_df = pd.DataFrame({'customer_id':customers, 'items':items})","metadata":{"execution":{"iopub.status.busy":"2022-05-06T14:32:10.116962Z","iopub.execute_input":"2022-05-06T14:32:10.117268Z","iopub.status.idle":"2022-05-06T14:32:10.122976Z","shell.execute_reply.started":"2022-05-06T14:32:10.117236Z","shell.execute_reply":"2022-05-06T14:32:10.122289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_freq = transactions.groupBy('article_id').count().sort(col('count').desc()).limit(12).collect()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T14:05:34.887681Z","iopub.execute_input":"2022-05-06T14:05:34.888006Z","iopub.status.idle":"2022-05-06T14:05:39.266128Z","shell.execute_reply.started":"2022-05-06T14:05:34.887962Z","shell.execute_reply":"2022-05-06T14:05:39.265114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"default = [i[0] for i in most_freq]","metadata":{"execution":{"iopub.status.busy":"2022-05-06T13:59:53.881148Z","iopub.execute_input":"2022-05-06T13:59:53.88148Z","iopub.status.idle":"2022-05-06T13:59:53.886275Z","shell.execute_reply.started":"2022-05-06T13:59:53.881448Z","shell.execute_reply":"2022-05-06T13:59:53.885486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cold_start = cold_start.withColumn('items', lit(' '.join(default))).select(*('customer_id', 'items')).toPandas()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T14:31:27.939883Z","iopub.execute_input":"2022-05-06T14:31:27.940171Z","iopub.status.idle":"2022-05-06T14:31:46.88833Z","shell.execute_reply.started":"2022-05-06T14:31:27.940142Z","shell.execute_reply":"2022-05-06T14:31:46.887465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cold_start.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-06T14:31:46.889993Z","iopub.execute_input":"2022-05-06T14:31:46.890296Z","iopub.status.idle":"2022-05-06T14:31:46.901217Z","shell.execute_reply.started":"2022-05-06T14:31:46.890256Z","shell.execute_reply":"2022-05-06T14:31:46.900169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_rcmnds = cold_start.append(with_history_df)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T14:33:46.650226Z","iopub.execute_input":"2022-05-06T14:33:46.65082Z","iopub.status.idle":"2022-05-06T14:33:46.787663Z","shell.execute_reply.started":"2022-05-06T14:33:46.650785Z","shell.execute_reply":"2022-05-06T14:33:46.786471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_rcmnds.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-06T14:33:47.281572Z","iopub.execute_input":"2022-05-06T14:33:47.281882Z","iopub.status.idle":"2022-05-06T14:34:00.365251Z","shell.execute_reply.started":"2022-05-06T14:33:47.281851Z","shell.execute_reply":"2022-05-06T14:34:00.364181Z"},"trusted":true},"execution_count":null,"outputs":[]}]}