{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-01T05:39:44.524008Z","iopub.execute_input":"2023-01-01T05:39:44.524473Z","iopub.status.idle":"2023-01-01T05:41:27.681414Z","shell.execute_reply.started":"2023-01-01T05:39:44.524384Z","shell.execute_reply":"2023-01-01T05:41:27.680292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyspark -q\nimport pyspark\nfrom pyspark.sql import SparkSession\nfrom pyspark.sql import SQLContext\nfrom pyspark.sql import functions as F\nfrom pyspark.sql import Window\nfrom pyspark.sql.types import StructType,StructField, StringType, IntegerType, ArrayType, DoubleType, BooleanType\n\nsc = SparkSession.builder.appName(\"Recommendations\").config(\"spark.sql.files.maxPartitionBytes\", 5000000).getOrCreate()\nspark = SparkSession(sc)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:01:49.414307Z","iopub.execute_input":"2023-01-01T06:01:49.415698Z","iopub.status.idle":"2023-01-01T06:02:33.476022Z","shell.execute_reply.started":"2023-01-01T06:01:49.415641Z","shell.execute_reply":"2023-01-01T06:02:33.475196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles = spark.read.option(\"header\",True) \\\n                .csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = spark.read.option(\"header\",True) \\\n                .csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = spark.read.option(\"header\",True) \\\n                .csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:08:00.437907Z","iopub.execute_input":"2023-01-01T06:08:00.438334Z","iopub.status.idle":"2023-01-01T06:08:06.473325Z","shell.execute_reply.started":"2023-01-01T06:08:00.438295Z","shell.execute_reply":"2023-01-01T06:08:06.472336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles = articles\\\n    .selectExpr('cast (article_id as int) article_id', 'cast (product_type_no as int) product_type_no', 'cast (graphical_appearance_no as int) graphical_appearance_no',\n                'cast (colour_group_code as int) colour_group_code ','cast (perceived_colour_value_id as int) perceived_colour_value_id',\n                'cast (department_no as int) department_no',  'cast (index_group_no as int) index_group_no',\n                'cast (section_no as int) section_no', 'cast (garment_group_no as int) garment_group_no')\\\n    .dropDuplicates()\n\narticles.show(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:08:17.478083Z","iopub.execute_input":"2023-01-01T06:08:17.478485Z","iopub.status.idle":"2023-01-01T06:08:20.604503Z","shell.execute_reply.started":"2023-01-01T06:08:17.478450Z","shell.execute_reply":"2023-01-01T06:08:20.603285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers = customers\\\n    .fillna({'age': '25'})\\\n    .drop('FN', 'Active', 'club_member_status', 'fashion_news_frequency', 'postal_code')","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:08:48.480366Z","iopub.execute_input":"2023-01-01T06:08:48.480783Z","iopub.status.idle":"2023-01-01T06:08:48.520751Z","shell.execute_reply.started":"2023-01-01T06:08:48.480708Z","shell.execute_reply":"2023-01-01T06:08:48.519567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"start_date = '2020-07-22'\n\nmin_week = 1\nmax_week = 9\napplication_week = max_week + 1\n#week: changed to tuesday\nprint(application_week)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:09:29.778950Z","iopub.execute_input":"2023-01-01T06:09:29.779335Z","iopub.status.idle":"2023-01-01T06:09:29.784928Z","shell.execute_reply.started":"2023-01-01T06:09:29.779303Z","shell.execute_reply":"2023-01-01T06:09:29.784061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions = transactions\\\n    .withColumn('article_id', transactions['article_id'].cast(IntegerType()))\\\n    .filter(F.col('t_dat') >= start_date)\\\n    .withColumn('week', F.when((F.col('t_dat') >= '2020-09-16') & (F.col('t_dat') <= '2020-09-22'), 9)\n                         .when((F.col('t_dat') >= '2020-09-09') & (F.col('t_dat') <= '2020-09-15'), 8)\n                         .when((F.col('t_dat') >= '2020-09-02') & (F.col('t_dat') <= '2020-09-08'), 7)\n                         .when((F.col('t_dat') >= '2020-08-26') & (F.col('t_dat') <= '2020-09-01'), 6)\n                         .when((F.col('t_dat') >= '2020-08-19') & (F.col('t_dat') <= '2020-08-25'), 5)\n                         .when((F.col('t_dat') >= '2020-08-12') & (F.col('t_dat') <= '2020-08-18'), 4)\n                         .when((F.col('t_dat') >= '2020-08-05') & (F.col('t_dat') <= '2020-08-11'), 3)\n                         .when((F.col('t_dat') >= '2020-07-29') & (F.col('t_dat') <= '2020-08-04'), 2)\n                         .when((F.col('t_dat') >= '2020-07-22') & (F.col('t_dat') <= '2020-07-28'), 1)\n                        .otherwise(999))\\\n    .drop('t_dat', 'price', 'sales_channel_id')\\\n    .orderBy(['week', 'customer_id'], ascending=True)\n\ntransactions.show(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:14:01.182347Z","iopub.execute_input":"2023-01-01T06:14:01.182764Z","iopub.status.idle":"2023-01-01T06:14:26.224579Z","shell.execute_reply.started":"2023-01-01T06:14:01.182730Z","shell.execute_reply":"2023-01-01T06:14:26.223506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_per_week = transactions\\\n    .groupBy('week').count().orderBy('week', ascending=True)\\\n\ntransactions_per_week.show(10)\ntransactions_per_week.unpersist()\n\n# check transactions loaded\n# remove data from memory","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:14:39.205512Z","iopub.execute_input":"2023-01-01T06:14:39.205837Z","iopub.status.idle":"2023-01-01T06:14:55.824011Z","shell.execute_reply.started":"2023-01-01T06:14:39.205811Z","shell.execute_reply":"2023-01-01T06:14:55.823198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.sql.functions import countDistinct\n\nunique_customers = transactions\\\n    .select(countDistinct('customer_id'))\n\nprint(\"customer_id in current perimeter : \"+ str(unique_customers.collect()[0][0]))\nunique_customers.unpersist()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:15:36.861529Z","iopub.execute_input":"2023-01-01T06:15:36.861872Z","iopub.status.idle":"2023-01-01T06:15:53.763929Z","shell.execute_reply.started":"2023-01-01T06:15:36.861841Z","shell.execute_reply":"2023-01-01T06:15:53.763043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of orders for each customer each week\n# shift week to +1 to make it a feature for next week\n\ncustomers_orders_lw = transactions\\\n    .groupBy('customer_id', 'week').count().orderBy('count', ascending=False)\\\n    .withColumnRenamed('count', 'lw_orders_count')\\\n    .withColumn('week', F.col('week')+1)\n\ncustomers_orders_lw.show(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:15:53.765362Z","iopub.execute_input":"2023-01-01T06:15:53.766623Z","iopub.status.idle":"2023-01-01T06:16:13.428007Z","shell.execute_reply.started":"2023-01-01T06:15:53.766583Z","shell.execute_reply":"2023-01-01T06:16:13.426587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#rank articles for each week\n\narticles_rank = transactions\\\n    .groupBy('article_id', 'week').count().orderBy('count', ascending=False)\\\n    .withColumnRenamed('count', 'articles_order_count')\n\nw_articles = Window.partitionBy(['week']).orderBy(articles_rank.articles_order_count.desc())\n\narticles_rank = articles_rank\\\n    .withColumn('rank', F.row_number().over(w_articles))\\\n    .filter(F.col('rank') <= 12)\\\n    .drop('articles_order_count')\\\n    .orderBy(['rank', 'week'])\n\narticles_top12 = articles_rank\\\n    .filter(F.col('rank') <= 12)\\\n    .drop('articles_order_count')\\\n    .orderBy(['rank', 'week'])\n\narticles_rank.show(20)\narticles_top12.show(25)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:23:09.431022Z","iopub.execute_input":"2023-01-01T06:23:09.431456Z","iopub.status.idle":"2023-01-01T06:23:46.808031Z","shell.execute_reply.started":"2023-01-01T06:23:09.431423Z","shell.execute_reply":"2023-01-01T06:23:46.807168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# after counting rows, I can drop duplicates\ntransactions = transactions.dropDuplicates()\\\n\ntransactions.show(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:23:46.809242Z","iopub.execute_input":"2023-01-01T06:23:46.810629Z","iopub.status.idle":"2023-01-01T06:24:05.704563Z","shell.execute_reply.started":"2023-01-01T06:23:46.810594Z","shell.execute_reply":"2023-01-01T06:24:05.703618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the goal of this transfrom is to shift last bought basket to next week in which the customer bought something\n# this is done to create negative observations in \"next week\" (from a copy of last purchased basked)\n# because the customer could've skipped some weeks, I need to put a number to each week partition and shift it +1\n\n# add a reference number for customers who bought something in a certain week (somewhat of a transaction identifier)\nrn_transactions = transactions\\\n    .select('customer_id', 'week')\\\n    .dropDuplicates()\n\nw_transactions = Window.partitionBy('customer_id').orderBy(rn_transactions.week.asc())\n","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:24:12.978911Z","iopub.execute_input":"2023-01-01T06:24:12.979345Z","iopub.status.idle":"2023-01-01T06:24:13.009021Z","shell.execute_reply.started":"2023-01-01T06:24:12.979309Z","shell.execute_reply":"2023-01-01T06:24:13.008228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# enumerate rows by week partition\nrn_transactions = rn_transactions\\\n    .withColumn('week_rn', F.row_number().over(w_transactions))\\\n    .select('customer_id', 'week', 'week_rn')\\\n    .orderBy(['customer_id', 'week'], ascending = True)\n\nrn_transactions.show(10)\n\nrn_transactions0 = rn_transactions.drop('article_id')\\\n    .withColumnRenamed('week', 'new_week')\\\n    .dropDuplicates()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:31:18.849571Z","iopub.execute_input":"2023-01-01T06:31:18.849941Z","iopub.status.idle":"2023-01-01T06:31:42.280116Z","shell.execute_reply.started":"2023-01-01T06:31:18.849911Z","shell.execute_reply":"2023-01-01T06:31:42.279231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# shift rn to next row\n# keep week info from rn_transactions0 and join on shifted week_rn\n\nlast_purchase = rn_transactions\\\n    .withColumn('week_rn', F.col('week_rn')+1)\\\n    .join(rn_transactions0, ['customer_id', 'week_rn'], 'inner')\\\n    .join(transactions, ['customer_id', 'week'], 'left')\\\n\nlast_purchase.show(10)\n\nlast_purchase = last_purchase\\\n    .drop('week')\\\n    .withColumnRenamed('new_week', 'week')\\\n    .orderBy(['customer_id', 'week'], ascending = True)\\\n    .select('customer_id', 'article_id', 'week')\n\nlast_purchase.show(20)\n\nlp_per_week = last_purchase\\\n    .groupBy('week').count().orderBy('week', ascending=True)\\\n","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:26:45.001326Z","iopub.execute_input":"2023-01-01T06:26:45.001677Z","iopub.status.idle":"2023-01-01T06:28:21.078908Z","shell.execute_reply.started":"2023-01-01T06:26:45.001635Z","shell.execute_reply":"2023-01-01T06:28:21.077845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lp_per_week.show(10)\nlp_per_week.unpersist()\nrn_transactions0.unpersist()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:28:21.080691Z","iopub.execute_input":"2023-01-01T06:28:21.081048Z","iopub.status.idle":"2023-01-01T06:28:59.002077Z","shell.execute_reply.started":"2023-01-01T06:28:21.081012Z","shell.execute_reply":"2023-01-01T06:28:59.001129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transactions_pos = transactions\\\n    .select('customer_id', 'article_id', 'week')\\\n    .withColumn('y', F.lit(1))","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:32:29.684040Z","iopub.execute_input":"2023-01-01T06:32:29.684432Z","iopub.status.idle":"2023-01-01T06:32:29.703140Z","shell.execute_reply.started":"2023-01-01T06:32:29.684400Z","shell.execute_reply":"2023-01-01T06:32:29.702441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# keep only negative obs by excluding stuff that the customer bought in next week (true y)\nlast_purchase = last_purchase\\\n    .join(transactions_pos, ['customer_id', 'article_id', 'week'], 'left')\\\n    .fillna({'y': 0})\\\n    .filter(F.col('y').isin(0))\\\n    .select('customer_id', 'article_id', 'week', 'y')\n\nlast_purchase.show(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:32:42.435504Z","iopub.execute_input":"2023-01-01T06:32:42.435852Z","iopub.status.idle":"2023-01-01T06:33:53.895920Z","shell.execute_reply.started":"2023-01-01T06:32:42.435822Z","shell.execute_reply":"2023-01-01T06:33:53.894984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\narticles_top12_pw = articles_top12\\\n    .withColumn('week', F.col('week')+1)\n\nlistona = transactions\\\n    .select('customer_id', 'week')\\\n    .dropDuplicates()\\\n    .join(articles_top12_pw, ['week'], 'left')\\\n    .join(transactions_pos, ['customer_id', 'article_id', 'week'], 'left')\\\n    .fillna({'y': 0})\\\n    .filter(F.col('y').isin(0))\\\n    .select('customer_id', 'article_id', 'week', 'y')\\\n    .orderBy('customer_id', 'week')\\\n\nlistona.show(10)\narticles_top12_pw.unpersist()\ntransactions_pos.unpersist()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:33:53.897449Z","iopub.execute_input":"2023-01-01T06:33:53.897802Z","iopub.status.idle":"2023-01-01T06:34:56.541487Z","shell.execute_reply.started":"2023-01-01T06:33:53.897764Z","shell.execute_reply":"2023-01-01T06:34:56.539829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# exclude from train the most remote week of observation since I don't generate a strategy for it\n# join all features\n\ntrain = transactions\\\n    .select('customer_id', 'article_id', 'week')\\\n    .withColumn('y', F.lit(1))\\\n    .unionByName(listona)\\\n    .unionByName(last_purchase)\\\n    .join(customers, 'customer_id', 'left')\\\n    .join(articles_rank, ['article_id', 'week'], 'left')\\\n    .join(articles, 'article_id', 'left')\\\n    .join(customers_orders_lw, ['customer_id', 'week'], 'left')\\\n    .orderBy(['week', 'customer_id'])\\\n    .filter(~F.col('week').isin(min_week))\\\n    .fillna({'rank': 999})\\\n    .fillna({'lw_orders_count': 0})\\\n    .orderBy(['week', 'customer_id'], ascending=True)\n\ntrain.show(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:34:56.542905Z","iopub.execute_input":"2023-01-01T06:34:56.543317Z","iopub.status.idle":"2023-01-01T06:38:15.078113Z","shell.execute_reply.started":"2023-01-01T06:34:56.543270Z","shell.execute_reply":"2023-01-01T06:38:15.077231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"application = spark.read.option(\"header\",True) \\\n                .csv(\"../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:38:15.080771Z","iopub.execute_input":"2023-01-01T06:38:15.081121Z","iopub.status.idle":"2023-01-01T06:38:15.338259Z","shell.execute_reply.started":"2023-01-01T06:38:15.081084Z","shell.execute_reply":"2023-01-01T06:38:15.337336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# top 12 sold from last week\narticles_top12_app = articles_top12\\\n    .filter(F.col('week').isin(max_week))\\\n    .drop('week')\\\n    .withColumn('week', F.lit(application_week))\\\n    .dropDuplicates()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:38:15.339215Z","iopub.execute_input":"2023-01-01T06:38:15.339587Z","iopub.status.idle":"2023-01-01T06:38:15.459079Z","shell.execute_reply.started":"2023-01-01T06:38:15.339553Z","shell.execute_reply":"2023-01-01T06:38:15.458060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# top 12 items last week list\n\ntop_12 = articles_top12_app.toPandas()\ntop_12_lw = top_12['article_id'].tolist()\n\nprint(top_12_lw)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:38:15.460179Z","iopub.execute_input":"2023-01-01T06:38:15.460537Z","iopub.status.idle":"2023-01-01T06:38:38.952643Z","shell.execute_reply.started":"2023-01-01T06:38:15.460503Z","shell.execute_reply":"2023-01-01T06:38:38.950826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# articles rank feature\nlast_week_rank = articles_rank\\\n    .filter(F.col('week').isin(max_week))\\\n    .drop('week')\\\n    .withColumn('week', F.lit(application_week))\\\n    .dropDuplicates()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:38:38.955432Z","iopub.execute_input":"2023-01-01T06:38:38.956053Z","iopub.status.idle":"2023-01-01T06:38:38.984921Z","shell.execute_reply.started":"2023-01-01T06:38:38.956010Z","shell.execute_reply":"2023-01-01T06:38:38.983756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#last purchased basket\n\nw_lp = Window.partitionBy('customer_id').orderBy(transactions.week.desc())\n\nlast_purchased_app = transactions\\\n    .select('customer_id', 'week')\\\n    .dropDuplicates()\\\n    .withColumn('rn', F.row_number().over(w_lp))\\\n    .filter(F.col('rn').isin(1))\\\n    .drop('rn')\\\n    .join(transactions, ['customer_id', 'week'], 'left')","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:38:38.988363Z","iopub.execute_input":"2023-01-01T06:38:38.988976Z","iopub.status.idle":"2023-01-01T06:38:39.061337Z","shell.execute_reply.started":"2023-01-01T06:38:38.988936Z","shell.execute_reply":"2023-01-01T06:38:39.060308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"application = application\\\n    .select('customer_id')\\\n    .join(customers, 'customer_id', 'left')\\\n    .dropDuplicates()\\\n    .withColumn('week', F.lit(application_week))\\\n    .join(articles_top12_app, 'week', 'left')\\\n    .select('customer_id', 'article_id', 'week')\\\n    .unionByName(last_purchased_app)\\\n    .dropDuplicates()\\\n    .join(customers, 'customer_id', 'left')\\\n    .join(last_week_rank, 'article_id', 'left')\\\n    .join(articles, 'article_id', 'left')\\\n    .join(customers_orders_lw, ['customer_id', 'week'], 'left')\\\n    .fillna({'rank': 999})\\\n    .fillna({'lw_orders_count': 0})\\\n    .drop('week')\\\n    .dropDuplicates()\n\napplication.show(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:38:39.062856Z","iopub.execute_input":"2023-01-01T06:38:39.063246Z","iopub.status.idle":"2023-01-01T06:41:31.273250Z","shell.execute_reply.started":"2023-01-01T06:38:39.063210Z","shell.execute_reply":"2023-01-01T06:41:31.271667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nlast_week_rank.unpersist()\narticles_top12_app.unpersist()\nlast_purchased_app.unpersist()\n\ntransactions.unpersist()\ncustomers.unpersist()\narticles.unpersist()\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:42:15.629630Z","iopub.execute_input":"2023-01-01T06:42:15.630177Z","iopub.status.idle":"2023-01-01T06:42:15.845883Z","shell.execute_reply.started":"2023-01-01T06:42:15.630113Z","shell.execute_reply":"2023-01-01T06:42:15.844665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport gc \n\ntrain.repartition(1).write.csv('/kaggle/working/train_df', header = 'true')\n\ntrain.unpersist()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:42:56.828140Z","iopub.execute_input":"2023-01-01T06:42:56.828577Z","iopub.status.idle":"2023-01-01T06:47:29.272046Z","shell.execute_reply.started":"2023-01-01T06:42:56.828543Z","shell.execute_reply":"2023-01-01T06:47:29.270635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"../\"))\nprint(os.listdir(\"../working/train_df\"))","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:47:29.274089Z","iopub.execute_input":"2023-01-01T06:47:29.274587Z","iopub.status.idle":"2023-01-01T06:47:29.282005Z","shell.execute_reply.started":"2023-01-01T06:47:29.274548Z","shell.execute_reply":"2023-01-01T06:47:29.280759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_t = os.listdir(\"../working/train_df\")\ntrim_t = [x for x in path_t if x.startswith('part')]\nstringa_t = ''.join(trim_t)\ntrain_path = '../working/train_df/'+stringa_t\n\nprint(train_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:47:29.283422Z","iopub.execute_input":"2023-01-01T06:47:29.284073Z","iopub.status.idle":"2023-01-01T06:47:29.414698Z","shell.execute_reply.started":"2023-01-01T06:47:29.284038Z","shell.execute_reply":"2023-01-01T06:47:29.413334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save application\napplication.repartition(1).write.csv('/kaggle/working/application', header = 'true')\n\napplication.unpersist()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:47:29.416780Z","iopub.execute_input":"2023-01-01T06:47:29.417912Z","iopub.status.idle":"2023-01-01T06:51:30.435525Z","shell.execute_reply.started":"2023-01-01T06:47:29.417869Z","shell.execute_reply":"2023-01-01T06:51:30.434201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom lightgbm.sklearn import LGBMRanker\n\ntrain = pd.read_csv(train_path)\ntrain.sort_values(['week', 'customer_id'], inplace=True)\ntrain.reset_index(drop=True, inplace=True)\nprint('train:', train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:51:30.438099Z","iopub.execute_input":"2023-01-01T06:51:30.438622Z","iopub.status.idle":"2023-01-01T06:51:48.046335Z","shell.execute_reply.started":"2023-01-01T06:51:30.438572Z","shell.execute_reply":"2023-01-01T06:51:48.044904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\n#train = train.rename(columns = lambda x:re.sub('[^A-Za-z0-9_]+', '', x))\n# columns renamed because for some reason one hot encoding creates invalid characters\ntrain_cols = list(train.columns)\nprint(train_cols)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:51:48.048031Z","iopub.execute_input":"2023-01-01T06:51:48.049032Z","iopub.status.idle":"2023-01-01T06:51:48.056958Z","shell.execute_reply.started":"2023-01-01T06:51:48.048983Z","shell.execute_reply":"2023-01-01T06:51:48.055232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"qids_train = train.groupby(['week', 'customer_id'])['article_id'].count().values\n\nX_train = train.drop([\"y\", 'customer_id', 'week'], axis=1)\ny_train = train[\"y\"]","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:51:48.058051Z","iopub.execute_input":"2023-01-01T06:51:48.058468Z","iopub.status.idle":"2023-01-01T06:51:50.681360Z","shell.execute_reply.started":"2023-01-01T06:51:48.058431Z","shell.execute_reply":"2023-01-01T06:51:50.680285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(qids_train)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:51:50.682898Z","iopub.execute_input":"2023-01-01T06:51:50.683315Z","iopub.status.idle":"2023-01-01T06:51:50.690986Z","shell.execute_reply.started":"2023-01-01T06:51:50.683280Z","shell.execute_reply":"2023-01-01T06:51:50.689326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# n_estimators is recommended as default to 100 by documentation\n\nmodel = LGBMRanker(\n    objective=\"lambdarank\",\n    metric=\"ndcg\",\n    boosting_type=\"dart\",\n    n_estimators=100,\n    importance_type='gain',\n    verbose=10,\n    random_state = 17\n)\n\nmodel.fit(\n    X=X_train,\n    y=y_train,\n    group=qids_train,\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:51:50.694443Z","iopub.execute_input":"2023-01-01T06:51:50.694890Z","iopub.status.idle":"2023-01-01T06:55:43.435731Z","shell.execute_reply.started":"2023-01-01T06:51:50.694848Z","shell.execute_reply":"2023-01-01T06:55:43.434556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature importance\nx_train_cols = list(X_train.columns)\n\nfor i in model.feature_importances_.argsort()[::-1]:\n    print(x_train_cols[i], model.feature_importances_[i]/model.feature_importances_.sum())","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:58:41.109482Z","iopub.execute_input":"2023-01-01T06:58:41.109971Z","iopub.status.idle":"2023-01-01T06:58:41.120164Z","shell.execute_reply.started":"2023-01-01T06:58:41.109941Z","shell.execute_reply":"2023-01-01T06:58:41.118771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"../\"))\nprint(os.listdir(\"../working/application\"))","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:58:57.126655Z","iopub.execute_input":"2023-01-01T06:58:57.127269Z","iopub.status.idle":"2023-01-01T06:58:57.134085Z","shell.execute_reply.started":"2023-01-01T06:58:57.127230Z","shell.execute_reply":"2023-01-01T06:58:57.133085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = os.listdir(\"../working/application\")\ntrim = [x for x in path if x.startswith('part')]\nstringa = ''.join(trim)\napp_path = '../working/application/'+stringa\n\nprint(app_path)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:59:06.872043Z","iopub.execute_input":"2023-01-01T06:59:06.872506Z","iopub.status.idle":"2023-01-01T06:59:06.878308Z","shell.execute_reply.started":"2023-01-01T06:59:06.872473Z","shell.execute_reply":"2023-01-01T06:59:06.877585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"application = pd.read_csv(app_path)\napplication.sort_values('customer_id', inplace=True)\napplication.reset_index(drop=True, inplace=True)\n\napplication_x = application.drop('customer_id', axis = 1)\nprint('application_x:', application_x.shape)\napp_cols = list(application_x.columns)\nprint(app_cols)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T06:59:18.476707Z","iopub.execute_input":"2023-01-01T06:59:18.477073Z","iopub.status.idle":"2023-01-01T07:00:03.499180Z","shell.execute_reply.started":"2023-01-01T06:59:18.477047Z","shell.execute_reply":"2023-01-01T07:00:03.498066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"application['prediction'] = model.predict(application_x)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:00:03.501092Z","iopub.execute_input":"2023-01-01T07:00:03.501453Z","iopub.status.idle":"2023-01-01T07:00:19.014552Z","shell.execute_reply.started":"2023-01-01T07:00:03.501416Z","shell.execute_reply":"2023-01-01T07:00:19.012795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"application.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:00:20.273522Z","iopub.execute_input":"2023-01-01T07:00:20.273843Z","iopub.status.idle":"2023-01-01T07:00:20.294665Z","shell.execute_reply.started":"2023-01-01T07:00:20.273817Z","shell.execute_reply":"2023-01-01T07:00:20.293521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_dict = application \\\n    .sort_values(['customer_id', 'prediction'], ascending=False) \\\n    .groupby('customer_id')['article_id'].apply(list).to_dict()","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:01:41.975800Z","iopub.execute_input":"2023-01-01T07:01:41.976315Z","iopub.status.idle":"2023-01-01T07:02:34.009879Z","shell.execute_reply.started":"2023-01-01T07:01:41.976277Z","shell.execute_reply":"2023-01-01T07:02:34.008680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:12:47.809749Z","iopub.execute_input":"2023-01-01T07:12:47.810378Z","iopub.status.idle":"2023-01-01T07:12:53.258872Z","shell.execute_reply.started":"2023-01-01T07:12:47.810334Z","shell.execute_reply":"2023-01-01T07:12:53.257727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nfor c_id in sub.customer_id:\n    pred = pred_dict.get(c_id, [])\n    pred = pred + top_12_lw\n    preds.append(pred[:12])","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:12:56.559325Z","iopub.execute_input":"2023-01-01T07:12:56.560299Z","iopub.status.idle":"2023-01-01T07:13:01.542796Z","shell.execute_reply.started":"2023-01-01T07:12:56.560252Z","shell.execute_reply":"2023-01-01T07:13:01.541048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = [' '.join(['0' + str(p) for p in ps]) for ps in preds]\nsub.prediction = preds","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:13:06.661497Z","iopub.execute_input":"2023-01-01T07:13:06.661923Z","iopub.status.idle":"2023-01-01T07:13:13.133648Z","shell.execute_reply.started":"2023-01-01T07:13:06.661889Z","shell.execute_reply":"2023-01-01T07:13:13.130519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_name = 'submission'\nsub.to_csv(f'{sub_name}.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-01T07:13:43.715764Z","iopub.execute_input":"2023-01-01T07:13:43.717020Z","iopub.status.idle":"2023-01-01T07:13:49.257751Z","shell.execute_reply.started":"2023-01-01T07:13:43.716971Z","shell.execute_reply":"2023-01-01T07:13:49.256024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}