{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Based on this notebook:   \nhttps://www.kaggle.com/abhilashawasthi/not-so-fancy-but-fast-benchmark  \nIncluding changes from this notebook:   \nhttps://www.kaggle.com/hengzheng/time-is-our-best-friend  \n  \nRewritten to be more pythonic and readable (and hence, extendable).  \n29 lines of code!!!\n\nUpdated: for customer, use season's data, but for popularity, use only more recent.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndata_path = '/kaggle/input/h-and-m-personalized-fashion-recommendations/'\nt_file, sub_file = 'transactions_train.csv', 'sample_submission.csv'\n\n# set dtype or pandas will drop the leading '0' and convert to int\nt = pd.read_csv(data_path + \"/\" + t_file, dtype={'article_id': str})\nsub = pd.read_csv(data_path + \"/\" + sub_file)\n\nt_season = t[t['t_dat'] > '2020-08-01'].copy()\nt_new = t[t['t_dat'] > '2020-09-07'].copy()\n\n# customer purchases (cp)\ncp = t_season.groupby([\"customer_id\", \"article_id\"])[[\"article_id\"]].count()\ncp.columns = [\"purchase_count\"]\ncp = cp.reset_index()\ncp = cp.sort_values([\"customer_id\", \"purchase_count\"], ascending=False)\ncp = cp.groupby(\"customer_id\").head(10)\ncp = cp.groupby(\"customer_id\")[\"article_id\"].agg(list)\n\n# popular purchases (pp)\npp = list(t_new['article_id'].value_counts().index[:12])\n\n# submission (sub)\nsub[\"prediction\"] = sub[\"customer_id\"].map(cp)\nsub[\"prediction\"] = sub[\"prediction\"].apply(lambda x: x if isinstance(x, list) else [])\nsub[\"prediction\"] = sub[\"prediction\"].apply(lambda x: x[:12] + pp[:12-len(x)])\nsub[\"prediction\"] = sub[\"prediction\"].apply(lambda x: \" \".join(x))\n\nsub.to_csv('simpler_submission.csv', index=False)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-11T16:38:46.780568Z","iopub.execute_input":"2022-02-11T16:38:46.780866Z","iopub.status.idle":"2022-02-11T16:39:59.572881Z","shell.execute_reply.started":"2022-02-11T16:38:46.780837Z","shell.execute_reply":"2022-02-11T16:39:59.572278Z"},"trusted":true},"execution_count":null,"outputs":[]}]}