{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### 라이브러리 확인","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nprint(\"Pandas version\", pd.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:35:59.334374Z","iopub.execute_input":"2022-07-15T07:35:59.335001Z","iopub.status.idle":"2022-07-15T07:35:59.362751Z","shell.execute_reply.started":"2022-07-15T07:35:59.334922Z","shell.execute_reply":"2022-07-15T07:35:59.361803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## RAPIDS - Data Science with GPU\n\n\nReference : https://rapids.ai/\n\ncudf : https://docs.rapids.ai/api/cudf/stable/\n\ncuml : https://docs.rapids.ai/api/cuml/stable/","metadata":{}},{"cell_type":"code","source":"import cudf\nprint(\"RAPIDS version\", cudf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:35:59.364746Z","iopub.execute_input":"2022-07-15T07:35:59.365155Z","iopub.status.idle":"2022-07-15T07:36:04.274707Z","shell.execute_reply.started":"2022-07-15T07:35:59.365119Z","shell.execute_reply":"2022-07-15T07:36:04.273648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load Transaction data\n\n- pandas로 데이터를 읽어온 뒤, 기본적인 데이터를 확인해봅니다.","metadata":{}},{"cell_type":"code","source":"# customers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\n# customers","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:04.276209Z","iopub.execute_input":"2022-07-15T07:36:04.276560Z","iopub.status.idle":"2022-07-15T07:36:04.282250Z","shell.execute_reply.started":"2022-07-15T07:36:04.276523Z","shell.execute_reply":"2022-07-15T07:36:04.281174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\n# articles","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:04.285434Z","iopub.execute_input":"2022-07-15T07:36:04.286885Z","iopub.status.idle":"2022-07-15T07:36:04.292277Z","shell.execute_reply.started":"2022-07-15T07:36:04.286845Z","shell.execute_reply":"2022-07-15T07:36:04.291320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del customers\n# del articles","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:04.293614Z","iopub.execute_input":"2022-07-15T07:36:04.294138Z","iopub.status.idle":"2022-07-15T07:36:04.303899Z","shell.execute_reply.started":"2022-07-15T07:36:04.294102Z","shell.execute_reply":"2022-07-15T07:36:04.302744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load data (pandas version)\n# train = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\",\n#                    dtype={'article_id' : str}, \n#                    parse_dates=[\"t_dat\"])\n# train","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:04.305346Z","iopub.execute_input":"2022-07-15T07:36:04.305788Z","iopub.status.idle":"2022-07-15T07:36:04.314201Z","shell.execute_reply.started":"2022-07-15T07:36:04.305752Z","shell.execute_reply":"2022-07-15T07:36:04.313267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check memory usage\n# mem_usage = train.memory_usage(deep=True).sum() / 1024 / 1024 / 1024\n# print(f\"Memory Usage : {mem_usage:.4} GiB\")","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:04.315380Z","iopub.execute_input":"2022-07-15T07:36:04.316006Z","iopub.status.idle":"2022-07-15T07:36:04.328053Z","shell.execute_reply.started":"2022-07-15T07:36:04.315969Z","shell.execute_reply":"2022-07-15T07:36:04.327125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reduce data size via cudf\n\nReference : https://www.kaggle.com/competitions/h-and-m-personalized-fashion-recommendations/discussion/308635","metadata":{}},{"cell_type":"code","source":"# Load data (cudf version)\ntrain = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\n\n# customer_id 데이터를 식별가능한 선에서 더 적은 사이즈로 변환합니다.\ntrain['customer_id'] = train.customer_id.str[-16:].str.hex_to_int().astype('int64')\n# article_id를 더 작은 사이즈의 data type으로 변환합니다.\ntrain['article_id'] = train.article_id.astype('int32')\n# t_dat column은 시간 정보로 변환합니다.\ntrain.t_dat = cudf.to_datetime(train.t_dat)\n\n# 나머지 column들은 제외합니다.\ntrain = train[['t_dat','customer_id','article_id']]\n\n# I/O를 줄이기 위해서 parquet 타입으로 변환합니다.\ntrain.to_parquet('train.pqt', index=False)\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:04.329605Z","iopub.execute_input":"2022-07-15T07:36:04.329955Z","iopub.status.idle":"2022-07-15T07:36:45.508377Z","shell.execute_reply.started":"2022-07-15T07:36:04.329920Z","shell.execute_reply":"2022-07-15T07:36:45.507462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check memory usage\nmem_usage = train.memory_usage(deep=True).sum() / 1024 / 1024 / 1024\nprint(f\"Memory Usage : {mem_usage:.4} GiB\")","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:45.512440Z","iopub.execute_input":"2022-07-15T07:36:45.512743Z","iopub.status.idle":"2022-07-15T07:36:45.524635Z","shell.execute_reply.started":"2022-07-15T07:36:45.512716Z","shell.execute_reply":"2022-07-15T07:36:45.523536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:45.526110Z","iopub.execute_input":"2022-07-15T07:36:45.527270Z","iopub.status.idle":"2022-07-15T07:36:45.668973Z","shell.execute_reply.started":"2022-07-15T07:36:45.527229Z","shell.execute_reply":"2022-07-15T07:36:45.667967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. 고객별 구매 트렌드 분석\n\n- 각 고객별로 최근 구매날짜기준, 직전주(last week)에 가장 많이 구매한 물품 목록을 찾습니다.","metadata":{}},{"cell_type":"code","source":"temp = train.groupby('customer_id')['t_dat'].max().reset_index()\ntemp.columns = ['customer_id', 'max_dat']\n#temp = temp.rename(columns={'t_dat' : 'max_dat'})\n\n# 고객별 시간 정보와 기존 데이터 결합\ntrain = cudf.merge(train, temp, on=\"customer_id\", how='left')\n# 최근 일주일 구매 내역 찾기\ntrain['diff_dat'] = (train.max_dat - train.t_dat).dt.days\ntrain = train[train.diff_dat < 7]\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:45.670443Z","iopub.execute_input":"2022-07-15T07:36:45.671375Z","iopub.status.idle":"2022-07-15T07:36:45.922262Z","shell.execute_reply.started":"2022-07-15T07:36:45.671333Z","shell.execute_reply":"2022-07-15T07:36:45.921155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. 고객별로 가장 많이 구매한 목록 찾기\n\n- 각 고객별로 가장 많은 구매한 article을 찾아봅니다.","metadata":{}},{"cell_type":"code","source":"# 고객별로 각 품목을 얼마나 구매했는지 계산합니다.\ntemp = train.groupby(['customer_id', 'article_id'])['t_dat'].count().reset_index()\ntemp.columns = ['customer_id', 'article_id', 'cnt']\n\n# # cudf는 GPU acceleration을 사용하므로, 데이터의 순서가 변경될 수 있기 때문에 재정렬을 해줍니다.\n# 기존 데이터와 결합\ntrain = cudf.merge(train, temp, on=['customer_id', 'article_id'], how='left')\ntrain = train.sort_values(by=['t_dat', 'cnt'], ascending=False)\ntrain = train.drop_duplicates(['customer_id', 'article_id'])\ntrain = train.sort_values(by=['t_dat', 'cnt'], ascending=False)\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:45.923997Z","iopub.execute_input":"2022-07-15T07:36:45.924362Z","iopub.status.idle":"2022-07-15T07:36:46.209772Z","shell.execute_reply.started":"2022-07-15T07:36:45.924325Z","shell.execute_reply":"2022-07-15T07:36:46.208723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. 지난주에 전체 고객이 가장 많이 산 물품 찾기\n\n- 예측 기간 기준으로 지난 주에 많이 산 물품을 찾습니다. (고객별로 차이가 있기 때문에 예측 기간을 기준으로 잡습니다)\n\n\n- 해당 유저가 cold-user이거나 특별한 구매 패턴을 보이지 않는다면, 최근 유행하는 품목을 추천해줍니다.","metadata":{}},{"cell_type":"code","source":"# parquet 데이터를 다시 불러옵니다.\ntrain = cudf.read_parquet('train.pqt')\n# 일주일 전 데이터 로드 (2020-9-23 기준으로 직전 주)\ntransactions_lastweek = train[train.t_dat >= cudf.to_datetime('2020-9-16')]\ntransactions_lastweek","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:46.211529Z","iopub.execute_input":"2022-07-15T07:36:46.211897Z","iopub.status.idle":"2022-07-15T07:36:46.432377Z","shell.execute_reply.started":"2022-07-15T07:36:46.211861Z","shell.execute_reply":"2022-07-15T07:36:46.431449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 예측은 12개의 품목을 해야하므로, 가장 많이 구매한 12개의 품목을 찾습니다.\ntop12 = transactions_lastweek.article_id.value_counts().index[:12]\nprint(\"Last week's top 12 popular items\")\ntop12","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:46.433722Z","iopub.execute_input":"2022-07-15T07:36:46.434723Z","iopub.status.idle":"2022-07-15T07:36:46.458162Z","shell.execute_reply.started":"2022-07-15T07:36:46.434685Z","shell.execute_reply":"2022-07-15T07:36:46.457159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# (OPTIONAL) 트렌드 분석을 기준으로 추천 결과를 생성해봅니다.\n# submission = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\n# submission = submission.drop(columns=\"prediction\")\n# submission['customer_id2'] = submission.customer_id.str[-16:].str.hex_to_int().astype('int64')\n\n# preds.columns = ['customer_id2', 'prediction']\n# submission = cudf.merge(submission, preds, on=\"customer_id2\", how='left').fillna('')\n# submission = submission.drop(columns='customer_id2')\n# submission['prediction'] = submission['prediction'] + top12\n# submission.prediction = submission.prediction.str.strip()\n# submission.prediction = submission.prediction.str[:11*12-1]\n# submission.to_csv('submission.csv', index=False)\n# submission","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:46.459561Z","iopub.execute_input":"2022-07-15T07:36:46.459983Z","iopub.status.idle":"2022-07-15T07:36:46.466541Z","shell.execute_reply.started":"2022-07-15T07:36:46.459944Z","shell.execute_reply":"2022-07-15T07:36:46.465556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4. 최근 3주의 트렌드를 이용한 분석 (예측 기간 기준!)\n\n\n- 최근 구매의 트렌드를 고객별이 아닌, 전체 기간을 기준으로 어떤 품목들을 많이 샀는지 체크합니다.\n\n\n- 각 고객별로 3주 구매내역에 대한 패턴을 분석합니다. (같은 논지를 활용하여 기간을 1, 2, 3개월로 변경 가능합니다.)","metadata":{}},{"cell_type":"code","source":"# data reload\ntrain = cudf.read_parquet('train.pqt')\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:46.468099Z","iopub.execute_input":"2022-07-15T07:36:46.468793Z","iopub.status.idle":"2022-07-15T07:36:46.690480Z","shell.execute_reply.started":"2022-07-15T07:36:46.468714Z","shell.execute_reply":"2022-07-15T07:36:46.688362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check memory usage\nmem_usage = train.memory_usage(deep=True).sum() / 1024 / 1024 / 1024\nprint(f\"Memory Usage : {mem_usage:.4} GiB\")","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:46.691689Z","iopub.execute_input":"2022-07-15T07:36:46.692302Z","iopub.status.idle":"2022-07-15T07:36:46.700219Z","shell.execute_reply.started":"2022-07-15T07:36:46.692261Z","shell.execute_reply":"2022-07-15T07:36:46.699259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 분석의 용이함을 위해 pandas dataframe으로 변환합니다.\ntrain = train.to_pandas() # cudf dataframe -> pandas dataframe\ntrain.article_id = train.article_id.astype(str)\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:36:46.701632Z","iopub.execute_input":"2022-07-15T07:36:46.702249Z","iopub.status.idle":"2022-07-15T07:37:18.217638Z","shell.execute_reply.started":"2022-07-15T07:36:46.702207Z","shell.execute_reply":"2022-07-15T07:37:18.216565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 예측 기간 기준 최근 3, 2, 1주의 데이터를 분석해봅니다.\ntransactions_3w = train[train['t_dat'] >= pd.to_datetime('2020-08-31')].copy()\ntransactions_3w\n# transactions_2w = train[train['t_dat'] >= pd.to_datetime('2020-09-07')].copy()\n# transactions_1w = train[train['t_dat'] >= pd.to_datetime('2020-09-15')].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:37:18.219169Z","iopub.execute_input":"2022-07-15T07:37:18.219638Z","iopub.status.idle":"2022-07-15T07:37:18.451515Z","shell.execute_reply.started":"2022-07-15T07:37:18.219600Z","shell.execute_reply":"2022-07-15T07:37:18.450167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4-1. 3주간 트렌드 분석\n\n\n- 3주 간의 내역을 기준으로, 고객별로 구매한 물품별 빈도를 dictionary로 저장합니다.\n\n\n\n< dictionary 구조>\n> {고객 : {구매한 물품 : 구매 빈도}}","metadata":{}},{"cell_type":"code","source":"purchase_in_3w = {}\n#purchase_in_3w = defaultidict(dict)\n\n## TO-DO : 고객별, 구매내역별 빈도를 저장하는 dictionary를 만들어서 가장 많이 구매한 물품 12개를 구해보세요.\nfor _, val in enumerate(zip(transactions_3w.customer_id, transactions_3w.article_id)):\n    c_id, a_id = val\n    if c_id not in purchase_in_3w:\n        purchase_in_3w[c_id] = {}  # defaultdict로 쉽게 할 수 있음\n        #purchase_in_3w = defaultdict(int)\n        \n    if a_id not in purchase_in_3w[c_id]:\n        purchase_in_3w[c_id][a_id] = 0  # defaultdict로 쉽게 할 수 있음\n        \n    purchase_in_3w[c_id][a_id] += 1\n    \n#purchase_in_3w","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:37:18.453256Z","iopub.execute_input":"2022-07-15T07:37:18.453715Z","iopub.status.idle":"2022-07-15T07:37:19.592750Z","shell.execute_reply.started":"2022-07-15T07:37:18.453675Z","shell.execute_reply":"2022-07-15T07:37:19.591426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cust_top_12s = []\n\nfor c_id in transactions_3w.customer_id:\n    cust_top12 = sorted(purchase_in_3w[c_id].items(), key=lambda x:x[1], reverse=True)\n    cust_top12 = [x[0] for x in cust_top12][:12]\n    cust_top_12s.append(cust_top12)\n    #print(f\"{c_id}\\t {cust_top12}\")\n    \ncust_top_12s[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:39:24.826343Z","iopub.execute_input":"2022-07-15T07:39:24.827386Z","iopub.status.idle":"2022-07-15T07:39:28.388711Z","shell.execute_reply.started":"2022-07-15T07:39:24.827341Z","shell.execute_reply":"2022-07-15T07:39:28.387561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_in_2w = {}\n# TO-DO","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:39:29.145120Z","iopub.execute_input":"2022-07-15T07:39:29.145465Z","iopub.status.idle":"2022-07-15T07:39:29.150250Z","shell.execute_reply.started":"2022-07-15T07:39:29.145435Z","shell.execute_reply":"2022-07-15T07:39:29.149091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"purchase_in_1w = {}\n#TO-DO","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:39:29.312370Z","iopub.execute_input":"2022-07-15T07:39:29.313176Z","iopub.status.idle":"2022-07-15T07:39:29.318471Z","shell.execute_reply.started":"2022-07-15T07:39:29.313129Z","shell.execute_reply":"2022-07-15T07:39:29.317213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # (OPTIONAL) make prediction\n\n# submission = cudf.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\n# submission['customer_id'] = submission.customer_id.str[-16:].str.hex_to_int().astype('int64')\n# submission = submission.to_pandas()\n\n# prediction_list = []\n\n# dummy_list = list((transactions_1w.article_id.value_counts()).index.astype(str))[:12]\n# dummy_pred = ' '.join(dummy_list)\n\n# for idx, cust_id in enumerate(submission.customer_id.values.reshape(-1, )):\n#     if cust_id in purchase_in_1w:\n#         l = sorted((purchase_in_1w[cust_id]).items(), key=lambda x:x[1], reverse=True)\n#         l = [y[0] for y in l]\n#         if len(l) > 12:\n#             s = ' '.join(l[:12])\n#         else:\n#             s = ' '.join(l + dummy_list_1w[:(12 - len(l))])\n            \n#     elif cust_id in purchase_in_2w:\n#         l = sorted((purchase_in_2w[cust_id]).items(), key=lambda x:x[1], reverse=True)\n#         l = [y[0] for y in l]\n#         if len(l) > 12:\n#             s = ' '.join(l[:12])\n#         else:\n#             s = ' '.join(l + dummy_list_2w[:(12 - len(l))])\n\n#     elif cust_id in purchase_in_3w:\n#         l = sorted((purchase_in_3w[cust_id]).items(), key=lambda x:x[1], reverse=True)\n#         l = [y[0] for y in l]\n#         if len(l) > 12:\n#             s= ' '.join(l[:12])\n#         else:\n#             s = ' '.join(l + dummy_list_3w[:(12 - len(l))])\n    \n#     else:\n#         s = dummy_pred\n\n#     prediction_list.append(s)\n    \n# submission['prediction'] = prediction_list\n# submission.to_csv(\"submission_last3w.csv\", index=False)\n# submission","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:39:32.925662Z","iopub.execute_input":"2022-07-15T07:39:32.926753Z","iopub.status.idle":"2022-07-15T07:39:32.932474Z","shell.execute_reply.started":"2022-07-15T07:39:32.926707Z","shell.execute_reply":"2022-07-15T07:39:32.931514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## References\n\n- https://www.kaggle.com/code/hengzheng/time-is-our-best-friend-v2\n\n- https://www.kaggle.com/code/cdeotte/recommend-items-purchased-together-0-021\n\n- https://www.kaggle.com/code/cdeotte/customers-who-bought-this-frequently-buy-this","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}