{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install pyarrow\n# !pip install fastparquet\n# !pip install gensim","metadata":{"execution":{"iopub.status.busy":"2022-05-09T00:58:38.361763Z","iopub.execute_input":"2022-05-09T00:58:38.362495Z","iopub.status.idle":"2022-05-09T00:58:38.366941Z","shell.execute_reply.started":"2022-05-09T00:58:38.362442Z","shell.execute_reply":"2022-05-09T00:58:38.365899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Overview","metadata":{"papermill":{"duration":0.025678,"end_time":"2022-04-23T02:52:48.647042","exception":false,"start_time":"2022-04-23T02:52:48.621364","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"（１）以前購入したもの（２）色違い（３）ペア購入商品（４）トレンド１（５）トレンド２（６）トレンド３（７）トレンド４（８）数年連続人気（９）協調フィルタリング１（１０）協調フィルタリング２の結果をそれぞれ計算し、アンサンブル。基本は予測（１）（２）（３）（９）（１０）を重視し、埋まらない部分をトレンドで埋める","metadata":{"papermill":{"duration":0.025728,"end_time":"2022-04-23T02:52:48.746194","exception":false,"start_time":"2022-04-23T02:52:48.720466","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# val_week : 104 ... calc CV\n#            105 ... make submission\n# use_item2vec : True ... use item2vec ! very slow\n#                False ... not use item 3 vec\n# is_load : True ... load result datas\n# is_save : True ... save result datas","metadata":{"execution":{"iopub.status.busy":"2022-05-09T00:58:38.371347Z","iopub.execute_input":"2022-05-09T00:58:38.371659Z","iopub.status.idle":"2022-05-09T00:58:38.380721Z","shell.execute_reply.started":"2022-05-09T00:58:38.371603Z","shell.execute_reply":"2022-05-09T00:58:38.379734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Libraries and Functions","metadata":{"papermill":{"duration":0.024089,"end_time":"2022-04-23T02:52:48.795238","exception":false,"start_time":"2022-04-23T02:52:48.771149","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import numpy as np, pandas as pd, datetime as dt\nimport matplotlib.pyplot as plt; plt.style.use('ggplot')\nimport seaborn as sns\nfrom collections import defaultdict\n\ndef iter_to_str(iterable):\n    return \" \".join(map(lambda x: str(0) + str(x), iterable))\n\ndef apk(actual, predicted, k=12):\n    if len(predicted) > k:\n        predicted = predicted[:k]\n    score, nhits = 0.0, 0.0\n    for i, p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            nhits += 1.0\n            score += nhits / (i + 1.0)\n    if not actual:\n        return 0.0\n    return score / min(len(actual), k)\n\ndef mapk(actual, predicted, k=12, return_apks=False):\n    assert len(actual) == len(predicted)\n    apks = [apk(ac, pr, k) for ac, pr in zip(actual, predicted) if 0 < len(ac)]\n    if return_apks:\n        return apks\n    return np.mean(apks)\n\ndef blend(dt, w=[], k=12):\n    if len(w) == 0:\n        w = [1] * (len(dt))\n    preds = []\n    for i in range(len(w)):\n        preds.append(dt[i].split())\n    res = {}\n    for i in range(len(preds)):\n        if w[i] < 0:\n            continue\n        for n, v in enumerate(preds[i]):\n            if v in res:\n                res[v] += (w[i] / (n + 1))\n            else:\n                res[v] = (w[i] / (n + 1))    \n    res = list(dict(sorted(res.items(), key=lambda item: -item[1])).keys())\n    return ' '.join(res[:k])\n\ndef prune(pred, ok_set, k=12):\n    pred = pred.split()\n    post = []\n    for item in pred:\n        if int(item) in ok_set and not item in post:\n            post.append(item)\n    return \" \".join(post[:k])\n\ndef validation(actual, predicted, grouping, score=0, index=-1, ignore=False, figsize=(12, 6)):\n    # actual, predicted : list of lists\n    # group : pandas Series\n    # score : pandas DataFrame\n    \n    vc = pd.Series(predicted).apply(len).value_counts()\n    print(\"Fill Rate = \", round(1 - sum(vc[k] * (12 - k) / 12 for k in (set(range(12)) & set(vc.index))) / len(actual), 3) * 100)\n    \n    \n    if ignore: return\n    ap12 = mapk(actual, predicted, return_apks=True)\n    map12 = round(np.mean(ap12), 6)\n    if isinstance(score, int): score = pd.DataFrame({g:[] for g in sorted(grouping.unique().tolist())})\n    if index == -1 : index = score.shape[0]\n    score.loc[index, \"All\"] = map12\n    plt.figure(figsize=figsize)\n    plt.subplot(1, 2, 1); sns.histplot(data=ap12, log_scale=(0, 10), bins=20); plt.title(f\"MAP@12 : {map12}\")\n    for g in grouping.unique():\n        map12 = round(mapk(actual[grouping == g], predicted[grouping == g]), 6)\n        score.loc[index, g] = map12\n    plt.subplot(1, 2, 2); score[[g for g in grouping.unique()[::-1]] + ['All']].loc[index].plot.barh(); plt.title(f\"MAP@12 of Groups\")\n    vc = pd.Series(predicted).apply(len).value_counts()\n    score.loc[index, \"Fill\"] = round(1 - sum(vc[k] * (12 - k) / 12 for k in (set(range(12)) & set(vc.index))) / len(actual), 3) * 100\n    display(score)\n    return score","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.166858,"end_time":"2022-04-23T02:52:49.987172","exception":false,"start_time":"2022-04-23T02:52:48.820314","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T00:58:38.387054Z","iopub.execute_input":"2022-05-09T00:58:38.387339Z","iopub.status.idle":"2022-05-09T00:58:39.697894Z","shell.execute_reply.started":"2022-05-09T00:58:38.387307Z","shell.execute_reply":"2022-05-09T00:58:39.696865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\ndef sort_item2vec(colname, max_n = 2, none = \"\") :\n    l = []\n    for i in tqdm(range(len(sub))) :\n        wid = sub.iloc[i]['customer_id']\n        oc = sub.iloc[i][colname].split()\n        if len(oc) == 0 :\n            l.append(none)\n        elif wid not in cust_buy :\n            l.append(\" \".join(oc))            \n        elif len(oc) <= 1 :\n            l.append(\" \".join(oc))\n        else :\n            ll = []\n            for e in oc : \n                buy = cust_buy[wid].split(\" \")\n                p = model.wv.similarity(buy, e).mean()\n                ll.append([p, e])\n            ll.sort(reverse=True)\n            l.append(\" \".join(list(map(lambda x: x[1], ll[:max_n]))))\n    #     if i == 10: break\n\n    return l\n\n\n# 負のものを削除\ndef remove_item2vec(colname, max_n = 2, none = \"\") :\n    l = []\n    for i in tqdm(range(len(sub))) :\n        wid = sub.iloc[i]['customer_id']\n        oc = sub.iloc[i][colname].split()\n        if len(oc) == 0 :\n            l.append(none)\n        elif wid not in cust_buy :\n            l.append(\" \".join(oc))            \n        elif len(oc) <= 1 :\n            l.append(\" \".join(oc))\n        else :\n            ll = []\n            rr = []\n            for e in oc : \n                buy = cust_buy[wid].split(\" \")\n                p = model.wv.similarity(buy, e).mean()\n                if p < 0 : \n                    rr.append(e)\n                else :\n                    ll.append(e)\n            l.append(\" \".join(ll+rr))\n    #     if i == 10: break\n\n    return l\n\n\ndef cut_items(name, n = 12) :\n    l = []\n    for i in tqdm(range(len(sub))) :\n        s = sub[name][i].split(\" \")\n        l.append(\" \".join(s[:n]))\n    return l","metadata":{"execution":{"iopub.status.busy":"2022-05-09T00:58:39.699718Z","iopub.execute_input":"2022-05-09T00:58:39.700092Z","iopub.status.idle":"2022-05-09T00:58:39.720923Z","shell.execute_reply.started":"2022-05-09T00:58:39.700044Z","shell.execute_reply":"2022-05-09T00:58:39.719848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_week = 104\nuse_item2vec = False\nis_load = False\nis_save = False","metadata":{"execution":{"iopub.status.busy":"2022-05-09T00:58:39.722194Z","iopub.execute_input":"2022-05-09T00:58:39.722526Z","iopub.status.idle":"2022-05-09T00:58:39.743056Z","shell.execute_reply.started":"2022-05-09T00:58:39.722482Z","shell.execute_reply":"2022-05-09T00:58:39.741471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### wod2vec","metadata":{}},{"cell_type":"code","source":"import json\nfrom  gensim import models\n\nif val_week == 104 :\n    print(\"item2vec = 104\")\n    model = models.Word2Vec.load('../input/handmitem2vec/word2vec_v2_week104.model')\n    with open('../input/handmitem2vec/cust_buy104.json', \"r\") as f :\n        cust_buy = json.load(f)\nelse :\n    print(\"item2vec = 105\")\n    model = models.Word2Vec.load('../input/handmitem2vec/word2vec_v2.model')\n    with open('../input/handmitem2vec/cust_buy.json', \"r\") as f :\n        cust_buy = json.load(f)\n","metadata":{"execution":{"iopub.status.busy":"2022-05-09T00:58:43.278692Z","iopub.execute_input":"2022-05-09T00:58:43.279432Z","iopub.status.idle":"2022-05-09T00:58:58.162741Z","shell.execute_reply.started":"2022-05-09T00:58:43.279380Z","shell.execute_reply":"2022-05-09T00:58:58.161721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data","metadata":{"papermill":{"duration":0.024177,"end_time":"2022-04-23T02:52:50.036764","exception":false,"start_time":"2022-04-23T02:52:50.012587","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# cdf = pd.read_parquet('../input/handmitem2vec/customers.parquet')\n# df = pd.read_parquet('../input/hm-parquets-of-datasets/transactions_train.parquet')\n# cdf = cdf[['customer_id', 'attribute']]\n# df = df.merge(cdf, on='customer_id', how='left')\n# df = df[df['attribute'] == 'Woman']\n# df","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:02:22.753609Z","iopub.execute_input":"2022-05-09T01:02:22.753952Z","iopub.status.idle":"2022-05-09T01:02:46.943047Z","shell.execute_reply.started":"2022-05-09T01:02:22.753915Z","shell.execute_reply":"2022-05-09T01:02:46.942331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_parquet('../input/hm-parquets-of-datasets/transactions_train.parquet')\nsub = pd.read_csv('../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv')\ncid = pd.DataFrame(sub.customer_id.apply(lambda s: int(s[-16:], 16)))","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":13.414428,"end_time":"2022-04-23T02:53:03.475349","exception":false,"start_time":"2022-04-23T02:52:50.060921","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T01:02:46.944731Z","iopub.execute_input":"2022-05-09T01:02:46.945079Z","iopub.status.idle":"2022-05-09T01:02:53.723780Z","shell.execute_reply.started":"2022-05-09T01:02:46.945035Z","shell.execute_reply":"2022-05-09T01:02:53.722888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Kangol x H&Mを除外（val_weekが105の場合）","metadata":{}},{"cell_type":"code","source":"# if val_week == 105 :\n#     print(\"remove Kangol x H&M\")\n#     articles_df = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\",dtype=str, encoding='utf8')\n#     xHM = []\n#     for i, e in enumerate(articles_df['detail_desc'].fillna(\"\")) :\n#         if 'Kangol x H&M' in e:\n#             #print(articles_df.iloc[i]['article_id'], e)\n#             xHM.append(int(articles_df.iloc[i]['article_id']))\n\n#     #xHM\n#     df = df[~df.article_id.isin(xHM)]","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:02:53.725146Z","iopub.execute_input":"2022-05-09T01:02:53.726009Z","iopub.status.idle":"2022-05-09T01:02:53.730675Z","shell.execute_reply.started":"2022-05-09T01:02:53.725973Z","shell.execute_reply":"2022-05-09T01:02:53.729825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Definition of Groups","metadata":{"papermill":{"duration":0.024442,"end_time":"2022-04-23T02:53:03.525992","exception":false,"start_time":"2022-04-23T02:53:03.50155","status":"completed"},"tags":[]}},{"cell_type":"code","source":"group = df.groupby('customer_id').sales_channel_id.mean().round().reset_index()\\\n    .merge(cid, on='customer_id', how='right').rename(columns={'sales_channel_id':'group'})\ngrouping = group.group.fillna(1.0)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":3.833687,"end_time":"2022-04-23T02:53:07.384015","exception":false,"start_time":"2022-04-23T02:53:03.550328","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T01:02:53.732627Z","iopub.execute_input":"2022-05-09T01:02:53.732937Z","iopub.status.idle":"2022-05-09T01:02:56.768112Z","shell.execute_reply.started":"2022-05-09T01:02:53.732901Z","shell.execute_reply":"2022-05-09T01:02:56.766980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### One-Week Hold Out","metadata":{"papermill":{"duration":0.024286,"end_time":"2022-04-23T02:53:07.433418","exception":false,"start_time":"2022-04-23T02:53:07.409132","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# id of week to be used in a validation; set 105 if you would like to create a submission\nval = df.loc[df.week == val_week].groupby('customer_id').article_id.apply(iter_to_str).reset_index()\\\n    .merge(cid, on='customer_id', how='right')\nactual = val.article_id.apply(lambda s: [] if pd.isna(s) else s.split())\nlast_date = df.loc[df.week < val_week].t_dat.max()","metadata":{"papermill":{"duration":4.626189,"end_time":"2022-04-23T02:53:12.083965","exception":false,"start_time":"2022-04-23T02:53:07.457776","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T01:02:56.769688Z","iopub.execute_input":"2022-05-09T01:02:56.770021Z","iopub.status.idle":"2022-05-09T01:03:02.827822Z","shell.execute_reply.started":"2022-05-09T01:02:56.769977Z","shell.execute_reply":"2022-05-09T01:03:02.826737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_date","metadata":{"papermill":{"duration":0.036854,"end_time":"2022-04-23T02:53:12.145693","exception":false,"start_time":"2022-04-23T02:53:12.108839","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T01:03:02.829462Z","iopub.execute_input":"2022-05-09T01:03:02.829836Z","iopub.status.idle":"2022-05-09T01:03:02.837335Z","shell.execute_reply.started":"2022-05-09T01:03:02.829788Z","shell.execute_reply":"2022-05-09T01:03:02.836323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## load Result","metadata":{}},{"cell_type":"code","source":"if is_load :\n    print(\"LOAD...\")\n    if val_week == 105: \n        sub = pd.read_parquet('submission_all.parquet')\n    else :\n        sub = pd.read_parquet('submission_all_104.parquet')\n    predicted = sub['last_purchase'].apply(lambda s: [] if pd.isna(s) else s.split())\n    score = validation(actual, predicted, grouping, index='Last Purchase', ignore=(val_week == 105))\n        \nsub","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:03:02.839053Z","iopub.execute_input":"2022-05-09T01:03:02.839414Z","iopub.status.idle":"2022-05-09T01:03:02.859847Z","shell.execute_reply.started":"2022-05-09T01:03:02.839371Z","shell.execute_reply":"2022-05-09T01:03:02.858736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ======================","metadata":{}},{"cell_type":"markdown","source":"### 最後に購入したアイテム(14日前から）","metadata":{"papermill":{"duration":0.02445,"end_time":"2022-04-23T02:53:12.195882","exception":false,"start_time":"2022-04-23T02:53:12.171432","status":"completed"},"tags":[]}},{"cell_type":"code","source":"init_date = last_date - dt.timedelta(days=9999)\ntrain = df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].copy()\ntrain = train.merge(train.groupby('customer_id').t_dat.max().reset_index().rename(columns={'t_dat':'l_dat'}),\n                   on = 'customer_id', how='left')\ntrain['d_dat'] = (train.l_dat - train.t_dat).dt.days\ntrain = train.loc[train.d_dat < 14].sort_values(['t_dat'], ascending=False).drop_duplicates(['customer_id', 'article_id'])\n#train = train.loc[train.d_dat < 9999].sort_values(['t_dat'], ascending=False).drop_duplicates(['customer_id', 'article_id'])\nsub['last_purchase'] = train.groupby('customer_id')\\\n    .article_id.apply(iter_to_str).reset_index()\\\n    .merge(cid, on='customer_id', how='right').article_id.fillna('')","metadata":{"papermill":{"duration":52.998734,"end_time":"2022-04-23T02:54:05.219324","exception":false,"start_time":"2022-04-23T02:53:12.22059","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T01:03:14.582439Z","iopub.execute_input":"2022-05-09T01:03:14.583119Z","iopub.status.idle":"2022-05-09T01:03:56.123696Z","shell.execute_reply.started":"2022-05-09T01:03:14.583075Z","shell.execute_reply":"2022-05-09T01:03:56.122699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub['last_purchase'] = cut_items('last_purchase', n = 6)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:04:37.635944Z","iopub.execute_input":"2022-05-09T01:04:37.636267Z","iopub.status.idle":"2022-05-09T01:04:37.660832Z","shell.execute_reply.started":"2022-05-09T01:04:37.636233Z","shell.execute_reply":"2022-05-09T01:04:37.659703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if use_item2vec:\n#     sub['last_purchase'] = swap_item2vec('last_purchase')","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:04:28.999532Z","iopub.execute_input":"2022-05-09T01:04:29.000131Z","iopub.status.idle":"2022-05-09T01:04:29.006457Z","shell.execute_reply.started":"2022-05-09T01:04:29.000081Z","shell.execute_reply":"2022-05-09T01:04:29.005477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = sub['last_purchase'].apply(lambda s: [] if pd.isna(s) else s.split())\nscore = validation(actual, predicted, grouping, index='Last Purchase', ignore=(val_week == 105))\n#Last Purchase\t0.012392\t0.023801\t0.020029\t26.9\n#Last Purchase\t0.012807\t0.024439\t0.020594\t28.7\n#Last Purchase\t0.011057\t0.020228\t0.017196\t14.2\n#Last Purchase\t0.010913\t0.020194\t0.017126\t14.2 item2vec\n#Last Purchase\t0.012333\t0.023044\t0.019503\t21.4","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:03:56.145409Z","iopub.execute_input":"2022-05-09T01:03:56.145842Z","iopub.status.idle":"2022-05-09T01:04:03.406920Z","shell.execute_reply.started":"2022-05-09T01:03:56.145799Z","shell.execute_reply":"2022-05-09T01:04:03.405983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 購入したアイテムの他の色（６日前まで）","metadata":{"papermill":{"duration":0.024202,"end_time":"2022-04-23T02:54:05.268336","exception":false,"start_time":"2022-04-23T02:54:05.244134","status":"completed"},"tags":[]}},{"cell_type":"code","source":"init_date = last_date - dt.timedelta(days=6)\n#init_date = last_date - dt.timedelta(days=14)\ntrain = df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].copy()\\\n    .groupby(['article_id']).t_dat.count().reset_index()\nadf = pd.read_parquet('../input/hm-parquets-of-datasets/articles.parquet')\nadf = adf.merge(train, on='article_id', how='left').rename(columns={'t_dat':'ct'})\\\n    .sort_values('ct', ascending=False).query('ct > 0')\n\nmap_to_col = defaultdict(list)\nfor aid in adf.article_id.tolist():\n    map_to_col[aid] = list(filter(lambda x: x != aid, adf[adf.product_code == aid // 1000].article_id.tolist()))[:1]\n\ndef map_to_variation(s):\n    f = lambda item: iter_to_str(map_to_col[int(item)])\n    return ' '.join(map(f, s.split()))\nsub['other_colors'] = sub['last_purchase'].fillna('').apply(map_to_variation)","metadata":{"papermill":{"duration":22.651535,"end_time":"2022-04-23T02:54:27.944334","exception":false,"start_time":"2022-04-23T02:54:05.292799","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T01:04:46.244685Z","iopub.execute_input":"2022-05-09T01:04:46.245021Z","iopub.status.idle":"2022-05-09T01:05:04.036053Z","shell.execute_reply.started":"2022-05-09T01:04:46.244986Z","shell.execute_reply":"2022-05-09T01:05:04.035101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 11日間の売れた商品リストに入っていないアイテムは削除する\ninit_date = last_date - dt.timedelta(days=11)\nsold_set = set(df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].article_id.tolist())\n\nsub['other_colors'] = sub['other_colors'].apply(prune, ok_set=sold_set)\n\n# l = []\n# for i in tqdm(range(len(sub))) :\n#     itms = sub['last_purchase'][i].split(\" \")\n#     t = set()\n#     for e in itms:\n# #         print(e)\n#         if e == \"\" : continue\n#         if int(e) not in sold_set : continue\n#         t.add(e)\n# #     break\n#     l.append(\" \".join(list(t)))\n\n# sub['last_purchase'] = l","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:05:04.038144Z","iopub.execute_input":"2022-05-09T01:05:04.038486Z","iopub.status.idle":"2022-05-09T01:05:07.118122Z","shell.execute_reply.started":"2022-05-09T01:05:04.038443Z","shell.execute_reply":"2022-05-09T01:05:07.117012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if use_item2vec:\n    sub['other_colors'] = sort_item2vec('other_colors', max_n=6)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:05:07.119727Z","iopub.execute_input":"2022-05-09T01:05:07.120244Z","iopub.status.idle":"2022-05-09T01:05:07.124899Z","shell.execute_reply.started":"2022-05-09T01:05:07.120194Z","shell.execute_reply":"2022-05-09T01:05:07.124188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub['other_colors'] = cut_items('other_colors', n=2)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:05:07.126598Z","iopub.execute_input":"2022-05-09T01:05:07.127034Z","iopub.status.idle":"2022-05-09T01:05:07.139613Z","shell.execute_reply.started":"2022-05-09T01:05:07.126991Z","shell.execute_reply":"2022-05-09T01:05:07.138841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = sub['other_colors'].apply(lambda s: [] if pd.isna(s) else s.split())\nscore = validation(actual, predicted, grouping, score, index='Other Colors', ignore=(val_week == 105))\n#0.008075(6) -> 0.007619(3) -> 0.007058(2) -> 0.005448(1)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:05:07.140813Z","iopub.execute_input":"2022-05-09T01:05:07.141519Z","iopub.status.idle":"2022-05-09T01:05:14.165948Z","shell.execute_reply.started":"2022-05-09T01:05:07.141481Z","shell.execute_reply":"2022-05-09T01:05:14.163716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ★ペアで購入した商品","metadata":{}},{"cell_type":"code","source":"init_date = last_date - dt.timedelta(days=28)\ntrain = df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].copy()\nu_art = train['article_id'].unique()\nlen(u_art)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:05:19.277896Z","iopub.execute_input":"2022-05-09T01:05:19.278223Z","iopub.status.idle":"2022-05-09T01:05:19.541208Z","shell.execute_reply.started":"2022-05-09T01:05:19.278191Z","shell.execute_reply":"2022-05-09T01:05:19.540206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.groupby(['t_dat', 'customer_id'], as_index=False)['article_id'].agg(list)\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:05:19.719350Z","iopub.execute_input":"2022-05-09T01:05:19.719987Z","iopub.status.idle":"2022-05-09T01:05:22.726171Z","shell.execute_reply.started":"2022-05-09T01:05:19.719937Z","shell.execute_reply":"2022-05-09T01:05:22.725023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = defaultdict(int)\nfor e in train['article_id'] :\n    l = []\n    for v in e :\n        l.append(v)\n    l.sort()\n    for i in range(len(l)) :\n        for j in range(i+1, len(l)) :\n            m[(l[i], l[j])]+=1\n            \nmm = defaultdict(list)\nfor e in m :\n    mm[e[0]].append((m[e], e[1]))\n\nfor e in mm :\n    mm[e].sort(reverse=True)\n    \nlen(mm)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-05-09T01:05:22.728196Z","iopub.execute_input":"2022-05-09T01:05:22.728449Z","iopub.status.idle":"2022-05-09T01:05:27.746538Z","shell.execute_reply.started":"2022-05-09T01:05:22.728418Z","shell.execute_reply":"2022-05-09T01:05:27.745662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"th = 5\nl = []\nfor i in tqdm(range(len(sub))) :\n#     ci = sub['customer_id'][i]\n#     buy = []\n#     if ci in cust_buy :\n#         buy = cust_buy[ci].split(\" \")\n    lp = sub['last_purchase'][i].split(\" \")\n    t = defaultdict(int)\n    if len(lp) != 0 :\n        for e in lp :\n            if e == '' : break\n            j = int(e)\n            for v in mm[j][:24] :\n                if v[0] < th : break # スコアが１以下はスキップ\n                aid = \"0\" + str(v[1])\n#                 if aid in buy : continue\n                t[aid] += v[0]\n    t = list(t.items())\n    t = sorted(t, key=lambda x: x[0] in t)\n    t = [x[0] for x in t]\n    l.append(\" \".join(t[:12]))\n#     if i == 2: break\nl[:10]","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:05:27.748554Z","iopub.execute_input":"2022-05-09T01:05:27.748892Z","iopub.status.idle":"2022-05-09T01:06:42.498966Z","shell.execute_reply.started":"2022-05-09T01:05:27.748851Z","shell.execute_reply":"2022-05-09T01:06:42.497675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['buy_together'] = l\nsub.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:06:42.501275Z","iopub.execute_input":"2022-05-09T01:06:42.501571Z","iopub.status.idle":"2022-05-09T01:06:42.637697Z","shell.execute_reply.started":"2022-05-09T01:06:42.501539Z","shell.execute_reply":"2022-05-09T01:06:42.636482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = sub['buy_together'].apply(lambda s: [] if pd.isna(s) else s.split())\nscore = validation(actual, predicted, grouping, score, index='Buy Together', ignore=(val_week == 105))\n#0.014198 0.014810 0.015348 0.015676","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:06:42.638748Z","iopub.execute_input":"2022-05-09T01:06:42.638981Z","iopub.status.idle":"2022-05-09T01:06:49.712352Z","shell.execute_reply.started":"2022-05-09T01:06:42.638953Z","shell.execute_reply":"2022-05-09T01:06:49.711605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 人気商品 ：年齢層、オンライン／オフライン（４日前まで）グループは1,2,na<-1","metadata":{"papermill":{"duration":0.024242,"end_time":"2022-04-23T02:54:27.993856","exception":false,"start_time":"2022-04-23T02:54:27.969614","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# cdf = pd.read_parquet('../input/handmitem2vec/customers.parquet')\n# cdf.fashion_news_frequency.unique()\n# cdf.head()","metadata":{"papermill":{"duration":0.717592,"end_time":"2022-04-23T02:54:28.792599","exception":false,"start_time":"2022-04-23T02:54:28.075007","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-05T12:54:04.77543Z","iopub.execute_input":"2022-05-05T12:54:04.775778Z","iopub.status.idle":"2022-05-05T12:54:04.778997Z","shell.execute_reply.started":"2022-05-05T12:54:04.775743Z","shell.execute_reply":"2022-05-05T12:54:04.778388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cdf = pd.read_parquet('../input/handmitem2vec/customers.parquet')\ncdf = cdf[['customer_id','age','attribute']]\n#listBin = [-1, 19, 29, 39, 49, 59, 69, 119]\nlistBin = [-1, 19, 34, 49, 119]\n#cdf['cust_feature'] = pd.cut(cdf['age'], listBin).astype(str).str.cat(cdf['attribute'].astype(str))\n#cdf['cust_feature'] = pd.qcut(cdf['age'], 10).astype(str).str.cat(cdf['attribute'].astype(str))\n#cdf['cust_feature'] = pd.cut(cdf['age'], listBin).astype(str)\ncdf['cust_feature'] = pd.qcut(cdf['age'], 10).astype(str)\ncdf = cdf.drop('age', axis=1)\ncdf = cdf.drop('attribute', axis=1)\ncdf['cust_feature'].unique()","metadata":{"papermill":{"duration":0.817427,"end_time":"2022-04-23T02:54:29.635547","exception":false,"start_time":"2022-04-23T02:54:28.81812","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T01:06:55.445320Z","iopub.execute_input":"2022-05-09T01:06:55.446288Z","iopub.status.idle":"2022-05-09T01:06:56.511400Z","shell.execute_reply.started":"2022-05-09T01:06:55.446245Z","shell.execute_reply":"2022-05-09T01:06:56.510402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 4日間の売れた商品リスト\n# init_date = last_date - dt.timedelta(days=4)\n# sold_set = set(df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].article_id.tolist())\n# len(sold_set)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:06:57.856574Z","iopub.execute_input":"2022-05-09T01:06:57.856922Z","iopub.status.idle":"2022-05-09T01:06:57.861988Z","shell.execute_reply.started":"2022-05-09T01:06:57.856891Z","shell.execute_reply":"2022-05-09T01:06:57.860957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"init_date = last_date - dt.timedelta(days=5 - 1)\ngroup_df = pd.concat([cid, group.group.fillna(1)], axis=1) # grouping can be changed\ngroup_df = group_df.merge(cdf, on='customer_id', how='right')\ngroup_df.columns = ['customer_id', 'group', 'cust_feature']\ngroup_df['cust_feature'] = group_df['group'].astype(str).str.cat(group_df['cust_feature'])\ntrain = df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].copy()\\\n    .merge(group_df, on='customer_id', how='left')\\\n    .groupby(['cust_feature', 'article_id']).t_dat.count().reset_index()\n\nitems = defaultdict(str)\nfor g in train.cust_feature.unique():\n#    items[g] = iter_to_str(train.loc[train.cust_feature == g].sort_values('t_dat', ascending=False).article_id.tolist()[:30])\n    items[g] = iter_to_str(train.loc[train.cust_feature == g].sort_values('t_dat', ascending=False).article_id.tolist()[:20])\n\nsub['trend_items'] = group_df.cust_feature.map(items)","metadata":{"papermill":{"duration":12.159356,"end_time":"2022-04-23T02:54:41.822092","exception":false,"start_time":"2022-04-23T02:54:29.662736","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T01:06:58.675097Z","iopub.execute_input":"2022-05-09T01:06:58.675435Z","iopub.status.idle":"2022-05-09T01:07:02.153710Z","shell.execute_reply.started":"2022-05-09T01:06:58.675393Z","shell.execute_reply":"2022-05-09T01:07:02.152854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if use_item2vec:\n    #normal = \"0706016001 0706016002 0372860001 0610776002 0759871002 0464297007 0372860002 0610776001 0399223001 0706016003 0720125001 0156231001\"\n    sub['trend_items'] = sort_item2vec('trend_items', max_n=12)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:02.155372Z","iopub.execute_input":"2022-05-09T01:07:02.155652Z","iopub.status.idle":"2022-05-09T01:07:02.160512Z","shell.execute_reply.started":"2022-05-09T01:07:02.155603Z","shell.execute_reply":"2022-05-09T01:07:02.159599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub['trend_items'] = remove_item2vec('trend_items', max_n=12)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:02.344686Z","iopub.execute_input":"2022-05-09T01:07:02.344999Z","iopub.status.idle":"2022-05-09T01:07:02.349625Z","shell.execute_reply.started":"2022-05-09T01:07:02.344964Z","shell.execute_reply":"2022-05-09T01:07:02.348588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = sub['trend_items'].apply(lambda s: [] if pd.isna(s) else s.split())\nscore = validation(actual, predicted, grouping, score, index='Trend Items', ignore=(val_week == 105))\n# Trend Items\t0.013180\t0.013154\t0.013163\t100.0","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:02.544918Z","iopub.execute_input":"2022-05-09T01:07:02.545204Z","iopub.status.idle":"2022-05-09T01:07:14.444000Z","shell.execute_reply.started":"2022-05-09T01:07:02.545174Z","shell.execute_reply":"2022-05-09T01:07:14.443138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 人気商品 ：属性、~年齢層（大まかな）~、オンライン／オフライン（４日前まで）グループは1,2,na<-1","metadata":{}},{"cell_type":"code","source":"cdf = pd.read_parquet('../input/handmitem2vec/customers.parquet')\ncdf = cdf[['customer_id','age','attribute']]\n#listBin = [-1, 19, 29, 39, 49, 59, 69, 119]\nlistBin = [-1, 19, 40, 119]\n#cdf['cust_feature'] = pd.cut(cdf['age'], listBin).astype(str).str.cat(cdf['attribute'].astype(str))\ncdf['cust_feature'] = cdf['attribute'].astype(str)\ncdf = cdf.drop('age', axis=1)\ncdf = cdf.drop('attribute', axis=1)\ncdf['cust_feature'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:14.445774Z","iopub.execute_input":"2022-05-09T01:07:14.446010Z","iopub.status.idle":"2022-05-09T01:07:14.973881Z","shell.execute_reply.started":"2022-05-09T01:07:14.445983Z","shell.execute_reply":"2022-05-09T01:07:14.972919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 4日間の売れた商品リスト\n# init_date = last_date - dt.timedelta(days=4)\n# sold_set = set(df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].article_id.tolist())\n# len(sold_set)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:14.975602Z","iopub.execute_input":"2022-05-09T01:07:14.976000Z","iopub.status.idle":"2022-05-09T01:07:14.980183Z","shell.execute_reply.started":"2022-05-09T01:07:14.975957Z","shell.execute_reply":"2022-05-09T01:07:14.979214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"init_date = last_date - dt.timedelta(days=5 - 1)\ngroup_df = pd.concat([cid, group.group.fillna(1)], axis=1) # grouping can be changed\ngroup_df = group_df.merge(cdf, on='customer_id', how='right')\ngroup_df.columns = ['customer_id', 'group', 'cust_feature']\ngroup_df['cust_feature'] = group_df['group'].astype(str).str.cat(group_df['cust_feature'])\ntrain = df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].copy()\\\n    .merge(group_df, on='customer_id', how='left')\\\n    .groupby(['cust_feature', 'article_id']).t_dat.count().reset_index()\n\nitems = defaultdict(str)\nfor g in train.cust_feature.unique():\n#    items[g] = iter_to_str(train.loc[train.cust_feature == g].sort_values('t_dat', ascending=False).article_id.tolist()[:30])\n    items[g] = iter_to_str(train.loc[train.cust_feature == g].sort_values('t_dat', ascending=False).article_id.tolist()[:20])\n\nsub['trend_items2'] = group_df.cust_feature.map(items)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:14.983096Z","iopub.execute_input":"2022-05-09T01:07:14.983430Z","iopub.status.idle":"2022-05-09T01:07:18.150286Z","shell.execute_reply.started":"2022-05-09T01:07:14.983386Z","shell.execute_reply":"2022-05-09T01:07:18.148710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if use_item2vec:\n    sub['trend_items2'] = sort_item2vec('trend_items2', max_n=12)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:18.152382Z","iopub.execute_input":"2022-05-09T01:07:18.153352Z","iopub.status.idle":"2022-05-09T01:07:18.158828Z","shell.execute_reply.started":"2022-05-09T01:07:18.153291Z","shell.execute_reply":"2022-05-09T01:07:18.157902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = sub['trend_items2'].apply(lambda s: [] if pd.isna(s) else s.split())\nscore = validation(actual, predicted, grouping, score, index='Trend Items2', ignore=(val_week == 105))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:18.160591Z","iopub.execute_input":"2022-05-09T01:07:18.161576Z","iopub.status.idle":"2022-05-09T01:07:28.560706Z","shell.execute_reply.started":"2022-05-09T01:07:18.161527Z","shell.execute_reply":"2022-05-09T01:07:28.559470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 人気商品 ： Active（４日前まで）グループは1,2,na<-1","metadata":{}},{"cell_type":"code","source":"# cdf = pd.read_parquet('../input/handmitem2vec/customers.parquet')\n# cdf = cdf[['customer_id','attribute','Active']]\n# cdf['cust_feature'] = cdf['attribute'].astype(str).str.cat(cdf['Active'].astype(str))\n# cdf = cdf.drop('attribute', axis=1)\n# cdf['cust_feature'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:28.562505Z","iopub.execute_input":"2022-05-09T01:07:28.562868Z","iopub.status.idle":"2022-05-09T01:07:28.568409Z","shell.execute_reply.started":"2022-05-09T01:07:28.562830Z","shell.execute_reply":"2022-05-09T01:07:28.567009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 4日間の売れた商品リスト\n# init_date = last_date - dt.timedelta(days=4)\n# sold_set = set(df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].article_id.tolist())\n# len(sold_set)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:28.570096Z","iopub.execute_input":"2022-05-09T01:07:28.570476Z","iopub.status.idle":"2022-05-09T01:07:28.586067Z","shell.execute_reply.started":"2022-05-09T01:07:28.570422Z","shell.execute_reply":"2022-05-09T01:07:28.585171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# init_date = last_date - dt.timedelta(days=5 - 1)\n# group_df = pd.concat([cid, group.group.fillna(1)], axis=1) # grouping can be changed\n# group_df = group_df.merge(cdf, on='customer_id', how='right')\n# group_df.columns = ['customer_id', 'group', 'Active', 'cust_feature']\n# group_df['cust_feature'] = group_df['group'].astype(str).str.cat(group_df['cust_feature'])\n\n# train = df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].copy()\\\n#     .merge(group_df, on='customer_id', how='left')\\\n#     .groupby(['cust_feature', 'article_id']).t_dat.count().reset_index()\n\n# items = defaultdict(str)\n# for g in train.cust_feature.unique():\n# #    items[g] = iter_to_str(train.loc[train.cust_feature == g].sort_values('t_dat', ascending=False).article_id.tolist()[:30])\n#     items[g] = iter_to_str(train.loc[train.cust_feature == g].sort_values('t_dat', ascending=False).article_id.tolist()[:20])\n\n# sub['trend_active'] = group_df.cust_feature.map(items)\n# sub['trend_active'] = sub['trend_active'].fillna(\"\")","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:28.587307Z","iopub.execute_input":"2022-05-09T01:07:28.588770Z","iopub.status.idle":"2022-05-09T01:07:28.603531Z","shell.execute_reply.started":"2022-05-09T01:07:28.588725Z","shell.execute_reply":"2022-05-09T01:07:28.602434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cdf = pd.read_parquet('../input/handmitem2vec/customers.parquet')\n# cdf = cdf[cdf['Active'] == 1]\n# cdf = cdf[['customer_id','attribute']]\n# cdf.columns = ['customer_id', 'cust_feature']\n# cdf['cust_feature'].unique()\n\ncdf = pd.read_parquet('../input/handmitem2vec/customers.parquet')\ncdf = cdf[['customer_id','age','attribute','Active']]\n#listBin = [-1, 19, 29, 39, 49, 59, 69, 119]\nlistBin = [-1, 19, 34, 49, 119]\ncdf['cust_feature'] = pd.cut(cdf['age'], listBin).astype(str).str.cat(cdf['attribute'].astype(str))#.str.cat(cdf['Active'].astype(str))\n#cdf['cust_feature'] = pd.qcut(cdf['age'], 10).astype(str).str.cat(cdf['attribute'].astype(str))\n#cdf['cust_feature'] = pd.cut(cdf['age'], listBin).astype(str)\n#cdf['cust_feature'] = pd.qcut(cdf['age'], 10).astype(str)\ncdf = cdf.drop('age', axis=1)\ncdf = cdf.drop('attribute', axis=1)\ncdf = cdf.drop('Active', axis=1)\n\ncdf['cust_feature'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:28.607948Z","iopub.execute_input":"2022-05-09T01:07:28.608383Z","iopub.status.idle":"2022-05-09T01:07:29.988700Z","shell.execute_reply.started":"2022-05-09T01:07:28.608341Z","shell.execute_reply":"2022-05-09T01:07:29.987698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 4日間の売れた商品リスト\n# init_date = last_date - dt.timedelta(days=4)\n# sold_set = set(df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].article_id.tolist())\n# len(sold_set)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:29.990143Z","iopub.execute_input":"2022-05-09T01:07:29.990401Z","iopub.status.idle":"2022-05-09T01:07:29.994397Z","shell.execute_reply.started":"2022-05-09T01:07:29.990369Z","shell.execute_reply":"2022-05-09T01:07:29.993342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"init_date = last_date - dt.timedelta(days=5 - 1)\ngroup_df = pd.concat([cid, group.group.fillna(1)], axis=1) # grouping can be changed\ngroup_df = group_df.merge(cdf, on='customer_id', how='left')\ngroup_df.columns = ['customer_id', 'group', 'cust_feature']\ngroup_df['cust_feature'] = group_df['group'].astype(str).str.cat(group_df['cust_feature'])\n\ntrain = df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].copy()\\\n    .merge(group_df, on='customer_id', how='left')\\\n    .groupby(['cust_feature', 'article_id']).t_dat.count().reset_index()\n\nitems = defaultdict(str)\nfor g in train.cust_feature.unique():\n    items[g] = iter_to_str(train.loc[train.cust_feature == g].sort_values('t_dat', ascending=False).article_id.tolist()[:20])\n\nsub['trend_ageatt'] = group_df.cust_feature.map(items)\nsub['trend_ageatt'] = sub['trend_ageatt'].fillna(\"\")","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:29.996073Z","iopub.execute_input":"2022-05-09T01:07:29.996618Z","iopub.status.idle":"2022-05-09T01:07:33.520140Z","shell.execute_reply.started":"2022-05-09T01:07:29.996582Z","shell.execute_reply":"2022-05-09T01:07:33.518994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if use_item2vec:\n    sub['trend_ageatt'] = sort_item2vec('trend_ageatt', max_n=12)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:33.521828Z","iopub.execute_input":"2022-05-09T01:07:33.522094Z","iopub.status.idle":"2022-05-09T01:07:33.528233Z","shell.execute_reply.started":"2022-05-09T01:07:33.522060Z","shell.execute_reply":"2022-05-09T01:07:33.526137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = sub['trend_ageatt'].apply(lambda s: [] if pd.isna(s) else s.split())\nscore = validation(actual, predicted, grouping, score, index='Trend age+attr', ignore=(val_week == 105))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:33.530191Z","iopub.execute_input":"2022-05-09T01:07:33.530469Z","iopub.status.idle":"2022-05-09T01:07:43.756376Z","shell.execute_reply.started":"2022-05-09T01:07:33.530438Z","shell.execute_reply":"2022-05-09T01:07:43.755236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Woman ","metadata":{}},{"cell_type":"code","source":"# cdf = pd.read_parquet('../input/handmitem2vec/customers.parquet')\n# #cdf = cdf[cdf['fashion_news_frequency'] >= 1]\n# cdf = cdf[cdf['attribute'] == 'Woman']\n# # cdf = cdf[cdf['Active'] == 1]\n# cdf","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:43.757741Z","iopub.execute_input":"2022-05-09T01:07:43.758235Z","iopub.status.idle":"2022-05-09T01:07:43.762238Z","shell.execute_reply.started":"2022-05-09T01:07:43.758190Z","shell.execute_reply":"2022-05-09T01:07:43.761284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cdf = pd.read_parquet('../input/handmitem2vec/customers.parquet')\n#cdf = cdf[cdf['fashion_news_frequency'] >= 1]\ncdf = cdf[cdf['attribute'] == 'Woman']\n# cdf = cdf[cdf['Active'] == 1]\ncdf = cdf[['customer_id','age']]\n# cdf.columns = ['customer_id', 'cust_feature']\n# cdf['cust_feature'].unique()\n#listBin = [-1, 19, 29, 39, 49, 59, 69, 119]\nlistBin = [-1, 19, 34, 49, 119]\n#cdf['cust_feature'] = pd.cut(cdf['age'], listBin).astype(str).str.cat(cdf['Active'].astype(str))\n#cdf['cust_feature'] = pd.qcut(cdf['age'], 10).astype(str).str.cat(cdf['attribute'].astype(str))\ncdf['cust_feature'] = pd.cut(cdf['age'], listBin).astype(str)\n#cdf['cust_feature'] = pd.qcut(cdf['age'], 10).astype(str)\ncdf = cdf.drop('age', axis=1)\ncdf['cust_feature'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:43.763885Z","iopub.execute_input":"2022-05-09T01:07:43.764871Z","iopub.status.idle":"2022-05-09T01:07:44.476453Z","shell.execute_reply.started":"2022-05-09T01:07:43.764789Z","shell.execute_reply":"2022-05-09T01:07:44.474575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 4日間の売れた商品リスト\n# init_date = last_date - dt.timedelta(days=4)\n# sold_set = set(df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].article_id.tolist())\n# len(sold_set)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:44.477628Z","iopub.execute_input":"2022-05-09T01:07:44.477880Z","iopub.status.idle":"2022-05-09T01:07:44.482364Z","shell.execute_reply.started":"2022-05-09T01:07:44.477844Z","shell.execute_reply":"2022-05-09T01:07:44.481664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"init_date = last_date - dt.timedelta(days=5 - 1)\ngroup_df = pd.concat([cid, group.group.fillna(1)], axis=1) # grouping can be changed\ngroup_df = group_df.merge(cdf, on='customer_id', how='left')\ngroup_df.columns = ['customer_id', 'group', 'cust_feature']\ngroup_df['cust_feature'] = group_df['group'].astype(str).str.cat(group_df['cust_feature'])\n\ntrain = df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].copy()\\\n    .merge(group_df, on='customer_id', how='left')\\\n    .groupby(['cust_feature', 'article_id']).t_dat.count().reset_index()\n\nitems = defaultdict(str)\nfor g in train.cust_feature.unique():\n    items[g] = iter_to_str(train.loc[train.cust_feature == g].sort_values('t_dat', ascending=False).article_id.tolist()[:20])\n\nsub['trend_woman'] = group_df.cust_feature.map(items)\nsub['trend_woman'] = sub['trend_woman'].fillna(\"\")","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:44.483317Z","iopub.execute_input":"2022-05-09T01:07:44.484187Z","iopub.status.idle":"2022-05-09T01:07:47.958267Z","shell.execute_reply.started":"2022-05-09T01:07:44.484151Z","shell.execute_reply":"2022-05-09T01:07:47.957209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if use_item2vec:\n    sub['trend_woman'] = sort_item2vec('trend_woman', max_n=12)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:47.959546Z","iopub.execute_input":"2022-05-09T01:07:47.959915Z","iopub.status.idle":"2022-05-09T01:07:47.964943Z","shell.execute_reply.started":"2022-05-09T01:07:47.959879Z","shell.execute_reply":"2022-05-09T01:07:47.963774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = sub['trend_woman'].apply(lambda s: [] if pd.isna(s) else s.split())\nscore = validation(actual, predicted, grouping, score, index='Trend Woman', ignore=(val_week == 105))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:47.966427Z","iopub.execute_input":"2022-05-09T01:07:47.966720Z","iopub.status.idle":"2022-05-09T01:07:57.847716Z","shell.execute_reply.started":"2022-05-09T01:07:47.966686Z","shell.execute_reply":"2022-05-09T01:07:57.846723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### item2vecだけでの予測","metadata":{}},{"cell_type":"code","source":"# # 11日間の売れた商品リスト\n# init_date = last_date - dt.timedelta(days=11)\n# sold_set = set(df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].article_id.tolist())","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:57.848918Z","iopub.execute_input":"2022-05-09T01:07:57.849159Z","iopub.status.idle":"2022-05-09T01:07:57.853391Z","shell.execute_reply.started":"2022-05-09T01:07:57.849129Z","shell.execute_reply":"2022-05-09T01:07:57.852442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# l = []\n# for i in tqdm(range(len(sub))) :\n#     cid = sub.iloc[i]['customer_id']\n#     if cid not in cust_buy :\n#         l.append(\"\")\n#         continue\n#     item = cust_buy[cid].split(\" \")\n#     r = model.wv.most_similar(item, topn=20)\n#     t = []\n#     for x, p in r :\n#         if int(x) in sold_set :\n#             t.append(x)\n#     l.append(\" \".join(t[:12]))\n# #     if i == 10 : break\n\n# sub['item2vec'] = l","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:57.854717Z","iopub.execute_input":"2022-05-09T01:07:57.854928Z","iopub.status.idle":"2022-05-09T01:07:57.867765Z","shell.execute_reply.started":"2022-05-09T01:07:57.854902Z","shell.execute_reply":"2022-05-09T01:07:57.866723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicted = sub['item2vec'].apply(lambda s: [] if pd.isna(s) else s.split())\n# score = validation(actual, predicted, grouping, score, index='item2vec', ignore=(val_week == 105))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:57.870095Z","iopub.execute_input":"2022-05-09T01:07:57.870469Z","iopub.status.idle":"2022-05-09T01:07:57.881673Z","shell.execute_reply.started":"2022-05-09T01:07:57.870414Z","shell.execute_reply":"2022-05-09T01:07:57.880650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cnt = 0\n# for i in range(len(actual)) :\n    \n#     if len(actual[i]) != 0 :\n#         cid = sub['customer_id'][i]\n# #         print(i, cid)\n# #         print(actual[i])\n#         if cid not in cust_buy : continue\n#         item = cust_buy[cid].split(\" \")\n# #         print(item[-1])\n#         r = model.wv.most_similar(item, topn=12)\n#         t = []\n#         for x, p in r :\n#             if x in actual[i] :\n#                 print(\"HIT\", i, x)\n#                 cnt+=1\n#             if int(x) in sold_set :\n#                 t.append(x)\n# #         print(r, t)\n# #         print(t)\n# #         cnt+=1\n# #         print(\"-\"*40)\n        \n#     if i == 100000: break\n# cnt/100000","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:57.882741Z","iopub.execute_input":"2022-05-09T01:07:57.883188Z","iopub.status.idle":"2022-05-09T01:07:57.896948Z","shell.execute_reply.started":"2022-05-09T01:07:57.883150Z","shell.execute_reply":"2022-05-09T01:07:57.895735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.035409,"end_time":"2022-04-23T02:54:41.883257","exception":false,"start_time":"2022-04-23T02:54:41.847848","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## kangol","metadata":{}},{"cell_type":"code","source":"# articles_df = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\",dtype=str, encoding='utf8')","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:57.898466Z","iopub.execute_input":"2022-05-09T01:07:57.898810Z","iopub.status.idle":"2022-05-09T01:07:57.914241Z","shell.execute_reply.started":"2022-05-09T01:07:57.898775Z","shell.execute_reply":"2022-05-09T01:07:57.913274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# xHM = []\n# for i, e in enumerate(articles_df['detail_desc'].fillna(\"\")) :\n#     if 'Kangol x H&M' in e:\n#         print(articles_df.iloc[i]['article_id'], e)\n#         xHM.append(articles_df.iloc[i]['article_id'])\n# #     elif 'Whooli Chen x H&M' in e :\n# #         print(articles_df.iloc[i]['article_id'], e)\n# #         xHM.append(articles_df.iloc[i]['article_id'])\n        \n# xHM\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-05-09T01:07:57.915348Z","iopub.execute_input":"2022-05-09T01:07:57.916077Z","iopub.status.idle":"2022-05-09T01:07:57.926518Z","shell.execute_reply.started":"2022-05-09T01:07:57.916043Z","shell.execute_reply":"2022-05-09T01:07:57.925916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ll = \" \".join(xHM[-12:])\n# sub['kangol'] = [ll for _ in range(len(sub))]","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:57.927982Z","iopub.execute_input":"2022-05-09T01:07:57.928431Z","iopub.status.idle":"2022-05-09T01:07:57.940332Z","shell.execute_reply.started":"2022-05-09T01:07:57.928400Z","shell.execute_reply":"2022-05-09T01:07:57.939670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub['kangol'] = sort_item2vec('kangol', max_n=12)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:57.944688Z","iopub.execute_input":"2022-05-09T01:07:57.945336Z","iopub.status.idle":"2022-05-09T01:07:57.953244Z","shell.execute_reply.started":"2022-05-09T01:07:57.945296Z","shell.execute_reply":"2022-05-09T01:07:57.952181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicted = sub['kangol'].apply(lambda s: [] if pd.isna(s) else s.split())\n# score = validation(actual, predicted, grouping, score, index='kangol', ignore=(val_week == 105))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:57.954812Z","iopub.execute_input":"2022-05-09T01:07:57.955864Z","iopub.status.idle":"2022-05-09T01:07:57.966837Z","shell.execute_reply.started":"2022-05-09T01:07:57.955821Z","shell.execute_reply":"2022-05-09T01:07:57.966053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:57.968074Z","iopub.execute_input":"2022-05-09T01:07:57.968682Z","iopub.status.idle":"2022-05-09T01:07:57.979665Z","shell.execute_reply.started":"2022-05-09T01:07:57.968623Z","shell.execute_reply":"2022-05-09T01:07:57.978576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 数年間定番アイテム","metadata":{}},{"cell_type":"code","source":"# dayly sale\n\n# 2018\nday = 20\nstart = dt.datetime(2018,9,day,0,0)\nend = dt.datetime(2018,9,day,23,59)\nx = df.loc[(df.t_dat >= start) & (df.t_dat <= end)].article_id.tolist()\n\npopular = set(x)\nprint(len(x))\n\nfor day in range(20, 31, 1) :\n    start = dt.datetime(2018,9,day,0,0)\n    end = dt.datetime(2018,9,day,23,59)\n    x = df.loc[(df.t_dat >= start) & (df.t_dat <= end)].article_id.tolist()\n    popular &= set(x)\n    print(start, end, len(x), len(popular))\n\n# 2019\nfor day in range(15, 30, 1) :\n    start = dt.datetime(2019,9,day,0,0)\n    end = dt.datetime(2019,9,day,23,59)\n    x = df.loc[(df.t_dat >= start) & (df.t_dat <= end)].article_id.tolist()\n    popular &= set(x)\n    print(start, end, len(x), len(popular))\n\n# 2020\nfor day in range(15, 23, 1) :\n    start = dt.datetime(2020,9,day,0,0)\n    end = dt.datetime(2020,9,day,23,59)\n    x = df.loc[(df.t_dat >= start) & (df.t_dat <= end)].article_id.tolist()\n    popular &= set(x)\n    print(start, end, len(x), len(popular))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:07:57.981010Z","iopub.execute_input":"2022-05-09T01:07:57.981271Z","iopub.status.idle":"2022-05-09T01:08:05.646930Z","shell.execute_reply.started":"2022-05-09T01:07:57.981230Z","shell.execute_reply":"2022-05-09T01:08:05.645919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nfrom collections import defaultdict\n\nl = []\nfor i in tqdm(range(len(sub))):\n    c_id = sub['customer_id'][i]\n    if c_id not in cust_buy :\n        l.append(\"\")\n        continue\n        \n    t = defaultdict(int)\n    li = cust_buy[c_id].split(\" \")\n    #li.reverse()\n#     print(cid, li)\n    for e in li:\n        if int(e) in popular :\n            t[e] += 1\n#     if len(t) != 0 : print(cid, t)\n#     print(t)\n    ti = []\n    for e in t :\n        ti.append([t[e],e])\n    ti.sort(reverse=True)\n    ti = list(map(lambda x: x[1], ti)) \n    l.append(\" \".join(ti))\n#     if i == 100: break\nsub['popular_items'] = l ","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:08:05.648687Z","iopub.execute_input":"2022-05-09T01:08:05.648959Z","iopub.status.idle":"2022-05-09T01:08:38.498121Z","shell.execute_reply.started":"2022-05-09T01:08:05.648928Z","shell.execute_reply":"2022-05-09T01:08:38.497158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = sub['popular_items'].apply(lambda s: [] if pd.isna(s) else s.split())\n#score = validation(actual, predicted, grouping, index='anytime popular', ignore=(val_week == 105))\nscore = validation(actual, predicted, grouping, score, index='Popular Items', ignore=(val_week == 105))\n#Popular Items\t0.003576\t0.002567\t0.002901\t3.4","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:08:38.501674Z","iopub.execute_input":"2022-05-09T01:08:38.501957Z","iopub.status.idle":"2022-05-09T01:08:46.786232Z","shell.execute_reply.started":"2022-05-09T01:08:38.501927Z","shell.execute_reply":"2022-05-09T01:08:46.785585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 嗜好の似たカスタマーが購入したもの","metadata":{}},{"cell_type":"code","source":"if val_week == 105 :\n    uucf = pd.read_csv('../input/hm-cf-data/uucf-2-0814.csv')\nelse :\n    print(\"104\")\n#    uucf = pd.read_csv('../input/uucf104.csv')\n    uucf = pd.read_csv('../input/hm-cf-data/uucf104-2-0814.csv')\n\nuucf.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:08:46.787225Z","iopub.execute_input":"2022-05-09T01:08:46.788142Z","iopub.status.idle":"2022-05-09T01:08:52.054110Z","shell.execute_reply.started":"2022-05-09T01:08:46.788099Z","shell.execute_reply":"2022-05-09T01:08:52.053002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = []\nfor i in tqdm(range(len(uucf))) :\n    s = uucf['prediction'][i]\n    if s[0] == '[' :\n        s = \" \".join(eval(s))\n    else : s = \"\"\n    l.append(s)\nsub['uucf'] = l\ndel uucf","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:08:52.057700Z","iopub.execute_input":"2022-05-09T01:08:52.057953Z","iopub.status.idle":"2022-05-09T01:09:09.710557Z","shell.execute_reply.started":"2022-05-09T01:08:52.057924Z","shell.execute_reply":"2022-05-09T01:09:09.709354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = sub['uucf'].apply(lambda s: [] if pd.isna(s) else s.split())\nscore = validation(actual, predicted, grouping, score, index='uucf Items', ignore=(val_week == 105))\n# 8352 9737 8653\n# 0.011568","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:09.711922Z","iopub.execute_input":"2022-05-09T01:09:09.712256Z","iopub.status.idle":"2022-05-09T01:09:16.289959Z","shell.execute_reply.started":"2022-05-09T01:09:09.712223Z","shell.execute_reply":"2022-05-09T01:09:16.288802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## another uucf","metadata":{}},{"cell_type":"code","source":"from collections import defaultdict\n\ndef remove_(in_df, num = 10000) :\n    m = defaultdict(int)\n    target, tmax = \"\", 0\n    for i in tqdm(range(len(in_df))) :\n        p = in_df['prediction'][i]\n        m[p] += 1\n        if m[p] > tmax :\n            tmax = m[p]\n            target = p\n\n    l = []\n    skip = 0\n    for i in tqdm(range(len(in_df))) :\n        p = in_df['prediction'][i]\n        if m[p] >= num:\n            l.append(\"\")\n            skip += 1\n        else :\n            l.append(p)\n    print(\"skip = \", skip)\n    return l","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:16.291538Z","iopub.execute_input":"2022-05-09T01:09:16.291806Z","iopub.status.idle":"2022-05-09T01:09:16.300560Z","shell.execute_reply.started":"2022-05-09T01:09:16.291776Z","shell.execute_reply":"2022-05-09T01:09:16.299617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if val_week == 105 :\n    uucf = pd.read_csv('../input/hm-cf-data/cf-v2.csv').fillna(\"\")\nelse :\n    print(\"104\")\n#    uucf = pd.read_csv('../input/uucf104.csv')\n    uucf = pd.read_csv('../input/hm-cf-data/cf104-v2.csv').fillna(\"\")\n\n# uucf['prediction'] = remove_(uucf, 10)\n\nsub['uucf2'] = uucf['prediction']\ndisplay(uucf.head(10))\ndel uucf","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:16.302124Z","iopub.execute_input":"2022-05-09T01:09:16.302393Z","iopub.status.idle":"2022-05-09T01:09:20.780780Z","shell.execute_reply.started":"2022-05-09T01:09:16.302361Z","shell.execute_reply":"2022-05-09T01:09:20.779727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = sub['uucf2'].apply(lambda s: [] if pd.isna(s) else s.split())\nscore = validation(actual, predicted, grouping, score, index='uucf2 Items', ignore=(val_week == 105))\n# 8352 9737 8653\n# 0.011568 0.011552 0.011568","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:20.782158Z","iopub.execute_input":"2022-05-09T01:09:20.782489Z","iopub.status.idle":"2022-05-09T01:09:27.674487Z","shell.execute_reply.started":"2022-05-09T01:09:20.782441Z","shell.execute_reply":"2022-05-09T01:09:27.673342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:27.676320Z","iopub.execute_input":"2022-05-09T01:09:27.676695Z","iopub.status.idle":"2022-05-09T01:09:27.704375Z","shell.execute_reply.started":"2022-05-09T01:09:27.676625Z","shell.execute_reply":"2022-05-09T01:09:27.703702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 11日間の売れた商品リスト\n# init_date = last_date - dt.timedelta(days=11)\n# last_sell = df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)]\n# sell =  last_sell.groupby('customer_id')\\\n#      .article_id.apply(iter_to_str).reset_index()\\\n#      .merge(cid, on='customer_id', how='right').article_id.fillna('')","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:27.705649Z","iopub.execute_input":"2022-05-09T01:09:27.706571Z","iopub.status.idle":"2022-05-09T01:09:27.710369Z","shell.execute_reply.started":"2022-05-09T01:09:27.706529Z","shell.execute_reply":"2022-05-09T01:09:27.709677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# last_buy = defaultdict(list)\n# for i in tqdm(range(len(sub))) :\n#     x = set(sell[i].split(\" \")) - set('')\n#     last_buy[sub['customer_id'][i]] = x","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:27.711469Z","iopub.execute_input":"2022-05-09T01:09:27.712063Z","iopub.status.idle":"2022-05-09T01:09:27.730124Z","shell.execute_reply.started":"2022-05-09T01:09:27.712025Z","shell.execute_reply":"2022-05-09T01:09:27.729056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pickle\n# with open(\"../input/handmitem2vec/cust_similar.pkl\", \"rb\") as f :\n#     cust_similar = pickle.load(f)\n# len(cust_similar)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:27.733605Z","iopub.execute_input":"2022-05-09T01:09:27.734185Z","iopub.status.idle":"2022-05-09T01:09:27.743427Z","shell.execute_reply.started":"2022-05-09T01:09:27.734145Z","shell.execute_reply":"2022-05-09T01:09:27.742711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# l = []\n# for i in tqdm(range(len(sub))): \n#     c = sub['customer_id'][i]\n#     x = cust_similar[c]\n#     y = set()\n#     if c in cust_buy :\n#         y = set(cust_buy[c].split(\" \"))\n#     t = []\n#     for e, _ in x :\n#         if e in y : continue # 既に購入しているものはスキップ\n#         if last_buy[e] == {''}: continue\n#         t += last_buy[e]\n#     l.append(\" \".join(set(t)))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:27.744485Z","iopub.execute_input":"2022-05-09T01:09:27.745084Z","iopub.status.idle":"2022-05-09T01:09:27.757768Z","shell.execute_reply.started":"2022-05-09T01:09:27.745040Z","shell.execute_reply":"2022-05-09T01:09:27.756878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub['similar_customers'] = l","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:27.759032Z","iopub.execute_input":"2022-05-09T01:09:27.759270Z","iopub.status.idle":"2022-05-09T01:09:27.770527Z","shell.execute_reply.started":"2022-05-09T01:09:27.759243Z","shell.execute_reply":"2022-05-09T01:09:27.769513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#if use_item2vec:\n#    sub['similar_customers'] = sort_item2vec('similar_customers', max_n=12)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:27.771798Z","iopub.execute_input":"2022-05-09T01:09:27.772037Z","iopub.status.idle":"2022-05-09T01:09:27.784185Z","shell.execute_reply.started":"2022-05-09T01:09:27.772007Z","shell.execute_reply":"2022-05-09T01:09:27.783121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicted = sub['similar_customers'].apply(lambda s: [] if pd.isna(s) else s.split())\n# score = validation(actual, predicted, grouping, score, index='Similar Customers', ignore=(val_week == 105))","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:27.785731Z","iopub.execute_input":"2022-05-09T01:09:27.786441Z","iopub.status.idle":"2022-05-09T01:09:27.796623Z","shell.execute_reply.started":"2022-05-09T01:09:27.786406Z","shell.execute_reply":"2022-05-09T01:09:27.795744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## save","metadata":{}},{"cell_type":"code","source":"if is_save :\n    print(\"SAVE.....\")\n    if val_week == 105: \n        sub.to_parquet('submission_all.parquet', index=False)\n    else :\n        sub.to_parquet('submission_all_104.parquet', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:27.798444Z","iopub.execute_input":"2022-05-09T01:09:27.799123Z","iopub.status.idle":"2022-05-09T01:09:27.810035Z","shell.execute_reply.started":"2022-05-09T01:09:27.799073Z","shell.execute_reply":"2022-05-09T01:09:27.809265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## memory clear","metadata":{}},{"cell_type":"code","source":"# import gc\n# del model\n# del cust_buy\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-05-09T01:09:27.811140Z","iopub.execute_input":"2022-05-09T01:09:27.811817Z","iopub.status.idle":"2022-05-09T01:09:27.824846Z","shell.execute_reply.started":"2022-05-09T01:09:27.811778Z","shell.execute_reply":"2022-05-09T01:09:27.823733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 予測をブレンド","metadata":{"papermill":{"duration":0.025811,"end_time":"2022-04-23T02:54:41.936402","exception":false,"start_time":"2022-04-23T02:54:41.910591","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# 11日間の売れた商品リスト\ninit_date = last_date - dt.timedelta(days=2)\nsold_set = set(df.loc[(df.t_dat >= init_date) & (df.t_dat <= last_date)].article_id.tolist())","metadata":{"papermill":{"duration":0.320912,"end_time":"2022-04-23T02:54:42.284143","exception":false,"start_time":"2022-04-23T02:54:41.963231","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T01:09:27.828192Z","iopub.execute_input":"2022-05-09T01:09:27.829160Z","iopub.status.idle":"2022-05-09T01:09:28.054224Z","shell.execute_reply.started":"2022-05-09T01:09:27.829108Z","shell.execute_reply":"2022-05-09T01:09:28.053367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 11日間に販売実績のある商品だけ対象\n# targets = ['last_purchase', 'popular_items', 'other_colors', \n#            'trend_woman', 'trend_ageatt', 'trend_items2', 'trend_items' ]\n# weights = [1000, 5, 10, \n#            2, 2, 1.5, 1]\n\n\n# targets = ['last_purchase', 'other_colors', 'popular_items', \n#            'trend_woman', 'trend_ageatt', 'trend_items2',  'uucf', 'trend_items' ]\n# weights = [1000, 10, 5, \n#            1, 1, 1.3, 3, 0.5]\n\n# targets = ['last_purchase', 'uucf', 'other_colors', 'popular_items', \n#            'trend_woman', 'trend_ageatt', 'trend_items2', 'trend_items' ]\n# weights = [1000, 10, 10, 10, \n#            1, 1, 1.3, 0.5]\ntargets = ['last_purchase', 'other_colors', 'popular_items', 'uucf', 'uucf2', 'buy_together',\n           'trend_woman', 'trend_ageatt', 'trend_items2', 'trend_items' ]\nweights = [10, 2 ,2, 2, 2, 1, \n           1, 1, 1.3, 0.5]\n\n\nsub['prediction'] = sub[targets].apply(blend, w=weights, axis=1, k=100).apply(prune, ok_set=sold_set)\npredicted = sub.prediction.apply(lambda s: [] if pd.isna(s) else s.split())\nscore = validation(actual, predicted, grouping, score, index='Prediction', ignore=(val_week == 105))\n\n#score が定義されていないとき（loadした時）\n#score = validation(actual, predicted, grouping, index='Prediction', ignore=(val_week == 105))\n\n\n#Prediction\t0.019646\t0.031350\t0.027481\t100.0 [100, 5, 10, 2, 1] without itemvec\n#Prediction\t0.020333\t0.032050\t0.028177\t100.0 [100, 5, 10, 2, 1] with itemvec  \n\n#Prediction\t0.019675\t0.031394\t0.027520\t100.0 [100, 5, 10, 2, 1, 2]\n#Prediction\t0.020852\t0.032008\t0.028320\t100.0 [100, 5, 10, 2, 1, 2] test\n\n#Prediction\t0.019834\t0.031329\t0.027529\t100.0\n# 7 predictions with item2vec \n# first 3-items : 22754 23247 \n#       4-items : 15229 15732 15836 15873\n#     all items : 28536 \n# 0.028359 0.028392 28464 0.028675 0.029264 0.029302 0.029368 0.029332 0.029333","metadata":{"papermill":{"duration":107.621413,"end_time":"2022-04-23T02:56:29.931535","exception":false,"start_time":"2022-04-23T02:54:42.310122","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-09T01:09:28.056277Z","iopub.execute_input":"2022-05-09T01:09:28.056668Z","iopub.status.idle":"2022-05-09T01:12:47.937566Z","shell.execute_reply.started":"2022-05-09T01:09:28.056602Z","shell.execute_reply":"2022-05-09T01:12:47.936686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# targets = ['last_purchase', 'buy_together', 'other_colors', 'popular_items', 'uucf', 'uucf2']\n# weights = [10, 1, 2, 2 ,2, 1]\n\n# sub['customer'] = sub[targets].apply(blend, w=weights, axis=1, k=100).apply(prune, ok_set=sold_set)\n# predicted = sub.customer.apply(lambda s: [] if pd.isna(s) else s.split())\n# score = validation(actual, predicted, grouping, score, index='Customer', ignore=(val_week == 105))\n# # 24625 0.024752 0.024641 0.024755 ","metadata":{"execution":{"iopub.status.busy":"2022-05-05T13:00:20.389568Z","iopub.execute_input":"2022-05-05T13:00:20.389789Z","iopub.status.idle":"2022-05-05T13:00:20.394619Z","shell.execute_reply.started":"2022-05-05T13:00:20.38976Z","shell.execute_reply":"2022-05-05T13:00:20.393776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# targets = ['trend_woman', 'trend_ageatt', 'trend_items2', 'trend_items']\n# weights = [1,1,1,1]\n# sub['trend'] = sub[targets].apply(blend, w=weights, axis=1, k=100).apply(prune, ok_set=sold_set)\n# predicted = sub.trend.apply(lambda s: [] if pd.isna(s) else s.split())\n# score = validation(actual, predicted, grouping, score, index='Trend', ignore=(val_week == 105))","metadata":{"execution":{"iopub.status.busy":"2022-05-05T13:00:20.396313Z","iopub.execute_input":"2022-05-05T13:00:20.396607Z","iopub.status.idle":"2022-05-05T13:00:20.412562Z","shell.execute_reply.started":"2022-05-05T13:00:20.396566Z","shell.execute_reply":"2022-05-05T13:00:20.411717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# targets = ['customer', 'trend']\n# weights = [4,1]\n# sub['prediction'] = sub[targets].apply(blend, w=weights, axis=1, k=100).apply(prune, ok_set=sold_set)\n# predicted = sub.prediction.apply(lambda s: [] if pd.isna(s) else s.split())\n# score = validation(actual, predicted, grouping, score, index='Prediction', ignore=(val_week == 105))","metadata":{"execution":{"iopub.status.busy":"2022-05-05T13:00:20.414042Z","iopub.execute_input":"2022-05-05T13:00:20.414325Z","iopub.status.idle":"2022-05-05T13:00:20.423554Z","shell.execute_reply.started":"2022-05-05T13:00:20.414284Z","shell.execute_reply":"2022-05-05T13:00:20.422851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(sub['prediction'][0].split())\n#sns.barplot(data=score, x='All', y=score.index)","metadata":{"papermill":{"duration":0.035039,"end_time":"2022-04-23T02:56:29.99383","exception":false,"start_time":"2022-04-23T02:56:29.958791","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-05T13:00:20.425093Z","iopub.execute_input":"2022-05-05T13:00:20.425381Z","iopub.status.idle":"2022-05-05T13:00:20.437782Z","shell.execute_reply.started":"2022-05-05T13:00:20.425341Z","shell.execute_reply":"2022-05-05T13:00:20.43696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if val_week == 105: sub[['customer_id', 'prediction']].to_csv('submission.csv', index=False)","metadata":{"papermill":{"duration":13.19832,"end_time":"2022-04-23T02:56:43.218549","exception":false,"start_time":"2022-04-23T02:56:30.020229","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-05-05T13:00:20.439132Z","iopub.execute_input":"2022-05-05T13:00:20.4394Z","iopub.status.idle":"2022-05-05T13:00:20.44754Z","shell.execute_reply.started":"2022-05-05T13:00:20.43937Z","shell.execute_reply":"2022-05-05T13:00:20.446972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 全部埋まっているかチェック","metadata":{}},{"cell_type":"code","source":"for i in tqdm(range(len(sub))) :\n    x = sub['prediction'][i].split(\" \")\n    if len(x) != 12 :\n        print(sub['customer_id'][i], x)\n        break","metadata":{"execution":{"iopub.status.busy":"2022-05-05T13:00:20.448405Z","iopub.execute_input":"2022-05-05T13:00:20.448961Z","iopub.status.idle":"2022-05-05T13:00:33.375431Z","shell.execute_reply.started":"2022-05-05T13:00:20.448921Z","shell.execute_reply":"2022-05-05T13:00:33.374495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}