{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Buy2Buy!!!","metadata":{}},{"cell_type":"markdown","source":"実際に購入された商品、カートに入れられた商品のみを用いて、それぞれの商品のペアを作成する。\n\nこれにより、よく一緒に購入される商品ランキングが作れる。","metadata":{}},{"cell_type":"markdown","source":"# 1.ライブラリのインポート","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport datetime\nimport time\nimport gc\nimport glob\nimport numpy as np\nimport copy","metadata":{"execution":{"iopub.status.busy":"2022-12-27T10:44:10.115351Z","iopub.execute_input":"2022-12-27T10:44:10.115791Z","iopub.status.idle":"2022-12-27T10:44:10.122445Z","shell.execute_reply.started":"2022-12-27T10:44:10.115759Z","shell.execute_reply":"2022-12-27T10:44:10.121246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2.行列の作成","metadata":{}},{"cell_type":"code","source":"%%time\n# 与えられたパスのファイルを読み込む関数\ndef read_file(f):\n    return pd.DataFrame( data_cache[f] )\n\n\n# ファイルのの容量をできるだけ小さくする関数。\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = (df.ts/1000).astype('int32')\n    df['type'] = df['type'].map(type_labels).astype('int8')\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-12-27T10:44:12.994533Z","iopub.execute_input":"2022-12-27T10:44:12.995044Z","iopub.status.idle":"2022-12-27T10:45:18.716927Z","shell.execute_reply.started":"2022-12-27T10:44:12.995004Z","shell.execute_reply":"2022-12-27T10:45:18.715594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_cache = {}\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('../input/otto-chunk-data-inparquet-format/*_parquet/*')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f in files:\n    data_cache[f] = read_file_to_cache(f)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"READ_CT = 5\nCHUNK = int( np.ceil( len(files)/6 ))\nprint(f'We will process {len(files)} files, in groups of {READ_CT} and chunks of {CHUNK}.')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# メモリエラーにならない最小の回数に分けて処理をする。\nDISK_PIECES = 1\nSIZE = 1.86e6/DISK_PIECES\n\nfor PART in range(DISK_PIECES):\n    print()\n    print('### DISK PART',PART+1)\n    \n    for j in range(6):\n        a = j * CHUNK\n        b = min((j + 1) * CHUNK, len(files))\n        print(f'Processing file {a} thru {b-1} in groups of {j+1} / 6')\n        \n        for k in range(a, b, READ_CT):\n            df = [read_file(files[k])]\n            for i in range(1, READ_CT):\n                if k + i < b:\n                    df.append(read_file(files[k+i]))\n                    \n            # リスト型として保存していたデータフレームをデータフレーム型として結合する。\n            df = pd.concat(df, ignore_index=True, axis=0)\n            \n            # カートに入れられた商品と、購入された商品だけを取り出す。\n            df = df.loc[df['type'].isin([1, 2])]\n            \n            # セッションは昇順、タイムスタンプは降順で並び替える。\n            df = df.sort_values(['session','ts'],ascending=[True,False])\n            df = df.reset_index(drop=True)\n            \n            # セッションごとに通し番号を付ける。\n            # すべての商品に対して計算するとメモリエラーになるので、直近30件のみを用いて行列を作成する。\n            df['n'] = df.groupby('session').cumcount()\n            df = df.loc[df['n']<30].drop(['n'], axis=1)\n            df = df.merge(df, on='session')\n            \n            # このようにmergeをすると、aid_xとaid_yが同じ行が作成されてしまう。\n            # その行を削除すると同時に、次の行動までの時間が2週間以上空いている場合も削除する。\n            df = df.loc[((df.ts_x - df.ts_y).abs() < 14 * 24 * 60 * 60) & (df.aid_x != df.aid_y)]\n            \n            # メモリエラーにならないようにaid_xが指定の値の間のものを順番に処理していく。\n            df = df.loc[(PART * SIZE <= df.aid_x) & (df.aid_x < (PART + 1) * SIZE)]\n            \n            # 重複行の削除。\n            df = df[['session', 'aid_x', 'aid_y', 'type_y']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n            \n            # 重みはすべて1にする。\n            df['wgt'] = 1\n            \n            # 必要な列のみ取り出す。\n            df = df[['aid_x', 'aid_y', 'wgt']]\n            \n            # 使用メモリを抑えるために型変換。\n            df.wgt = df.wgt.astype('float32')\n            \n            # 今、同じaid_xとaid_yのペアが複数存在している可能性がある。\n            # なので、同じペアならすべての重みを足し合わせる。\n            df = df.groupby(['aid_x', 'aid_y']).wgt.sum()\n            \n            # COMBINE INNER CHUNKS\n            if k == a:\n                tmp2 = df\n            else:\n                tmp2 = tmp2.add(df, fill_value=0)\n        # COMBINE OUTER CHUNKS\n        if a == 0:\n            tmp = tmp2\n        else:\n            tmp = tmp.add(tmp2, fill_value=0)\n        del df, tmp2\n        gc.collect()\n        \n    tmp = tmp.reset_index()\n    \n    #aid_xは昇順、wgtは降順で並び替えて、インデックスを振りなおす。\n    tmp = tmp.sort_values(['aid_x', 'wgt'], ascending=[True, False])\n    tmp = tmp.reset_index(drop=True)\n    \n    # ある商品aid_xごとに通し番号を付ける。\n    # ある商品aid_xに対し、スコアの高い順に上位20個のみを保存する。\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n<20].drop(['n'], axis=1)\n    tmp.to_parquet(f'top15_buy2buy_{PART}.parquet')","metadata":{},"execution_count":null,"outputs":[]}]}