{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":38760,"databundleVersionId":4493939,"sourceType":"competition"},{"sourceId":4436180,"sourceType":"datasetVersion","datasetId":2597726}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np\n\nimport glob\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom datetime import datetime\n\nfrom tqdm import tqdm\n\ntype_labels = {'clicks':1, 'carts':2, 'orders':3}","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:23:07.282963Z","iopub.execute_input":"2024-06-05T03:23:07.284007Z","iopub.status.idle":"2024-06-05T03:23:07.778767Z","shell.execute_reply.started":"2024-06-05T03:23:07.283954Z","shell.execute_reply":"2024-06-05T03:23:07.777342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load(which):    \n    dfs = []\n\n    test_files = glob.glob('/kaggle/input/otto-chunk-data-inparquet-format/'+which+'_parquet/*')\n    \n    for e, chunk_file in enumerate(test_files):\n        chunk = pd.read_parquet(chunk_file)\n        chunk.ts = (chunk.ts/1000).astype('int32')\n        chunk['type'] = chunk['type'].map(type_labels).astype('int8')\n        dfs.append(chunk)\n        \n    return pd.concat(dfs).reset_index(drop=True) #.astype({\"ts\": \"datetime64[ms]\"})","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:23:07.781366Z","iopub.execute_input":"2024-06-05T03:23:07.781875Z","iopub.status.idle":"2024-06-05T03:23:07.790676Z","shell.execute_reply.started":"2024-06-05T03:23:07.781839Z","shell.execute_reply":"2024-06-05T03:23:07.789363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = load('train')","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:23:07.792207Z","iopub.execute_input":"2024-06-05T03:23:07.792691Z","iopub.status.idle":"2024-06-05T03:24:18.792816Z","shell.execute_reply.started":"2024-06-05T03:23:07.792647Z","shell.execute_reply":"2024-06-05T03:24:18.791568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:24:18.795125Z","iopub.execute_input":"2024-06-05T03:24:18.795592Z","iopub.status.idle":"2024-06-05T03:24:18.818121Z","shell.execute_reply.started":"2024-06-05T03:24:18.795552Z","shell.execute_reply":"2024-06-05T03:24:18.816705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# To reduce runtime, only consider data from the last three days as train_df.","metadata":{}},{"cell_type":"code","source":"datetime.strptime(\"2022-08-25 00:00:00\", \"%Y-%m-%d %H:%M:%S\").timestamp()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:24:18.822144Z","iopub.execute_input":"2024-06-05T03:24:18.822583Z","iopub.status.idle":"2024-06-05T03:24:18.830984Z","shell.execute_reply.started":"2024-06-05T03:24:18.82255Z","shell.execute_reply":"2024-06-05T03:24:18.829783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df[train_df['ts']>=1661385600].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:24:18.832532Z","iopub.execute_input":"2024-06-05T03:24:18.832904Z","iopub.status.idle":"2024-06-05T03:24:20.667606Z","shell.execute_reply.started":"2024-06-05T03:24:18.832866Z","shell.execute_reply":"2024-06-05T03:24:20.666495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['datetime'] = train_df['ts'].apply(lambda x: datetime.fromtimestamp(x))","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:26:32.848569Z","iopub.execute_input":"2024-06-05T03:26:32.849878Z","iopub.status.idle":"2024-06-05T03:27:45.797846Z","shell.execute_reply.started":"2024-06-05T03:26:32.848989Z","shell.execute_reply":"2024-06-05T03:27:45.796358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Choose the last day as the validation set.","metadata":{}},{"cell_type":"code","source":"train_df['datetime'].dt.day.unique()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:28:17.622943Z","iopub.execute_input":"2024-06-05T03:28:17.623447Z","iopub.status.idle":"2024-06-05T03:28:18.647928Z","shell.execute_reply.started":"2024-06-05T03:28:17.623408Z","shell.execute_reply":"2024-06-05T03:28:18.646712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['day'] = train_df['datetime'].dt.day","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:28:23.430841Z","iopub.execute_input":"2024-06-05T03:28:23.431285Z","iopub.status.idle":"2024-06-05T03:28:24.327327Z","shell.execute_reply.started":"2024-06-05T03:28:23.431224Z","shell.execute_reply":"2024-06-05T03:28:24.325836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val_offline = train_df[train_df['day']==28].reset_index(drop=True)\ndf_train_offline = train_df[train_df['day']<28].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:28:25.532397Z","iopub.execute_input":"2024-06-05T03:28:25.532869Z","iopub.status.idle":"2024-06-05T03:28:28.192724Z","shell.execute_reply.started":"2024-06-05T03:28:25.532835Z","shell.execute_reply":"2024-06-05T03:28:28.191335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A very simple itemCF for now; later, improvements and optimizations will be considered, such as time factors, click order, and user activity.","metadata":{}},{"cell_type":"code","source":"df_train_offline","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:28:33.864468Z","iopub.execute_input":"2024-06-05T03:28:33.864887Z","iopub.status.idle":"2024-06-05T03:28:33.88268Z","shell.execute_reply.started":"2024-06-05T03:28:33.864853Z","shell.execute_reply":"2024-06-05T03:28:33.881076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_offline['aid'].nunique()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:28:41.971991Z","iopub.execute_input":"2024-06-05T03:28:41.972453Z","iopub.status.idle":"2024-06-05T03:28:42.53851Z","shell.execute_reply.started":"2024-06-05T03:28:41.97241Z","shell.execute_reply":"2024-06-05T03:28:42.537337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session_aids_df = df_train_offline[['session','aid']].groupby('session',as_index=False).agg(list)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:28:45.229906Z","iopub.execute_input":"2024-06-05T03:28:45.230308Z","iopub.status.idle":"2024-06-05T03:30:17.465616Z","shell.execute_reply.started":"2024-06-05T03:28:45.230277Z","shell.execute_reply":"2024-06-05T03:30:17.464118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session_aids_df","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:31:02.357238Z","iopub.execute_input":"2024-06-05T03:31:02.357793Z","iopub.status.idle":"2024-06-05T03:31:02.384751Z","shell.execute_reply.started":"2024-06-05T03:31:02.357749Z","shell.execute_reply":"2024-06-05T03:31:02.383322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:31:12.368577Z","iopub.execute_input":"2024-06-05T03:31:12.37223Z","iopub.status.idle":"2024-06-05T03:31:12.376857Z","shell.execute_reply.started":"2024-06-05T03:31:12.372174Z","shell.execute_reply":"2024-06-05T03:31:12.375693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_Sim = defaultdict()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T03:31:13.600561Z","iopub.execute_input":"2024-06-05T03:31:13.600991Z","iopub.status.idle":"2024-06-05T03:31:13.606561Z","shell.execute_reply.started":"2024-06-05T03:31:13.600956Z","shell.execute_reply":"2024-06-05T03:31:13.605186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,row in tqdm( session_aids_df.iterrows(), total=len(session_aids_df) ):\n    \n    aid_list = row['aid']\n    \n    for item in aid_list:\n        \n        item_Sim.setdefault(item,dict())\n        \n        for related_item in aid_list:\n            if item == related_item:\n                continue\n        \n            item_Sim[item].setdefault(related_item,0)\n            \n            item_Sim[item][related_item] += 1\n    ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To be continued","metadata":{}},{"cell_type":"code","source":"def recommend_items(item_Sim, last_item, top_n=20, cache={}):\n    \"\"\"\n    根据用户的最后一个交互商品推荐商品，并使用缓存\n    \n    参数:\n    - item_Sim: 商品共现矩阵\n    - last_item: 用户的最后一个交互商品\n    - top_n: 推荐的商品数量\n    - cache: 缓存字典\n    \n    返回:\n    - 推荐的商品列表\n    \"\"\"\n    if last_item in cache:\n        return cache[last_item]\n    \n    recommendations = {}\n\n    if last_item in item_Sim:\n        for related_item, count in item_Sim[last_item].items():\n            recommendations[related_item] = count\n\n    sorted_recommendations = sorted(recommendations.items(), key=lambda x: x[1], reverse=True)\n    recommended_items = [item for item, score in sorted_recommendations[:top_n]]\n    \n    # 将结果存储在缓存中\n    cache[last_item] = recommended_items\n    \n    return recommended_items","metadata":{"execution":{"iopub.status.busy":"2024-06-05T04:11:55.126646Z","iopub.execute_input":"2024-06-05T04:11:55.127101Z","iopub.status.idle":"2024-06-05T04:11:55.136396Z","shell.execute_reply.started":"2024-06-05T04:11:55.127067Z","shell.execute_reply":"2024-06-05T04:11:55.135024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm\n\n# 假设 df_train_offline 和 session_aids_df 已经预处理完成\n# df_train_offline 包含实际的用户交互记录\n# session_aids_df 包含每个 session 下所有的 aid 列表\n\n# 限制数据集为前一万个 session\nlimited_session_aids_df = session_aids_df.head(10000)\n\n# 初始化评分列表\nscores = []\n\n# 初始化缓存字典\ncache = {}\n\n# 遍历每个 session\nfor i, row in tqdm(limited_session_aids_df.iterrows(), total=len(limited_session_aids_df)):\n    session_id = row['session']\n    actual_aids = df_train_offline[df_train_offline['session'] == session_id]['aid'].tolist()\n    actual_types = df_train_offline[df_train_offline['session'] == session_id]['type'].tolist()\n    \n    if len(actual_aids) > 0:\n        last_item = actual_aids[-1]\n        \n        # 获取推荐商品\n        recommended_items = recommend_items(item_Sim, last_item, top_n=20, cache=cache)\n        \n        # 初始化各类型的计数\n        Rclicks = 0\n        Rcarts = 0\n        Rorders = 0\n        \n        # 计算推荐商品的类型计数\n        for item, item_type in zip(actual_aids, actual_types):\n            if item in recommended_items:\n                if item_type == 1:\n                    Rclicks += 1\n                elif item_type == 2:\n                    Rcarts += 1\n                elif item_type == 3:\n                    Rorders += 1\n        \n        # 计算最终评分\n        score = 0.10 * Rclicks + 0.30 * Rcarts + 0.60 * Rorders\n    else:\n        score = 0\n    \n    scores.append(score)\n\n# 打印平均评分\naverage_score = sum(scores) / len(scores)\nprint(f'Average Score: {average_score}')","metadata":{"execution":{"iopub.status.busy":"2024-06-05T04:24:44.090789Z","iopub.execute_input":"2024-06-05T04:24:44.091281Z","iopub.status.idle":"2024-06-05T04:31:29.88962Z","shell.execute_reply.started":"2024-06-05T04:24:44.091227Z","shell.execute_reply":"2024-06-05T04:31:29.888181Z"},"trusted":true},"execution_count":null,"outputs":[]}]}