{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":38760,"databundleVersionId":4493939},{"sourceType":"datasetVersion","sourceId":4436180,"datasetId":2597726,"databundleVersionId":4495742},{"sourceType":"datasetVersion","sourceId":4645695,"datasetId":2700220,"databundleVersionId":4707633},{"sourceType":"datasetVersion","sourceId":4630155,"datasetId":2693897,"databundleVersionId":4691925},{"sourceType":"datasetVersion","sourceId":4483558,"datasetId":2623568,"databundleVersionId":4543798}],"dockerImageVersionId":30301,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2026-03-04T20:16:26.154100Z","iopub.execute_input":"2026-03-04T20:16:26.154707Z","iopub.status.idle":"2026-03-04T20:16:26.728711Z","shell.execute_reply.started":"2026-03-04T20:16:26.154583Z","shell.execute_reply":"2026-03-04T20:16:26.727289Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Credits\n\nWe thank many Kagglers who have shared ideas. We use fast Handcrafted model from Aldparis [here](https://www.kaggle.com/code/adaubas/otto-fast-handcrafted-model-recall-20/notebook). We use co-visitation matrix idea from Vladimir here. We use groupby sort logic from Sinan in comment section here. We use duplicate prediction removal logic from Radek here. We use multiple visit logic from Pietro here. We use type weighting logic from Ingvaras here. We use leaky test data from my previous notebook here. And some ideas may have originated from Tawara here and KJ here. We use Colum2131's parquets here. Above image is from Ravi's discussion about candidate rerank models here","metadata":{}},{"cell_type":"code","source":"VER = 1\nimport pandas as pd, numpy as np\nimport pickle, glob, gc\n\nfrom collections import Counter\nimport itertools\n\n# multiprocessing \nimport psutil\nN_CORES = psutil.cpu_count()     # Available CPU cores\nprint(f\"N Cores : {N_CORES}\")\nfrom multiprocessing import Pool","metadata":{"execution":{"iopub.status.busy":"2026-03-04T20:16:26.731086Z","iopub.execute_input":"2026-03-04T20:16:26.731723Z","iopub.status.idle":"2026-03-04T20:16:26.739155Z","shell.execute_reply.started":"2026-03-04T20:16:26.731686Z","shell.execute_reply":"2026-03-04T20:16:26.737750Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Validation","metadata":{}},{"cell_type":"code","source":"type_labels = {'clicks':0, 'carts':1, 'orders':2}\n\ndef load_test(files):    \n    dfs = []\n    for e, chunk_file in enumerate(glob.glob(files)):\n        chunk = pd.read_parquet(chunk_file)\n        chunk.ts = (chunk.ts/1000).astype('int32')\n        chunk['type'] = chunk['type'].map(type_labels).astype('int8')\n        dfs.append(chunk)\n    return pd.concat(dfs).reset_index(drop=True) #.astype({\"ts\": \"datetime64[ms]\"})\n\nvalid = load_test('../input/otto-validation/test_parquet/*')\nprint('Valid data has shape',valid.shape)\nvalid.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-04T20:16:26.740885Z","iopub.execute_input":"2026-03-04T20:16:26.741275Z","iopub.status.idle":"2026-03-04T20:16:30.043928Z","shell.execute_reply.started":"2026-03-04T20:16:26.741222Z","shell.execute_reply":"2026-03-04T20:16:30.042562Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\nDISK_PIECES = 4\n# LOAD THREE CO-VISITATION MATRICES\ndef pqt_to_dict(df):\n    return df.groupby('aid_x').aid_y.apply(list).to_dict()\n\ntop_20_clicks = pqt_to_dict( pd.read_parquet(f'../input/otto-co-visitation-matrices/top_20_valid_clicks_v{VER}_0.pqt') )\nfor k in range(1, DISK_PIECES): \n    top_20_clicks.update( pqt_to_dict( pd.read_parquet(f'../input/otto-co-visitation-matrices/top_20_valid_clicks_v{VER}_{k}.pqt') ) )\n\n\ntop_20_buys = pqt_to_dict( pd.read_parquet(f'../input/otto-co-visitation-matrices/top_15_valid_carts_orders_v{VER}_0.pqt') )\nfor k in range(1, DISK_PIECES): \n    top_20_buys.update( pqt_to_dict( pd.read_parquet(f'../input/otto-co-visitation-matrices/top_15_valid_carts_orders_v{VER}_{k}.pqt') ) )\n    \ntop_20_buy2buy = pqt_to_dict( pd.read_parquet(f'../input/otto-co-visitation-matrices/top_15_valid_buy2buy_v{VER}_0.pqt') )\n\n# TOP CLICKS AND ORDERS IN TEST\ntop_clicks = valid.loc[valid['type']==0, 'aid'].value_counts().index.values[:20]\ntop_orders = valid.loc[valid['type']==2, 'aid'].value_counts().index.values[:20]\n\nprint('Here are size of our 3 co-visitation matrices:')\nprint( len( top_20_clicks ), len( top_20_buy2buy ), len( top_20_buys ) )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nPIECES = 5\nvalid_bysession_list = []\nfor PART in range(PIECES):\n    with open(f'../input/otto-valid-test-list/valid_group_tolist_{PART}_{VER}.pkl', 'rb') as f:\n        valid_bysession_list.extend(pickle.load(f))\nprint(len(valid_bysession_list))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def df_parallelize_run(func, t_split):\n    num_cores = np.min([N_CORES, len(t_split)])\n    pool = Pool(num_cores)\n    df = pool.map(func, t_split)\n    pool.close()\n    pool.join()\n    \n    return df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#type_weight_multipliers = {'clicks': 1, 'carts': 6, 'orders': 3}\ntype_weight_multipliers = {0: 1, 1: 6, 2: 3}\n\ndef suggest_clicks(df):\n    session = df[0]\n    aids = df[1]\n    types = df[2]\n    unique_aids = list(dict.fromkeys(aids[::-1]))\n    # Rerank candidates using weights\n    if len(unique_aids) >= 20:\n        weights = np.logspace(0.1, 1, len(aids), base=2, endpoint=True) - 1\n        aids_temp = Counter()\n        # Rerank based on repeat items and type of items\n        for aid, w, t in zip(aids, weights, types):\n            aids_temp[aid] += w * type_weight_multipliers[t]\n        sorted_aids = [k for k, v in aids_temp.most_common(20)]\n        return session, sorted_aids\n    \n    # Use \"clicks\" co-visitation matrix\n    aids2 = list(itertools.chain(*[top_20_clicks[aid] for aid in unique_aids if aid in top_20_clicks]))\n    # Rerank candidates\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2).most_common(20) if aid2 not in unique_aids]\n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n    \n    # Use top20 test clicks\n    return session, result + list(top_clicks)[:20 - len(result)]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\n# Predict on all sessions in parallel\ntemp = df_parallelize_run(suggest_clicks, valid_bysession_list)\nval_clicks = pd.Series([f[1] for f in temp], index=[f[0] for f in temp])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_clicks","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_clicks[11098528]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def suggest_buys(df):\n    # USE USER HISTORY AIDS AND TYPES\n    session = df[0]\n    aids = df[1]\n    types = df[2]\n\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    unique_buys = list(dict.fromkeys( [f for i, f in enumerate(aids) if types[i] in [1, 2]][::-1] ))\n\n    # RERANK CANDIDATES USING WEIGHTS\n    if len(unique_aids)>=20:\n        \n        weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter() \n        # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n        for aid,w,t in zip(aids,weights,types): \n            aids_temp[aid] += w * type_weight_multipliers[t]\n        # RERANK CANDIDATES USING \"BUY2BUY\" CO-VISITATION MATRIX\n        aids3 = list(itertools.chain(*[top_20_buy2buy[aid] for aid in unique_buys if aid in top_20_buy2buy]))\n        for aid in aids3: aids_temp[aid] += 0.1\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return session, sorted_aids\n            \n    # USE \"CART ORDER\" CO-VISITATION MATRIX\n    aids2 = list(itertools.chain(*[top_20_buys[aid] for aid in unique_aids if aid in top_20_buys]))\n    # USE \"BUY2BUY\" CO-VISITATION MATRIX\n    aids3 = list(itertools.chain(*[top_20_buy2buy[aid] for aid in unique_buys if aid in top_20_buy2buy]))\n    # RERANK CANDIDATES\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2 + aids3).most_common(20) if aid2 not in unique_aids] \n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n    # USE TOP20 TEST ORDERS\n    return session, result + list(top_orders)[:20-len(result)]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\n# Predict on all sessions in parallel\ntemp = df_parallelize_run(suggest_buys, valid_bysession_list)\nval_buys = pd.Series([f[1]  for f in temp], index=[f[0] for f in temp])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_buys","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(val_clicks[11098530],\"\\n\", val_buys[11098530])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nvalid_labels = pd.read_parquet('../input/otto-validation/test_labels.parquet')\nvalid_labels","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"benchmark = {\"clicks\":0.5255597442145808, \"carts\":0.4093328152483512, \"orders\":0.6487936598117477, \"all\":.5646320148830121}\n# version 1\n#weights = {'clicks': 0.10, 'carts': 0.30, 'orders': 0.60}\n# version 2\n#weights = {'clicks': 0.30, 'carts': 0.10, 'orders': 0.60}\n# version 3\nweights = {'clicks': 0.30, 'carts': 0.05, 'orders': 0.65}\n\ndef hits(b):\n    # b[0] : session id\n    # b[1] : ground truth\n    # b[2] : aids prediction \n    return b[0], len(set(b[1]).intersection(set(b[2]))), np.clip(len(b[1]), 0, 20)\n\ndef otto_metric_piece(values, typ, verbose=True):\n    \"\"\"计算单一指标的recall\n    c1\n              session                                             labels\n    0        11098528  [11830, 1732105, 588923, 884502, 1157882, 5717...\n    1        11098529  [1105029, 295362, 132016, 459126, 890962, 1135...\n    2        11098530  [409236, 264500, 1603001, 963957, 254154, 5830..\n    \"\"\"\n    c1 = pd.DataFrame(values, columns=[\"labels\"]).reset_index().rename({\"index\":\"session\"}, axis=1)\n    \"\"\"a 加入了两列：type == order和 ground_truth\n             session    type                                       ground_truth                labels\n    0       11098528  orders  [990658, 950341, 1462506, 1561739, 907564, 369...            [11830, 1732105, 588923, 884502, 1157882, 5717...\n    1       11098530  orders                                           [409236]            [409236, 264500, 1603001, 963957, 254154, 5830...\n    2       11098531  orders                                          [1365569]            [396199, 1271998, 452188, 1728212, 1365569, 62... \n    \"\"\"\n    a = valid_labels.loc[valid_labels['type'] == typ].merge(c1, how='left', on=['session'])\n    b = [[a0, a1, a2] for a0, a1, a2 in zip(a['session'], a['ground_truth'], a['labels'])]\n    c = df_parallelize_run(hits, b)\n    \"\"\"c\n    [[11098528        1       11]\n     [11098530        1        1]\n     [11098531        1        1]\n    \"\"\"\n    c = np.array(c)\n    recall = c[:, 1].sum() / c[:, 2].sum()\n    print('{} recall = {:.5f} (vs {:.5f} in benchmark)'.format(typ ,recall, benchmark[typ]))\n    \n    return recall\n    \n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n_ = otto_metric_piece(val_buys, \"orders\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n_ = otto_metric_piece(val_buys, \"carts\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 计算三项type的recall","metadata":{}},{"cell_type":"code","source":"def otto_metric(clicks, carts, orders, verbose = True):\n    score = 0\n    score += weights[\"clicks\"] * otto_metric_piece(clicks, \"clicks\", verbose = verbose)\n    score += weights['carts'] * otto_metric_piece(carts, \"carts\", verbose = verbose)\n    score += weights[\"orders\"] * otto_metric_piece(orders, \"orders\", verbose = verbose)\n    if verbose:\n        print('=============')\n        print('Overall Recall = {:.5f} (vs {:.5f} in benchmark)'.format(score, benchmark[\"all\"]))\n        print('=============')\n    \n    return score","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n_ = otto_metric(val_clicks, val_buys, val_buys)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del temp\n_ = gc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# FREE MEMORY\ndel valid_bysession_list, val_clicks, val_buys\ndel top_20_clicks, top_20_buy2buy, top_20_buys, top_clicks, top_orders, valid\n_ = gc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Test","metadata":{}},{"cell_type":"markdown","source":"Here a submission file is created.","metadata":{}},{"cell_type":"code","source":"test = load_test('../input/otto-chunk-data-inparquet-format/test_parquet/*')\nprint('Test data has shape',test.shape)\ntest.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ntop_20_clicks = pqt_to_dict( pd.read_parquet(f'../input/otto-co-visitation-matrices/top_20_test_clicks_v{VER}_0.pqt') )\nfor k in range(1, DISK_PIECES): \n    top_20_clicks.update( pqt_to_dict( pd.read_parquet(f'../input/otto-co-visitation-matrices/top_20_test_clicks_v{VER}_{k}.pqt') ) )\n\n\ntop_20_buys = pqt_to_dict( pd.read_parquet(f'../input/otto-co-visitation-matrices/top_15_test_carts_orders_v{VER}_0.pqt') )\nfor k in range(1, DISK_PIECES): \n    top_20_buys.update( pqt_to_dict( pd.read_parquet(f'../input/otto-co-visitation-matrices/top_15_test_carts_orders_v{VER}_{k}.pqt') ) )\n    \ntop_20_buy2buy = pqt_to_dict( pd.read_parquet(f'../input/otto-co-visitation-matrices/top_15_test_buy2buy_v{VER}_0.pqt') )\n\n# TOP CLICKS AND ORDERS IN TEST\ntop_clicks = test.loc[test['type']==0, 'aid'].value_counts().index.values[:20]\ntop_orders = test.loc[test['type']==2, 'aid'].value_counts().index.values[:20]\n\nprint('Here are size of our 3 co-visitation matrices:')\nprint( len( top_20_clicks ), len( top_20_buy2buy ), len( top_20_buys ) )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nPIECES = 5\ntest_bysession_list = []\nfor PART in range(PIECES):\n    with open(f'../input/otto-valid-test-list/test_group_tolist_{PART}_{VER}.pkl', 'rb') as f:\n        test_bysession_list.extend(pickle.load(f))\nprint(len(test_bysession_list))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\n# Predict on all sessions in parallel\ntemp = df_parallelize_run(suggest_clicks, test_bysession_list)\nclicks_pred_df = pd.Series([f[1] for f in temp], index=[f[0] for f in temp])\nclicks_pred_df = clicks_pred_df.add_suffix(\"_clicks\")\nclicks_pred_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\n# Predict on all sessions in parallel\ntemp = df_parallelize_run(suggest_clicks, test_bysession_list)\nclicks_pred_df = pd.Series([f[1] for f in temp], index=[f[0] for f in temp])\nclicks_pred_df = clicks_pred_df.add_suffix(\"_clicks\")\nclicks_pred_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\n# Predict on all sessions in parallel\ntemp = df_parallelize_run(suggest_buys, test_bysession_list)\nbuys_pred_df = pd.Series([f[1] for f in temp], index=[f[0] for f in temp])\norders_pred_df = buys_pred_df.add_suffix(\"_orders\")\ncarts_pred_df = buys_pred_df.add_suffix(\"_carts\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\n# Predict on all sessions in parallel\ntemp = df_parallelize_run(suggest_buys, test_bysession_list)\nbuys_pred_df = pd.Series([f[1] for f in temp], index=[f[0] for f in temp])\norders_pred_df = buys_pred_df.add_suffix(\"_orders\")\ncarts_pred_df = buys_pred_df.add_suffix(\"_carts\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_df = pd.concat([clicks_pred_df, orders_pred_df, carts_pred_df]).reset_index()\npred_df.columns = [\"session_type\", \"labels\"]\npred_df[\"labels\"] = pred_df.labels.apply(lambda x: \" \".join(map(str,x)))\npred_df.to_csv(\"submission.csv\", index=False)\npred_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}