{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":38760,"databundleVersionId":4493939,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":4474043,"sourceType":"datasetVersion","datasetId":2601572}],"dockerImageVersionId":30302,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nfrom collections import defaultdict, Counter\nimport cudf, cupy\nimport gc\nimport glob\nimport itertools\nprint('Using RAPIDS version',cudf.__version__)","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2025-10-28T10:20:53.693217Z","iopub.execute_input":"2025-10-28T10:20:53.694238Z","iopub.status.idle":"2025-10-28T10:20:56.563772Z","shell.execute_reply.started":"2025-10-28T10:20:53.694116Z","shell.execute_reply":"2025-10-28T10:20:56.562809Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cache_data_to_memory(files):\n    \"\"\"\n    Cache all parquet files to CPU RAM for faster GPU processing.\n    \"\"\"\n    print(\"Caching data to CPU RAM...\")\n    data_cache = {}\n    for f in files:\n        data_cache[f] = pd.read_parquet(f)\n    \n    print(f'Cached {len(files)} files to memory.')\n    return data_cache\n\ndef read_file(f):\n    return cudf.DataFrame(data_cache[f])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:23:17.643947Z","iopub.execute_input":"2025-10-28T10:23:17.644299Z","iopub.status.idle":"2025-10-28T10:23:17.649475Z","shell.execute_reply.started":"2025-10-28T10:23:17.644273Z","shell.execute_reply":"2025-10-28T10:23:17.648511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Split your large files into smaller chunks\ndef split_large_files(input_files, output_dir='./chunked_data', rows_per_chunk=5_000_000):\n    \"\"\"Split large parquet files into smaller chunks\"\"\"\n    import os\n    os.makedirs(output_dir, exist_ok=True)\n    \n    chunk_files = []\n    for file_path in input_files:\n        print(f\"Splitting {file_path}...\")\n        df = pd.read_parquet(file_path)\n        \n        # Split into chunks\n        num_chunks = int(np.ceil(len(df) / rows_per_chunk))\n        for i in range(num_chunks):\n            start_idx = i * rows_per_chunk\n            end_idx = min((i + 1) * rows_per_chunk, len(df))\n            chunk_df = df.iloc[start_idx:end_idx]\n            \n            # Save chunk\n            chunk_file = f'{output_dir}/chunk_{os.path.basename(file_path)}_{i}.parquet'\n            chunk_df.to_parquet(chunk_file)\n            chunk_files.append(chunk_file)\n            # print(f\"  Created {chunk_file} with {len(chunk_df)} rows\")\n    \n    return chunk_files\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:20:56.572107Z","iopub.execute_input":"2025-10-28T10:20:56.572437Z","iopub.status.idle":"2025-10-28T10:20:56.588732Z","shell.execute_reply.started":"2025-10-28T10:20:56.572405Z","shell.execute_reply":"2025-10-28T10:20:56.587831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_covisit_matrix_gpu(\n    data_cache,\n    type1_filter=None,\n    type2_filter=None,\n    type1_weights=None,\n    type2_weights=None,\n    max_session_length=30,\n    max_days_elapsed=1,\n    top_k=15,\n    disk_pieces=6,\n    read_ct=5,\n    output_prefix='covisit_matrix'\n):\n    \"\"\"\n    Build co-visit matrix using cuDF GPU acceleration from cached data.\n\n    \"\"\"\n    \n    # Set default weights\n    if type1_weights is None:\n        type1_weights = {0: 1, 1: 1, 2: 1}\n    if type2_weights is None:\n        type2_weights = {0: 1, 1: 1, 2: 1}\n    \n\n    files = list(data_cache.keys())\n    \n    # CHUNK PARAMETERS\n    CHUNK = int(np.ceil(len(files) / 6))\n    print(f'Processing {len(files)} files, in groups of {read_ct} and chunks of {CHUNK}.')\n    \n    # Calculate size for disk pieces\n    SIZE = 1.86e6 / disk_pieces\n    \n    # PROCESS IN PARTS FOR MEMORY MANAGEMENT\n    for PART in range(disk_pieces):\n        print(f'\\n### DISK PART {PART + 1}')\n        \n        # OUTER CHUNKS\n        for j in range(6):\n            a = j * CHUNK\n            b = min((j + 1) * CHUNK, len(files))\n            print(f'Processing files {a} thru {b-1} in groups of {read_ct}...')\n            \n            # INNER CHUNKS\n            for k in range(a, b, read_ct):\n                # READ FILES\n                df = [read_file(files[k])]\n                \n                for i in range(1, read_ct):\n                    if k + i < b:\n                        df.append(read_file(files[k + i]))\n                df = cudf.concat(df, ignore_index=True, axis=0)\n                df = df.sort_values(['session', 'ts'], ascending=[True, False])\n                \n                # USE TAIL OF SESSION\n                df = df.reset_index(drop=True)\n                df['n'] = df.groupby('session').cumcount()\n                df = df.loc[df.n < max_session_length].drop('n', axis=1)\n                \n                # CREATE PAIRS\n                df = df.merge(df, on='session', suffixes=('_x', '_y'))\n                df = df.loc[((df.ts_x - df.ts_y).abs() < max_days_elapsed * 24 * 60 * 60) & (df.aid_x != df.aid_y)]\n                \n                # FILTER BY TYPE\n                if type1_filter is not None:\n                    df = df.loc[df.type_x.isin(type1_filter)]\n                  \n                if type2_filter is not None:\n                    df = df.loc[df.type_y.isin(type2_filter)]\n                  \n                \n                # MEMORY MANAGEMENT - COMPUTE IN PARTS\n                df = df.loc[(df.aid_x >= PART * SIZE) & (df.aid_x < (PART + 1) * SIZE)]\n               \n                # ASSIGN WEIGHTS (multiply both type weights)\n                df = df[['session', 'aid_x', 'aid_y', 'type_x', 'type_y']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n                \n                # Map weights for both types\n                df['wgt_x'] = df.type_x.map(type1_weights).fillna(1)\n                df['wgt_y'] = df.type_y.map(type2_weights).fillna(1)\n                df['wgt'] = (df.wgt_x * df.wgt_y).astype('float32')\n                \n                df = df[['aid_x', 'aid_y', 'wgt']]\n                df = df.groupby(['aid_x', 'aid_y']).wgt.sum()\n                \n                # COMBINE INNER CHUNKS\n                if k == a:\n                    tmp2 = df\n                else:\n                    tmp2 = tmp2.add(df, fill_value=0)\n                print(k, ', ', end='')\n            \n            print()\n            # COMBINE OUTER CHUNKS\n            if a == 0:\n                tmp = tmp2\n            else:\n                tmp = tmp.add(tmp2, fill_value=0)\n            del tmp2, df\n            gc.collect()\n        \n        # CONVERT MATRIX TO DATAFRAME\n        tmp = tmp.reset_index()\n        tmp = tmp.sort_values(['aid_x', 'wgt'], ascending=[True, False])\n        \n        # SAVE TOP K\n        tmp = tmp.reset_index(drop=True)\n        tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n        tmp = tmp.loc[tmp.n < top_k].drop('n', axis=1)\n        \n        # SAVE PART TO DISK\n        output_file = f'{output_prefix}_{PART}.pqt'\n        tmp.to_pandas().to_parquet(output_file)\n        print(f'Saved {output_file}')\n        \n        del tmp\n        gc.collect()\n    \n    print(f'\\nCompleted! Output saved as {output_prefix}_*.pqt')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:20:56.590046Z","iopub.execute_input":"2025-10-28T10:20:56.590465Z","iopub.status.idle":"2025-10-28T10:20:56.609891Z","shell.execute_reply.started":"2025-10-28T10:20:56.590441Z","shell.execute_reply":"2025-10-28T10:20:56.609004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split your files first\noriginal_files = glob.glob('../input/otto-full-optimized-memory-footprint/*.parquet')\nchunked_files = split_large_files(original_files, rows_per_chunk=5_000_000)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:20:56.612439Z","iopub.execute_input":"2025-10-28T10:20:56.612673Z","iopub.status.idle":"2025-10-28T10:22:20.561964Z","shell.execute_reply.started":"2025-10-28T10:20:56.612652Z","shell.execute_reply":"2025-10-28T10:22:20.561210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Now cache the chunked files\ndata_cache = cache_data_to_memory(chunked_files)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:23:22.999962Z","iopub.execute_input":"2025-10-28T10:23:23.000325Z","iopub.status.idle":"2025-10-28T10:23:27.692715Z","shell.execute_reply.started":"2025-10-28T10:23:23.000294Z","shell.execute_reply":"2025-10-28T10:23:27.691675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nDISK_PIECES=4\n# Process with appropriate parameters for ~20 chunks\nbuild_covisit_matrix_gpu(\n    data_cache=data_cache,\n    type1_filter=[0,1,2],\n    type2_filter=[1,2],\n    type1_weights={0: 1,1: 2,2: 2},\n    type2_weights={1:1, 2: 3},\n    disk_pieces=DISK_PIECES,  # Now you can use more pieces\n    read_ct=3,      # Process 3 chunks at a time\n    output_prefix='covisit_all_to_cart-order'\n)\n\nbuild_covisit_matrix_gpu(\n    data_cache=data_cache,\n    type1_filter=[0,1,2],\n    type2_filter=[0],\n    type1_weights={0: 1,1: 2,2: 2},\n    type2_weights={0:1},\n    disk_pieces=DISK_PIECES,  # Now you can use more pieces\n    read_ct=3,      # Process 3 chunks at a time\n    output_prefix='covisit_all_to_click'\n)\n\nbuild_covisit_matrix_gpu(\n    data_cache=data_cache,\n    type1_filter=[0],\n    type2_filter=[1,2],\n    type1_weights={0: 1,},\n    type2_weights={1: 1, 2: 3},\n    disk_pieces=DISK_PIECES,  # Now you can use more pieces\n    read_ct=3,      # Process 3 chunks at a time\n    output_prefix='covisit_click_to_buy'\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:23:41.732904Z","iopub.execute_input":"2025-10-28T10:23:41.733659Z","iopub.status.idle":"2025-10-28T10:31:31.372698Z","shell.execute_reply.started":"2025-10-28T10:23:41.733626Z","shell.execute_reply":"2025-10-28T10:31:31.371820Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clean up cache when done\ndel data_cache\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:31:31.374730Z","iopub.execute_input":"2025-10-28T10:31:31.375366Z","iopub.status.idle":"2025-10-28T10:31:31.474510Z","shell.execute_reply.started":"2025-10-28T10:31:31.375329Z","shell.execute_reply":"2025-10-28T10:31:31.473600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_parquet('/kaggle/input/otto-full-optimized-memory-footprint/test.parquet')\nprint('Test data has shape',test_df.shape)\ntest_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:31:31.475492Z","iopub.execute_input":"2025-10-28T10:31:31.475735Z","iopub.status.idle":"2025-10-28T10:31:31.652071Z","shell.execute_reply.started":"2025-10-28T10:31:31.475713Z","shell.execute_reply":"2025-10-28T10:31:31.651000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ndef pqt_to_dict(df):\n    return df.groupby('aid_x').aid_y.apply(list).to_dict()\n    \ntop_20_all_to_cart_order = pqt_to_dict( pd.read_parquet(f'/kaggle/working/covisit_all_to_cart-order_0.pqt') )\nfor k in range(1,DISK_PIECES): \n    top_20_all_to_cart_order.update(pqt_to_dict( pd.read_parquet(f'/kaggle/working/covisit_all_to_cart-order_{k}.pqt') ) )\n\ntop_20_all_to_click = pqt_to_dict( pd.read_parquet(f'/kaggle/working/covisit_all_to_click_0.pqt') )\nfor k in range(1,DISK_PIECES): \n    top_20_all_to_click.update(pqt_to_dict( pd.read_parquet(f'/kaggle/working/covisit_all_to_click_{k}.pqt') ) )\n\ntop_20_click_to_buy = pqt_to_dict( pd.read_parquet(f'/kaggle/working/covisit_click_to_buy_0.pqt') )\nfor k in range(1,DISK_PIECES): \n    top_20_click_to_buy.update(pqt_to_dict( pd.read_parquet(f'/kaggle/working/covisit_click_to_buy_{k}.pqt') ) )\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:32:36.999783Z","iopub.execute_input":"2025-10-28T10:32:37.000469Z","iopub.status.idle":"2025-10-28T10:33:55.833829Z","shell.execute_reply.started":"2025-10-28T10:32:37.000436Z","shell.execute_reply":"2025-10-28T10:33:55.832881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# TOP CLICKS AND ORDERS IN TEST\ntop_clicks = test_df.loc[test_df['type']=='0','aid'].value_counts().index.values[:20]\ntop_carts  = test_df.loc[test_df['type']=='1','aid'].value_counts().index.values[:20]\ntop_orders = test_df.loc[test_df['type']=='2','aid'].value_counts().index.values[:20]\n\nprint('Here are size of our 3 co-visitation matrices:')\nprint( len( top_20_all_to_cart_order ), len(top_20_all_to_click), len(top_20_click_to_buy)  )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:33:59.502974Z","iopub.execute_input":"2025-10-28T10:33:59.503346Z","iopub.status.idle":"2025-10-28T10:33:59.531996Z","shell.execute_reply.started":"2025-10-28T10:33:59.503314Z","shell.execute_reply":"2025-10-28T10:33:59.531032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"type_weight_multipliers = {0: 1, 1: 3, 2: 6}\ndef suggest_clicks(df):\n    aids = df.aid.tolist()\n    types = df.type.tolist()\n    unique_aids = list(dict.fromkeys(aids[::-1]))  # Reverse for recency\n    \n    # Long sessions: rerank by recency + type\n    if len(unique_aids) >= 20:\n        weights = np.logspace(0.1, 1, len(aids), base=2, endpoint=True) - 1\n        aids_temp = Counter()\n        \n        for aid, w, t in zip(aids, weights, types):\n            aids_temp[aid] += w * type_weight_multipliers[t]\n        \n        return [k for k, v in aids_temp.most_common(20)]\n    \n    # Short sessions: use co-visitation\n    aids2 = list(itertools.chain(*[top_20_clicks[aid] for aid in unique_aids if aid in top_20_clicks]))\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2).most_common(20) if aid2 not in unique_aids]\n    \n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n    return result + list(top_clicks)[:20 - len(result)]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:34:11.338901Z","iopub.execute_input":"2025-10-28T10:34:11.339875Z","iopub.status.idle":"2025-10-28T10:34:11.348867Z","shell.execute_reply.started":"2025-10-28T10:34:11.339828Z","shell.execute_reply":"2025-10-28T10:34:11.347790Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# %%time\n# pred_df_clicks = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n#     lambda x: suggest_clicks(x)\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:22:20.856825Z","iopub.status.idle":"2025-10-28T10:22:20.857168Z","shell.execute_reply.started":"2025-10-28T10:22:20.856982Z","shell.execute_reply":"2025-10-28T10:22:20.856997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# clicks_pred_df = pd.DataFrame(pred_df_clicks.add_suffix(\"_clicks\"), columns=[\"labels\"]).reset_index()\n\n# orders_pred_df = pd.DataFrame(pred_df_clicks.add_suffix(\"_orders\"), columns=[\"labels\"]).reset_index()\n# carts_pred_df =  pd.DataFrame(pred_df_clicks.add_suffix(\"_carts\"),  columns=[\"labels\"]).reset_index()\n\n# clicks_pred_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:22:20.858502Z","iopub.status.idle":"2025-10-28T10:22:20.858937Z","shell.execute_reply.started":"2025-10-28T10:22:20.858703Z","shell.execute_reply":"2025-10-28T10:22:20.858724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pred_df = pd.concat([clicks_pred_df, orders_pred_df, carts_pred_df])\n# # pred_df=clicks_pred_df\n# pred_df.columns = [\"session_type\", \"labels\"]\n# pred_df[\"labels\"] = pred_df.labels.apply(lambda x: \" \".join(map(str,x)))\n# pred_df.to_csv(\"submission.csv\", index=False)\n# pred_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-28T10:22:20.860479Z","iopub.status.idle":"2025-10-28T10:22:20.860915Z","shell.execute_reply.started":"2025-10-28T10:22:20.860685Z","shell.execute_reply":"2025-10-28T10:22:20.860708Z"}},"outputs":[],"execution_count":null}]}