{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Thanks for EDWARD CROOKENDEN  & CARNO ZHAO  to do the amazing jobs for this compettions","metadata":{}},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nimport os\nimport random\nimport numpy as np\nimport json\nfrom datetime import timedelta\nfrom collections import Counter\nfrom tqdm.notebook import tqdm\nfrom heapq import nlargest\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_theme()\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:04:06.802208Z","iopub.execute_input":"2022-12-11T05:04:06.802707Z","iopub.status.idle":"2022-12-11T05:04:07.460276Z","shell.execute_reply.started":"2022-12-11T05:04:06.802610Z","shell.execute_reply":"2022-12-11T05:04:07.459075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Paths ###\n\nDATA_PATH = Path('../input/otto-recommender-system')\nTRAIN_PATH = DATA_PATH/'train.jsonl'\nTEST_PATH = DATA_PATH/'test.jsonl'\nSAMPLE_SUB_PATH = Path('../input/otto-recommender-system/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:04:18.980302Z","iopub.execute_input":"2022-12-11T05:04:18.981423Z","iopub.status.idle":"2022-12-11T05:04:18.987737Z","shell.execute_reply.started":"2022-12-11T05:04:18.981382Z","shell.execute_reply":"2022-12-11T05:04:18.986513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(TRAIN_PATH, 'r') as f:\n    print(f\"We have {len(f.readlines()):,} lines in the training data\")","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:04:35.022100Z","iopub.execute_input":"2022-12-11T05:04:35.022525Z","iopub.status.idle":"2022-12-11T05:06:45.626701Z","shell.execute_reply.started":"2022-12-11T05:04:35.022487Z","shell.execute_reply":"2022-12-11T05:06:45.625360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_size = 150000\n\nchunks = pd.read_json(TRAIN_PATH, lines=True, chunksize = sample_size)\n\nfor c in chunks:\n    sample_train_df = c\n    break","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:06:45.628725Z","iopub.execute_input":"2022-12-11T05:06:45.629066Z","iopub.status.idle":"2022-12-11T05:06:56.956515Z","shell.execute_reply.started":"2022-12-11T05:06:45.629034Z","shell.execute_reply":"2022-12-11T05:06:56.955442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_df.set_index('session', drop=True, inplace=True)\nsample_train_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:06:56.957739Z","iopub.execute_input":"2022-12-11T05:06:56.958163Z","iopub.status.idle":"2022-12-11T05:06:57.065520Z","shell.execute_reply.started":"2022-12-11T05:06:56.958133Z","shell.execute_reply":"2022-12-11T05:06:57.064408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Sample the first session in the df\nexample_session = sample_train_df.iloc[0].item()\nprint(f'This session was {len(example_session)} actions long \\n')\nprint(f'The first action in the session: \\n {example_session[0]} \\n')\n\n# Time of session\ntime_elapsed = example_session[-1][\"ts\"] - example_session[0][\"ts\"]\n# The timestamp is in milliseconds since 00:00:00 UTC on 1 January 1970\nprint(f'The first session elapsed: {str(timedelta(milliseconds=time_elapsed))} \\n')\n\n# Count the frequency of actions within the session\naction_counts = {}\nfor action in example_session:\n    action_counts[action['type']] = action_counts.get(action['type'], 0) + 1  \nprint(f'The first session contains the following frequency of actions: {action_counts}')","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:06:57.067755Z","iopub.execute_input":"2022-12-11T05:06:57.068087Z","iopub.status.idle":"2022-12-11T05:06:57.076359Z","shell.execute_reply.started":"2022-12-11T05:06:57.068056Z","shell.execute_reply":"2022-12-11T05:06:57.075298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Extract information from each session and add it to the df ###\n\naction_counts_list, article_id_counts_list, session_length_time_list, session_length_action_list = ([] for i in range(4))\noverall_action_counts = {}\noverall_article_id_counts = {}\n\nfor i, row in tqdm(sample_train_df.iterrows(), total=len(sample_train_df)):\n    \n    actions = row['events']\n    \n    # Get the frequency of actions and article_ids\n    action_counts = {}\n    article_id_counts = {}\n    for action in actions:\n        action_counts[action['type']] = action_counts.get(action['type'], 0) + 1\n        article_id_counts[action['aid']] = article_id_counts.get(action['aid'], 0) + 1\n        overall_action_counts[action['type']] = overall_action_counts.get(action['type'], 0) + 1\n        overall_article_id_counts[action['aid']] = overall_article_id_counts.get(action['aid'], 0) + 1\n        \n    # Get the length of the session\n    session_length_time = actions[-1]['ts'] - actions[0]['ts']\n    \n    # Add to list\n    action_counts_list.append(action_counts)\n    article_id_counts_list.append(article_id_counts)\n    session_length_time_list.append(session_length_time)\n    session_length_action_list.append(len(actions))\n    \nsample_train_df['action_counts'] = action_counts_list\nsample_train_df['article_id_counts'] = article_id_counts_list\nsample_train_df['session_length_unix'] = session_length_time_list\nsample_train_df['session_length_hours'] = sample_train_df['session_length_unix']*2.77778e-7  # Convert to hours\nsample_train_df['session_length_action'] = session_length_action_list","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:06:57.077551Z","iopub.execute_input":"2022-12-11T05:06:57.077904Z","iopub.status.idle":"2022-12-11T05:07:24.153874Z","shell.execute_reply.started":"2022-12-11T05:06:57.077871Z","shell.execute_reply":"2022-12-11T05:07:24.152645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Actions ###\n\ntotal_actions = sum(overall_action_counts.values())\n\nplt.figure(figsize=(8,6))\nsns.barplot(x=list(overall_action_counts.keys()), y=[i/total_actions for i in overall_action_counts.values()]);\nplt.title(f'Action frequency', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.xlabel('Category', fontsize=12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:07:24.155332Z","iopub.execute_input":"2022-12-11T05:07:24.155696Z","iopub.status.idle":"2022-12-11T05:07:24.422255Z","shell.execute_reply.started":"2022-12-11T05:07:24.155662Z","shell.execute_reply":"2022-12-11T05:07:24.421081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(24, 10))\n\np = sns.distplot(sample_train_df['session_length_action'], color=\"y\", bins= 70, ax=ax[0], kde=False)\np.set_xlabel(\"Number of actions\", fontsize = 16)\np.set_ylabel(\"Density\", fontsize = 16)\np.set_title(\"Distribution of the number of actions taken in each session\", fontsize = 14)\np.axvline(sample_train_df['session_length_action'].mean(), color='r', linestyle='--', label=\"Mean\")\n\np = sns.distplot(sample_train_df['session_length_hours'], color=\"b\", bins= 70, ax=ax[1], kde=False)\np.set_xlabel(\"Hours\", fontsize = 16)\np.set_ylabel(\"Density\", fontsize = 16)\np.set_title(\"Length of each session\", fontsize = 16);","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:07:24.424417Z","iopub.execute_input":"2022-12-11T05:07:24.424888Z","iopub.status.idle":"2022-12-11T05:07:25.404414Z","shell.execute_reply.started":"2022-12-11T05:07:24.424843Z","shell.execute_reply":"2022-12-11T05:07:25.403033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'{round(len(sample_train_df[sample_train_df[\"session_length_action\"]<10])/len(sample_train_df),3)*100}% of the sessions had less than 10 actions')","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:07:25.406151Z","iopub.execute_input":"2022-12-11T05:07:25.406684Z","iopub.status.idle":"2022-12-11T05:07:25.485508Z","shell.execute_reply.started":"2022-12-11T05:07:25.406641Z","shell.execute_reply":"2022-12-11T05:07:25.484335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"article_id_freq = list(overall_article_id_counts.values())\ncut_off = [i for i in article_id_freq if i<30]\n\nplt.figure(figsize=(8,6))\nsns.distplot(cut_off, bins=30, kde=False);\nplt.title(f'Article ID frequency', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.xlabel('Article', fontsize=12);","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:07:25.490373Z","iopub.execute_input":"2022-12-11T05:07:25.490713Z","iopub.status.idle":"2022-12-11T05:07:26.015405Z","shell.execute_reply.started":"2022-12-11T05:07:25.490683Z","shell.execute_reply":"2022-12-11T05:07:26.014283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Look at the most interacted with articles ###\nprint(f'Frequency of most common articles: {sorted(list(overall_article_id_counts.values()))[-5:]} \\n')\nres = nlargest(5, overall_article_id_counts, key = overall_article_id_counts.get)\nprint(f'IDs for those common articles: {res}')","metadata":{"execution":{"iopub.status.busy":"2022-12-11T05:07:26.016838Z","iopub.execute_input":"2022-12-11T05:07:26.017361Z","iopub.status.idle":"2022-12-11T05:07:26.475171Z","shell.execute_reply.started":"2022-12-11T05:07:26.017326Z","shell.execute_reply":"2022-12-11T05:07:26.473990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport heapq\nimport pickle\nimport numba as nb\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\n\ntail = 30\nparallel = 1024\ntopn = 20\nops_weights = np.array([1.0, 6.0, 3.0])\nOP_WEIGHT = 0; TIME_WEIGHT = 1\nparallel = 1024\ntest_ops_weights = np.array([1.0, 6.0, 3.0])","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-13T14:40:14.958526Z","iopub.execute_input":"2022-11-13T14:40:14.959427Z","iopub.status.idle":"2022-11-13T14:40:15.80325Z","shell.execute_reply.started":"2022-11-13T14:40:14.959316Z","shell.execute_reply":"2022-11-13T14:40:15.80203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data\n\n#### Meta DataFrame\n\n`train.csv/test.csv` columns:\n\n- `session`: ordered session ids\n- `length`: length of each session\n- `start_time`: start time stamp (unit=1sec) of this session\n\n#### Data Array\n\n`train.npz/test.npz`:\n\nAn npz contains three 1-dim arrays: \n\n- `aids`: aid array\n- `ts`: time stamp array (minused by start_time of corresponding session)\n- `ops`: type array, `clicks=0, carts=1, orders=2`\n\nAn array is the concatenation of each session, ordered by session id\n\ne.g.\n\n```python\naids = np.concatenate([\n        [1, 2, 3], # session 0 time-ordered aids\n        [4, 5, 6], # session 1 time-ordered aids\n        ...])\n```\n","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/otto-data/train.csv\")\ndf_test = pd.read_csv(\"../input/otto-data/test.csv\")\ndf = pd.concat([df, df_test]).reset_index(drop = True)\nnpz = np.load(\"../input/otto-data/train.npz\")\nnpz_test = np.load(\"../input/otto-data/test.npz\")\naids = np.concatenate([npz['aids'], npz_test['aids']])\nts = np.concatenate([npz['ts'], npz_test['ts']])\nops = np.concatenate([npz['ops'], npz_test['ops']])\n\ndf[\"idx\"] = np.cumsum(df.length) - df.length\ndf[\"end_time\"] = df.start_time + ts[df.idx + df.length - 1]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T14:40:15.807279Z","iopub.execute_input":"2022-11-13T14:40:15.807863Z","iopub.status.idle":"2022-11-13T14:40:54.20763Z","shell.execute_reply.started":"2022-11-13T14:40:15.807826Z","shell.execute_reply":"2022-11-13T14:40:54.206425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Numba functions\n\nTo fully utilize the ability of numba's jit, and to avoid memory errors, we process `parallel=1024` sessions in one jit function call.\n\n- Firstly, we count the `(aid1, aid2)` pairs in each session in parallel. \n\n- Then, we merge these pairs into a nested counter `{aid1: {aid2: n}, ...}`\n\n- Finally, we find the top-k `aid2` from `aid1`'s counter","metadata":{}},{"cell_type":"code","source":"# get pair dict {(aid1, aid2): weight} for each session\n# The maximum time span between two points is 1 day = 24 * 60 * 60 sec\n@nb.jit(nopython = True, cache = True)\ndef get_single_pairs(pairs, aids, ts, ops, idx, length, start_time, ops_weights, mode):\n    max_idx = idx + length\n    min_idx = max(max_idx - tail, idx)\n    for i in range(min_idx, max_idx):\n        for j in range(i + 1, max_idx):\n            if ts[j] - ts[i] >= 24 * 60 * 60: break\n            if aids[i] == aids[j]: continue\n            if mode == OP_WEIGHT:\n                w1 = ops_weights[ops[j]]\n                w2 = ops_weights[ops[i]]\n            elif mode == TIME_WEIGHT:\n                w1 = 1 + 3 * (ts[i] + start_time - 1659304800) / (1662328791 - 1659304800)\n                w2 = 1 + 3 * (ts[j] + start_time - 1659304800) / (1662328791 - 1659304800)\n            pairs[(aids[i], aids[j])] = w1\n            pairs[(aids[j], aids[i])] = w2\n\n# get pair dict of each session in parallel\n# merge pairs into a nested dict format (cnt)\n@nb.jit(nopython = True, parallel = True, cache = True)\ndef get_pairs(aids, ts, ops, row, cnts, ops_weights, mode):\n    par_n = len(row)\n    pairs = [{(0, 0): 0.0 for _ in range(0)} for _ in range(par_n)]\n    for par_i in nb.prange(par_n):\n        _, idx, length, start_time = row[par_i]\n        get_single_pairs(pairs[par_i], aids, ts, ops, idx, length, start_time, ops_weights, mode)\n    for par_i in range(par_n):\n        for (aid1, aid2), w in pairs[par_i].items():\n            if aid1 not in cnts: cnts[aid1] = {0: 0.0 for _ in range(0)}\n            cnt = cnts[aid1]\n            if aid2 not in cnt: cnt[aid2] = 0.0\n            cnt[aid2] += w\n    \n# util function to get most common keys from a counter dict using min-heap\n# overwrite == 1 means the later item with equal weight is more important\n# otherwise, means the former item with equal weight is more important\n# the result is ordered from higher weight to lower weight\n@nb.jit(nopython = True, cache = True)\ndef heap_topk(cnt, overwrite, cap):\n    q = [(0.0, 0, 0) for _ in range(0)]\n    for i, (k, n) in enumerate(cnt.items()):\n        if overwrite == 1:\n            heapq.heappush(q, (n, i, k))\n        else:\n            heapq.heappush(q, (n, -i, k))\n        if len(q) > cap:\n            heapq.heappop(q)\n    return [heapq.heappop(q)[2] for _ in range(len(q))][::-1]\n   \n# save top-k aid2 for each aid1's cnt\n@nb.jit(nopython = True, cache = True)\ndef get_topk(cnts, topk, k):\n    for aid1, cnt in cnts.items():\n        topk[aid1] = np.array(heap_topk(cnt, 1, k))","metadata":{"execution":{"iopub.status.busy":"2022-11-13T14:40:54.209284Z","iopub.execute_input":"2022-11-13T14:40:54.209692Z","iopub.status.idle":"2022-11-13T14:40:54.471035Z","shell.execute_reply.started":"2022-11-13T14:40:54.209657Z","shell.execute_reply":"2022-11-13T14:40:54.469904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train\n\nHere we use topk counting from [Deotte's notebooks](https://www.kaggle.com/code/cdeotte/candidate-rerank-model-lb-0-573). I dropped some of his ideas to make the whole pipeline simple but still strong.\n\n- **mode 0**: Counters are weighted by operation types, with `op_weights = [1.0, 6.0, 3.0]`. This will be used for carts and orders prediction.\n\n- **mode 1**: Counters are weighted by operation time.  This will be used for clicks prediction.\n\nFor each mode, the eastimated running time is **7min~8min** in kaggle notebook","metadata":{}},{"cell_type":"code","source":"topks = {}\n\n# for two modes\nfor mode in [OP_WEIGHT, TIME_WEIGHT]:\n    # get nested counter\n    cnts = nb.typed.Dict.empty(\n        key_type = nb.types.int64,\n        value_type = nb.typeof(nb.typed.Dict.empty(key_type = nb.types.int64, value_type = nb.types.float64)))\n    max_idx = len(df)\n    for idx in tqdm(range(0, max_idx, parallel)):\n        row = df.iloc[idx:min(idx + parallel, max_idx)][['session', 'idx', 'length', 'start_time']].values\n        get_pairs(aids, ts, ops, row, cnts, ops_weights, mode)\n\n    # get topk from counter\n    topk = nb.typed.Dict.empty(\n            key_type = nb.types.int64,\n            value_type = nb.types.int64[:])\n    get_topk(cnts, topk, topn)\n\n    del cnts; gc.collect()\n    topks[mode] = topk","metadata":{"execution":{"iopub.status.busy":"2022-11-13T14:40:54.474038Z","iopub.execute_input":"2022-11-13T14:40:54.474491Z","iopub.status.idle":"2022-11-13T14:57:30.234723Z","shell.execute_reply.started":"2022-11-13T14:40:54.474443Z","shell.execute_reply":"2022-11-13T14:57:30.233456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Inference Numba Functions\n\nThe logic is a simpler version of [Deotte's notebook](https://www.kaggle.com/code/cdeotte/candidate-rerank-model-lb-0-573).\n\nSame as training functions, we process `parallel=1024` sessions in one jit function call. For each session, if the unique aids is more than 20, we rerank them by a time-decay weight. Otherwise, we use current aids to recall new aids from their topk co-visitation candidates, weighted by the query's operation type (`test_ops_weights=[1.0, 6.0, 3.0]`). \n\nWe used **mode0** topk to generate predictions for carts and orders, and **mode1** topk to generate predictions for clicks. ","metadata":{}},{"cell_type":"code","source":"@nb.jit(nopython = True, cache = True)\ndef inference_(aids, ops, row, result, topk, test_ops_weights, seq_weight):\n    for session, idx, length in row:\n        unique_aids = nb.typed.Dict.empty(key_type = nb.types.int64, value_type = nb.types.float64)\n        cnt = nb.typed.Dict.empty(key_type = nb.types.int64, value_type = nb.types.float64)\n        \n        candidates = aids[idx:idx + length][::-1]\n        candidates_ops = ops[idx:idx + length][::-1]\n        for a in candidates:\n            unique_aids[a] = 0\n                \n        if len(unique_aids) >= 20:\n            sequence_weight = np.power(2, np.linspace(seq_weight, 1, len(candidates)))[::-1] - 1\n            for a, op, w in zip(candidates, candidates_ops, sequence_weight):\n                if a not in cnt: cnt[a] = 0\n                cnt[a] += w * test_ops_weights[op]\n            result_candidates = heap_topk(cnt, 0, 20)\n        else:\n            result_candidates = list(unique_aids)\n            for a in result_candidates:\n                if a not in topk: continue\n                for b in topk[a]:\n                    if b in unique_aids: continue\n                    if b not in cnt: cnt[b] = 0\n                    cnt[b] += 1\n            result_candidates.extend(heap_topk(cnt, 0, 20 - len(result_candidates)))\n        result[session] = np.array(result_candidates)\n        \n@nb.jit(nopython = True)\ndef inference(aids, ops, row, \n              result_clicks, result_buy,\n              topk_clicks, topk_buy,\n              test_ops_weights):\n    inference_(aids, ops, row, result_clicks, topk_clicks, test_ops_weights, 0.1)\n    inference_(aids, ops, row, result_buy, topk_buy, test_ops_weights, 0.5)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:01:48.75753Z","iopub.execute_input":"2022-11-13T15:01:48.757954Z","iopub.status.idle":"2022-11-13T15:01:48.78381Z","shell.execute_reply.started":"2022-11-13T15:01:48.757921Z","shell.execute_reply":"2022-11-13T15:01:48.782726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result place holder\nresult_clicks = nb.typed.Dict.empty(\n    key_type = nb.types.int64,\n    value_type = nb.types.int64[:])\nresult_buy = nb.typed.Dict.empty(\n    key_type = nb.types.int64,\n    value_type = nb.types.int64[:])\nfor idx in tqdm(range(len(df) - len(df_test), len(df), parallel)):\n    row = df.iloc[idx:min(idx + parallel, len(df))][['session', 'idx', 'length']].values\n    inference(aids, ops, row, result_clicks, result_buy, topks[TIME_WEIGHT], topks[OP_WEIGHT], test_ops_weights)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:01:49.54435Z","iopub.execute_input":"2022-11-13T15:01:49.544781Z","iopub.status.idle":"2022-11-13T15:01:51.991856Z","shell.execute_reply.started":"2022-11-13T15:01:49.544743Z","shell.execute_reply":"2022-11-13T15:01:51.990734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs = []\nop_names = [\"clicks\", \"carts\", \"orders\"]\nfor result, op in zip([result_clicks, result_buy, result_buy], op_names):\n\n    sub = pd.DataFrame({\"session_type\": result.keys(), \"labels\": result.values()})\n    sub.session_type = sub.session_type.astype(str) + f\"_{op}\"\n    sub.labels = sub.labels.apply(lambda x: \" \".join(x.astype(str)))\n    subs.append(sub)\n    \nsub = pd.concat(subs).reset_index(drop = True)\nsub.to_csv('submission.csv', index = False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:01:51.993623Z","iopub.execute_input":"2022-11-13T15:01:51.99398Z","iopub.status.idle":"2022-11-13T15:01:52.756408Z","shell.execute_reply.started":"2022-11-13T15:01:51.993947Z","shell.execute_reply":"2022-11-13T15:01:52.755261Z"},"trusted":true},"execution_count":null,"outputs":[]}]}