{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## OTTO – Multi-Objective Recommender System","metadata":{}},{"cell_type":"markdown","source":"## 1. Setup","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\nimport pathlib\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport datatable as dt\nfrom datetime import timedelta\n\nfrom plotly.offline import init_notebook_mode, iplot, plot\nimport plotly as py\ninit_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.figure_factory as ff","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-14T14:52:26.081754Z","iopub.execute_input":"2022-11-14T14:52:26.082283Z","iopub.status.idle":"2022-11-14T14:52:29.995197Z","shell.execute_reply.started":"2022-11-14T14:52:26.082168Z","shell.execute_reply":"2022-11-14T14:52:29.993924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_dataset_directory = pathlib.Path('../input/otto-recommender-system')\npickled_dataset_directory = pathlib.Path('../input/otto-multi-objective-recommender-system-pickle')\n\ndf_train = pd.read_pickle(pickled_dataset_directory / 'train.pkl')\ndf_test = pd.read_pickle(pickled_dataset_directory / 'test.pkl')\n\nprint(f'Training Shape: {df_train.shape} - Memory Usage: {df_train.memory_usage().sum() / 1024 ** 2:.2f} MB')\nprint(f'Test Shape: {df_test.shape} - Memory Usage: {df_test.memory_usage().sum() / 1024 ** 2:.2f} MB')","metadata":{"execution":{"iopub.status.busy":"2022-11-14T14:52:32.137535Z","iopub.execute_input":"2022-11-14T14:52:32.138128Z","iopub.status.idle":"2022-11-14T14:53:07.623736Z","shell.execute_reply.started":"2022-11-14T14:52:32.138071Z","shell.execute_reply":"2022-11-14T14:53:07.622302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  时间间隔与动作关系探索","metadata":{}},{"cell_type":"code","source":"df_train_by_session = df_train.groupby('session')\ndf_train[\"prev_ts\"] = df_train_by_session[\"ts\"].shift(1)\ndf_train[\"interval\"]  = df_train[\"ts\"] - df_train[\"prev_ts\"]\ndf_train[\"interval\"].fillna(timedelta(0), inplace=True)\ndel df_train_by_session","metadata":{"execution":{"iopub.status.busy":"2022-11-14T14:53:34.885642Z","iopub.execute_input":"2022-11-14T14:53:34.886551Z","iopub.status.idle":"2022-11-14T14:53:57.100859Z","shell.execute_reply.started":"2022-11-14T14:53:34.886509Z","shell.execute_reply":"2022-11-14T14:53:57.099643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_by_type = df_train.groupby(\"type\")\nprint(df_train_by_type[\"interval\"].mean())\ndel df_train_by_type","metadata":{"execution":{"iopub.status.busy":"2022-11-14T14:54:29.257962Z","iopub.execute_input":"2022-11-14T14:54:29.259051Z","iopub.status.idle":"2022-11-14T14:54:35.002738Z","shell.execute_reply.started":"2022-11-14T14:54:29.259003Z","shell.execute_reply":"2022-11-14T14:54:35.001810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"从上面结果可以推断：\n1. 添加购物车不可能出现一天活动的开头部分\n2. 但是购买可能出现在一天活动的开头部分","metadata":{}},{"cell_type":"code","source":"df_train_by_type_aid = df_train.groupby([\"type\",\"aid\"], group_keys=True).count()","metadata":{"execution":{"iopub.status.busy":"2022-11-14T15:10:31.521987Z","iopub.execute_input":"2022-11-14T15:10:31.522460Z","iopub.status.idle":"2022-11-14T15:11:19.483104Z","shell.execute_reply.started":"2022-11-14T15:10:31.522429Z","shell.execute_reply":"2022-11-14T15:11:19.481743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_by_type_aid.loc[0].sort_values(\"session\", ascending=False).head(20)","metadata":{"execution":{"iopub.status.busy":"2022-11-14T15:24:02.920962Z","iopub.execute_input":"2022-11-14T15:24:02.921382Z","iopub.status.idle":"2022-11-14T15:24:03.326614Z","shell.execute_reply.started":"2022-11-14T15:24:02.921347Z","shell.execute_reply":"2022-11-14T15:24:03.325780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_by_type_aid.loc[1].sort_values(\"session\", ascending=False).head(20)","metadata":{"execution":{"iopub.status.busy":"2022-11-14T15:24:21.275537Z","iopub.execute_input":"2022-11-14T15:24:21.276088Z","iopub.status.idle":"2022-11-14T15:24:21.454112Z","shell.execute_reply.started":"2022-11-14T15:24:21.276040Z","shell.execute_reply":"2022-11-14T15:24:21.452784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_by_type_aid.loc[2].sort_values(\"session\", ascending=False).head(20)","metadata":{"execution":{"iopub.status.busy":"2022-11-14T15:24:42.134069Z","iopub.execute_input":"2022-11-14T15:24:42.134475Z","iopub.status.idle":"2022-11-14T15:24:42.241083Z","shell.execute_reply.started":"2022-11-14T15:24:42.134445Z","shell.execute_reply":"2022-11-14T15:24:42.239775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Introduction\n\nThis is a large dataset of e-commerce sessions. Sessions consist of user events such as clicking to a product, adding a product to cart or ordering a product. Each session belongs to a unique user and duration of sessions are different from each other.\n\n* `session`: unique ID of the user session\n* `aid`: unique ID of the product\n* `ts`: timestamp of the event\n* `type`: category of the event\n\nIt is a multi-objective ranking problem since there are 3 categories (click, cart, order) to recommend for each session. Recommendations for each category are evaluated on recall@20 and scores are multiplied with their corresponding weights.\n\n* 216.7M events in training set and 6.9M events in test set\n* 12.8M sessions in training set and 1.6m sessions in test set\n* 1.8M products in training set and 783K products in test set\n* 194.7M clicks in training set and 6.2M clicks in test set\n* 16.8M carts in training set and 570K carts in test set\n* 5M orders in training set and 65K orders in test set","metadata":{}},{"cell_type":"code","source":"train_events = df_train.shape[0]\ntest_events = df_test.shape[0]\nprint(f'Number of Events - Training: {train_events} | Test: {test_events}')\n\ntrain_unique_sessions = df_train['session'].unique()\ntest_unique_sessions = df_test['session'].unique()\nprint(f'Number of Unique Sessions - Training: {len(train_unique_sessions)} | Test: {len(test_unique_sessions)}')\ndel train_unique_sessions, test_unique_sessions\n\ntrain_unique_aids = df_train['aid'].unique()\ntest_unique_aids = df_test['aid'].unique()\noverlapping_aids = set(train_unique_aids).intersection(set(test_unique_aids))\nprint(f'Number of Unique Products - Training: {len(train_unique_aids)} | Test: {len(test_unique_aids)} - ({len(overlapping_aids)} Overlapping Products)')\ndel train_unique_aids, test_unique_aids, overlapping_aids\n\ntrain_clicks = df_train[df_train['type'] == 0].shape[0]\ntest_clicks = df_test[df_test['type'] == 0].shape[0]\nprint(f'Number of Clicks - Training: {train_clicks} | Test: {test_clicks}')\n\ntrain_carts = df_train[df_train['type'] == 1].shape[0]\ntest_carts = df_test[df_test['type'] == 1].shape[0]\nprint(f'Number of Carts - Training: {train_carts} | Test: {test_carts}')\n\ntrain_orders = df_train[df_train['type'] == 2].shape[0]\ntest_orders = df_test[df_test['type'] == 2].shape[0]\nprint(f'Number of Orders - Training: {train_orders} | Test: {test_orders}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_start = df_train['ts'].min().strftime('%Y.%m.%d %X')\ntrain_end = df_train['ts'].max().strftime('%Y.%m.%d %X')\ntest_start = df_test['ts'].min().strftime('%Y.%m.%d %X')\ntest_end = df_test['ts'].max().strftime('%Y.%m.%d %X')\nprint(f'Events Time Range - Training: {train_start}-{train_end} | Test: {test_start}-{test_end}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Training and test set are split by time. Training set is 4 weeks of user events and test set is the following week after training set. There aren't any overlapping events between training and test set because some of the training set events are trimmed in order to prevent leakage. Proportions of event types are similar in training and test set which can be seen from the visualization below.","metadata":{}},{"cell_type":"code","source":"def visualize_categorical_feature_distribution(df, feature):\n\n    \"\"\"\n    Visualize distribution of given categorical column in given dataframe\n    \n    Parameters\n    ----------\n    df: pandas.DataFrame of shape (n_samples, 4)\n        Dataframe with session, aid, ts, type columns\n        \n    feature: str\n        Name of the categorical feature\n    \n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n    feat_value_count = df[feature].value_counts()\n    trace1 = go.Bar(\n                x = feat_value_count.index,\n                y = feat_value_count.values,\n                name = \"type\",\n                marker = dict(color = 'skyblue',\n                             line=dict(color='rgb(0,0,0)',width=1.5)))\n    data = [trace1]\n    layout = go.Layout()\n    fig = go.Figure(data = data, layout = layout)\n    iplot(fig)\n    \n\nvisualize_categorical_feature_distribution(df=df_train, feature='type')\nvisualize_categorical_feature_distribution(df=df_test, feature='type')","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Users and Products\n\nAs we already know, there are 12.8M unique user sessions in training set and 1.6m unique user sessions in test set. However, their statistics are also different between training and test set. In training set, mean session duration is 4 times longer than test set. Shortest session in training set is 2 because it wouldn't be possible to create ground-truth on single event sessions. Longest session in training set is 500 which looks like a threshold for trimming extremely long sessions.\n\nDiscrepancy in session statistics may not be a problem because training set is 4 weeks while test set is only a single week. Mean session duration being 4 times longer in training set is also an artifact of that.\n\nDistributions of session durations can be seen from the visualization below. Both densities are on log scale for interpretability.","metadata":{}},{"cell_type":"code","source":"def visualize_continuous_feature_distribution(df_train, df_test, feature, path=None):\n\n    \"\"\"\n    Visualize distribution of given continuous column in given dataframe\n    \n    Parameters\n    ----------\n    df_train: pandas.DataFrame of shape (n_samples, 4)\n        Training dataframe with session, aid, ts, type columns\n        \n    df_test: pandas.DataFrame of shape (n_samples, 4)\n        Test dataframe with session, aid, ts, type columns\n        \n    feature: str\n        Name of the continuous feature\n    \n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n    \n    fig, ax = plt.subplots(figsize=(24, 6), dpi=100)\n\n    sns.kdeplot(df_train[feature], label='train', fill=True, log_scale=True, ax=ax)\n    sns.kdeplot(df_test[feature], label='test', fill=True, log_scale=True, ax=ax)\n    ax.tick_params(axis='x', labelsize=12.5)\n    ax.tick_params(axis='y', labelsize=12.5)\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.legend(prop={'size': 15})\n    title = f'''\n    {feature}\n    Mean - Train: {df_train[feature].mean():.2f} |  Test: {df_test[feature].mean():.2f}\n    Median - Train: {df_train[feature].median():.2f} |  Test: {df_test[feature].median():.2f}\n    Std - Train: {df_train[feature].std():.2f} |  Test: {df_test[feature].std():.2f}\n    Min - Train: {df_train[feature].min():.2f} |  Test: {df_test[feature].min():.2f}\n    Max - Train: {df_train[feature].max():.2f} |  Test: {df_test[feature].max():.2f}\n    '''\n    ax.set_title(title, size=20, pad=15)\n    \n    if path is None:\n        plt.show()\n    else:\n        plt.savefig(path)\n        plt.close(fig)","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_session_groupby = df_train.groupby('session')\ndf_train_sessions = train_session_groupby[['session']].count().reset_index(drop=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_session_groupby = df_test.groupby('session')\ndf_test_sessions = test_session_groupby[['session']].count()\nvisualize_continuous_feature_distribution(df_train=df_train_sessions, df_test=df_test_sessions, feature='session')\ndel train_session_groupby, df_train_sessions, test_session_groupby, df_test_sessions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 1.8M unique products in training set and 783K unique products in test set. Products in test set are a subset of products in training set so there aren't any unseen products in test set. Distribution of product occurences is different in training and test set as well because of the previously mentioned reason. Training set product occurences distribution is more normal while test set product occurences distribution is multi-modal which could be an artifact of session truncation.\n\nDistributions of product occurences can be seen from the visualization below. Both densities are on log scale for interpretability.","metadata":{}},{"cell_type":"code","source":"train_product_groupby = df_train.groupby('aid')\ndf_train_products = train_product_groupby[['aid']].count().reset_index(drop=True)\ntest_product_groupby = df_test.groupby('aid')\ndf_test_products = test_product_groupby[['aid']].count().reset_index(drop=True)\nvisualize_continuous_feature_distribution(df_train=df_train_products, df_test=df_test_products, feature='aid')\ndel train_product_groupby, df_train_products, test_product_groupby, df_test_products","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_session_type_groupby = df_train.groupby(['session', 'type'])\ndf_train_sessions_types = train_session_type_groupby[['session']].count().rename(columns={'session': 'count'}).reset_index()\ndf_train_sessions_types.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_sessions_types['total'] = df_train_sessions_types.groupby('session')['count'].transform('sum')\ndf_train_sessions_types.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_sessions_types.sort_values(by=['total', 'session', 'type'], ascending=False, inplace=True)\ndf_train_sessions_types['rate'] = df_train_sessions_types['count'] / df_train_sessions_types['total']\ndf_train_sessions_types.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Events\n\nIn training set 93% of sessions are clicks, 16% of sessions are carts and 10% of sessions are orders on average. Those numbers don't add up to 100 because they are averages calculated on all sessions. They have 12, 10 and 8 standard deviations respectively. In test set those numbers are little bit higher for averages and standard deviations because sessions are truncated randomly. Events are likely over/under represented in test sessions.\n\n(0 -> clicks, 1 -> carts, 2 -> orders)","metadata":{}},{"cell_type":"code","source":"train_session_type_groupby = df_train.groupby(['session', 'type'])\ndf_train_sessions_types = train_session_type_groupby[['session']].count().rename(columns={'session': 'count'}).reset_index()\ndf_train_sessions_types['total'] = df_train_sessions_types.groupby('session')['count'].transform('sum')\ndf_train_sessions_types.sort_values(by=['total', 'session', 'type'], ascending=False, inplace=True)\ndf_train_sessions_types['rate'] = df_train_sessions_types['count'] / df_train_sessions_types['total']\ndf_train_sessions_type_rate_means = df_train_sessions_types.groupby('type')['rate'].mean().to_dict()\ndf_train_sessions_type_rate_stds = df_train_sessions_types.groupby('type')['rate'].std().to_dict()\n\ntest_session_type_groupby = df_test.groupby(['session', 'type'])\ndf_test_sessions_types = test_session_type_groupby[['session']].count().rename(columns={'session': 'count'}).reset_index()\ndf_test_sessions_types['total'] = df_test_sessions_types.groupby('session')['count'].transform('sum')\ndf_test_sessions_types.sort_values(by=['total', 'session', 'type'], ascending=False, inplace=True)\ndf_test_sessions_types['rate'] = df_test_sessions_types['count'] / df_test_sessions_types['total']\ndf_test_sessions_type_rate_means = df_test_sessions_types.groupby('type')['rate'].mean().to_dict()\ndf_test_sessions_type_rate_stds = df_test_sessions_types.groupby('type')['rate'].std().to_dict()\n\nprint(\nf'''\nSession Event Type Percentages on Average\nTraining: 0: {df_train_sessions_type_rate_means[0]:.4f}(±{df_train_sessions_type_rate_stds[0]:.4f}) | 1: {df_train_sessions_type_rate_means[1]:.4f}(±{df_train_sessions_type_rate_stds[1]:.4f}) | 2: {df_train_sessions_type_rate_means[2]:.4f}(±{df_train_sessions_type_rate_stds[2]:.4f})\nTest: 0: {df_test_sessions_type_rate_means[0]:.4f}(±{df_test_sessions_type_rate_stds[0]:.4f}) | 1: {df_test_sessions_type_rate_means[1]:.4f}(±{df_test_sessions_type_rate_stds[1]:.4f}) | 2: {df_test_sessions_type_rate_means[2]:.4f}(±{df_test_sessions_type_rate_stds[2]:.4f})\n'''\n)\ndel df_train_sessions_type_rate_means, df_train_sessions_type_rate_stds, df_test_sessions_type_rate_means, df_test_sessions_type_rate_stds","metadata":{"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some of the sessions might not have any clicks, carts or orders but the statistics below doesn't include their missing event types. The highlight is, some of the sessions in test set can only have clicks, carts or orders, but training set sessions are more heterogeneous. Homogeneous sessions in training set are click-only sessions. This discrepancy could be a problem during modelling.","metadata":{}},{"cell_type":"code","source":"df_train_sessions_type_rate_mins = df_train_sessions_types.groupby('type')['rate'].min().to_dict()\ndf_train_sessions_type_rate_maxs = df_train_sessions_types.groupby('type')['rate'].max().to_dict()\ndf_test_sessions_type_rate_mins = df_test_sessions_types.groupby('type')['rate'].min().to_dict()\ndf_test_sessions_type_rate_maxs = df_test_sessions_types.groupby('type')['rate'].max().to_dict()\n\nprint(\nf'''\nSession Event Type Least and Most Percentages\nTraining: 0: {df_train_sessions_type_rate_mins[0]:.4f}-{df_train_sessions_type_rate_maxs[0]:.4f} | 1: {df_train_sessions_type_rate_mins[1]:.4f}-{df_train_sessions_type_rate_maxs[1]:.4f} | 2: {df_train_sessions_type_rate_mins[2]:.4f}-{df_train_sessions_type_rate_maxs[2]:.4f}\nTest: 0: {df_test_sessions_type_rate_mins[0]:.4f}-{df_test_sessions_type_rate_maxs[0]:.4f} | 1: {df_test_sessions_type_rate_mins[1]:.4f}-{df_test_sessions_type_rate_maxs[1]:.4f} | 2: {df_test_sessions_type_rate_mins[2]:.4f}-{df_test_sessions_type_rate_maxs[2]:.4f}\n'''\n)\ndel df_train_sessions_type_rate_mins, df_train_sessions_type_rate_maxs, df_test_sessions_type_rate_mins, df_test_sessions_type_rate_maxs","metadata":{"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Another important thing to consider is how sessions start and end. There are two interesting things here. First, sessions are supposed to start with clicks but very few of them start with carts or orders. Those sessions are less than 1% and they might be truncated from their beginning so they could fit into the selected time frame. The other thing is, sessions in test set are less likely to end with an order because of the truncation. Truncated timesteps in test set are more likely order events.  ","metadata":{}},{"cell_type":"code","source":"df_train_session_firsts = (df_train.groupby('session')['type'].first().value_counts() / df_train['session'].nunique()).to_dict()\ndf_train_session_lasts = (df_train.groupby('session')['type'].last().value_counts() / df_train['session'].nunique()).to_dict()\ndf_test_session_firsts = (df_test.groupby('session')['type'].first().value_counts() / df_test['session'].nunique()).to_dict()\ndf_test_session_lasts = (df_test.groupby('session')['type'].last().value_counts() / df_test['session'].nunique()).to_dict()\n\nprint(\nf'''\nSession Event Type First and Last Percentages\nTraining: 0: {df_train_session_firsts[0]:.4f}-{df_train_session_lasts[0]:.4f} | 1: {df_train_session_firsts[1]:.4f}-{df_train_session_lasts[1]:.4f} | 2: {df_train_session_firsts[2]:.4f}-{df_train_session_lasts[2]:.4f}\nTest: 0: {df_test_session_firsts[0]:.4f}-{df_test_session_lasts[0]:.4f} | 1: {df_test_session_firsts[1]:.4f}-{df_test_session_lasts[1]:.4f} | 2: {df_test_session_firsts[2]:.4f}-{df_test_session_lasts[2]:.4f}\n'''\n)\ndel df_train_session_firsts, df_train_session_lasts, df_test_session_firsts, df_test_session_lasts","metadata":{"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A natural sequence is clicking to a product, adding it to the cart and ordering it. It can be seen from the visualization of session 3 below that cart events lead to order events multiple times. However, this doesn't apply to all sessions.","metadata":{}},{"cell_type":"code","source":"def visualize_session(df, session, path=None):\n    \n    \"\"\"\n    Visualize given session in given dataframe as a sequence\n    \n    Parameters\n    ----------\n    df: pandas.DataFrame of shape (n_samples, 4)\n        Training or test dataframe with session, aid, ts, type columns\n\n    feature: int\n        Unique ID of the session\n    \n    path: path-like str or None\n        Path of the output file or None (if path is None, plot is displayed with selected backend)\n    \"\"\"\n    \n    df_session = df.loc[df['session'] == session, :]\n    trace1 = go.Scatter(\n                    x = df_session.ts,\n                    y = df_session.type,\n                    mode = \"lines + markers\",\n                    name = f\"session {session}\",\n                    marker = dict(color = 'rgba(16, 112, 2, 0.8)'),\n                    text= df_session.aid)\n    data = [trace1]\n    layout = dict(title = f'''\n                    Session: {session}\n                    Clicks: {(df_session['type'] == 0).sum()} |  Carts: {(df_session['type'] == 1).sum()} | Orders: {(df_session['type'] == 2).sum()}\n                    ''',\n              xaxis= dict(title= 'Timestamps',ticklen= 5,zeroline= False),\n              yaxis= dict(title= 'Events',ticklen= 5,zeroline= False)\n             )\n    fig = dict(data = data, layout = layout)\n    iplot(fig)\n   \n\nvisualize_session(df=df_train, session=3)","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In training set, session 39 starts with cart event and session 747 starts with order event which are anomalies mentioned before.","metadata":{}},{"cell_type":"code","source":"visualize_session(df=df_train, session=39)\nvisualize_session(df=df_train, session=747)","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Ground-truth\n\nGround-truth of both training and test set can be created as long as a session has more than 1 timesteps. Ground-truth of clicks and other events are created differently. Ground-truth click of a timestep is the next click in that session. Ground-truth of carts or orders of a timestep is the collection of all next unique carts or orders in that session.","metadata":{}},{"cell_type":"code","source":"def get_labels(aids, event_types):\n    \n    \"\"\"\n    Create ground-truth labels from given session aids and event types\n    \n    Parameters\n    ----------\n    aids: pandas.Series of shape (n_events)\n        Session aids\n\n    event_types: pandas.Series of shape (n_events)\n        Session event types\n        \n    Returns\n    -------\n    labels: list of shape (n_events)\n        Ground-truth labels\n    \"\"\"\n    \n    previous_click = None\n    previous_carts = set()\n    previous_orders = set()\n    labels = []\n\n    for aid, event_type in zip(reversed(aids.values), reversed(event_types.values)):\n        \n        label = {}\n        \n        if event_type == 0:\n            previous_click = aid\n        elif event_type == 1:\n            previous_carts.add(aid)\n        elif event_type == 2:\n            previous_orders.add(aid)\n            \n        label[0] = previous_click \n        label[1] = previous_carts.copy() if len(previous_carts) > 0 else np.nan\n        label[2] = previous_orders.copy() if len(previous_orders) > 0 else np.nan\n        labels.append(label)\n        \n    labels = labels[:-1][::-1]\n    labels.append({0: np.nan, 1: np.nan, 2: np.nan})\n    \n    return labels\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Labels of session 747 are extracted with the function above. It can be seen that after a cart or order event, corresponding aid is dropped from the labels.","metadata":{}},{"cell_type":"code","source":"df_session747 = df_train.loc[df_train['session'] == 747, :]\nsession747_labels = get_labels(aids=df_session747['aid'], event_types=df_session747['type'])\ndf_session747.loc[:, 'label'] = session747_labels\ndf_session747","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session747_labels","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}