{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":38760,"databundleVersionId":4493939,"sourceType":"competition"}],"dockerImageVersionId":30579,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:01:47.342305Z","iopub.execute_input":"2023-11-18T00:01:47.343545Z","iopub.status.idle":"2023-11-18T00:01:49.337853Z","shell.execute_reply.started":"2023-11-18T00:01:47.343456Z","shell.execute_reply":"2023-11-18T00:01:49.336204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunks = pd.read_json(\"/kaggle/input/otto-recommender-system/train.jsonl\", lines=True, chunksize=2_000)\n\nfor chunk in chunks:\n    df_sample = chunk\n    break","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:01:49.340451Z","iopub.execute_input":"2023-11-18T00:01:49.341820Z","iopub.status.idle":"2023-11-18T00:01:49.903385Z","shell.execute_reply.started":"2023-11-18T00:01:49.341760Z","shell.execute_reply":"2023-11-18T00:01:49.902245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:01:49.905453Z","iopub.execute_input":"2023-11-18T00:01:49.906316Z","iopub.status.idle":"2023-11-18T00:01:50.044055Z","shell.execute_reply.started":"2023-11-18T00:01:49.906272Z","shell.execute_reply":"2023-11-18T00:01:50.043068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_data(df: pd.DataFrame):\n    df_c = df.explode(\"events\", ignore_index=True)\n    df_c = pd.concat([df_c.drop([\"events\"], axis=1), df_c[\"events\"].apply(pd.Series)], axis=1)\n    return df_c","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:02:03.701549Z","iopub.execute_input":"2023-11-18T00:02:03.701932Z","iopub.status.idle":"2023-11-18T00:02:03.709823Z","shell.execute_reply.started":"2023-11-18T00:02:03.701903Z","shell.execute_reply":"2023-11-18T00:02:03.708021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_pp = preprocess_data(df_sample)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:02:05.264197Z","iopub.execute_input":"2023-11-18T00:02:05.264653Z","iopub.status.idle":"2023-11-18T00:02:46.013867Z","shell.execute_reply.started":"2023-11-18T00:02:05.264620Z","shell.execute_reply":"2023-11-18T00:02:46.012349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sample_pp","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:02:46.015787Z","iopub.execute_input":"2023-11-18T00:02:46.016135Z","iopub.status.idle":"2023-11-18T00:02:46.031369Z","shell.execute_reply.started":"2023-11-18T00:02:46.016106Z","shell.execute_reply":"2023-11-18T00:02:46.030018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Quantiles of events and distribution of events through time","metadata":{}},{"cell_type":"code","source":"def get_ts_distribution(df, act_type, q, act_type_colname=\"type\", ts_colname=\"ts\"):\n    if act_type != \"all\":\n        df_typed = df[df[act_type_colname] == act_type]\n    else:\n        df_typed = df.copy()\n    quantile_typed = df_typed.ts.quantile(q)\n    \n    sns.histplot(df_typed, x=ts_colname, bins=20)\n    plt.title(f\"`{act_type}` event histplot\")\n    plt.legend()\n    plt.show()\n    \n    return dict(zip(q, quantile_typed))","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:52:05.239969Z","iopub.execute_input":"2023-11-18T00:52:05.240386Z","iopub.status.idle":"2023-11-18T00:52:05.246768Z","shell.execute_reply.started":"2023-11-18T00:52:05.240356Z","shell.execute_reply":"2023-11-18T00:52:05.245538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q = np.around(np.arange(0.1, 1, 0.1), 1)\nq","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:52:05.744941Z","iopub.execute_input":"2023-11-18T00:52:05.745672Z","iopub.status.idle":"2023-11-18T00:52:05.753960Z","shell.execute_reply.started":"2023-11-18T00:52:05.745636Z","shell.execute_reply":"2023-11-18T00:52:05.752562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"global_quantiles = get_ts_distribution(df_sample_pp, \"all\", q)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:52:06.175098Z","iopub.execute_input":"2023-11-18T00:52:06.176018Z","iopub.status.idle":"2023-11-18T00:52:06.507304Z","shell.execute_reply.started":"2023-11-18T00:52:06.175973Z","shell.execute_reply":"2023-11-18T00:52:06.505796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, there are a lot of data in the start of the period. Possibly it is some big sale. Maybe it is helpful to exclude sale period","metadata":{}},{"cell_type":"code","source":"TS_MIN = 1659500000000\ndf_sample_pp_exclude_sale = df_sample_pp[df_sample_pp[\"ts\"] >= TS_MIN]","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:52:10.176921Z","iopub.execute_input":"2023-11-18T00:52:10.177414Z","iopub.status.idle":"2023-11-18T00:52:10.192773Z","shell.execute_reply.started":"2023-11-18T00:52:10.177379Z","shell.execute_reply":"2023-11-18T00:52:10.191448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Per different event types","metadata":{}},{"cell_type":"code","source":"clicks_quantiles = get_ts_distribution(df_sample_pp, \"clicks\", q)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:52:10.983160Z","iopub.execute_input":"2023-11-18T00:52:10.983848Z","iopub.status.idle":"2023-11-18T00:52:11.333007Z","shell.execute_reply.started":"2023-11-18T00:52:10.983813Z","shell.execute_reply":"2023-11-18T00:52:11.331581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"carts_quantiles = get_ts_distribution(df_sample_pp, \"carts\", q)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:52:11.631294Z","iopub.execute_input":"2023-11-18T00:52:11.631739Z","iopub.status.idle":"2023-11-18T00:52:11.922439Z","shell.execute_reply.started":"2023-11-18T00:52:11.631707Z","shell.execute_reply":"2023-11-18T00:52:11.921099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"orders_quantiles = get_ts_distribution(df_sample_pp, \"orders\", q)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:52:16.813341Z","iopub.execute_input":"2023-11-18T00:52:16.813840Z","iopub.status.idle":"2023-11-18T00:52:17.099144Z","shell.execute_reply.started":"2023-11-18T00:52:16.813797Z","shell.execute_reply":"2023-11-18T00:52:17.097835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ","metadata":{}},{"cell_type":"code","source":"from datetime import datetime\n\n\ndf_q_clicks = pd.DataFrame.from_dict(clicks_quantiles, orient=\"index\", columns=[\"clicks_ts\"])\ndf_q_carts = pd.DataFrame.from_dict(carts_quantiles, orient=\"index\", columns=[\"carts_ts\"])\ndf_q_orders = pd.DataFrame.from_dict(orders_quantiles, orient=\"index\", columns=[\"orders_ts\"])\ndf_q_global = pd.DataFrame.from_dict(global_quantiles, orient=\"index\", columns=[\"global_ts\"])\n\n\ndf_q = df_q_clicks.join(df_q_carts).join(df_q_orders).join(df_q_global)\n\ndf_q[\"max_min_deviation_hours\"] = (df_q.max(axis=1) - df_q.min(axis=1)) / 1000 / 60 / 60\ndf_q = df_q.reset_index(drop=False, names=[\"q\"])\n\ndf_q.style.background_gradient(subset=[\"clicks_ts\", \"carts_ts\", \"orders_ts\", \"global_ts\"], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:31:54.428296Z","iopub.execute_input":"2023-11-18T00:31:54.428710Z","iopub.status.idle":"2023-11-18T00:31:54.478153Z","shell.execute_reply.started":"2023-11-18T00:31:54.428681Z","shell.execute_reply":"2023-11-18T00:31:54.476347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Global max_min deviation (sampled) in hours:","metadata":{}},{"cell_type":"code","source":"(df_sample_pp.ts.max() - df_sample_pp.ts.min()) / 1000 / 60 / 60","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:32:03.843424Z","iopub.execute_input":"2023-11-18T00:32:03.843828Z","iopub.status.idle":"2023-11-18T00:32:03.852979Z","shell.execute_reply.started":"2023-11-18T00:32:03.843799Z","shell.execute_reply":"2023-11-18T00:32:03.851914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see, the quantiles deviation between actions of different types are not that big. It means that we can choose one single timestamp threshold","metadata":{}},{"cell_type":"markdown","source":"## Length of action sequences for different sessions","metadata":{}},{"cell_type":"code","source":"df_gb_all_cnts = df_sample_pp.groupby(\"session\").count()[\"aid\"]\nplt.title(\"Histplot of length of sessions\")\nsns.histplot(df_gb_all_cnts)\n\nprint(\"Mean:\", df_gb_all_cnts.mean())\n\nplt.ylabel(\"Count\")\nplt.xlabel(\"Number of items per session\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:35:26.801796Z","iopub.execute_input":"2023-11-18T00:35:26.802177Z","iopub.status.idle":"2023-11-18T00:35:27.177169Z","shell.execute_reply.started":"2023-11-18T00:35:26.802147Z","shell.execute_reply":"2023-11-18T00:35:27.175591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that mostly we have not that much actions per session, so it is harder to predict for users that have small amount of actions","metadata":{}},{"cell_type":"markdown","source":"# Distibution of events globally","metadata":{}},{"cell_type":"code","source":"df_sample_pp.groupby(\"type\")[\"session\"].agg(lambda x: len(x) / len(df_sample_pp))","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:36:40.550631Z","iopub.execute_input":"2023-11-18T00:36:40.551769Z","iopub.status.idle":"2023-11-18T00:36:40.590356Z","shell.execute_reply.started":"2023-11-18T00:36:40.551713Z","shell.execute_reply":"2023-11-18T00:36:40.589264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If we exclude possible sale","metadata":{}},{"cell_type":"code","source":"df_sample_pp_exclude_sale.groupby(\"type\")[\"session\"].agg(lambda x: len(x) / len(df_sample_pp_exclude_sale))","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:38:30.742885Z","iopub.execute_input":"2023-11-18T00:38:30.743861Z","iopub.status.idle":"2023-11-18T00:38:30.779854Z","shell.execute_reply.started":"2023-11-18T00:38:30.743810Z","shell.execute_reply":"2023-11-18T00:38:30.778363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Number of clicks is more that 90% of all data. It is easier to predict them (and harder to predict other targets)","metadata":{}},{"cell_type":"markdown","source":"## Length of session","metadata":{}},{"cell_type":"code","source":"df_session_length = (\n    df_sample_pp\n    .groupby(\"session\")[\"ts\"]\n    .agg(lambda x: (x.max() - x.min()) / 1000 / 60 / 60)\n)\n\nsns.histplot(df_session_length)\nplt.xlabel(\"Hours\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:39:46.939186Z","iopub.execute_input":"2023-11-18T00:39:46.940197Z","iopub.status.idle":"2023-11-18T00:39:47.393605Z","shell.execute_reply.started":"2023-11-18T00:39:46.940152Z","shell.execute_reply":"2023-11-18T00:39:47.392557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that sessions mostly are not short, so it seems that it is ok to divide them to train and test\n\nIn this situation session is not just short event, but it something that can actually define separete user","metadata":{}},{"cell_type":"markdown","source":"## Interactions with items","metadata":{}},{"cell_type":"code","source":"df_items_interactions = df_sample_pp.groupby([\"aid\", \"type\"])[\"session\"].count().reset_index()\n\nsns.histplot(df_items_interactions, x=\"session\", hue=\"type\", kde=True, bins=100)\nplt.xlim(0, 20)\n\nplt.title(\"Number of interactions per item\")\nplt.xlabel(\"Number of interactions\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:50:43.625366Z","iopub.execute_input":"2023-11-18T00:50:43.625852Z","iopub.status.idle":"2023-11-18T00:50:45.135228Z","shell.execute_reply.started":"2023-11-18T00:50:43.625817Z","shell.execute_reply":"2023-11-18T00:50:45.133935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_items_interactions.groupby(\"type\")[\"session\"].mean()","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:40:20.811809Z","iopub.execute_input":"2023-11-18T00:40:20.813029Z","iopub.status.idle":"2023-11-18T00:40:20.831964Z","shell.execute_reply.started":"2023-11-18T00:40:20.812991Z","shell.execute_reply":"2023-11-18T00:40:20.830782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that there is few interactions for different items (mean number is less than 2). So adjacency matrix should be really sparse (relative to items)\n\nIt is happening also because of sampled data\n\nIt would be interesting to see how parameters relative to real exposures of items (unfortunately we do not have such data)","metadata":{}},{"cell_type":"markdown","source":"## Sequential user roadmap","metadata":{}},{"cell_type":"markdown","source":"Here we want to check how many orders were previously clicked and then added to cart","metadata":{}},{"cell_type":"code","source":"def get_orders_fraction_with_correct_user_roadmap(df: pd.DataFrame, keep=\"last\"):\n    df_orders = df[df[\"type\"] == \"orders\"].drop_duplicates(subset=[\"session\", \"aid\"], keep=keep)\n    df_carts = df[df[\"type\"] == \"carts\"].drop_duplicates(subset=[\"session\", \"aid\"], keep=keep)\n    df_clicks = df[df[\"type\"] == \"clicks\"].drop_duplicates(subset=[\"session\", \"aid\"], keep=keep)\n\n    df_orders_carts = (\n        df_orders\n        .merge(\n            df_carts,\n            on=[\"session\", \"aid\"],\n            how=\"inner\",\n            suffixes=(\"_orders\", \"_carts\")\n        )\n    )\n    df_orders_carts = df_orders_carts[\n        df_orders_carts[\"ts_orders\"] > df_orders_carts[\"ts_carts\"]\n    ]\n    df_orders_carts.drop_duplicates(subset=[\"session\", \"aid\"], inplace=True, keep=keep)\n    \n    df_orders_carts_clicks = (\n        df_orders_carts\n        .merge(\n            df_clicks,\n            on=[\"session\", \"aid\"],\n            how=\"inner\",\n        )\n    )\n    df_orders_carts_clicks = df_orders_carts_clicks[\n        df_orders_carts_clicks[\"ts_carts\"] > df_orders_carts_clicks[\"ts\"]\n    ]\n    df_orders_carts_clicks.drop_duplicates(subset=[\"session\", \"aid\"], inplace=True, keep=keep)\n    \n    df_orders_clicks = (\n        df_orders\n        .merge(\n            df_clicks,\n            on=[\"session\", \"aid\"],\n            how=\"inner\",\n            suffixes=(\"_orders\", \"_clicks\")\n        )\n    )\n    df_orders_clicks = df_orders_clicks[\n        df_orders_clicks[\"ts_orders\"] > df_orders_clicks[\"ts_clicks\"]\n    ]\n    df_orders_clicks.drop_duplicates(subset=[\"session\", \"aid\"], inplace=True, keep=keep)\n    \n    df_carts_clicks = (\n        df_carts\n        .merge(\n            df_clicks,\n            on=[\"session\", \"aid\"],\n            how=\"inner\",\n            suffixes=(\"_carts\", \"_clicks\")\n        )\n    )\n    df_carts_clicks = df_carts_clicks[\n        df_carts_clicks[\"ts_carts\"] > df_carts_clicks[\"ts_clicks\"]\n    ]\n    df_carts_clicks.drop_duplicates(subset=[\"session\", \"aid\"], inplace=True, keep=keep)\n    \n    return (\n        {\n            \"orders\": len(df_orders),\n            \"orders_with_prev_cart\": len(df_orders_carts),\n            \"orders_with_prev_click\": len(df_orders_clicks),\n            \"orders_with_prev_cart_and_prev_click\": len(df_orders_carts_clicks),\n        },\n        {\n            \"carts\": len(df_carts),\n            \"carts_with_prev_click\": len(df_carts_clicks),\n        },\n        {\n            \"orders_with_prev_cart / orders\": len(df_orders_carts) / len(df_orders),\n            \"orders_with_prev_click / orders\": len(df_orders_clicks) / len(df_orders),\n            \"orders_with_prev_cart_and_prev_click / orders\": len(df_orders_carts_clicks) / len(df_orders),\n            \"carts_with_prev_clicks / carts\": len(df_carts_clicks) / len(df_carts)\n        }\n    )","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:43:31.139708Z","iopub.execute_input":"2023-11-18T00:43:31.140112Z","iopub.status.idle":"2023-11-18T00:43:31.156606Z","shell.execute_reply.started":"2023-11-18T00:43:31.140084Z","shell.execute_reply":"2023-11-18T00:43:31.155337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"orders_dict, carts_dict, fr_dict = get_orders_fraction_with_correct_user_roadmap(df_sample_pp)","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:44:14.174978Z","iopub.execute_input":"2023-11-18T00:44:14.175412Z","iopub.status.idle":"2023-11-18T00:44:14.385709Z","shell.execute_reply.started":"2023-11-18T00:44:14.175382Z","shell.execute_reply":"2023-11-18T00:44:14.384466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame.from_dict(orders_dict, orient=\"index\", columns=[\"count\"])","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:45:12.103533Z","iopub.execute_input":"2023-11-18T00:45:12.103917Z","iopub.status.idle":"2023-11-18T00:45:12.112987Z","shell.execute_reply.started":"2023-11-18T00:45:12.103890Z","shell.execute_reply":"2023-11-18T00:45:12.112207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame.from_dict(carts_dict, orient=\"index\", columns=[\"count\"])","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:45:20.373291Z","iopub.execute_input":"2023-11-18T00:45:20.374739Z","iopub.status.idle":"2023-11-18T00:45:20.385794Z","shell.execute_reply.started":"2023-11-18T00:45:20.374694Z","shell.execute_reply":"2023-11-18T00:45:20.384421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame.from_dict(fr_dict, orient=\"index\", columns=[\"fraction\"])","metadata":{"execution":{"iopub.status.busy":"2023-11-18T00:46:03.563787Z","iopub.execute_input":"2023-11-18T00:46:03.564171Z","iopub.status.idle":"2023-11-18T00:46:03.575954Z","shell.execute_reply.started":"2023-11-18T00:46:03.564143Z","shell.execute_reply":"2023-11-18T00:46:03.574559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see only 28% of orders for this data sample were previously clicked and then added to cart, so consequent prediction is bad decision for such data. The better way is to predict targets independently.\n\nOther statistics are also proves that.\n\nPossible reasons:\n- time constraints that leads to truncation of previous clicks and additions to cart in the given data\n- business logic that allows users to directly add item to cart or even to order item without any click","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}