{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n\n\n<div align=\"center\"><p style=\"font-family: 'Mochiy Pop P One';font-size:32px; font-weight: bold; color:mediumblue\">OTTO – Multi-Objective Recommender System - EDA </p></div>\n\n\n\n\n<li><strong>train.jsonl</strong> - the training data, which contains full session data\n    <ul>\n        <li><code>session</code> - the unique session id</li>\n        <li><code>events</code> - the time ordered sequence of events in the session\n            <ul>\n                <li><code>aid</code> - the article id (product code) of the associated event</li>\n                <li><code>ts</code> - the Unix timestamp of the event</li>\n                <li><code>type</code> - the event type, i.e., whether a product was clicked, added to the user's cart, or ordered during the session</li>\n            </ul>\n        </li>\n    </ul>\n</li>\n    \n    \n\n    \n    ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport gc\nimport copy\nimport json\nimport sys\nfrom pathlib import Path\nfrom datetime import datetime, timedelta, date\nimport time\nfrom dateutil.relativedelta import relativedelta \n\nimport pyarrow.parquet as pq\nimport pyarrow as pa\n\nfrom tqdm import tqdm\n\nimport matplotlib.pyplot as plt\nimport matplotlib.lines as mlines\nimport pandas as pd\nimport statsmodels.api as sm\nimport seaborn as sns\npd.options.display.max_rows = 100\npd.options.display.max_columns = 100\n\npd.options.display.float_format = '{:,.2f}'.format\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport pytorch_lightning as pl\nrandom_seed=1\npl.seed_everything(random_seed)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-02T15:35:20.517243Z","iopub.execute_input":"2022-11-02T15:35:20.517669Z","iopub.status.idle":"2022-11-02T15:35:22.454485Z","shell.execute_reply.started":"2022-11-02T15:35:20.517640Z","shell.execute_reply":"2022-11-02T15:35:22.452529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<div align=\"center\"><p style=\"font-family: 'Mochiy Pop P One';font-size:25px; font-weight: bold; color:mediumblue\">Load data  </p></div>\n\n<div align=\"left\" style=\"font-family: 'Mochiy Pop P One';font-size:16px; font-weight: normal; color:dimgray\">\n    \nThere are over 12 million (12,899,779) samples in training data. \n\n<br>      \n<br>\n    \n</div>   ","metadata":{}},{"cell_type":"code","source":"train_file = \"../input/otto-recommender-system/train.jsonl\"","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:01:25.410931Z","iopub.execute_input":"2022-11-02T15:01:25.411256Z","iopub.status.idle":"2022-11-02T15:01:25.417186Z","shell.execute_reply.started":"2022-11-02T15:01:25.411223Z","shell.execute_reply":"2022-11-02T15:01:25.415902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnum_lines = sum(1 for line in open(train_file))","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:01:25.418782Z","iopub.execute_input":"2022-11-02T15:01:25.419118Z","iopub.status.idle":"2022-11-02T15:03:01.767932Z","shell.execute_reply.started":"2022-11-02T15:01:25.419092Z","shell.execute_reply":"2022-11-02T15:03:01.766252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunksize = 100_000\nnum_chunks = int(np.ceil(num_lines / chunksize))\n\nprint(f'number of samples in train: {num_lines:,}\\nnumber of chunks: {num_chunks:,}')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:56:56.547127Z","iopub.execute_input":"2022-11-02T15:56:56.547584Z","iopub.status.idle":"2022-11-02T15:56:56.554213Z","shell.execute_reply.started":"2022-11-02T15:56:56.547545Z","shell.execute_reply":"2022-11-02T15:56:56.552882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nchunks = pd.read_json(train_file, lines=True, chunksize=chunksize)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:03:01.779781Z","iopub.execute_input":"2022-11-02T15:03:01.780084Z","iopub.status.idle":"2022-11-02T15:03:01.798960Z","shell.execute_reply.started":"2022-11-02T15:03:01.780057Z","shell.execute_reply":"2022-11-02T15:03:01.797418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsession_stats = []\ntype_stats = []\ncounter = 0\nfor chunk in tqdm(chunks):\n    chunk.set_index(keys=['session'], inplace=True )\n    chunk['cnt'] = chunk['events'].apply(lambda x: len(x))\n    chunk['total_seconds'] = chunk['events'].apply(lambda x: (x[-1]['ts']-x[0]['ts'])/1e3)\n    session_stats.append(chunk[['cnt', 'total_seconds']])\n    cnt = chunk['events'].apply(lambda x: pd.DataFrame(x)['type'].value_counts().to_dict())\n    type_stats.append(pd.DataFrame(cnt.to_dict()).T)\n    counter = counter + 1\n    if counter>=10:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:03:01.800049Z","iopub.execute_input":"2022-11-02T15:03:01.800686Z","iopub.status.idle":"2022-11-02T15:19:05.366536Z","shell.execute_reply.started":"2022-11-02T15:03:01.800659Z","shell.execute_reply":"2022-11-02T15:19:05.364835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunk.sort_values(by='cnt').tail(10)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:38:19.887643Z","iopub.execute_input":"2022-11-02T15:38:19.888162Z","iopub.status.idle":"2022-11-02T15:38:20.187126Z","shell.execute_reply.started":"2022-11-02T15:38:19.888123Z","shell.execute_reply":"2022-11-02T15:38:20.186029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.DataFrame(chunk.loc[923610]['events'])","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:38:27.801436Z","iopub.execute_input":"2022-11-02T15:38:27.801974Z","iopub.status.idle":"2022-11-02T15:38:27.809468Z","shell.execute_reply.started":"2022-11-02T15:38:27.801936Z","shell.execute_reply":"2022-11-02T15:38:27.808394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.iloc[:5]","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:38:28.596402Z","iopub.execute_input":"2022-11-02T15:38:28.597490Z","iopub.status.idle":"2022-11-02T15:38:28.608920Z","shell.execute_reply.started":"2022-11-02T15:38:28.597431Z","shell.execute_reply":"2022-11-02T15:38:28.606626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n<div align=\"center\"><p style=\"font-family: 'Mochiy Pop P One';font-size:25px; font-weight: bold; color:mediumblue\">Explore data  </p></div>\n\n<div align=\"left\" style=\"font-family: 'Mochiy Pop P One';font-size:16px; font-weight: normal; color:dimgray\">\n    \nSample half a million rows from the 12,899,779 training data, and look at the following areas:\n<br>      \n<br>\n1. Distribution of <mark>Number of articles (products)</mark> in each session (with duplicates).  \n<br>\n2. Distribution of <mark>Duration of each session</mark>.\n<br>\n2. Distribution of <mark>Number of clicks items, carts itmes, orders items</mark>.\n    \n</div>   ","metadata":{}},{"cell_type":"code","source":"df_session_stats = pd.concat(session_stats)\ndf_type_stats = pd.concat(type_stats)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:33:57.259681Z","iopub.execute_input":"2022-11-02T15:33:57.260160Z","iopub.status.idle":"2022-11-02T15:33:57.325170Z","shell.execute_reply.started":"2022-11-02T15:33:57.260121Z","shell.execute_reply":"2022-11-02T15:33:57.323560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_session_stats.index.name='session'\ndf_type_stats.index.name='session'","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:33:57.690685Z","iopub.execute_input":"2022-11-02T15:33:57.691110Z","iopub.status.idle":"2022-11-02T15:33:57.699257Z","shell.execute_reply.started":"2022-11-02T15:33:57.691074Z","shell.execute_reply":"2022-11-02T15:33:57.696882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_session_stats.shape, df_type_stats.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:33:58.776098Z","iopub.execute_input":"2022-11-02T15:33:58.776573Z","iopub.status.idle":"2022-11-02T15:33:58.785481Z","shell.execute_reply.started":"2022-11-02T15:33:58.776536Z","shell.execute_reply":"2022-11-02T15:33:58.783992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div align=\"left\" style=\"font-family: 'Mochiy Pop P One';font-size:18px; font-weight: bold; color:dimgray\">\n    \n  The following table shows how many items (including <code>clicks</code>, <code>carts</code>, and<code>orders</code>) in each session, and total duration in seconds for each session.\n\n    \n</div>   ","metadata":{}},{"cell_type":"code","source":"df_session_stats.iloc[:5]","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:34:00.301917Z","iopub.execute_input":"2022-11-02T15:34:00.302357Z","iopub.status.idle":"2022-11-02T15:34:00.316100Z","shell.execute_reply.started":"2022-11-02T15:34:00.302314Z","shell.execute_reply":"2022-11-02T15:34:00.314290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div align=\"left\" style=\"font-family: 'Mochiy Pop P One';font-size:18px; font-weight: bold; color:dimgray\">\n    \n\n1. Distribution of <mark>Number of articles (products)</mark> in each session (with duplicates).  \n\n    \n</div>   ","metadata":{}},{"cell_type":"code","source":"background_color = \"#fbfbfb\"\nfig, axs = plt.subplots(1, 2, figsize=(12, 5), sharex=False, sharey=False, dpi=100,facecolor=background_color)\nsns.kdeplot(df_session_stats['cnt'], ax=axs[0], color='lightgray', ec='lightgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False)\nsns.kdeplot(df_session_stats['cnt'], ax=axs[1], color='lightgray', ec='lightgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False, log_scale=True)\naxs[0].grid(which='major', axis='x', zorder=0, color='gray', linestyle=':', dashes=(1,5))\naxs[0].set_xlabel('Num. of articles (products)', fontfamily='monospace')\naxs[1].grid(which='major', axis='x', zorder=0, color='gray', linestyle=':', dashes=(1,5))\naxs[1].set_xlabel('Num. of articles (products)', fontfamily='monospace')\naxs[1].set_ylabel('Density of Log Value', fontfamily='monospace')\n\nplt.tight_layout()\nplt.subplots_adjust(left=None, bottom=None, right=None, top=None, wspace=0.2, hspace=0.2) # useful for adjusting space between subplots\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:35:30.741249Z","iopub.execute_input":"2022-11-02T15:35:30.741671Z","iopub.status.idle":"2022-11-02T15:35:38.816197Z","shell.execute_reply.started":"2022-11-02T15:35:30.741643Z","shell.execute_reply":"2022-11-02T15:35:38.815017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div align=\"left\" style=\"font-family: 'Mochiy Pop P One';font-size:18px; font-weight: bold; color:dimgray\">\n    \n\n2. Distribution of <mark>Duration of each session</mark>.\n\n    \n</div>   ","metadata":{}},{"cell_type":"code","source":"background_color = \"#fbfbfb\"\nfig, axs = plt.subplots(1, 2, figsize=(12, 5), sharex=False, sharey=False, dpi=100,facecolor=background_color)\nsns.kdeplot(df_session_stats['total_seconds'], ax=axs[0], color='lightgray', ec='lightgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False)\nsns.kdeplot(np.log(df_session_stats['total_seconds']+1), ax=axs[1], color='lightgray', ec='lightgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False)\naxs[0].grid(which='major', axis='x', zorder=0, color='gray', linestyle=':', dashes=(1,5))\naxs[0].set_xlabel('Duration of a session (in seconds)', fontfamily='monospace')\naxs[1].grid(which='major', axis='x', zorder=0, color='gray', linestyle=':', dashes=(1,5))\naxs[1].set_xlabel('Duration of a session (in seconds)', fontfamily='monospace')\naxs[1].set_ylabel('Density of Log Value', fontfamily='monospace')\n\nplt.tight_layout()\nplt.subplots_adjust(left=None, bottom=None, right=None, top=None, wspace=0.2, hspace=0.2) # useful for adjusting space between subplots\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:35:38.818286Z","iopub.execute_input":"2022-11-02T15:35:38.819405Z","iopub.status.idle":"2022-11-02T15:35:45.617039Z","shell.execute_reply.started":"2022-11-02T15:35:38.819357Z","shell.execute_reply":"2022-11-02T15:35:45.615482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div align=\"left\" style=\"font-family: 'Mochiy Pop P One';font-size:18px; font-weight: bold; color:dimgray\">\n    \nThe following table shows how many items <mark>CLICKED</mark> (column <code>clicks</code>), <mark>ADDED TO CARTS</mark> (column <code>carts</code>), <mark>ORDERED</mark> (column <code>orders</code>).\n\n    \n</div>   ","metadata":{}},{"cell_type":"code","source":"df_type_stats[df_type_stats['orders'].notna()].iloc[:5]","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:35:45.622538Z","iopub.execute_input":"2022-11-02T15:35:45.624181Z","iopub.status.idle":"2022-11-02T15:35:45.652284Z","shell.execute_reply.started":"2022-11-02T15:35:45.624113Z","shell.execute_reply":"2022-11-02T15:35:45.649153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"background_color = \"#fbfbfb\"\nfig, axs = plt.subplots(2, 3, figsize=(12, 6), sharex=False, sharey=False, dpi=100,facecolor=background_color)\n# sns.kdeplot(df_type_stats['clicks'], ax=axs[0,0], color='dimgray', ec='dimgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False)\n# sns.kdeplot(df_type_stats['carts'], ax=axs[0,1], color='dimgray', ec='dimgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False)\n# sns.kdeplot(df_type_stats['orders'], ax=axs[0,2], color='dimgray', ec='dimgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False)\n# sns.kdeplot(df_type_stats['clicks'], ax=axs[1,0], color='dimgray', ec='dimgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False, log_scale=True)\n# sns.kdeplot(df_type_stats['carts'], ax=axs[1,1], color='dimgray', ec='dimgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False, log_scale=True)\n# sns.kdeplot(df_type_stats['orders'], ax=axs[1,2], color='dimgray', ec='dimgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False, log_scale=True)\n\ncols = ['clicks','carts','orders']\nrows = ['', 'of Log Value']\nfor j in [0, 1, 2]:\n    sns.kdeplot(df_type_stats[cols[j]], ax=axs[0,j], color='dimgray', ec='dimgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False)\n    sns.kdeplot(df_type_stats[cols[j]], ax=axs[1,j], color='dimgray', ec='dimgray', shade=True, linewidth=.5, alpha=0.9, zorder=3, legend=False, log_scale=True)\n\n    for i in [0, 1]:\n        axs[i,j].grid(which='major', axis='x', zorder=0, color='dimgray', linestyle=':', dashes=(1,5))\n        axs[i,j].set_xlabel(f'{cols[j]} ', fontfamily='monospace')\n        axs[i,j].set_ylabel(f'Density {rows[i]}', fontfamily='monospace')\n\nplt.tight_layout()\nplt.subplots_adjust(left=None, bottom=None, right=None, top=None, wspace=0.2, hspace=0.2) # useful for adjusting space between subplots\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:36:28.900472Z","iopub.execute_input":"2022-11-02T15:36:28.900882Z","iopub.status.idle":"2022-11-02T15:36:42.972510Z","shell.execute_reply.started":"2022-11-02T15:36:28.900853Z","shell.execute_reply":"2022-11-02T15:36:42.971022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div align=\"left\" style=\"font-family: 'Mochiy Pop P One';font-size:18px; font-weight: bold; color:dimgray\">\n    \nSession 923610\n\n    \n</div>   ","metadata":{}},{"cell_type":"code","source":"sample['ts2'] = sample['ts'].apply(lambda x: datetime.fromtimestamp(x / 1e3) )\nsample['diff_total_seconds'] = sample['ts'].diff()/1e3","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:38:37.486525Z","iopub.execute_input":"2022-11-02T15:38:37.487300Z","iopub.status.idle":"2022-11-02T15:38:37.496601Z","shell.execute_reply.started":"2022-11-02T15:38:37.487252Z","shell.execute_reply":"2022-11-02T15:38:37.494852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:51:39.502946Z","iopub.execute_input":"2022-11-02T15:51:39.503410Z","iopub.status.idle":"2022-11-02T15:51:39.522872Z","shell.execute_reply.started":"2022-11-02T15:51:39.503371Z","shell.execute_reply":"2022-11-02T15:51:39.521473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14, 6), dpi=100)\n\nplt.plot(sample['ts2'], sample['diff_total_seconds'], marker='', markersize=2,  linewidth=2)\n    \n\nplt.title('Seconds between two consecutive events',  fontsize=14, loc='left')\n\nplt.grid(visible=True, which='major', axis='both', color='lightgray', linestyle='-', linewidth=0.2)\nplt.gca().spines[['top', 'right']].set_visible(False)\nplt.gca().spines[['bottom', 'left']].set_edgecolor('whitesmoke')\n\nplt.gcf().autofmt_xdate()\nplt.yticks(fontsize=9, )#rotation=90\nplt.xticks(fontsize=9, )#rotation=90\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T15:51:52.294928Z","iopub.execute_input":"2022-11-02T15:51:52.295390Z","iopub.status.idle":"2022-11-02T15:51:52.534754Z","shell.execute_reply.started":"2022-11-02T15:51:52.295357Z","shell.execute_reply":"2022-11-02T15:51:52.533506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"References:\n\n- [Read a chunk of jsonl](https://www.kaggle.com/code/inversion/read-a-chunk-of-jsonl)\n- [https://www.kaggle.com/code/edwardcrookenden/otto-getting-started-eda-baseline](https://www.kaggle.com/code/edwardcrookenden/otto-getting-started-eda-baseline)","metadata":{}}]}