{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n!pip install polars \nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport polars as pl\n### Imports ###\n\nimport pandas as pd\nfrom pathlib import Path\nimport os\nimport random\nimport numpy as np\nimport json\nfrom datetime import timedelta\nfrom collections import Counter\nfrom tqdm.notebook import tqdm\nfrom heapq import nlargest\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_theme()\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\n\n\n#df = pl.read_ndjson(files[2])\n        #df2 = df.groupby('vendor_name').agg([pl.mean('Passenger_Count')])\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-14T16:25:50.2983Z","iopub.execute_input":"2023-02-14T16:25:50.298709Z","iopub.status.idle":"2023-02-14T16:26:06.425674Z","shell.execute_reply.started":"2023-02-14T16:25:50.298676Z","shell.execute_reply":"2023-02-14T16:26:06.424413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nfiles = []\nimport os\ni=0\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        file_name = (os.path.join(dirname, filename))\n        files.append(file_name)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.256751Z","iopub.status.idle":"2023-02-14T16:25:42.257598Z","shell.execute_reply.started":"2023-02-14T16:25:42.257297Z","shell.execute_reply":"2023-02-14T16:25:42.257324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/otto-full-optimized-memory-footprint/train.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.259039Z","iopub.status.idle":"2023-02-14T16:25:42.259831Z","shell.execute_reply.started":"2023-02-14T16:25:42.259545Z","shell.execute_reply":"2023-02-14T16:25:42.259571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(files)","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.261266Z","iopub.status.idle":"2023-02-14T16:25:42.262087Z","shell.execute_reply.started":"2023-02-14T16:25:42.261805Z","shell.execute_reply":"2023-02-14T16:25:42.261831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(20)","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.272334Z","iopub.execute_input":"2023-02-14T16:25:42.273121Z","iopub.status.idle":"2023-02-14T16:25:42.29467Z","shell.execute_reply.started":"2023-02-14T16:25:42.273074Z","shell.execute_reply":"2023-02-14T16:25:42.292781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['session'] == 10]","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.296318Z","iopub.status.idle":"2023-02-14T16:25:42.297656Z","shell.execute_reply.started":"2023-02-14T16:25:42.297277Z","shell.execute_reply":"2023-02-14T16:25:42.297314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.299683Z","iopub.status.idle":"2023-02-14T16:25:42.300759Z","shell.execute_reply.started":"2023-02-14T16:25:42.300433Z","shell.execute_reply":"2023-02-14T16:25:42.300466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's look at an example session and print out some basic info\n\n# Sample the first session in the df\nexample_session = df[df['session'] == 0]\nprint(f'This session was {(example_session.shape[0])} actions long \\n')\nprint(f'The first action in the session: \\n {example_session.iloc[0]} \\n')\n\n# Time of session\ntime_elapsed = example_session.iloc[-1][\"ts\"] - example_session.iloc[0][\"ts\"]\n# The timestamp is in milliseconds since 00:00:00 UTC on 1 January 1970\nprint(f'The first session elapsed: {time_elapsed} \\n')\nprint(time_elapsed)\n#print(f'The first session elapsed: {str(timedelta(seconds=int(time_elapsed)))} \\n')\n\n# Count the frequency of actions within the session\naction_counts = {}\n#for action in example_session:\n#    action_counts[action['type']] = action_counts.get(action['type'], 0) + 1  \n\naction_counts = dict(example_session['type'].value_counts())\nprint(f'The first session contains the following frequency of actions: {action_counts}')","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.305821Z","iopub.execute_input":"2023-02-14T16:25:42.306623Z","iopub.status.idle":"2023-02-14T16:25:42.332066Z","shell.execute_reply.started":"2023-02-14T16:25:42.306573Z","shell.execute_reply":"2023-02-14T16:25:42.330213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_df = df.where(df.session < (10000))\nsample_train_df.set_index('session', drop=True, inplace=True)\ndel df","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.333717Z","iopub.status.idle":"2023-02-14T16:25:42.334653Z","shell.execute_reply.started":"2023-02-14T16:25:42.334305Z","shell.execute_reply":"2023-02-14T16:25:42.334335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\naction_counts_list, article_id_counts_list, session_length_time_list, session_length_action_list = ([] for i in range(4))\noverall_action_counts = {}\noverall_article_id_counts = {}\n\n\n\n\nfor i, row in tqdm(sample_train_df.iterrows(), total=len(sample_train_df)):\n    print(row)\n    #actions = row['events']\n    \n    # Get the frequency of actions and article_ids\n    action_counts = {}\n    article_id_counts = {}\n    for action in row:\n        action_counts[action['type']] = action_counts.get(action['type'], 0) + 1\n        article_id_counts[action['aid']] = article_id_counts.get(action['aid'], 0) + 1\n        overall_action_counts[action['type']] = overall_action_counts.get(action['type'], 0) + 1\n        overall_article_id_counts[action['aid']] = overall_article_id_counts.get(action['aid'], 0) + 1\n        \n    action_counts = dict(example_session['type'].value_counts())\n    article_id_counts = dict(example_session['type'].value_counts())\n    overall_action_counts = dict(example_session['type'].value_counts())\n    overall_article_id_counts = dict(example_session['type'].value_counts())\n\n        \n    # Get the length of the session\n    session_length_time = actions[-1]['ts'] - actions[0]['ts']\n    \n    # Add to list\n    action_counts_list.append(action_counts)\n    article_id_counts_list.append(article_id_counts)\n    session_length_time_list.append(session_length_time)\n    session_length_action_list.append(len(actions))\n    \nsample_train_df['action_counts'] = action_counts_list\nsample_train_df['article_id_counts'] = article_id_counts_list\nsample_train_df['session_length_unix'] = session_length_time_list\nsample_train_df['session_length_hours'] = sample_train_df['session_length_unix']*2.77778e-7  # Convert to hours\nsample_train_df['session_length_action'] = session_length_action_list","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.33632Z","iopub.status.idle":"2023-02-14T16:25:42.337466Z","shell.execute_reply.started":"2023-02-14T16:25:42.337104Z","shell.execute_reply":"2023-02-14T16:25:42.337139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" dict(sample_train_df['type'].value_counts())\n    \naction_counts['type'] =  dict(sample_train_df['type'].value_counts())\narticle_id_counts['aid'] =  dict(sample_train_df['aid'].value_counts())\noverall_action_counts['type'] =  dict(sample_train_df['type'].value_counts())\noverall_article_id_counts['aid'] =  dict(sample_train_df['aid'].value_counts())\n    ","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.339759Z","iopub.status.idle":"2023-02-14T16:25:42.340644Z","shell.execute_reply.started":"2023-02-14T16:25:42.340314Z","shell.execute_reply":"2023-02-14T16:25:42.340344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"overall_action_counts","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.350757Z","iopub.execute_input":"2023-02-14T16:25:42.351611Z","iopub.status.idle":"2023-02-14T16:25:42.371221Z","shell.execute_reply.started":"2023-02-14T16:25:42.351561Z","shell.execute_reply":"2023-02-14T16:25:42.36957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n\ntotal_actions = sum(overall_action_counts.get('type').values())\noverall_action_counts\noverall_action_counts.get('type').values()\ncliks,carts,orders = overall_action_counts.get('type').values()\n\nplt.figure(figsize=(8,6))\nsns.barplot(x=list(['cliks','carts','orders']), y=[i/total_actions for i in [cliks,carts,orders]])\n\nplt.title(f'Action frequency', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.xlabel('Category', fontsize=12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.372798Z","iopub.status.idle":"2023-02-14T16:25:42.373703Z","shell.execute_reply.started":"2023-02-14T16:25:42.373385Z","shell.execute_reply":"2023-02-14T16:25:42.37342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(24, 10))\n\np = sns.distplot(sample_train_df['session_length_action'], color=\"y\", bins= 70, ax=ax[0], kde=False)\np.set_xlabel(\"Number of actions\", fontsize = 16)\np.set_ylabel(\"Density\", fontsize = 16)\np.set_title(\"Distribution of the number of actions taken in each session\", fontsize = 14)\np.axvline(sample_train_df['session_length_action'].mean(), color='r', linestyle='--', label=\"Mean\")\n\np = sns.distplot(sample_train_df['session_length_hours'], color=\"b\", bins= 70, ax=ax[1], kde=False)\np.set_xlabel(\"Hours\", fontsize = 16)\np.set_ylabel(\"Density\", fontsize = 16)\np.set_title(\"Length of each session\", fontsize = 16);","metadata":{"execution":{"iopub.status.busy":"2023-02-14T16:25:42.404216Z","iopub.execute_input":"2023-02-14T16:25:42.405038Z","iopub.status.idle":"2023-02-14T16:25:42.428872Z","shell.execute_reply.started":"2023-02-14T16:25:42.404985Z","shell.execute_reply":"2023-02-14T16:25:42.426971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}