{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-17T17:07:59.328054Z","iopub.execute_input":"2022-11-17T17:07:59.329041Z","iopub.status.idle":"2022-11-17T17:07:59.357289Z","shell.execute_reply.started":"2022-11-17T17:07:59.328923Z","shell.execute_reply":"2022-11-17T17:07:59.355950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:05.206223Z","iopub.execute_input":"2022-11-17T17:08:05.206643Z","iopub.status.idle":"2022-11-17T17:08:05.216637Z","shell.execute_reply.started":"2022-11-17T17:08:05.206609Z","shell.execute_reply":"2022-11-17T17:08:05.215536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I will try to read the smaller jsonl file which is test.jsonl using pandas builtin methods. Let's see where do we reach.","metadata":{}},{"cell_type":"code","source":"df = pd.read_json('/kaggle/input/otto-recommender-system/test.jsonl', lines=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:11.837847Z","iopub.execute_input":"2022-11-17T17:08:11.838801Z","iopub.status.idle":"2022-11-17T17:08:36.399300Z","shell.execute_reply.started":"2022-11-17T17:08:11.838744Z","shell.execute_reply":"2022-11-17T17:08:36.397976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Print As many output we want from the same cell","metadata":{}},{"cell_type":"code","source":"from IPython.core.interactiveshell import InteractiveShell\nInteractiveShell.ast_node_interactivity = \"all\"","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:46.455220Z","iopub.execute_input":"2022-11-17T17:08:46.455722Z","iopub.status.idle":"2022-11-17T17:08:46.462630Z","shell.execute_reply.started":"2022-11-17T17:08:46.455678Z","shell.execute_reply":"2022-11-17T17:08:46.460810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:47.224522Z","iopub.execute_input":"2022-11-17T17:08:47.225016Z","iopub.status.idle":"2022-11-17T17:08:47.279836Z","shell.execute_reply.started":"2022-11-17T17:08:47.224969Z","shell.execute_reply":"2022-11-17T17:08:47.278512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increase the size of text in events column to have a better view of the data","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_colwidth', 1000)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:49.133933Z","iopub.execute_input":"2022-11-17T17:08:49.134422Z","iopub.status.idle":"2022-11-17T17:08:49.139872Z","shell.execute_reply.started":"2022-11-17T17:08:49.134381Z","shell.execute_reply":"2022-11-17T17:08:49.138941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:50.030794Z","iopub.execute_input":"2022-11-17T17:08:50.031559Z","iopub.status.idle":"2022-11-17T17:08:50.072801Z","shell.execute_reply.started":"2022-11-17T17:08:50.031521Z","shell.execute_reply":"2022-11-17T17:08:50.071624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I am planning to push each dictionary in the list to a new row. If I could do that then I will parse eash dictionary and create the columns out of each dictionary.\nBefore I do that, just wanted to check what is the distribution of count of events in each session.","metadata":{}},{"cell_type":"code","source":"df['events_length'] = df['events'].map(lambda x: len(x))","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:51.457947Z","iopub.execute_input":"2022-11-17T17:08:51.458645Z","iopub.status.idle":"2022-11-17T17:08:52.321752Z","shell.execute_reply.started":"2022-11-17T17:08:51.458580Z","shell.execute_reply":"2022-11-17T17:08:52.320796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['events_length'].describe(percentiles = [0.01, 0.05, 0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 0.95, 0.99])","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:52.323677Z","iopub.execute_input":"2022-11-17T17:08:52.324074Z","iopub.status.idle":"2022-11-17T17:08:52.393728Z","shell.execute_reply.started":"2022-11-17T17:08:52.324017Z","shell.execute_reply":"2022-11-17T17:08:52.392671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can make these numbers in the distribution look much better using simple pandas command.","metadata":{"execution":{"iopub.status.busy":"2022-11-16T18:14:20.813538Z","iopub.execute_input":"2022-11-16T18:14:20.814009Z","iopub.status.idle":"2022-11-16T18:14:20.821769Z","shell.execute_reply.started":"2022-11-16T18:14:20.813971Z","shell.execute_reply":"2022-11-16T18:14:20.820206Z"}}},{"cell_type":"code","source":"pd.set_option('display.float_format', lambda x: '%.2f' % x)","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:53.462761Z","iopub.execute_input":"2022-11-17T17:08:53.463217Z","iopub.status.idle":"2022-11-17T17:08:53.469241Z","shell.execute_reply.started":"2022-11-17T17:08:53.463181Z","shell.execute_reply":"2022-11-17T17:08:53.467913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['events_length'].describe(percentiles = [0.01, 0.05, 0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 0.95, 0.99])","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:54.116198Z","iopub.execute_input":"2022-11-17T17:08:54.117330Z","iopub.status.idle":"2022-11-17T17:08:54.190062Z","shell.execute_reply.started":"2022-11-17T17:08:54.117287Z","shell.execute_reply":"2022-11-17T17:08:54.188871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* More than 40% session has only 1 event\n* 99% sessions doesn't cross more than 40 events\n* 90% sessions doesn't cross more than 10 events","metadata":{}},{"cell_type":"markdown","source":"A simple histogram of event couns","metadata":{}},{"cell_type":"code","source":"import matplotlib\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:56.174138Z","iopub.execute_input":"2022-11-17T17:08:56.174650Z","iopub.status.idle":"2022-11-17T17:08:56.180402Z","shell.execute_reply.started":"2022-11-17T17:08:56.174587Z","shell.execute_reply":"2022-11-17T17:08:56.179190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xValues = df['events_length']\nplt.hist(xValues, density = True, label = 'EventsCount', histtype = 'bar')\nplt.legend(prop ={'size': 10})\nplt.title('EventsCount Distribution\\n\\n', fontweight =\"bold\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:56.912359Z","iopub.execute_input":"2022-11-17T17:08:56.913201Z","iopub.status.idle":"2022-11-17T17:08:57.218857Z","shell.execute_reply.started":"2022-11-17T17:08:56.913148Z","shell.execute_reply":"2022-11-17T17:08:57.217677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Each call of plot before plot.show() is printing some value. I don't like them. To avoid this we will assign these values to some variable.","metadata":{}},{"cell_type":"code","source":"xValues = df['events_length']\ncounts, bins, bars = plt.hist(xValues, density = True, label = 'EventsCount', histtype = 'bar')\nleg = plt.legend(prop ={'size': 10})\ntitle = plt.title('EventsCount Distribution\\n\\n', fontweight =\"bold\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:08:58.542584Z","iopub.execute_input":"2022-11-17T17:08:58.543080Z","iopub.status.idle":"2022-11-17T17:08:58.799795Z","shell.execute_reply.started":"2022-11-17T17:08:58.543041Z","shell.execute_reply":"2022-11-17T17:08:58.798514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A clean plot only.\n\nI think we can make the plot look much better and hence better representation of data if we drop the outliers. Let's try.","metadata":{}},{"cell_type":"code","source":"xValues = df.query('events_length<40')['events_length']\ncounts, bins, bars = plt.hist(xValues, density = True, label = 'EventsCount', histtype = 'bar')\nleg = plt.legend(prop ={'size': 10})\ntitle = plt.title('EventsCount Distribution\\n\\n', fontweight =\"bold\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:09:00.042728Z","iopub.execute_input":"2022-11-17T17:09:00.044244Z","iopub.status.idle":"2022-11-17T17:09:00.473833Z","shell.execute_reply.started":"2022-11-17T17:09:00.044186Z","shell.execute_reply":"2022-11-17T17:09:00.472624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"A plot which represnts the distribution in much better way.","metadata":{}},{"cell_type":"markdown","source":"Let's try parsing of the dictionary and create separate rows for each dictionary in the list.","metadata":{}},{"cell_type":"markdown","source":"**Illustration**","metadata":{}},{"cell_type":"code","source":"examples_data = df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:17:42.124845Z","iopub.execute_input":"2022-11-17T17:17:42.125268Z","iopub.status.idle":"2022-11-17T17:17:42.131269Z","shell.execute_reply.started":"2022-11-17T17:17:42.125237Z","shell.execute_reply":"2022-11-17T17:17:42.129700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(examples_data.events.tolist(), index = examples_data['session'])","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:26:26.511393Z","iopub.execute_input":"2022-11-17T17:26:26.511783Z","iopub.status.idle":"2022-11-17T17:26:26.556000Z","shell.execute_reply.started":"2022-11-17T17:26:26.511753Z","shell.execute_reply":"2022-11-17T17:26:26.554749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacked_data = pd.DataFrame(examples_data.events.tolist(), index = examples_data['session']).stack().reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:20:24.748762Z","iopub.execute_input":"2022-11-17T17:20:24.749958Z","iopub.status.idle":"2022-11-17T17:20:24.763922Z","shell.execute_reply.started":"2022-11-17T17:20:24.749910Z","shell.execute_reply":"2022-11-17T17:20:24.762812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stacked_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:21:13.063496Z","iopub.execute_input":"2022-11-17T17:21:13.064428Z","iopub.status.idle":"2022-11-17T17:21:13.076625Z","shell.execute_reply.started":"2022-11-17T17:21:13.064387Z","shell.execute_reply":"2022-11-17T17:21:13.075646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_data = pd.DataFrame(list(stacked_data[0]), index = stacked_data['session']).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:22:52.783251Z","iopub.execute_input":"2022-11-17T17:22:52.783760Z","iopub.status.idle":"2022-11-17T17:22:52.792983Z","shell.execute_reply.started":"2022-11-17T17:22:52.783723Z","shell.execute_reply":"2022-11-17T17:22:52.791680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:22:55.138693Z","iopub.execute_input":"2022-11-17T17:22:55.139481Z","iopub.status.idle":"2022-11-17T17:22:55.151155Z","shell.execute_reply.started":"2022-11-17T17:22:55.139440Z","shell.execute_reply":"2022-11-17T17:22:55.149945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is how we want the data. Let me try this on complete test data.","metadata":{}},{"cell_type":"code","source":"(pd.DataFrame(df.events.tolist(), index = df['session'])).head()","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:27:03.506031Z","iopub.execute_input":"2022-11-17T17:27:03.506491Z","iopub.status.idle":"2022-11-17T17:28:54.855316Z","shell.execute_reply.started":"2022-11-17T17:27:03.506457Z","shell.execute_reply":"2022-11-17T17:28:54.854196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ofcourse, this throws memory exceeded message. The data is too large to run such tricks here.","metadata":{"execution":{"iopub.status.busy":"2022-11-17T17:29:41.567287Z","iopub.execute_input":"2022-11-17T17:29:41.567763Z","iopub.status.idle":"2022-11-17T17:29:41.574563Z","shell.execute_reply.started":"2022-11-17T17:29:41.567726Z","shell.execute_reply":"2022-11-17T17:29:41.573169Z"}}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}