{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os, pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom plotly import tools\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nfrom sklearn import model_selection, preprocessing, metrics\nfrom lightgbm import LGBMClassifier\nimport lightgbm as lgb\nimport numpy as np\nfrom pathlib import Path\nimport gc\n# List the files in the mounted path\npd.set_option(\"display.max_columns\", None)\ndata_path = Path('/kaggle/input/otto-recommender-system/')\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-03T03:30:47.411254Z","iopub.execute_input":"2023-01-03T03:30:47.411813Z","iopub.status.idle":"2023-01-03T03:30:47.424320Z","shell.execute_reply.started":"2023-01-03T03:30:47.411770Z","shell.execute_reply":"2023-01-03T03:30:47.423067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading data","metadata":{}},{"cell_type":"markdown","source":"Referring to https://www.kaggle.com/code/junjitakeshima/otto-easy-understanding-for-beginner-en","metadata":{}},{"cell_type":"code","source":"Sample = pd.read_csv(\"/kaggle/input/otto-recommender-system/sample_submission.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:30:50.656897Z","iopub.execute_input":"2023-01-03T03:30:50.657313Z","iopub.status.idle":"2023-01-03T03:30:55.523233Z","shell.execute_reply.started":"2023-01-03T03:30:50.657282Z","shell.execute_reply":"2023-01-03T03:30:55.522033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Sample.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:30:59.015978Z","iopub.execute_input":"2023-01-03T03:30:59.016538Z","iopub.status.idle":"2023-01-03T03:30:59.031748Z","shell.execute_reply.started":"2023-01-03T03:30:59.016489Z","shell.execute_reply":"2023-01-03T03:30:59.030475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_size = 100_000\n\nchunks = pd.read_json(data_path / 'train.jsonl' , lines=True, chunksize = sample_size)\n\nfor chunk in chunks :\n    train_df = chunk\n    break\n\ntrain_df.set_index('session', drop=True, inplace=True)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:31:01.307062Z","iopub.execute_input":"2023-01-03T03:31:01.307504Z","iopub.status.idle":"2023-01-03T03:31:13.089756Z","shell.execute_reply.started":"2023-01-03T03:31:01.307466Z","shell.execute_reply":"2023-01-03T03:31:13.088270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:31:59.735332Z","iopub.execute_input":"2023-01-03T03:31:59.735769Z","iopub.status.idle":"2023-01-03T03:31:59.845761Z","shell.execute_reply.started":"2023-01-03T03:31:59.735730Z","shell.execute_reply":"2023-01-03T03:31:59.844738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.iloc[4,0]","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:32:40.680999Z","iopub.execute_input":"2023-01-03T03:32:40.681419Z","iopub.status.idle":"2023-01-03T03:32:40.693358Z","shell.execute_reply.started":"2023-01-03T03:32:40.681383Z","shell.execute_reply":"2023-01-03T03:32:40.692172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.iloc[1,0]","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:34:33.617993Z","iopub.execute_input":"2023-01-03T03:34:33.618426Z","iopub.status.idle":"2023-01-03T03:34:33.630661Z","shell.execute_reply.started":"2023-01-03T03:34:33.618391Z","shell.execute_reply":"2023-01-03T03:34:33.629308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:35:23.179143Z","iopub.execute_input":"2023-01-03T03:35:23.180297Z","iopub.status.idle":"2023-01-03T03:35:23.672796Z","shell.execute_reply.started":"2023-01-03T03:35:23.180242Z","shell.execute_reply":"2023-01-03T03:35:23.671358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = pd.DataFrame()\nchunks = pd.read_json(data_path / 'train.jsonl', lines=True, chunksize=100)\n\nfor chunk in chunks:\n    event_dict = {'session': [], 'aid': [], 'ts': [], 'type': []}\n    \n    for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n        for event in events:\n            event_dict['session'].append(session)\n            event_dict['aid'].append(event['aid'])\n            event_dict['ts'].append(event['ts'])\n            event_dict['type'].append(event['type'])\n    train_df = pd.DataFrame(event_dict)\n    \n    break\n        \ntrain_df = train_df.reset_index(drop=True)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:46:04.986610Z","iopub.execute_input":"2023-01-03T03:46:04.987056Z","iopub.status.idle":"2023-01-03T03:46:07.091650Z","shell.execute_reply.started":"2023-01-03T03:46:04.987019Z","shell.execute_reply":"2023-01-03T03:46:07.090475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:46:19.138716Z","iopub.execute_input":"2023-01-03T03:46:19.140244Z","iopub.status.idle":"2023-01-03T03:46:19.162919Z","shell.execute_reply.started":"2023-01-03T03:46:19.140196Z","shell.execute_reply":"2023-01-03T03:46:19.161752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:53:27.767467Z","iopub.execute_input":"2023-01-03T03:53:27.767876Z","iopub.status.idle":"2023-01-03T03:53:27.784567Z","shell.execute_reply.started":"2023-01-03T03:53:27.767826Z","shell.execute_reply":"2023-01-03T03:53:27.783054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T03:55:48.752226Z","iopub.execute_input":"2023-01-03T03:55:48.752818Z","iopub.status.idle":"2023-01-03T03:55:48.768536Z","shell.execute_reply.started":"2023-01-03T03:55:48.752767Z","shell.execute_reply":"2023-01-03T03:55:48.767244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.groupby(['session']).agg({'aid': 'nunique'})","metadata":{"execution":{"iopub.status.busy":"2023-01-03T04:11:46.326634Z","iopub.execute_input":"2023-01-03T04:11:46.327098Z","iopub.status.idle":"2023-01-03T04:11:46.344187Z","shell.execute_reply.started":"2023-01-03T04:11:46.327059Z","shell.execute_reply":"2023-01-03T04:11:46.342519Z"},"trusted":true},"execution_count":null,"outputs":[]}]}