{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom matplotlib import pyplot as plt\n\n# make sure fastkaggle is install and imported\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \ntry: import fastkaggle\nexcept ModuleNotFoundError:\n    os.system(\"pip install -Uq fastkaggle\")\n\nfrom fastkaggle import *\n\n# use fastdebug.utils \nif iskaggle: os.system(\"pip install nbdev snoop\")\n\n# if iskaggle:\n#     path = \"../input/fastdebugutils0\"\n#     import sys\n#     sys.path\n#     sys.path.insert(1, path)\n#     import utils as fu\n#     from utils import *\n# else: \n#     from fastdebug.utils import *\n#     import fastdebug.utils as fu","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-09T19:29:50.385556Z","iopub.execute_input":"2022-12-09T19:29:50.386006Z","iopub.status.idle":"2022-12-09T19:30:00.799681Z","shell.execute_reply.started":"2022-12-09T19:29:50.385968Z","shell.execute_reply":"2022-12-09T19:30:00.798221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The original [notebook](https://www.kaggle.com/code/danielliao/eda-an-overview-of-the-full-dataset) is done by DANIELLIAO. I experimented to learn the techniques.\n\nimport utils and otto dataset","metadata":{}},{"cell_type":"code","source":"# Here are the files\n!ls ../input/otto-full-optimized-memory-footprint/","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:21:35.616610Z","iopub.execute_input":"2022-12-09T19:21:35.616977Z","iopub.status.idle":"2022-12-09T19:21:36.821661Z","shell.execute_reply.started":"2022-12-09T19:21:35.616945Z","shell.execute_reply":"2022-12-09T19:21:36.820266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:21:36.823231Z","iopub.execute_input":"2022-12-09T19:21:36.824562Z","iopub.status.idle":"2022-12-09T19:21:42.459241Z","shell.execute_reply.started":"2022-12-09T19:21:36.824521Z","shell.execute_reply":"2022-12-09T19:21:42.458113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.tail()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:21:42.463031Z","iopub.execute_input":"2022-12-09T19:21:42.463512Z","iopub.status.idle":"2022-12-09T19:21:42.483918Z","shell.execute_reply.started":"2022-12-09T19:21:42.463458Z","shell.execute_reply":"2022-12-09T19:21:42.482519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:21:42.485749Z","iopub.execute_input":"2022-12-09T19:21:42.486894Z","iopub.status.idle":"2022-12-09T19:21:42.501948Z","shell.execute_reply.started":"2022-12-09T19:21:42.486834Z","shell.execute_reply":"2022-12-09T19:21:42.500389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"An we see that the session_ids are not overlapping between train and test so it will be impossible to map the users (even if we have seen them before in train). We have to assume each session is from a different user.","metadata":{}},{"cell_type":"markdown","source":"The type column has been encoded as integers. To translate between the integer and original representations, please use the following.","metadata":{}},{"cell_type":"code","source":"!pip install pickle5\n\nimport pickle5 as pickle\n\nwith open('../input/otto-full-optimized-memory-footprint/id2type.pkl', \"rb\") as fh:\n    id2type = pickle.load(fh)\nwith open('../input/otto-full-optimized-memory-footprint/type2id.pkl', \"rb\") as fh:\n    type2id = pickle.load(fh)","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:21:42.503614Z","iopub.execute_input":"2022-12-09T19:21:42.503982Z","iopub.status.idle":"2022-12-09T19:21:55.056911Z","shell.execute_reply.started":"2022-12-09T19:21:42.503951Z","shell.execute_reply":"2022-12-09T19:21:55.055679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using id2type we can convert from integer to string representation (and we can use type2id to go in the other direction)","metadata":{}},{"cell_type":"code","source":"id2type, type2id","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:21:55.058393Z","iopub.execute_input":"2022-12-09T19:21:55.058779Z","iopub.status.idle":"2022-12-09T19:21:55.067539Z","shell.execute_reply.started":"2022-12-09T19:21:55.058743Z","shell.execute_reply":"2022-12-09T19:21:55.066236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id2type","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:21:55.069077Z","iopub.execute_input":"2022-12-09T19:21:55.069400Z","iopub.status.idle":"2022-12-09T19:21:55.080756Z","shell.execute_reply.started":"2022-12-09T19:21:55.069371Z","shell.execute_reply":"2022-12-09T19:21:55.079443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"### Comparing with training and testing set","metadata":{}},{"cell_type":"markdown","source":"test user / train user","metadata":{}},{"cell_type":"code","source":"train.session.unique().shape[0], test.session.unique().shape[0], test.session.unique().shape[0]/train.session.unique().shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:21:55.082214Z","iopub.execute_input":"2022-12-09T19:21:55.082627Z","iopub.status.idle":"2022-12-09T19:21:58.673416Z","shell.execute_reply.started":"2022-12-09T19:21:55.082596Z","shell.execute_reply":"2022-12-09T19:21:58.671899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Is there any new items in test not in train? -- No, there isn't any new items in testing set while not in the training set.","metadata":{}},{"cell_type":"code","source":"len(set(test.aid.tolist()) - set(train.aid.tolist()))","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:21:58.676953Z","iopub.execute_input":"2022-12-09T19:21:58.677333Z","iopub.status.idle":"2022-12-09T19:22:51.628609Z","shell.execute_reply.started":"2022-12-09T19:21:58.677302Z","shell.execute_reply":"2022-12-09T19:22:51.624122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Distribution","metadata":{}},{"cell_type":"markdown","source":"np - log1p - return natural log and also be super accurate in floating point","metadata":{}},{"cell_type":"code","source":"train.groupby('session')['aid'].count().describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:22:51.637764Z","iopub.execute_input":"2022-12-09T19:22:51.640984Z","iopub.status.idle":"2022-12-09T19:23:20.076018Z","shell.execute_reply.started":"2022-12-09T19:22:51.640931Z","shell.execute_reply":"2022-12-09T19:23:20.074696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.groupby('session')['aid'].count().apply(np.log1p).hist()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:23:20.077815Z","iopub.execute_input":"2022-12-09T19:23:20.078960Z","iopub.status.idle":"2022-12-09T19:23:33.457042Z","shell.execute_reply.started":"2022-12-09T19:23:20.078921Z","shell.execute_reply":"2022-12-09T19:23:33.455453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.groupby('session')['aid'].count().describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:23:33.458991Z","iopub.execute_input":"2022-12-09T19:23:33.459463Z","iopub.status.idle":"2022-12-09T19:23:33.939854Z","shell.execute_reply.started":"2022-12-09T19:23:33.459426Z","shell.execute_reply":"2022-12-09T19:23:33.938431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.groupby('session')['aid'].count().apply(np.log1p).hist()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:23:33.941868Z","iopub.execute_input":"2022-12-09T19:23:33.942331Z","iopub.status.idle":"2022-12-09T19:23:34.671986Z","shell.execute_reply.started":"2022-12-09T19:23:33.942288Z","shell.execute_reply":"2022-12-09T19:23:34.670722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.groupby('session')['aid'].count()","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:23:34.673678Z","iopub.execute_input":"2022-12-09T19:23:34.675035Z","iopub.status.idle":"2022-12-09T19:23:35.088547Z","shell.execute_reply.started":"2022-12-09T19:23:34.674982Z","shell.execute_reply":"2022-12-09T19:23:35.087532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Timestamp","metadata":{}},{"cell_type":"markdown","source":"The training dataset starts from 2022-07-31 22:00 to 2022-08-28 21:59:59, it last 4 weeks.\n\nThe testing dataset starts from 2022-08-28 22:00 to 2022-09-04 21:59:59, it last 1 week.\n\nThe problem is that the data doesn't come from the same period. In most geographies the beginning of September is the start of the school year!\n\nThat is the period where people are coming back from vacation, commerce resumes after a slowdown during the vacation season. while some school or office products are sold well.","metadata":{}},{"cell_type":"markdown","source":"|Dataset|Sun|Mon|Tue|Wed|Thu|Fri|Sat|\n|--|--|--|--|--|--|--|--|\n|training =>|31|1|2|3|4|5|6|\n| |7|8|9|10|11|12|13|\n| |14|15|16|17|18|19|20|\n| |21|22|23|24|25|26|27|\n|Testing =>|28|29|30|31|1|2|3|\n| |4|--|--|--|--|--|--|\n","metadata":{}},{"cell_type":"code","source":"import datetime\nprint(\"Train and test with /1000: \")\nprint(datetime.datetime.fromtimestamp(train.ts.min()/1000), datetime.datetime.fromtimestamp(train.ts.max()/1000))\nprint(datetime.datetime.fromtimestamp(test.ts.min()/1000), datetime.datetime.fromtimestamp(test.ts.max()/1000))\nprint(\"Train and test without /1000:\")\nprint(datetime.datetime.fromtimestamp(train.ts.min()), datetime.datetime.fromtimestamp(train.ts.max()))\nprint(datetime.datetime.fromtimestamp(test.ts.min()), datetime.datetime.fromtimestamp(test.ts.max()))","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:23:35.089934Z","iopub.execute_input":"2022-12-09T19:23:35.090255Z","iopub.status.idle":"2022-12-09T19:23:35.928805Z","shell.execute_reply.started":"2022-12-09T19:23:35.090227Z","shell.execute_reply":"2022-12-09T19:23:35.927468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compute the time difference","metadata":{}},{"cell_type":"code","source":"from tqdm.notebook import tqdm\nsample_data = train.head(100)\n### Extract information from each session and add it to the df ###\n\naction_counts_list, article_id_counts_list, session_length_time_list, session_length_action_list = ([] for i in range(4))\noverall_action_counts = {}\noverall_article_id_counts = {}\n\nfor i, row in tqdm(sample_data.iterrows(), total=len(sample_data)):\n    print(row[2])","metadata":{"execution":{"iopub.status.busy":"2022-12-09T20:53:21.474784Z","iopub.execute_input":"2022-12-09T20:53:21.475213Z","iopub.status.idle":"2022-12-09T20:53:21.525230Z","shell.execute_reply.started":"2022-12-09T20:53:21.475180Z","shell.execute_reply":"2022-12-09T20:53:21.524009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Todo：\n\n1. daytime, nighttime\n2. weekday, weekend\n3. Should sample mini batch data by `session + aid` like this:\n\n|session|aid|click|cart|order|\n|--|--|--|--|--|\n|1|1|1|1|0|\n|1|2|1|0|0|\n|2|1|1|1|1|","metadata":{}},{"cell_type":"markdown","source":"### Comparing with types between training and testing dataset","metadata":{}},{"cell_type":"markdown","source":"In the testing set, the proportion of orders is lower than it in training set. The reason may be the time window in testing set is shorter than training set, and usually customers need more time to consider.","metadata":{}},{"cell_type":"code","source":"labels = 'Clicks', 'Carts', 'Orders'\ntrain_events = [train.loc[train.type == 0].shape[0], train.loc[train.type == 1].shape[0], train.loc[train.type == 2].shape[0]]\ntest_events = [test.loc[test.type == 0].shape[0], test.loc[test.type == 1].shape[0], test.loc[test.type == 2].shape[0]]\n\nfig = plt.figure(figsize=(4,3),dpi=144)\nax = fig.add_subplot(121)\n\ncts = train_events\nax.pie(cts, labels=labels, autopct='%1.1f%%')\n\nax = fig.add_subplot(122)\nax.pie(test_events, labels=labels, autopct='%1.1f%%')","metadata":{"execution":{"iopub.status.busy":"2022-12-09T19:36:35.219610Z","iopub.execute_input":"2022-12-09T19:36:35.220102Z","iopub.status.idle":"2022-12-09T19:36:45.104554Z","shell.execute_reply.started":"2022-12-09T19:36:35.220069Z","shell.execute_reply":"2022-12-09T19:36:45.102975Z"},"trusted":true},"execution_count":null,"outputs":[]}]}