{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Otto: Rename Columns\n\nHere is a quick notebook to help!\n\nAfter working with the data I constantly had to remind myself what the columns meant since the names are not obvious.","metadata":{}},{"cell_type":"code","source":"# bring in the packages we will need\nimport numpy as np\nimport pandas as pd","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2022-11-19T13:33:50.491924Z","iopub.execute_input":"2022-11-19T13:33:50.492746Z","iopub.status.idle":"2022-11-19T13:33:50.520594Z","shell.execute_reply.started":"2022-11-19T13:33:50.492640Z","shell.execute_reply":"2022-11-19T13:33:50.519377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# grab the data path\nclass CFG:\n    data_path = '/kaggle/input/otto-recommender-system/'","metadata":{"execution":{"iopub.status.busy":"2022-11-19T13:33:50.522890Z","iopub.execute_input":"2022-11-19T13:33:50.523292Z","iopub.status.idle":"2022-11-19T13:33:50.528655Z","shell.execute_reply.started":"2022-11-19T13:33:50.523257Z","shell.execute_reply":"2022-11-19T13:33:50.527362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert json to csv","metadata":{}},{"cell_type":"code","source":"# if you want to convert your json to a dataframe\n# grab this great snippet of code from \n# https://www.kaggle.com/code/columbia2131/otto-read-a-chunk-of-jsonl-to-manageable-df/\n\ntrain = pd.DataFrame()\nchunks = pd.read_json(CFG.data_path + 'train.jsonl', lines=True, chunksize=100_000)\n\n\nfor e, chunk in enumerate(chunks):\n    event_dict = {\n        'session': [],\n        'aid': [],\n        'ts': [],\n        'type': [],\n    }\n    if e < 2:\n        # train_sessions = pd.concat([train_sessions, chunk])\n        for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n            for event in events:\n                event_dict['session'].append(session)\n                event_dict['aid'].append(event['aid'])\n                event_dict['ts'].append(event['ts'])\n                event_dict['type'].append(event['type'])\n        chunk_session = pd.DataFrame(event_dict)\n        train = pd.concat([train, chunk_session])\n    else:\n        break\n        \ntrain = train.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T13:33:50.530425Z","iopub.execute_input":"2022-11-19T13:33:50.530916Z","iopub.status.idle":"2022-11-19T13:34:44.887956Z","shell.execute_reply.started":"2022-11-19T13:33:50.530866Z","shell.execute_reply":"2022-11-19T13:34:44.886576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# View the data as a dataframe","metadata":{}},{"cell_type":"code","source":"# view the dataframe\ndisplay(train)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T13:34:44.891280Z","iopub.execute_input":"2022-11-19T13:34:44.892067Z","iopub.status.idle":"2022-11-19T13:34:44.916067Z","shell.execute_reply.started":"2022-11-19T13:34:44.891926Z","shell.execute_reply":"2022-11-19T13:34:44.915076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The columns and definitions\n\n**session** - the unique session id [a better name is **customer_id**]\n\n**aid** - the article id [better name is **product_code**] of the associated event\n\n**ts** - the Unix [**timestamp**] of the event\n\n**type** - the [**event type**], i.e., whether a product was clicked, added to the user's cart, or ordered during the session","metadata":{}},{"cell_type":"code","source":"# change the column names the more obvious ones highlighted above\ntrain.rename(index=str, columns={'session': 'customer_id',\n                              'aid' : 'product_code',\n                              'ts' : 'time_stamp',\n                              'type' : 'event_type'}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T13:34:44.917755Z","iopub.execute_input":"2022-11-19T13:34:44.918472Z","iopub.status.idle":"2022-11-19T13:34:47.922405Z","shell.execute_reply.started":"2022-11-19T13:34:44.918428Z","shell.execute_reply":"2022-11-19T13:34:47.921084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# View the new column titles","metadata":{}},{"cell_type":"code","source":"# view the first 10 rows\ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T13:34:47.924164Z","iopub.execute_input":"2022-11-19T13:34:47.924540Z","iopub.status.idle":"2022-11-19T13:34:47.936559Z","shell.execute_reply.started":"2022-11-19T13:34:47.924503Z","shell.execute_reply":"2022-11-19T13:34:47.935234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.to_csv(\"otto-rename-columns.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-19T13:34:47.937931Z","iopub.execute_input":"2022-11-19T13:34:47.938392Z","iopub.status.idle":"2022-11-19T13:35:12.200110Z","shell.execute_reply.started":"2022-11-19T13:34:47.938357Z","shell.execute_reply":"2022-11-19T13:35:12.199033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Upvote if this helped 🤓","metadata":{}}]}