{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-24T19:13:20.603878Z","iopub.execute_input":"2021-06-24T19:13:20.604759Z","iopub.status.idle":"2021-06-24T19:13:20.616290Z","shell.execute_reply.started":"2021-06-24T19:13:20.604615Z","shell.execute_reply":"2021-06-24T19:13:20.615446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls ../input/","metadata":{"execution":{"iopub.status.busy":"2021-06-24T19:13:20.617704Z","iopub.execute_input":"2021-06-24T19:13:20.618299Z","iopub.status.idle":"2021-06-24T19:13:21.354759Z","shell.execute_reply.started":"2021-06-24T19:13:20.618254Z","shell.execute_reply":"2021-06-24T19:13:21.353602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/mlb-player-digital-engagement-forecasting/train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-24T19:13:21.356633Z","iopub.execute_input":"2021-06-24T19:13:21.356937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop('date', axis=1).columns","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%timeit\ntrain.at[1, 'date']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Progress bar\nfrom tqdm import tqdm\n# ast, a module to parse string of list and dict\nimport ast\n\n    \n\ntarget_columns = ['nextDayPlayerEngagement']  # You can add more columns in this list but beware RAM usage # noqa E501\n\n# target_columns = train.drop('date', axis=1).columns\n\nfor col in target_columns:\n    print(f\"Parse columns {col}\")\n    target_df = pd.DataFrame()\n    target_df['date'] = train['date']\n    for i in range(train.shape[0]):\n        cell = train.iloc[i][col]\n        if cell is not None:\n            # If a cell is not None make list\n            try:\n                value = ast.literal_eval(cell)\n            except ValueError:\n                continue\n            for k, v in enumerate(value):\n                prefix_col = f'{col}_{k}'\n                \n                for key, value in v.items():\n                    new_col = f'{prefix_col}_{key}'\n                    if new_col not in target_df.columns:\n                        train[new_col] = None\n                    target_df.at[i, new_col] = value\n        # Save to csv\n    target_df.to_csv(f'{col}.csv')\n\ntarget_df.head(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}