{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook I shuffle the whole dataset and split it in 10 new feather files that I use for a more unbiased continued learning.","metadata":{}},{"cell_type":"code","source":"#import some lib\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport os\nimport gc\npd.set_option('display.max_columns', 500)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-09T11:27:16.578151Z","iopub.execute_input":"2022-10-09T11:27:16.578587Z","iopub.status.idle":"2022-10-09T11:27:16.585108Z","shell.execute_reply.started":"2022-10-09T11:27:16.578549Z","shell.execute_reply":"2022-10-09T11:27:16.583660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [\n    'ball_pos_x', 'ball_pos_y','ball_pos_z', 'ball_vel_x', 'ball_vel_y', 'ball_vel_z', \n    'p0_pos_x', 'p0_pos_y', 'p0_pos_z', 'p0_vel_x', 'p0_vel_y', 'p0_vel_z', 'p0_boost', 'p0_na',\n    'p1_pos_x', 'p1_pos_y', 'p1_pos_z', 'p1_vel_x', 'p1_vel_y', 'p1_vel_z', 'p1_boost', 'p1_na',\n    'p2_pos_x', 'p2_pos_y', 'p2_pos_z', 'p2_vel_x', 'p2_vel_y', 'p2_vel_z', 'p2_boost', 'p2_na',\n    'p3_pos_x', 'p3_pos_y', 'p3_pos_z', 'p3_vel_x', 'p3_vel_y', 'p3_vel_z', 'p3_boost', 'p3_na',\n    'p4_pos_x', 'p4_pos_y', 'p4_pos_z', 'p4_vel_x', 'p4_vel_y', 'p4_vel_z', 'p4_boost', 'p4_na',\n    'p5_pos_x', 'p5_pos_y', 'p5_pos_z', 'p5_vel_x', 'p5_vel_y', 'p5_vel_z', 'p5_boost', 'p5_na',\n    'boost0_timer', 'boost1_timer', 'boost2_timer', 'boost3_timer',\n    'boost4_timer', 'boost5_timer']","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:11:14.816963Z","iopub.execute_input":"2022-10-09T11:11:14.817358Z","iopub.status.idle":"2022-10-09T11:11:14.831839Z","shell.execute_reply.started":"2022-10-09T11:11:14.817326Z","shell.execute_reply":"2022-10-09T11:11:14.830429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nDEBUG = False\ninput_path = Path('../input/fast-loading-high-compression-with-feather/feather_data')\n\ndef read_train():\n    dfs = []\n    for i in range(10):\n        dfs.append(pd.read_feather(input_path / f'train_{i}_compressed.ftr'))\n    result = pd.concat(dfs)\n    if DEBUG:\n        result = result.sample(frac=0.05)\n    return result.sample(frac=1, random_state=42)\n\ntrain_df = read_train()\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:11:17.487160Z","iopub.execute_input":"2022-10-09T11:11:17.487592Z","iopub.status.idle":"2022-10-09T11:12:21.830635Z","shell.execute_reply.started":"2022-10-09T11:11:17.487554Z","shell.execute_reply":"2022-10-09T11:12:21.829662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs_shuffled = np.array_split(train_df, 10)","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:16:25.516033Z","iopub.execute_input":"2022-10-09T11:16:25.516461Z","iopub.status.idle":"2022-10-09T11:16:27.811496Z","shell.execute_reply.started":"2022-10-09T11:16:25.516426Z","shell.execute_reply":"2022-10-09T11:16:27.810294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check whether the specified path exists or not\n# if not then create \n\ndef makedir_check(path):\n    exist_dir = os.path.exists(path)\n    if exist_dir:\n        print(\"The directory already exists!\")\n    else:\n        os.makedirs(path)\n        print(f\"The directory, {path}, has been created!\")","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:27:21.856952Z","iopub.execute_input":"2022-10-09T11:27:21.857368Z","iopub.status.idle":"2022-10-09T11:27:21.863398Z","shell.execute_reply.started":"2022-10-09T11:27:21.857329Z","shell.execute_reply":"2022-10-09T11:27:21.862101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"makedir_check('feather_data')","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:27:29.327636Z","iopub.execute_input":"2022-10-09T11:27:29.328105Z","iopub.status.idle":"2022-10-09T11:27:29.333752Z","shell.execute_reply.started":"2022-10-09T11:27:29.328066Z","shell.execute_reply":"2022-10-09T11:27:29.332883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, df in enumerate(dfs_shuffled):\n    #save to feather\n    df.reset_index(drop=True).to_feather(f\"feather_data/shuffled_train_{i}.ftr\")\n    print(f\"Saved chunk {i} of the shuffled train dataframe as feather\")","metadata":{"execution":{"iopub.status.busy":"2022-10-09T11:33:19.364516Z","iopub.execute_input":"2022-10-09T11:33:19.364940Z","iopub.status.idle":"2022-10-09T11:33:29.896212Z","shell.execute_reply.started":"2022-10-09T11:33:19.364906Z","shell.execute_reply":"2022-10-09T11:33:29.895002Z"},"trusted":true},"execution_count":null,"outputs":[]}]}