{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"markdown","source":"## Problem with big size of train.csv\nNot everyone has enough resource for work with big size file csv. Next script can help someone solve this problem. It create split files for each earthquake. Size of biggest file = 1.7 Gb"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd\n# train_file_locate - variable with path to train.csv For example 'data/train.csv'\ndef split_to_separate_earthquake_files(train_file_locate):\n    print('Begin')\n    chunksize = 10 ** 7\n    i = 1\n    n = 0\n    time_to_falture_pr = chunksize\n    dt = pd.DataFrame\n    frames = []\n    for chunk in pd.read_csv(train_file_locate, chunksize=chunksize):\n        n += 1\n        print('read chank_'+str(n))\n        time_to_falture_begin = chunk['time_to_failure'][(n-1)*chunksize]\n        time_to_falture_end = chunk['time_to_failure'][(n-1)*chunksize + chunk.shape[0] - 1]\n        if (time_to_falture_begin > time_to_falture_end) and (time_to_falture_begin < time_to_falture_pr) :\n            frames.append(chunk)\n        else:\n            if time_to_falture_begin > time_to_falture_pr:\n                print('saving earthquake_'+str(i))   \n                fr_to_csv = pd.concat(frames)\n                fr_to_csv.to_csv('data/earthquake_'+str(i)+'.csv')\n                print('saved earthquake_'+str(i))            \n                dt = pd.DataFrame\n                frames = []\n                i += 1\n            else:\n                if time_to_falture_begin > time_to_falture_end:\n                    frames.append(chunk)\n                else:\n                    frames.append(chunk.loc[chunk['time_to_failure'] < time_to_falture_end])\n                    print('saving_2 earthquake_'+str(i)) \n                    fr_to_csv = pd.concat(frames)\n                    fr_to_csv.to_csv('data/earthquake_'+str(i)+'.csv')\n                    print('saved_2 earthquake_'+str(i))            \n                    dt = pd.DataFrame\n                    frames = []\n                    i += 1\n                    frames.append(chunk.loc[chunk['time_to_failure'] < time_to_falture_end])\n        time_to_falture_pr = time_to_falture_end\n    print('Ready')\n    # return count of earthquake\n    return i","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"706582d283952d32a0f987e52f35b625e12be113"},"cell_type":"markdown","source":"## Hope it help someone"},{"metadata":{"trusted":true,"_uuid":"a2a89957537680e43dacdde508f97b477a578d26"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b65d6d066ec97c159cc6bda53ab504e55ad8f602"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b6fbeb8dc223f1b5849501da32413477c962b27c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}