{
  "id": 80762,
  "title": "Fast data Reading based on different earthquakes",
  "url": "/competitions/LANL-Earthquake-Prediction/discussion/80762",
  "author_name": "Vasilis Stamatopoulos",
  "post_date": "2019-02-16T08:03:36.947000",
  "votes": 2,
  "comment_count": 0,
  "views": 0,
  "content": "<p>Here are two scripts that transform the columns in .bin files, which you can call by passing an index number from 0 to 15 (one corresponding to each earthquake).\nHope this helps you in reading the train set\n<strong>Script 1</strong>\nRun this script first to create the .bin files</p>\n\n<pre><code>import time\nimport os.path\nimport numpy as np\nimport pandas as pd\n\nCOLUMN_TO_TYPE = {\n    'acoustic_data': np.int16,\n    'time_to_failure': np.float64\n\n}\n\ndef prepare_data(directory, name, output_columns):\n    start_time = time.time()\n    file_path = os.path.join(directory, '{}.csv'.format(name))\n    print('reading {}  '.format(file_path))\n\n    dtypes = {column: COLUMN_TO_TYPE[column] for column in output_columns}\n\n    data = pd.read_csv(file_path, usecols =output_columns, dtype=dtypes, engine='c')\n    print(\"{:6.4f} secs\".format((time.time() - start_time)))\n\n    for column in output_columns:\n        output_file_name = '{}_{}.bin'.format(name, column)\n        #print('dumping {}  '.format(output_file_name), end='')\n        start_time = time.time()\n        mmap = np.memmap(output_file_name, dtype=COLUMN_TO_TYPE[column], mode='w+', shape=(data.shape[0]))\n        mmap[:] = data[column].values\n        del mmap\n        print(\"{:6.4f} secs\".format((time.time() - start_time)))\n\n\ndef main():\n    directory = '../input'\n    output_columns = ['acoustic_data', 'time_to_failure']\n    prepare_data(directory, 'train', output_columns)\n\n\nif __name__ == '__main__':\n    main()\n</code></pre>\n\n<p><strong>Script 2</strong>\n import the following each time you want to read and run this\ninfo = init_reading()\nread_object_info(info, i, as_pandas=True, columns=None) #where i == [0,16)</p>\n\n<pre><code>import numpy as np\nimport pandas as pd\nimport os.path\nimport time\n\nCOLUMN_TO_TYPE = {\n    'acoustic_data': np.int16,\n    'time_to_failure': np.float64\n\n}\n\npart1_directory = r'dataset/column_files'\n\n\nCOLUMN_TO_FOLDER = {\n    'acoustic_data': part1_directory,\n    'time_to_failure': part1_directory\n}\n\n\n\ndef init_reading():\n\n    info = {\n        \"0\": [0,   5656574],\n        \"1\": [50085878, 104677356],\n        \"2\": [104677356, 138772453],\n        \"3\": [138772453, 187641820],\n        \"4\": [187641820, 218652630],\n        \"5\": [218652630, 245829585],\n        \"6\": [245829585, 307838917],\n        \"7\": [307838917, 338276287],\n        \"8\": [338276287, 375377848],\n        \"9\": [375377848, 419368880],\n        \"10\": [419368880, 461811623],\n        \"11\": [461811623, 495800225],\n        \"12\": [495800225, 528777115],\n        \"13\": [528777115, 585568144],\n        \"14\": [585568144, 621985673],\n        \"15\": [621985673, 629145479]\n    }\n\n    mmaps = {}\n    for column, dtype in COLUMN_TO_TYPE.items():\n        directory = COLUMN_TO_FOLDER[column]\n        file_path = os.path.join(directory, 'train_set_{}.bin'.format(column))\n        mmap = np.memmap(file_path, dtype=COLUMN_TO_TYPE[column], mode='r', shape=(629145479,))\n        mmaps[column] = mmap\n\n    info['mmaps'] = mmaps\n\n    return info\n\n\n\ndef read_object_info(info, object_id, as_pandas=True, columns=None):\n    start = info[str(object_id)][0]\n    end = info[str(object_id)][1]\n\n    data = read_object_by_index_range(info, start, end, as_pandas, columns)\n    return data\n\n\ndef read_object_by_index_range(info, start, end, as_pandas=True, columns=None):\n    data = {}\n    for column, mmap in info['mmaps'].items():\n        if columns is None or column in columns:\n            data[column] = mmap[start: end]\n\n    if as_pandas:\n        data = pd.DataFrame(data)\n\n    return data\n</code></pre>\n\n<p>Hope these help</p>",
  "messages": [
    {
      "id": 472581,
      "postDate": "2019-02-16T08:03:36.947Z",
      "content": "<p>Here are two scripts that transform the columns in .bin files, which you can call by passing an index number from 0 to 15 (one corresponding to each earthquake).\nHope this helps you in reading the train set\n<strong>Script 1</strong>\nRun this script first to create the .bin files</p>\n\n<pre><code>import time\nimport os.path\nimport numpy as np\nimport pandas as pd\n\nCOLUMN_TO_TYPE = {\n    'acoustic_data': np.int16,\n    'time_to_failure': np.float64\n\n}\n\ndef prepare_data(directory, name, output_columns):\n    start_time = time.time()\n    file_path = os.path.join(directory, '{}.csv'.format(name))\n    print('reading {}  '.format(file_path))\n\n    dtypes = {column: COLUMN_TO_TYPE[column] for column in output_columns}\n\n    data = pd.read_csv(file_path, usecols =output_columns, dtype=dtypes, engine='c')\n    print(\"{:6.4f} secs\".format((time.time() - start_time)))\n\n    for column in output_columns:\n        output_file_name = '{}_{}.bin'.format(name, column)\n        #print('dumping {}  '.format(output_file_name), end='')\n        start_time = time.time()\n        mmap = np.memmap(output_file_name, dtype=COLUMN_TO_TYPE[column], mode='w+', shape=(data.shape[0]))\n        mmap[:] = data[column].values\n        del mmap\n        print(\"{:6.4f} secs\".format((time.time() - start_time)))\n\n\ndef main():\n    directory = '../input'\n    output_columns = ['acoustic_data', 'time_to_failure']\n    prepare_data(directory, 'train', output_columns)\n\n\nif __name__ == '__main__':\n    main()\n</code></pre>\n\n<p><strong>Script 2</strong>\n import the following each time you want to read and run this\ninfo = init_reading()\nread_object_info(info, i, as_pandas=True, columns=None) #where i == [0,16)</p>\n\n<pre><code>import numpy as np\nimport pandas as pd\nimport os.path\nimport time\n\nCOLUMN_TO_TYPE = {\n    'acoustic_data': np.int16,\n    'time_to_failure': np.float64\n\n}\n\npart1_directory = r'dataset/column_files'\n\n\nCOLUMN_TO_FOLDER = {\n    'acoustic_data': part1_directory,\n    'time_to_failure': part1_directory\n}\n\n\n\ndef init_reading():\n\n    info = {\n        \"0\": [0,   5656574],\n        \"1\": [50085878, 104677356],\n        \"2\": [104677356, 138772453],\n        \"3\": [138772453, 187641820],\n        \"4\": [187641820, 218652630],\n        \"5\": [218652630, 245829585],\n        \"6\": [245829585, 307838917],\n        \"7\": [307838917, 338276287],\n        \"8\": [338276287, 375377848],\n        \"9\": [375377848, 419368880],\n        \"10\": [419368880, 461811623],\n        \"11\": [461811623, 495800225],\n        \"12\": [495800225, 528777115],\n        \"13\": [528777115, 585568144],\n        \"14\": [585568144, 621985673],\n        \"15\": [621985673, 629145479]\n    }\n\n    mmaps = {}\n    for column, dtype in COLUMN_TO_TYPE.items():\n        directory = COLUMN_TO_FOLDER[column]\n        file_path = os.path.join(directory, 'train_set_{}.bin'.format(column))\n        mmap = np.memmap(file_path, dtype=COLUMN_TO_TYPE[column], mode='r', shape=(629145479,))\n        mmaps[column] = mmap\n\n    info['mmaps'] = mmaps\n\n    return info\n\n\n\ndef read_object_info(info, object_id, as_pandas=True, columns=None):\n    start = info[str(object_id)][0]\n    end = info[str(object_id)][1]\n\n    data = read_object_by_index_range(info, start, end, as_pandas, columns)\n    return data\n\n\ndef read_object_by_index_range(info, start, end, as_pandas=True, columns=None):\n    data = {}\n    for column, mmap in info['mmaps'].items():\n        if columns is None or column in columns:\n            data[column] = mmap[start: end]\n\n    if as_pandas:\n        data = pd.DataFrame(data)\n\n    return data\n</code></pre>\n\n<p>Hope these help</p>",
      "rawMarkdown": "Here are two scripts that transform the columns in .bin files, which you can call by passing an index number from 0 to 15 (one corresponding to each earthquake).\nHope this helps you in reading the train set\n**Script 1**\nRun this script first to create the .bin files\n\n    import time\n    import os.path\n    import numpy as np\n    import pandas as pd\n    \n    COLUMN_TO_TYPE = {\n        'acoustic_data': np.int16,\n        'time_to_failure': np.float64\n       \n    }\n    \n    def prepare_data(directory, name, output_columns):\n        start_time = time.time()\n        file_path = os.path.join(directory, '{}.csv'.format(name))\n        print('reading {}  '.format(file_path))\n    \n        dtypes = {column: COLUMN_TO_TYPE[column] for column in output_columns}\n    \n        data = pd.read_csv(file_path, usecols =output_columns, dtype=dtypes, engine='c')\n        print(\"{:6.4f} secs\".format((time.time() - start_time)))\n    \n        for column in output_columns:\n            output_file_name = '{}_{}.bin'.format(name, column)\n            #print('dumping {}  '.format(output_file_name), end='')\n            start_time = time.time()\n            mmap = np.memmap(output_file_name, dtype=COLUMN_TO_TYPE[column], mode='w+', shape=(data.shape[0]))\n            mmap[:] = data[column].values\n            del mmap\n            print(\"{:6.4f} secs\".format((time.time() - start_time)))\n    \n    \n    def main():\n        directory = '../input'\n        output_columns = ['acoustic_data', 'time_to_failure']\n        prepare_data(directory, 'train', output_columns)\n    \n    \n    if __name__ == '__main__':\n        main()\n\n\n\n**Script 2**\n import the following each time you want to read and run this\ninfo = init_reading()\nread_object_info(info, i, as_pandas=True, columns=None) #where i == [0,16)\n\n\n    import numpy as np\n    import pandas as pd\n    import os.path\n    import time\n    \n    COLUMN_TO_TYPE = {\n        'acoustic_data': np.int16,\n        'time_to_failure': np.float64\n       \n    }\n    \n    part1_directory = r'dataset/column_files'\n    \n    \n    COLUMN_TO_FOLDER = {\n        'acoustic_data': part1_directory,\n        'time_to_failure': part1_directory\n    }\n    \n    \n    \n    def init_reading():\n       \n        info = {\n            \"0\": [0,   5656574],\n            \"1\": [50085878, 104677356],\n            \"2\": [104677356, 138772453],\n            \"3\": [138772453, 187641820],\n            \"4\": [187641820, 218652630],\n            \"5\": [218652630, 245829585],\n            \"6\": [245829585, 307838917],\n            \"7\": [307838917, 338276287],\n            \"8\": [338276287, 375377848],\n            \"9\": [375377848, 419368880],\n            \"10\": [419368880, 461811623],\n            \"11\": [461811623, 495800225],\n            \"12\": [495800225, 528777115],\n            \"13\": [528777115, 585568144],\n            \"14\": [585568144, 621985673],\n            \"15\": [621985673, 629145479]\n        }\n    \n        mmaps = {}\n        for column, dtype in COLUMN_TO_TYPE.items():\n            directory = COLUMN_TO_FOLDER[column]\n            file_path = os.path.join(directory, 'train_set_{}.bin'.format(column))\n            mmap = np.memmap(file_path, dtype=COLUMN_TO_TYPE[column], mode='r', shape=(629145479,))\n            mmaps[column] = mmap\n    \n        info['mmaps'] = mmaps\n    \n        return info\n    \n    \n    \n    def read_object_info(info, object_id, as_pandas=True, columns=None):\n        start = info[str(object_id)][0]\n        end = info[str(object_id)][1]\n    \n        data = read_object_by_index_range(info, start, end, as_pandas, columns)\n        return data\n    \n    \n    def read_object_by_index_range(info, start, end, as_pandas=True, columns=None):\n        data = {}\n        for column, mmap in info['mmaps'].items():\n            if columns is None or column in columns:\n                data[column] = mmap[start: end]\n    \n        if as_pandas:\n            data = pd.DataFrame(data)\n    \n        return data\n\n\nHope these help",
      "votes": 2
    }
  ],
  "comments": [],
  "raw_markdown_by_id": {
    "472581": "Here are two scripts that transform the columns in .bin files, which you can call by passing an index number from 0 to 15 (one corresponding to each earthquake).\nHope this helps you in reading the train set\n**Script 1**\nRun this script first to create the .bin files\n\n    import time\n    import os.path\n    import numpy as np\n    import pandas as pd\n    \n    COLUMN_TO_TYPE = {\n        'acoustic_data': np.int16,\n        'time_to_failure': np.float64\n       \n    }\n    \n    def prepare_data(directory, name, output_columns):\n        start_time = time.time()\n        file_path = os.path.join(directory, '{}.csv'.format(name))\n        print('reading {}  '.format(file_path))\n    \n        dtypes = {column: COLUMN_TO_TYPE[column] for column in output_columns}\n    \n        data = pd.read_csv(file_path, usecols =output_columns, dtype=dtypes, engine='c')\n        print(\"{:6.4f} secs\".format((time.time() - start_time)))\n    \n        for column in output_columns:\n            output_file_name = '{}_{}.bin'.format(name, column)\n            #print('dumping {}  '.format(output_file_name), end='')\n            start_time = time.time()\n            mmap = np.memmap(output_file_name, dtype=COLUMN_TO_TYPE[column], mode='w+', shape=(data.shape[0]))\n            mmap[:] = data[column].values\n            del mmap\n            print(\"{:6.4f} secs\".format((time.time() - start_time)))\n    \n    \n    def main():\n        directory = '../input'\n        output_columns = ['acoustic_data', 'time_to_failure']\n        prepare_data(directory, 'train', output_columns)\n    \n    \n    if __name__ == '__main__':\n        main()\n\n\n\n**Script 2**\n import the following each time you want to read and run this\ninfo = init_reading()\nread_object_info(info, i, as_pandas=True, columns=None) #where i == [0,16)\n\n\n    import numpy as np\n    import pandas as pd\n    import os.path\n    import time\n    \n    COLUMN_TO_TYPE = {\n        'acoustic_data': np.int16,\n        'time_to_failure': np.float64\n       \n    }\n    \n    part1_directory = r'dataset/column_files'\n    \n    \n    COLUMN_TO_FOLDER = {\n        'acoustic_data': part1_directory,\n        'time_to_failure': part1_directory\n    }\n    \n    \n    \n    def init_reading():\n       \n        info = {\n            \"0\": [0,   5656574],\n            \"1\": [50085878, 104677356],\n            \"2\": [104677356, 138772453],\n            \"3\": [138772453, 187641820],\n            \"4\": [187641820, 218652630],\n            \"5\": [218652630, 245829585],\n            \"6\": [245829585, 307838917],\n            \"7\": [307838917, 338276287],\n            \"8\": [338276287, 375377848],\n            \"9\": [375377848, 419368880],\n            \"10\": [419368880, 461811623],\n            \"11\": [461811623, 495800225],\n            \"12\": [495800225, 528777115],\n            \"13\": [528777115, 585568144],\n            \"14\": [585568144, 621985673],\n            \"15\": [621985673, 629145479]\n        }\n    \n        mmaps = {}\n        for column, dtype in COLUMN_TO_TYPE.items():\n            directory = COLUMN_TO_FOLDER[column]\n            file_path = os.path.join(directory, 'train_set_{}.bin'.format(column))\n            mmap = np.memmap(file_path, dtype=COLUMN_TO_TYPE[column], mode='r', shape=(629145479,))\n            mmaps[column] = mmap\n    \n        info['mmaps'] = mmaps\n    \n        return info\n    \n    \n    \n    def read_object_info(info, object_id, as_pandas=True, columns=None):\n        start = info[str(object_id)][0]\n        end = info[str(object_id)][1]\n    \n        data = read_object_by_index_range(info, start, end, as_pandas, columns)\n        return data\n    \n    \n    def read_object_by_index_range(info, start, end, as_pandas=True, columns=None):\n        data = {}\n        for column, mmap in info['mmaps'].items():\n            if columns is None or column in columns:\n                data[column] = mmap[start: end]\n    \n        if as_pandas:\n            data = pd.DataFrame(data)\n    \n        return data\n\n\nHope these help"
  }
}