{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"pip install progressbar2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\"\"\"\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport progressbar\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nlabel_df = pd.read_csv('/kaggle/input/predict-volcanic-eruptions-ingv-oe/train.csv')\nlabels = dict(zip(label_df['segment_id'].astype(str), label_df['time_to_eruption']))\n#print(labels)\n\n\ndef normalize_df(df):\n    return (df-df.min())/(df.max()-df.min())\n\n\ndef read_segment(segment_id, train):\n    #print(f'Read {segment_id}')\n    if train:\n        segment_path = f'/kaggle/input/predict-volcanic-eruptions-ingv-oe/train/{segment_id}.csv'\n    else:\n        segment_path = f'/kaggle/input/predict-volcanic-eruptions-ingv-oe/test/{segment_id}.csv'\n        \n    segment = pd.read_csv(segment_path)\n    #print(segment.isnull().any())\n    #display(segment)\n    segment = segment.fillna(segment.mean())\n    #display(segment)\n    nan_columns = segment.isnull().any().values\n    is_any_col_nan = (sum(nan_columns*1)>0)\n    segment = segment.fillna(0)\n    \n    segment = normalize_df(segment)\n\n    if is_any_col_nan:\n        drop=True\n    else:\n        drop=False\n        # drop this datapoint\n    # read df\n    # fill NaN\n    # convert to numpy tensor\n    return segment, drop\n\n\ndef create_data(segment_ids, eruption_times, train):\n    max_val = 0\n    dropped=0\n    #X = np.empty((4431, 60001, 10))\n    y = eruption_times\n    if not os.path.exists(\"/kaggle/data/segments\"):\n        os.makedirs(\"/kaggle/data/segments\")\n    with progressbar.ProgressBar(max_value=len(segment_ids)) as bar:\n        for idx, seg_id in progressbar.progressbar(enumerate(segment_ids)):\n            segment, drop = read_segment(seg_id, train)\n            if drop:\n                dropped+=1\n\n            bar.update(idx)\n\n                \n            segment.to_csv(f\"/kaggle/data/segments/{str(seg_id)}.csv\", index=False)\n    print(f\"{dropped} data points with missing sensors\")\n   \n    label_df.to_csv(f\"/kaggle/working/data/train.csv\")\n    \ncreate_data(list(labels.keys())[:], label_df['time_to_eruption'], True)\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport progressbar\nfrom multiprocessing import Pool\nfrom functools import partial\nimport tqdm\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nlabel_df = pd.read_csv('/kaggle/input/predict-volcanic-eruptions-ingv-oe/train.csv')\nlabels = dict(zip(label_df['segment_id'].astype(str), label_df['time_to_eruption']))\n#print(labels)\n\n\ndef normalize_df(df):\n    return (df-df.min())/(df.max()-df.min())\n\n\ndef read_segment(segment_id, train):\n    #print(f'Read {segment_id}')\n    if train:\n        segment_path = f'/kaggle/input/predict-volcanic-eruptions-ingv-oe/train/{segment_id}.csv'\n    else:\n        segment_path = f'/kaggle/input/predict-volcanic-eruptions-ingv-oe/test/{segment_id}.csv'\n        \n    segment = pd.read_csv(segment_path)\n    #print(segment.isnull().any())\n    #display(segment)\n    segment = segment.fillna(segment.mean())\n    #display(segment)\n    nan_columns = segment.isnull().any().values\n    is_any_col_nan = (sum(nan_columns*1)>0)\n    segment = segment.fillna(0)\n    \n    segment = normalize_df(segment).astype('float16')\n\n    if is_any_col_nan:\n        drop=True\n    else:\n        drop=False\n        # drop this datapoint\n    # read df\n    # fill NaN\n    # convert to numpy tensor\n    write_segment(segment_id, segment, train)\n    return segment, drop\n\n\ndef write_segment(seg_id, segment, train):\n    segment.to_csv(f\"/kaggle/working/data/segments/{str(seg_id)}.csv\", index=False)\n    #print(f'written: {seg_id}')\n\n\ndef create_data(segment_ids, eruption_times, train):\n    max_val = 0\n    dropped=0\n    #X = np.empty((4431, 60001, 10))\n    y = eruption_times\n    if not os.path.exists(\"/kaggle/working/data/segments\"):\n        os.makedirs(\"/kaggle/working/data/segments\")\n\n    with Pool(4) as p:\n        #r = list(tqdm.tqdm(p.imap(_foo, range(30)), total=30))\n        result = list(tqdm.tqdm(p.imap(partial(read_segment, train=True), segment_ids), total=len(segment_ids)))\n    print(f\"{dropped} data points with missing sensors\")\n   \n    label_df.to_csv(f\"/kaggle/working/data/train.csv\")\n    \ncreate_data(list(labels.keys())[:], label_df['time_to_eruption'], True)\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls /kaggle/input/predict-volcanic-eruptions-ingv-oe/train | wc\n!ls /kaggle/input/predict-volcanic-eruptions-ingv-oe/test | wc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls /kaggle/working\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df=pd.read_csv('/kaggle/working/data/segments/139656908.csv')\ndisplay(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}