{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nimport sys\nimport matplotlib.pyplot as plt\nfrom catboost import CatBoostRegressor\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f4b5a2e0f96e58b607d4b50a1982d3a18972c3a"},"cell_type":"code","source":"# constants\ntrace_length = 150000\nstep_size = 25000\nexpected_passes = 629145480//step_size","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"15e31ae7d122f14aa065a775ee61875b8e496653"},"cell_type":"code","source":"columns = ['freq_'+str(i) for i in range(1200)]\ntransformed_data = []\ntarget = []\nprevious_df = pd.DataFrame()\nchunk_no = 0\nwindow_no = 0\nfor train_df in tqdm(pd.read_csv('../input/train.csv', chunksize =2* 10 ** 5, dtype={'acoustic_data': np.int16, 'time_to_failure': np.float32})):\n    chunk_no += 1\n#     print (\"processing chunk number \", chunk_no)\n    if not previous_df.empty:\n        train_df = pd.concat([previous_df, train_df], axis=0)\n    start_index = min(train_df.index.values)\n    while start_index + trace_length <= max(train_df.index.values):\n        data_list = []\n        interim_index = start_index\n        for _ in range(3):\n            power_spectrum = np.fft.fft(train_df.loc[interim_index:interim_index+50000, 'acoustic_data'].values)\n            power_spectrum = np.absolute(power_spectrum)\n            data_list.extend(list(power_spectrum[:10000][::25]))\n            interim_index += 50000\n        transformed_data.append(data_list)\n        target.append(np.mean(train_df.loc[start_index:start_index+150000, 'time_to_failure'].values))\n        start_index += step_size\n        window_no += 1\n#         sys.stdout.flush()\n#         print (\"windows processed \", window_no)\n#     if window_no > 10000:\n#         print (\"10000 windows processed\")\n#         break\n        \n        \n    previous_df = train_df.loc[start_index:, :]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9fca30fe8e8d7c848617c0ec2e41356f2926287e"},"cell_type":"code","source":"X_train = pd.DataFrame(columns=columns, data=transformed_data)\ny_train = pd.Series(data=target)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1573eaccd65443d3fb5c504fcc1e79ec661f017"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X_train, y_train, shuffle=True, test_size=0.2, \n                                                    random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d891fab3a9e920119c2eec66165545dc557cba17"},"cell_type":"code","source":"params = {\n    'iterations': 2000,\n    'learning_rate': 0.1,\n    'eval_metric': 'MAE',\n    'random_seed': 42,\n    'logging_level': 'Silent',\n    'use_best_model': False\n}\nearlystop_params = params.copy()\nearlystop_params.update({\n    'od_type': 'Iter',\n    'od_wait': 100\n})\nearlystop_model = CatBoostRegressor(**earlystop_params)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59c01451acb0ab2cbd5e031817d24b144bbc6021"},"cell_type":"code","source":"earlystop_model.fit(\n    X_train, y_train,\n    eval_set=(X_test, y_test),\n    logging_level='Verbose',  # you can uncomment this for text output\n    plot=True\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eba18fcf61c89f639fcd825069b8ecec77bc6129"},"cell_type":"code","source":"transformed_data = []\nsubmission = pd.read_csv('../input/sample_submission.csv', index_col='seg_id')\nfor i, seg_id in enumerate(tqdm(submission.index)):\n    seg = pd.read_csv('../input/test/' + seg_id + '.csv')\n    interim_index = 0\n    data_list = []\n    for _ in range(3):\n        power_spectrum = np.fft.fft(seg.loc[interim_index:interim_index+50000, 'acoustic_data'].values)\n        power_spectrum = np.absolute(power_spectrum)\n        data_list.extend(list(power_spectrum[:10000][::25]))\n        interim_index += 50000\n    transformed_data.append(data_list)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"04fb1c5cf8be00cb451efa463e7df1203fdfbdb5"},"cell_type":"code","source":"test_transformed = pd.DataFrame(columns=columns, data=transformed_data, index=submission.index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"787efcc3ebb453e5dc35cedfa8602e7b74579eeb"},"cell_type":"code","source":"predictions = earlystop_model.predict(test_transformed)\nsubmission['time_to_failure'] = predictions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac2ae667eec22ca856690ce3cbfb386cd6737f31"},"cell_type":"code","source":"submission.to_csv('submission.csv')","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}