{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"324103e214a934f73a437043a3915ce2fd901be6"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import NuSVR\nfrom sklearn.metrics import mean_absolute_error","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e10c213da62f94d923a6d3b41e7f619bdd9cf8f8"},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d7caa1f6a8eb4d1b1257f5b5a0a9bc7683e7a260"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25ad08c168c282e44f4445316ace3d57a21b9bcd"},"cell_type":"code","source":"pd.options.display.precision = 9\ntrain.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aee87a43c261472ce89dad9b8d0b163b59df4d1e"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e1344e0a7349989371ac61833b61a1612a85162d"},"cell_type":"code","source":"from scipy import stats","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d57554e286b034f5d3925bf573c92433a4ed1144"},"cell_type":"code","source":"rows = 100_000\nsegments = int(np.floor(train.shape[0] / rows))\n\nX_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['ave', 'std', 'max', 'min', 'sum', 'range'])\ny_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['time_to_failure'])\n\nfor segment in tqdm(range(segments)):\n    seg = train.iloc[segment*rows:segment*rows+rows]\n    x = seg['acoustic_data'].values\n    y = seg['time_to_failure'].values[-1]\n    \n    y_train.loc[segment, 'time_to_failure'] = y\n    \n    X_train.loc[segment, 'ave'] = x.mean()\n    X_train.loc[segment, 'std'] = x.std()\n    X_train.loc[segment, 'max'] = x.max()\n    X_train.loc[segment, 'min'] = x.min()\n    X_train.loc[segment, 'sum'] = x.sum()\n    X_train.loc[segment, 'range'] = x.max()-x.min()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2fc8555dc701e87cde9e5eba242153d90cc1f11f"},"cell_type":"code","source":"X_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ca381b25ca46411c94b4b468ac7d461868f08a5"},"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ebdb2a1f2b441f88068e3af1c7a3fef588773265"},"cell_type":"code","source":"svm = NuSVR()\nsvm.fit(X_train_scaled, y_train.values.flatten())\ny_pred = svm.predict(X_train_scaled)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe737bbc981d03940a5064747fda09efee6bb8f8"},"cell_type":"code","source":"submission = pd.read_csv('../input/sample_submission.csv', index_col='seg_id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f3383a031e1fa8be29a0d1869fa7949ba47e56d"},"cell_type":"code","source":"X_test = pd.DataFrame(columns=X_train.columns, dtype=np.float64, index=submission.index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ab1fdbef86746b8ba3a935056ca733a21c5389bf"},"cell_type":"code","source":"X_test.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4c8ab1a8f187b307d1a73ecdd5374700a9d449d2"},"cell_type":"code","source":"for seg_id in X_test.index:\n    seg = pd.read_csv('../input/test/' + seg_id + '.csv')\n    \n    x = seg['acoustic_data'].values\n    \n    X_test.loc[seg_id, 'ave'] = x.mean()\n    X_test.loc[seg_id, 'std'] = x.std()\n    X_test.loc[seg_id, 'max'] = x.max()\n    X_test.loc[seg_id, 'min'] = x.min()\n    X_test.loc[seg_id, 'sum'] = x.sum()\n    X_test.loc[seg_id, 'range'] = x.max()-x.min()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da067f2dbb1ef5b42eda38c663d4fd6ed301a9ed"},"cell_type":"code","source":"X_test_scaled = scaler.transform(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"140ad2fbc6f9c9f0ff1724b84fa2fa1573a5dc1a"},"cell_type":"code","source":"#X_test_scaled.fillna(method='bfill') ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f25456a888bf9ccd4722decc049468b671f62c3b"},"cell_type":"code","source":"submission['time_to_failure'] = svm.predict(X_test_scaled)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc8bf00be7ad6c2942ef9e0289b242d8fa37cf81"},"cell_type":"code","source":"submission.to_csv('submission100k.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9a989843fff5ee2140a0035bf4810b7cc22a1e11"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}