{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import NuSVR\nfrom sklearn.metrics import mean_absolute_error","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2cc91e3d3a151e2cb52cb014a55492ed77e8028"},"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import NuSVR\nfrom sklearn.metrics import mean_absolute_error","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"699b2c1f8a4b435230476338a3f288f07a70108b"},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d45a79f2d09727875c06d0f20b5ae30ad63e7bbd"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c9d7f73a50a6bd560c0193ccb1a517b757026f9f"},"cell_type":"code","source":"#pandas doesn't show us all the decimals\npd.options.display.precision=15","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"55c649f165a6e319c802de68ecca3a921e53f5c1"},"cell_type":"code","source":"#much better!\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51a32428401eb653987389bb8222fedd9613a7a1"},"cell_type":"code","source":"# Create a training file with simple derived features\n\nrows = 150_000\nsegments = int(np.floor(train.shape[0] / rows))\n\nX_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['ave', 'std', 'max', 'min'])\ny_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['time_to_failure'])\n\nfor segment in tqdm(range(segments)):\n    seg = train.iloc[segment*rows:segment*rows+rows]\n    x = seg['acoustic_data'].values\n    y = seg['time_to_failure'].values[-1]\n    \n    y_train.loc[segment, 'time_to_failure'] = y\n    \n    X_train.loc[segment, 'ave'] = x.mean()\n    X_train.loc[segment, 'std'] = x.std()\n    X_train.loc[segment, 'max'] = x.max()\n    X_train.loc[segment, 'min'] = x.min()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f981eb25e260a1f1ceb552d187a84c7ca6bee2af"},"cell_type":"code","source":"rows=150_000\nsegments=int(np.floor(train.shape[0]/rows))\nX_train=pd.DataFrame(index=range(segments),dtype=np.float64,columns=['ave','std','max','min'])\ny_train=pd.DataFrame(index=range(segments),dtype=np.float64,columns=['time_to_failure'])\nfor segment in tqdm(range(segments)):\n    seg=train.iloc[segment*rows:segment*rows+rows]\n    x=seg['acoustic_data'].values\n    y=seg['time_to_failure'].values[-1]\n    y_train.loc[segment,'time_to_failure']=y\n    X_train.loc[segment,'ave']=x.mean()\n    X_train.loc[segment,'std']=x.std()\n    X_train.loc[segment,'max']=x.max()\n    X_train.loc[segment,'min']=x.min()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f2a19ae4a12200ce8e4ef865b6b1db13da762f4c"},"cell_type":"code","source":"X_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e1edfd7e8cc19de3932ed06d335504b384d760e7"},"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ca34860a0b3ac445be0965225a18b1d4b42a3fe5"},"cell_type":"code","source":"scaler=StandardScaler()\nscaler.fit(X_train)\nX_train_scaled=scaler.transform(X_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2282756b0797c7b547ca6df1e4025a925564ef78"},"cell_type":"code","source":"\nsvm=NuSVR()\nsvm.fit(X_train_scaled,y_train.values.flatten())\ny_pred=svm.predict(X_train_scaled)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f745f34fd3f2683cd72061061871574a4a9da9f9"},"cell_type":"code","source":"\nplt.figure(figsize=(6,6))\nplt.scatter(y_train.values.flatten(),y_pred)\nplt.xlim(0,20)\nplt.ylim(0,20)\nplt.xlabel('actual',fontsize=12)\nplt.ylabel('predicted',fontsize=12)\nplt.plot([(0,0),(20,20)],[(0,0),(20,20)])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dfcc93d8021c8485860cd44e4fddd18637aa1767"},"cell_type":"code","source":"score = mean_absolute_error(y_train.values.flatten(), y_pred)\nprint(f'Score: {score:0.3f}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f35b3227c87af8258bfbddb775a26f7830cc50f6"},"cell_type":"code","source":"submission = pd.read_csv('../input/sample_submission.csv', index_col='seg_id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f38f34e91fb079946629387f30af525701e331c5"},"cell_type":"code","source":"X_test = pd.DataFrame(columns=X_train.columns, dtype=np.float64, index=submission.index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"24a69122ed38604515667fd84e7e9458562299fb"},"cell_type":"code","source":"for seg_id in X_test.index:\n    seg = pd.read_csv('../input/test/' + seg_id + '.csv')\n    \n    x = seg['acoustic_data'].values\n    \n    X_test.loc[seg_id, 'ave'] = x.mean()\n    X_test.loc[seg_id, 'std'] = x.std()\n    X_test.loc[seg_id, 'max'] = x.max()\n    X_test.loc[seg_id, 'min'] = x.min()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"03f917e87feb9fc54596f76a8d25948f68cb8b7b"},"cell_type":"code","source":"X_test_scaled = scaler.transform(X_test)\nsubmission['time_to_failure'] = svm.predict(X_test_scaled)\nsubmission.to_csv('submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aae88750474ecd0eaf5eac7fb30f73c4f652ba53"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}