{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom scipy.stats import skew,kurtosis\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nfrom sklearn.linear_model import LinearRegression\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"934b84e9edd76c2bfd3a342ac1957c7fc9f55239"},"cell_type":"code","source":"from scipy.stats import kurtosis\na=np.array([1,5,8,6,4,2])\nkurtosis(a)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"96910e2b2bff97998b6a5403743b48c9b9704b7b"},"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import NuSVR\nfrom sklearn.metrics import mean_absolute_error\nfrom catboost import CatBoostRegressor","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c368d4a380435cdc2d4fe2eb9871236f1bc87b4f"},"cell_type":"code","source":"train.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a8b02413f09c8205f0d38c8448e4678ccf147c39"},"cell_type":"code","source":"# pandas doesn't show us all the decimals\npd.options.display.precision = 15","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d105ddc41071309306b9784ba43d0e6ce3297bf"},"cell_type":"code","source":"# much better!\ntrain.head()\n#train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"842ac200627c9c798964fb810b9c3950cbf82050"},"cell_type":"code","source":"# Create a training file with simple derived features\nfrom scipy.stats import kurtosis,skew\nrows = 150_000\nsegments = int(np.floor(train.shape[0] / rows))\n\nX_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n         columns=['ave', 'std', 'max', 'min','kurt','skew','25per','50per','75per'])\n#X_train=pd.DataFrame(columns=['ave','std','max','min'])\ny_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['time_to_failure'])\n#y_train=pd.DataFrame(columns=['time_to_failure'])\nfor segment in tqdm(range(segments)):\n    seg = train.iloc[segment*rows:segment*rows+rows]\n    x = seg['acoustic_data'].values\n    y = seg['time_to_failure'].values[-1]\n    #print(seg['acoustic_data'].values)\n    #print(seg['time_to_failure'].values[-1])\n    y_train.loc[segment, 'time_to_failure'] = y\n    \n    X_train.loc[segment, 'ave'] = x.mean()\n    X_train.loc[segment, 'std'] = x.std()\n    X_train.loc[segment, 'max'] = x.max()\n    X_train.loc[segment, 'min'] = x.min()\n    #X_train.loc[segment, 'mad'] = x.mad()\n    X_train.loc[segment, 'kurt'] = kurtosis(x)\n    X_train.loc[segment, 'skew'] = skew(x)\n    X_train.loc[segment, '25per'] = np.quantile(x,0.25)\n    X_train.loc[segment, '50per'] = np.quantile(x,0.50)\n    X_train.loc[segment, '75per'] = np.quantile(x,0.75)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2eb91f44301bc28510db46b8e06f722365226a8"},"cell_type":"code","source":"print(X_train.head())\nprint(y_train.head())\nprint(X_train.shape[0],y_train.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b71b22cf696be973dc1e8e84bf1b6ef4b0eaddb"},"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)\nprint(X_train_scaled)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"495bb6b3911525cc04e83b59844fdd43326bc558"},"cell_type":"code","source":"svm = NuSVR()\nsvm.fit(X_train_scaled, y_train.values.flatten())\ny_pred = svm.predict(X_train_scaled)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6bc42a5e069d2b1bcda5798d9d2090efed95999d"},"cell_type":"code","source":"plt.figure(figsize=(6, 6))\nplt.scatter(y_train.values.flatten(), y_pred)\nplt.xlim(0, 20)\nplt.ylim(0, 20)\nplt.xlabel('actual', fontsize=12)\nplt.ylabel('predicted', fontsize=12)\nplt.plot([(0, 0), (20, 20)], [(0, 0), (20, 20)])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7f55ddbc1974805d8d0ec1b3f2bc7ad9a723468e"},"cell_type":"code","source":"score = mean_absolute_error(y_train.values.flatten(), y_pred)\nprint(f'Score: {score:0.3f}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87c4dc7aabc5f1309ca6b67dc62fed2d24918e03"},"cell_type":"code","source":"linearRegressor = LinearRegression()\nlinearRegressor.fit(X_train_scaled, y_train.values.flatten())\ny_pred1 = linearRegressor.predict(X_train_scaled)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b09c96d4bd82bbc62332aeb090bf2380283113cf"},"cell_type":"code","source":"plt.figure(figsize=(6, 6))\nplt.scatter(y_train.values.flatten(), y_pred1)\nplt.xlim(0, 20)\nplt.ylim(0, 20)\nplt.xlabel('actual', fontsize=12)\nplt.ylabel('predicted', fontsize=12)\nplt.plot([(0, 0), (20, 20)], [(0, 0), (20, 20)])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0e21f7e99865a8b904f8dd0f269c23e17d1deb0"},"cell_type":"code","source":"score = mean_absolute_error(y_train.values.flatten(), y_pred1)\nprint(f'Score: {score:0.3f}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b74356643241fa7d5dcfab20834ae08475218b12"},"cell_type":"code","source":"catboostreg=CatBoostRegressor(iterations=10000,loss_function='MAE',boosting_type='Ordered')\ncatboostreg.fit(X_train,y_train,silent=True)\nprint(catboostreg.best_score_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6188e2fe74d8b0fd4c943d6687edfdf8bda91c95"},"cell_type":"code","source":"#no impact on training with Cat Boost Regressor on scaled data\ncatboostreg1=CatBoostRegressor(iterations=10000,loss_function='MAE',boosting_type='Ordered')\ncatboostreg1.fit(X_train_scaled,y_train,silent=True)\nprint(catboostreg1.best_score_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a4b66284e79c7a2dae3ea7a2856fff8d7fee7a9e"},"cell_type":"code","source":"submission = pd.read_csv('../input/sample_submission.csv', index_col='seg_id')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87640239d9f9a3fdda3705dc44f9e3541b41b0c4"},"cell_type":"code","source":"X_test = pd.DataFrame(columns=X_train.columns, dtype=np.float64, index=submission.index)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"009a093b49048a18e556595f0de5104d26fbeb37"},"cell_type":"code","source":"print(X_test.index)\nfor seg_id in X_test.index:\n    seg = pd.read_csv('../input/test/' + seg_id + '.csv')\n    \n    x = seg['acoustic_data'].values\n    #print(seg_id)\n    #print(x.mean())\n    X_test.loc[seg_id, 'ave'] = x.mean()\n    X_test.loc[seg_id, 'std'] = x.std()\n    X_test.loc[seg_id, 'max'] = x.max()\n    X_test.loc[seg_id, 'min'] = x.min()\n    X_test.loc[seg_id, 'kurt'] = kurtosis(x)\n    X_test.loc[seg_id, 'skew'] = skew(x)\n    X_test.loc[seg_id, '25per'] = np.quantile(x,0.25)\n    X_test.loc[seg_id, '50per'] = np.quantile(x,0.50)\n    X_test.loc[seg_id, '75per'] = np.quantile(x,0.75)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1770a6a62722f0cdb73d9b8afc31d4726758b39f"},"cell_type":"code","source":"X_test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d985214b42dd7abff775473f7c0b63b764d0405d"},"cell_type":"code","source":"X_test_scaled = scaler.transform(X_test)\nsubmission['time_to_failure'] = catboostreg.predict(X_test_scaled)\n#submission['time_to_failure'] = svm.predict(X_test_scaled)\nsubmission.to_csv('submission_cbg.csv')\nprint(\"Done\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}