{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd #veri işleme, CSV dosyası G / Ç","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:23:18.283435Z","iopub.execute_input":"2021-05-19T23:23:18.283731Z","iopub.status.idle":"2021-05-19T23:23:18.303780Z","shell.execute_reply.started":"2021-05-19T23:23:18.283685Z","shell.execute_reply":"2021-05-19T23:23:18.303169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nfrom catboost import CatBoostRegressor\nfrom sklearn.metrics import mean_absolute_error\nimport os","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:23:20.525051Z","iopub.execute_input":"2021-05-19T23:23:20.525546Z","iopub.status.idle":"2021-05-19T23:23:21.705899Z","shell.execute_reply.started":"2021-05-19T23:23:20.525505Z","shell.execute_reply":"2021-05-19T23:23:21.704978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#input içerisindekileri listeledik.\nprint(os.listdir(\"../input\"))","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:23:23.479713Z","iopub.execute_input":"2021-05-19T23:23:23.480197Z","iopub.status.idle":"2021-05-19T23:23:23.486180Z","shell.execute_reply.started":"2021-05-19T23:23:23.480152Z","shell.execute_reply":"2021-05-19T23:23:23.485272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Eğitim dosyasını okuyup train oluşturduk.\n#Eğitim verileri çok büyük bu yüzden hafızadan tasarruf etmek için veri türlerini belirttik.\ntrain = pd.read_csv('../input/train.csv', dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:23:26.140927Z","iopub.execute_input":"2021-05-19T23:23:26.141247Z","iopub.status.idle":"2021-05-19T23:27:24.168187Z","shell.execute_reply.started":"2021-05-19T23:23:26.141196Z","shell.execute_reply":"2021-05-19T23:27:24.167034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Ortalama değerlere yuvarlanmış acoustic_data grafiği oluşturduk.\ngrafik = train['acoustic_data'][:100]\ngrafik_smooth = grafik.rolling(10).mean()\n\nplt.plot(grafik)\nplt.plot(grafik_smooth)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:28:16.967920Z","iopub.execute_input":"2021-05-19T23:28:16.968271Z","iopub.status.idle":"2021-05-19T23:28:17.186972Z","shell.execute_reply.started":"2021-05-19T23:28:16.968208Z","shell.execute_reply":"2021-05-19T23:28:17.186097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Verilen değişken ile hedef değişken arasındaki ilişkiyi görmek için grafik ve her iki özelliğe dayalı olarak çizim yapma işlevi gerçekleştirdik.\nacoustic_data_train = train['acoustic_data'].values[::100]\ntime_to_failure_train = train['time_to_failure'].values[::100]\nfig, ax1 = plt.subplots(figsize=(18, 8))\nplt.title(\"Acoustic_data ve time_to_failure Eğilimleri\")\nplt.plot(acoustic_data_train, color='crimson')\nax1.set_ylabel('acoustic_data', color='crimson')\nplt.legend(['acoustic_data'])\nax2 = ax1.twinx()\nplt.plot(time_to_failure_train, color='b')\nax2.set_ylabel('time_to_failure', color='b')\nplt.legend(['time_to_failure'], loc=(0.9, 0.9))\nplt.grid(False)\n#eski çerçeveyi sildik\ndel acoustic_data_train\ndel time_to_failure_train","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:28:21.050944Z","iopub.execute_input":"2021-05-19T23:28:21.051241Z","iopub.status.idle":"2021-05-19T23:28:28.535506Z","shell.execute_reply.started":"2021-05-19T23:28:21.051204Z","shell.execute_reply":"2021-05-19T23:28:28.534399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pandas bize tüm ondalık sayıları göstermez\npd.options.display.precision = 10\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:28:34.123212Z","iopub.execute_input":"2021-05-19T23:28:34.123536Z","iopub.status.idle":"2021-05-19T23:28:34.147920Z","shell.execute_reply.started":"2021-05-19T23:28:34.123485Z","shell.execute_reply":"2021-05-19T23:28:34.146724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train verileri için özellik oluşturma\n# Basit türetilmiş özelliklere sahip bir eğitim dosyası oluşturduk\n\n\nrows = 150_000 #150_000\nsegments = int(np.floor(train.shape[0] / rows))\n\nX_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['ort', 'std', 'max', 'min'])\ny_train = pd.DataFrame(index=range(segments), dtype=np.float64,\n                       columns=['time_to_failure'])\n\nfor segment in tqdm(range(segments)):\n    seg = train.iloc[segment*rows:segment*rows+rows]\n    x = seg['acoustic_data'].values\n    y = seg['time_to_failure'].values[-1]\n    \n    y_train.loc[segment, 'time_to_failure'] = y\n    \n    X_train.loc[segment, 'ort'] = x.mean()\n    X_train.loc[segment, 'std'] = x.std()\n    X_train.loc[segment, 'max'] = x.max()\n    X_train.loc[segment, 'min'] = x.min()","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:28:38.182957Z","iopub.execute_input":"2021-05-19T23:28:38.183516Z","iopub.status.idle":"2021-05-19T23:28:47.017218Z","shell.execute_reply.started":"2021-05-19T23:28:38.183208Z","shell.execute_reply":"2021-05-19T23:28:47.016247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:29:13.665699Z","iopub.execute_input":"2021-05-19T23:29:13.666000Z","iopub.status.idle":"2021-05-19T23:29:13.682481Z","shell.execute_reply.started":"2021-05-19T23:29:13.665955Z","shell.execute_reply":"2021-05-19T23:29:13.681739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Tüm değerlerin eşit olarak katkıda bulunmasını sağlamak için değerlerin aynı birime getirilmesi gerektiğinden ölçeklendirdik.\nscaler = StandardScaler()\nscaler.fit(X_train)\nX_train_scaled = scaler.transform(X_train)\n","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:29:17.372118Z","iopub.execute_input":"2021-05-19T23:29:17.372427Z","iopub.status.idle":"2021-05-19T23:29:17.379632Z","shell.execute_reply.started":"2021-05-19T23:29:17.372389Z","shell.execute_reply":"2021-05-19T23:29:17.378656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_scaled","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:29:24.264833Z","iopub.execute_input":"2021-05-19T23:29:24.265099Z","iopub.status.idle":"2021-05-19T23:29:24.270783Z","shell.execute_reply.started":"2021-05-19T23:29:24.265065Z","shell.execute_reply":"2021-05-19T23:29:24.270005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Catboost \nc = CatBoostRegressor(loss_function='MAE')\nc.fit(X_train_scaled, y_train.values.flatten(), silent=True)\ny_pred = c.predict(X_train_scaled)","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:29:31.061129Z","iopub.execute_input":"2021-05-19T23:29:31.061757Z","iopub.status.idle":"2021-05-19T23:29:38.922525Z","shell.execute_reply.started":"2021-05-19T23:29:31.061706Z","shell.execute_reply":"2021-05-19T23:29:38.921615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Tahmin edilen ve Gerçek değerlerin görselleştirilmesi\nplt.figure(figsize=(8, 8))\nplt.scatter(y_train.values.flatten(), y_pred, c='crimson')\nplt.xlim(0, 20)\nplt.ylim(0, 20)\nplt.xlabel('Gerçek', fontsize=10)\nplt.ylabel('Tahmin Edilen', fontsize=10)\nplt.plot([(0, 0), (15, 15)], [(0, 0), (15, 15)])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:29:49.268026Z","iopub.execute_input":"2021-05-19T23:29:49.268368Z","iopub.status.idle":"2021-05-19T23:29:49.595999Z","shell.execute_reply.started":"2021-05-19T23:29:49.268313Z","shell.execute_reply":"2021-05-19T23:29:49.594580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Ortalama Mutlak Hata (iki sürekli değişken arasındaki farkın ölçüsüdür)\n#Düşük değerlere sahip tahminleyiciler daha iyi performans gösterir.\nscore = mean_absolute_error(y_train.values.flatten(), y_pred)\nprint(f'Score: {score:0.3f}')","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:29:55.283880Z","iopub.execute_input":"2021-05-19T23:29:55.284183Z","iopub.status.idle":"2021-05-19T23:29:55.292240Z","shell.execute_reply.started":"2021-05-19T23:29:55.284136Z","shell.execute_reply":"2021-05-19T23:29:55.291566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission(gönderim) dosyasını input dizininden okuma\nsubmission = pd.read_csv('../input/sample_submission.csv', index_col='seg_id')","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:29:58.708880Z","iopub.execute_input":"2021-05-19T23:29:58.709384Z","iopub.status.idle":"2021-05-19T23:29:58.732637Z","shell.execute_reply.started":"2021-05-19T23:29:58.709328Z","shell.execute_reply":"2021-05-19T23:29:58.731916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test verisini X_train ile aynı sütunları kullanarak oluşturduk.\nX_test = pd.DataFrame(columns=X_train.columns, dtype=np.float64, index=submission.index)","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:30:06.012044Z","iopub.execute_input":"2021-05-19T23:30:06.012584Z","iopub.status.idle":"2021-05-19T23:30:06.020132Z","shell.execute_reply.started":"2021-05-19T23:30:06.012533Z","shell.execute_reply":"2021-05-19T23:30:06.019343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test verileri için özellik oluşturduk.\nfor seg_id in X_test.index:\n    seg = pd.read_csv('../input/test/' + seg_id + '.csv')\n    \n    x = seg['acoustic_data'].values\n    \n    X_test.loc[seg_id, 'ort'] = x.mean()\n    X_test.loc[seg_id, 'std'] = x.std()\n    X_test.loc[seg_id, 'max'] = x.max()\n    X_test.loc[seg_id, 'min'] = x.min()","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:30:19.779431Z","iopub.execute_input":"2021-05-19T23:30:19.779785Z","iopub.status.idle":"2021-05-19T23:31:24.436623Z","shell.execute_reply.started":"2021-05-19T23:30:19.779737Z","shell.execute_reply":"2021-05-19T23:31:24.435605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submission(gönderim) dosyasını tahminlerle doldurduk.\nX_test_scaled = scaler.transform(X_test)\nsubmission['time_to_failure'] = c.predict(X_test_scaled)\n# csv dosyasına yazdık\nsubmission.to_csv('submission.csv')\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:32:21.014957Z","iopub.execute_input":"2021-05-19T23:32:21.015317Z","iopub.status.idle":"2021-05-19T23:32:21.041565Z","shell.execute_reply.started":"2021-05-19T23:32:21.015252Z","shell.execute_reply":"2021-05-19T23:32:21.040595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Catboost ile tahmin ve gerçek grafiği oluşturduk.\nplt.figure(figsize=(18, 8))\nplt.plot(y_train, color='g', label='y_train')\nplt.legend(['y_train'], loc=(0.88, 0.95))\nplt.plot(y_pred, color='b', label='cat')\nplt.title('Catboost');\nplt.suptitle('Tahminler vs Gerçek');\n","metadata":{"execution":{"iopub.status.busy":"2021-05-19T23:33:14.685917Z","iopub.execute_input":"2021-05-19T23:33:14.686894Z","iopub.status.idle":"2021-05-19T23:33:15.198997Z","shell.execute_reply.started":"2021-05-19T23:33:14.686833Z","shell.execute_reply":"2021-05-19T23:33:15.197777Z"},"trusted":true},"execution_count":null,"outputs":[]}]}