{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport os\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nplt.style.use('seaborn-white')\n%matplotlib inline\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nimport lightgbm as lgbm\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-05T19:43:27.825877Z","iopub.execute_input":"2022-12-05T19:43:27.826377Z","iopub.status.idle":"2022-12-05T19:43:29.802157Z","shell.execute_reply.started":"2022-12-05T19:43:27.826341Z","shell.execute_reply":"2022-12-05T19:43:29.800570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_list=[]\nfile_list_train=[]\nfile_list_test=[]\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        file_list.append(os.path.join(dirname, filename))\n\nPATH = '/kaggle/input/predict-volcanic-eruptions-ingv-oe/'\n\nfor dirname, _, filenames in os.walk('/kaggle/input/predict-volcanic-eruptions-ingv-oe/train'):\n    for filename in filenames:\n        file_list_train.append(os.path.join(dirname, filename))\n\nfor dirname, _, filenames in os.walk('/kaggle/input/predict-volcanic-eruptions-ingv-oe/test'):\n    for filename in filenames:\n        file_list_test.append(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:43:29.804464Z","iopub.execute_input":"2022-12-05T19:43:29.804844Z","iopub.status.idle":"2022-12-05T19:43:42.140624Z","shell.execute_reply.started":"2022-12-05T19:43:29.804812Z","shell.execute_reply":"2022-12-05T19:43:42.139259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2.Дослідження даних (Data Understanding) \n# Переглядаємо файл sample_submission\nprint(file_list[0])\nprint(pd.read_csv(file_list[0]))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:43:42.142060Z","iopub.execute_input":"2022-12-05T19:43:42.142444Z","iopub.status.idle":"2022-12-05T19:43:42.175283Z","shell.execute_reply.started":"2022-12-05T19:43:42.142413Z","shell.execute_reply":"2022-12-05T19:43:42.174252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Переглядаємо файл 800654756.csv в папці train (показники показів датчиків)\nprint(file_list_train[0])\nprint(pd.read_csv(file_list_train[0]))\n# Десять хвилин журналів від десяти різних датчиків, розташованих навколо вулкана","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:43:42.178070Z","iopub.execute_input":"2022-12-05T19:43:42.178449Z","iopub.status.idle":"2022-12-05T19:43:42.313196Z","shell.execute_reply.started":"2022-12-05T19:43:42.178408Z","shell.execute_reply":"2022-12-05T19:43:42.311768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Переглядаємо файл train.csv\nprint(file_list[1])\nprint(pd.read_csv(file_list[1]))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:43:42.315136Z","iopub.execute_input":"2022-12-05T19:43:42.315654Z","iopub.status.idle":"2022-12-05T19:43:42.333775Z","shell.execute_reply.started":"2022-12-05T19:43:42.315609Z","shell.execute_reply":"2022-12-05T19:43:42.332642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Aналіз статистики по відсутнім показам датчиків\n\n# file_list_test\nprint(len(file_list_test)) # довжина массиву file_list_test","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:43:42.334913Z","iopub.execute_input":"2022-12-05T19:43:42.335771Z","iopub.status.idle":"2022-12-05T19:43:42.341498Z","shell.execute_reply.started":"2022-12-05T19:43:42.335736Z","shell.execute_reply":"2022-12-05T19:43:42.340311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keys = list(pd.read_csv(file_list_test[0]).keys()) # Ключі(сенсори) массиву file_list_test\n# print(keys)\nnanCount = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0]\nfor index in range(len(file_list_test)):\n#     if(index % 200 == 0): \n#         print(index)\n    df = pd.read_csv(file_list_test[index]) #записуємо кожний cегмент з масиву в Dataframe\n    for key in df.keys():\n        if df[key].isna().sum() == 60001: # кількість відсутніх значень (nan)\n            nanCount[keys.index(key)] += 1 # підрахунок nan значень для кожного сенсору в сегменті даних\n# print(nanCount)\ndata={'sensors': keys, 'count': nanCount}\nmyFrame = pd.DataFrame(data)\nprint(myFrame)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:43:42.343366Z","iopub.execute_input":"2022-12-05T19:43:42.343755Z","iopub.status.idle":"2022-12-05T19:53:24.881503Z","shell.execute_reply.started":"2022-12-05T19:43:42.343721Z","shell.execute_reply":"2022-12-05T19:53:24.879307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Гістограма статистики по відсутнім показам датчиків\nmyFrame.plot(figsize =(11, 11), x=\"sensors\", y=\"count\", kind=\"bar\",  rot=5, fontsize=12 )","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:47:51.816188Z","iopub.execute_input":"2022-12-05T21:47:51.816632Z","iopub.status.idle":"2022-12-05T21:47:52.126217Z","shell.execute_reply.started":"2022-12-05T21:47:51.816594Z","shell.execute_reply":"2022-12-05T21:47:52.124991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(file_list_test[0])\nprint(pd.read_csv(file_list_test[0]))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:25.347624Z","iopub.execute_input":"2022-12-05T19:53:25.348756Z","iopub.status.idle":"2022-12-05T19:53:25.478147Z","shell.execute_reply.started":"2022-12-05T19:53:25.348713Z","shell.execute_reply":"2022-12-05T19:53:25.476610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"file_list_test: \", len(file_list_test))\nprint(\"file_list_train: \", len(file_list_train))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:25.483144Z","iopub.execute_input":"2022-12-05T19:53:25.483588Z","iopub.status.idle":"2022-12-05T19:53:25.490382Z","shell.execute_reply.started":"2022-12-05T19:53:25.483547Z","shell.execute_reply":"2022-12-05T19:53:25.489042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files_train = [file.split('/')[-1].split('.')[-2] for file in file_list_train]\nfiles_test = [file.split('/')[-1].split('.')[-2] for file in file_list_test]\nprint(files_train[0:10]) # Перші 10 файлів тренування \nprint(files_test[0:10]) # Перші 10 файлів тестування ","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:25.492472Z","iopub.execute_input":"2022-12-05T19:53:25.493572Z","iopub.status.idle":"2022-12-05T19:53:25.513223Z","shell.execute_reply.started":"2022-12-05T19:53:25.493520Z","shell.execute_reply":"2022-12-05T19:53:25.511665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Кількість перетинів індексів файлів\n\ntest_set = set(files_test)\ntrain_set = set(files_train)\ninter = test_set.intersection(train_set) # Метод повертає набір, що містить схожість між двома або більше наборами\nprint(inter)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:25.515567Z","iopub.execute_input":"2022-12-05T19:53:25.516530Z","iopub.status.idle":"2022-12-05T19:53:25.534693Z","shell.execute_reply.started":"2022-12-05T19:53:25.516476Z","shell.execute_reply":"2022-12-05T19:53:25.533388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Множина перетину порожня. Тобто, індекси файлів не грають в даному випадку суттєвої ролі, а головне - покази датчиків та відповідні значення часу","metadata":{}},{"cell_type":"code","source":"# Читаємо дані тренування у DataFrame\ntrain = pd.read_csv(PATH+'train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:25.536199Z","iopub.execute_input":"2022-12-05T19:53:25.536558Z","iopub.status.idle":"2022-12-05T19:53:25.555602Z","shell.execute_reply.started":"2022-12-05T19:53:25.536526Z","shell.execute_reply":"2022-12-05T19:53:25.554292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Зображуємо на діаграмі розподілення часу. \nsns.distplot(train['time_to_eruption'], hist=True, kde=True, bins=100, color='green', hist_kws={'edgecolor':'black'})","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:25.557039Z","iopub.execute_input":"2022-12-05T19:53:25.557405Z","iopub.status.idle":"2022-12-05T19:53:26.027404Z","shell.execute_reply.started":"2022-12-05T19:53:25.557371Z","shell.execute_reply":"2022-12-05T19:53:26.026438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Описова статистика ознаки часу \n\ntrain['time_to_eruption'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:26.028712Z","iopub.execute_input":"2022-12-05T19:53:26.029778Z","iopub.status.idle":"2022-12-05T19:53:26.044336Z","shell.execute_reply.started":"2022-12-05T19:53:26.029727Z","shell.execute_reply":"2022-12-05T19:53:26.043350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Можна припустити, що час вимірюється у мілісекундах. Тоді, максимальне значення тривалості - близько 13 годин. Відсутність від’ємних значень тривалості свідчить про відсутність помилок у даних. ","metadata":{}},{"cell_type":"code","source":"# Візуалізуємо покази датчиків файлу train/800654756.csv.\ndf_segment_id = pd.read_csv(PATH+'/train/800654756.csv')\n\ndf_segment_id.plot(figsize=(20, 20), subplots=True, layout=(10, 1), rot = 0,\n                  lw=1, title='segment_id #800654756')","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:26.045890Z","iopub.execute_input":"2022-12-05T19:53:26.046321Z","iopub.status.idle":"2022-12-05T19:53:29.475894Z","shell.execute_reply.started":"2022-12-05T19:53:26.046286Z","shell.execute_reply":"2022-12-05T19:53:29.474429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cейсмосенсори вимірюють фізичну величину, коливання.\nПокази сенсорів є часовими рядами, тобто процесом який розвивається у часі, і наступне значення поєднане з попередніми. Також, покази деяких сенсорів можуть бути відсутніми, що буде потребувати додаткової обробки на етапі підготовки даних. ","metadata":{}},{"cell_type":"markdown","source":"Можна припустити, що амплітуда та частота коливань показів датчиків безпосередньо впливає на час до виверження. Для цього зробемо перевірку, візуалізуємо покази, які відповідають мінімальному та максимальному часу. ","metadata":{}},{"cell_type":"code","source":"display(train.sort_values('time_to_eruption', axis=0, ascending=True).iloc[[0, -1], :])\nsegment_id_min = 601524801\nsegment_id_max = 1923243961\n\ndf_segment_id_min = pd.read_csv(PATH+'/train/'+str(segment_id_min)+'.csv')\ndf_segment_id_max = pd.read_csv(PATH+'/train/'+str(segment_id_max)+'.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:29.477539Z","iopub.execute_input":"2022-12-05T19:53:29.478275Z","iopub.status.idle":"2022-12-05T19:53:29.743593Z","shell.execute_reply.started":"2022-12-05T19:53:29.478235Z","shell.execute_reply":"2022-12-05T19:53:29.742511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Отримали індекс файлу з мінімальним часом – segment_id_min = 601524801,\nмаксимальним часом – segment_id_max = 1923243961 ","metadata":{}},{"cell_type":"code","source":"# Візуалізуємо покази датчиків файлу train/601524801.csv.\n\ndf_segment_id_min.plot(figsize=(20,20), subplots=True, layout=(10,1), rot=0, lw=1, title='segment_id #601524801 (min)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:29.745151Z","iopub.execute_input":"2022-12-05T19:53:29.745541Z","iopub.status.idle":"2022-12-05T19:53:32.331529Z","shell.execute_reply.started":"2022-12-05T19:53:29.745506Z","shell.execute_reply":"2022-12-05T19:53:32.330022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Візуалізуємо покази датчиків файлу train/1923243961.csv.\n\ndf_segment_id_max.plot(figsize=(20,20), subplots=True, layout=(10,1), rot=0, lw=1, title='segment_id #1923243961 (max)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:32.333229Z","iopub.execute_input":"2022-12-05T19:53:32.333632Z","iopub.status.idle":"2022-12-05T19:53:35.155387Z","shell.execute_reply.started":"2022-12-05T19:53:32.333598Z","shell.execute_reply":"2022-12-05T19:53:35.154035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Як бачимо покази, що відповідають максимальному часу відрізняються за характеристиками амплітуди та частоти від показів, які відповідають мінімальному часу. Конкретна інтерпретація цих показів неможливо без знань типів датчиків та не входить до нашої задачі.\n* Головний висновок – для прогнозу можна використовувати не сам сигнал в необробленому вигляді – а його статистичні характеристики. \n","metadata":{}},{"cell_type":"code","source":"# 3. Підготовка даних для навчання (Data Preparation)\n# Функція обчислень медіани, дисперсії, стандартного відхилення, зсуву, \n# мінімального та максимального значень, квантилів тощо \n\ndef build_features(signal, ts, sensor_id):\n    X = pd.DataFrame()\n    f = np.fft.fft(signal) # Обчислення одновимірного дискретного перетворення Фур'є.\n    f_real = np.real(f) # Повертає дійсну частину складного аргументу.\n    X.loc[ts, f'{sensor_id}_sum'] = signal.sum()\n    X.loc[ts, f'{sensor_id}_mean'] = signal.mean()\n    X.loc[ts, f'{sensor_id}_std'] = signal.std()\n    X.loc[ts, f'{sensor_id}_var'] = signal.var()\n    X.loc[ts, f'{sensor_id}_max'] = signal.max()\n    X.loc[ts, f'{sensor_id}_min'] = signal.min()\n    X.loc[ts, f'{sensor_id}_skew'] = signal.skew()\n    X.loc[ts, f'{sensor_id}_mad'] = signal.mad()\n    X.loc[ts, f'{sensor_id}_kurtosis'] = signal.kurtosis()\n    X.loc[ts, f'{sensor_id}_quantile99'] = np.quantile(signal, 0.99)\n    X.loc[ts, f'{sensor_id}_quantile95'] = np.quantile(signal, 0.95)\n    X.loc[ts, f'{sensor_id}_quantile85'] = np.quantile(signal, 0.85)\n    X.loc[ts, f'{sensor_id}_quantile75'] = np.quantile(signal, 0.75)\n    X.loc[ts, f'{sensor_id}_quantile55'] = np.quantile(signal, 0.55)\n    X.loc[ts, f'{sensor_id}_quantile45'] = np.quantile(signal, 0.45)\n    X.loc[ts, f'{sensor_id}_quantile25'] = np.quantile(signal, 0.25)\n    X.loc[ts, f'{sensor_id}_quantile15'] = np.quantile(signal, 0.15)\n    X.loc[ts, f'{sensor_id}_quantile05'] = np.quantile(signal, 0.05)\n    X.loc[ts, f'{sensor_id}_quantile01'] = np.quantile(signal, 0.01)\n    X.loc[ts, f'{sensor_id}_fft_real_mean'] = f_real.mean()\n    X.loc[ts, f'{sensor_id}_fft_real_std'] = f_real.std()\n    X.loc[ts, f'{sensor_id}_fft_real_max'] = f_real.max()\n    X.loc[ts, f'{sensor_id}_fft_real_min'] = f_real.min()\n    \n    return X","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:35.157802Z","iopub.execute_input":"2022-12-05T19:53:35.158247Z","iopub.status.idle":"2022-12-05T19:53:35.172866Z","shell.execute_reply.started":"2022-12-05T19:53:35.158207Z","shell.execute_reply":"2022-12-05T19:53:35.171850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set = list()\nseg=0\nprint(enumerate(train.segment_id))\n\nfor seg, segment_id in enumerate(train.segment_id): # цикл сегментів\n    signals = pd.read_csv(PATH+'/train/'+str(segment_id)+'.csv')\n    train_row=[]\n    \n    if seg % 200 == 0:\n        print('Обробка segment_id={}'.format(seg))\n    \n    for sensor in range(0, 10): # цикл для розрахунку обчислень медіани, дисперсії, стандартного відхилення, зсуву, мінімального та максимального значень, квантилів тощо \n        sensor_id = f'sensor_{sensor+1}'\n        train_row.append(build_features(signals[sensor_id].fillna(0), segment_id, sensor_id))\n    \n    train_row = pd.concat(train_row, axis=1) #Об'єднання об'єкту pandas вздовж стовбців.\n    train_set.append(train_row) # додавання результату розрахунків до списку\n    seg+=1\n    \ntrain_set = pd.concat(train_set)\n# print(train_set)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T19:53:35.174717Z","iopub.execute_input":"2022-12-05T19:53:35.175454Z","iopub.status.idle":"2022-12-05T20:38:57.602554Z","shell.execute_reply.started":"2022-12-05T19:53:35.175412Z","shell.execute_reply":"2022-12-05T20:38:57.599976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set = train_set.reset_index() # Скидаємо індекс\ntrain_set = train_set.rename(columns={'index': 'segment_id'})\n# print(train_set)\ntrain_set = pd.merge(train_set, train, on='segment_id') # Слиття Dataframe за столбцем segment_id\n# print(train_set)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T20:38:57.605661Z","iopub.execute_input":"2022-12-05T20:38:57.606399Z","iopub.status.idle":"2022-12-05T20:38:57.658420Z","shell.execute_reply.started":"2022-12-05T20:38:57.606346Z","shell.execute_reply":"2022-12-05T20:38:57.657057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_set.head(3))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T20:38:57.660161Z","iopub.execute_input":"2022-12-05T20:38:57.660586Z","iopub.status.idle":"2022-12-05T20:38:57.680623Z","shell.execute_reply.started":"2022-12-05T20:38:57.660549Z","shell.execute_reply":"2022-12-05T20:38:57.678973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Дані для тесту \ntest_files = []\nfor dirname, _, filenames in os.walk(PATH+'/test/'):\n    for filename in filenames:\n        test_files.append(filename[:-4])\n\ntest = pd.DataFrame(test_files, columns=['segment_id'])","metadata":{"execution":{"iopub.status.busy":"2022-12-05T20:38:57.682627Z","iopub.execute_input":"2022-12-05T20:38:57.683168Z","iopub.status.idle":"2022-12-05T20:39:00.425566Z","shell.execute_reply.started":"2022-12-05T20:38:57.683118Z","shell.execute_reply":"2022-12-05T20:39:00.424286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set = list()\nseg=0\n\nfor seg, segment_id in enumerate(test.segment_id):  # цикл сегментів\n    signals = pd.read_csv(PATH+'/test/'+str(segment_id)+'.csv')\n    test_row=[]\n    \n    if seg%200 == 0:\n        print('Processing segment_id={}'.format(seg))\n        \n    for sensor in range(0, 10):  # цикл для розрахунку обчислень медіани, дисперсії, стандартного відхилення, зсуву, мінімального та максимального значень, квантилів тощо \n        sensor_id = f'sensor_{sensor+1}'\n        test_row.append(build_features(signals[sensor_id].fillna(0), segment_id, sensor_id))\n    \n    test_row = pd.concat(test_row, axis=1) # Об'єднання об'єкту pandas вздовж стовбців.\n    test_set.append(test_row) # додавання результату розрахунків до списку\n    seg+=1\n    \ntest_set = pd.concat(test_set)\n# print(test_set)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T20:39:00.427321Z","iopub.execute_input":"2022-12-05T20:39:00.427895Z","iopub.status.idle":"2022-12-05T21:23:59.762266Z","shell.execute_reply.started":"2022-12-05T20:39:00.427859Z","shell.execute_reply":"2022-12-05T21:23:59.760863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set=test_set.reset_index() # Скидаємо індекс\ntest_set=test_set.rename(columns={'index': 'segment_id'})\n# print(test_set)\ntest_set=pd.merge(test_set, test, on=\"segment_id\") # Слиття Dataframe за столбцем segment_id\n# print(test_set)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:23:59.764171Z","iopub.execute_input":"2022-12-05T21:23:59.764564Z","iopub.status.idle":"2022-12-05T21:23:59.823708Z","shell.execute_reply.started":"2022-12-05T21:23:59.764529Z","shell.execute_reply":"2022-12-05T21:23:59.822164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_set.drop(['segment_id', 'time_to_eruption'], axis=1) # Видалення столбців з тренувального набору\ny = train_set['time_to_eruption']\n\n#Розділення даних тренування на набори для безпосередньо тренування та валідації \nX_train, X_valid, y_train, y_valid = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:23:59.825559Z","iopub.execute_input":"2022-12-05T21:23:59.826277Z","iopub.status.idle":"2022-12-05T21:23:59.859673Z","shell.execute_reply.started":"2022-12-05T21:23:59.826227Z","shell.execute_reply":"2022-12-05T21:23:59.858615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Вхідні дані тренування (статистичні характеристики показів датчиків) \n\nprint(X_train.head(3))\nprint('np.shape(X_train) = ', np.shape(X_train))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:23:59.866878Z","iopub.execute_input":"2022-12-05T21:23:59.868380Z","iopub.status.idle":"2022-12-05T21:23:59.887375Z","shell.execute_reply.started":"2022-12-05T21:23:59.868308Z","shell.execute_reply":"2022-12-05T21:23:59.886047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Вихідні дані тренування (дані часу) \n\nprint(y_train.head(3))\nprint('np.shape(y_train) = ', np.shape(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:23:59.888646Z","iopub.execute_input":"2022-12-05T21:23:59.889047Z","iopub.status.idle":"2022-12-05T21:23:59.905711Z","shell.execute_reply.started":"2022-12-05T21:23:59.889011Z","shell.execute_reply":"2022-12-05T21:23:59.904314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 4. Моделювання (Modeling) \n# модель регресії методом випадкових лісів\nfrom sklearn.ensemble import RandomForestRegressor\nmodel = RandomForestRegressor(max_depth=20, random_state=0) # модель випадкового лісу с максимальною глибиною 20\nmodel.fit(X_train, y_train) # навчаємо модель","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:23:59.907383Z","iopub.execute_input":"2022-12-05T21:23:59.907985Z","iopub.status.idle":"2022-12-05T21:24:52.227443Z","shell.execute_reply.started":"2022-12-05T21:23:59.907932Z","shell.execute_reply":"2022-12-05T21:24:52.226054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Отриману модель можна використовувати для оцінки точності на наборі валідації.","metadata":{}},{"cell_type":"code","source":"# 5. Оцінка точності моделі (Evaluation)\n# Передбачення побудованої моделі\nimport math\n\ny_pred = model.predict(X_valid)\nfrom sklearn.metrics import mean_squared_error # Середньоквадратична помилка регресії.\nmse = mean_squared_error(y_valid, y_pred)\n\nfrom sklearn.metrics import r2_score\nconf_mat = r2_score(y_valid, y_pred)\n\n\nprint(\"- M. sq. error :\", mse)\nprint(\"- Std. dev.    :\", math.sqrt(mse))\nprint(\"- R^2 score    :\", conf_mat, \"\\n\")\n\nfig=plt.figure()\nmulreg = fig.add_subplot(1,1,1)\nmulreg.scatter(y_valid, y_pred, color='g')\nmulreg.set_title('Nonlinear Regression ')","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:27:36.235256Z","iopub.execute_input":"2022-12-05T21:27:36.235935Z","iopub.status.idle":"2022-12-05T21:27:36.544176Z","shell.execute_reply.started":"2022-12-05T21:27:36.235879Z","shell.execute_reply":"2022-12-05T21:27:36.542643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(test_set.drop(columns=['segment_id'], axis=1)) # Передбачення для тестового набору даних\n\nsubmission = pd.DataFrame()\nsubmission['segment_id'] = test_set['segment_id']\nsubmission['time_to_eruption'] = predictions\nsubmission.to_csv('submission.csv', header=True, index=False) # Створення вихідного файлу submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:27:59.681492Z","iopub.execute_input":"2022-12-05T21:27:59.682717Z","iopub.status.idle":"2022-12-05T21:27:59.885876Z","shell.execute_reply.started":"2022-12-05T21:27:59.682661Z","shell.execute_reply":"2022-12-05T21:27:59.884467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from math import isnan\nsensor_result = {}\nfor i in range(1,11):\n    sensor_result[f\"sensor_{i}\"] =  0\n    \nprint(sensor_result)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:28:02.216054Z","iopub.execute_input":"2022-12-05T21:28:02.217268Z","iopub.status.idle":"2022-12-05T21:28:02.225813Z","shell.execute_reply.started":"2022-12-05T21:28:02.217217Z","shell.execute_reply":"2022-12-05T21:28:02.224211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in file_list_test:\n    df_test_stat = pd.read_csv(i)\n    for j in range(1,11):\n        if isnan(df_test_stat[f\"sensor_{j}\"].mean()):\n            sensor_result[f\"sensor_{j}\"] += 1\n\nprint(sensor_result)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:28:04.980268Z","iopub.execute_input":"2022-12-05T21:28:04.980706Z","iopub.status.idle":"2022-12-05T21:36:52.997576Z","shell.execute_reply.started":"2022-12-05T21:28:04.980669Z","shell.execute_reply":"2022-12-05T21:36:52.995938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20,5))\nplt.bar(sensor_result.keys(), sensor_result.values(), width=.5, color='green')\nplt.title(\"Кількість відсутніх даних від кожного датчика\")","metadata":{"execution":{"iopub.status.busy":"2022-12-05T21:37:32.358686Z","iopub.execute_input":"2022-12-05T21:37:32.360102Z","iopub.status.idle":"2022-12-05T21:37:32.660533Z","shell.execute_reply.started":"2022-12-05T21:37:32.360044Z","shell.execute_reply":"2022-12-05T21:37:32.659239Z"},"trusted":true},"execution_count":null,"outputs":[]}]}