{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:35:59.114394Z","iopub.execute_input":"2022-07-22T14:35:59.114989Z","iopub.status.idle":"2022-07-22T14:35:59.137092Z","shell.execute_reply.started":"2022-07-22T14:35:59.114918Z","shell.execute_reply":"2022-07-22T14:35:59.136431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"from random import randint\nimport pandas as pd\nimport os\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n","metadata":{"_uuid":"e59b8a497f38a3af0c01bdac42381f8bc95d23b0","execution":{"iopub.status.busy":"2022-07-22T14:39:46.890836Z","iopub.execute_input":"2022-07-22T14:39:46.891503Z","iopub.status.idle":"2022-07-22T14:39:47.892329Z","shell.execute_reply.started":"2022-07-22T14:39:46.891455Z","shell.execute_reply":"2022-07-22T14:39:47.891138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cat '../input/readme.txt'","metadata":{"_uuid":"8f40ed0f0ed4abfcc7e3bfd7afde307dc0eed230","execution":{"iopub.status.busy":"2022-07-22T14:39:54.674529Z","iopub.execute_input":"2022-07-22T14:39:54.674866Z","iopub.status.idle":"2022-07-22T14:39:55.473358Z","shell.execute_reply.started":"2022-07-22T14:39:54.674805Z","shell.execute_reply":"2022-07-22T14:39:55.472345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load DATA","metadata":{"_uuid":"0ed62791761388c1ad88e3b87f7c5fed2d4b1272"}},{"cell_type":"code","source":"# Create column names\noperational_settings = ['op_setting_{}'.format(i + 1) for i in range (3)]\nsensor_columns = ['sensor_{}'.format(i + 1) for i in range(27)]\nfeatures = operational_settings + sensor_columns\nmetadata = ['engine_no', 'time_in_cycles']\nlist_columns = metadata + features\nlist_columns","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:39:59.436436Z","iopub.execute_input":"2022-07-22T14:39:59.436770Z","iopub.status.idle":"2022-07-22T14:39:59.446138Z","shell.execute_reply.started":"2022-07-22T14:39:59.436706Z","shell.execute_reply":"2022-07-22T14:39:59.445326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_directory = os.path.join(os.getcwd(), '../input')\nlist_file_train = [x for x in sorted(os.listdir(data_directory)) if 'train' in x]\nlist_file_train","metadata":{"execution":{"iopub.status.busy":"2022-07-22T14:40:07.216287Z","iopub.execute_input":"2022-07-22T14:40:07.216546Z","iopub.status.idle":"2022-07-22T14:40:07.225983Z","shell.execute_reply.started":"2022-07-22T14:40:07.216510Z","shell.execute_reply":"2022-07-22T14:40:07.225134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dictionnary = {}\n\nfor file_train in list_file_train:\n    data_set_name = file_train.replace('train_', '').replace('.txt', '')\n    #print(data_set_name)\n    file_test = 'test_' + data_set_name + '.txt'\n    rul_test = 'RUL_' + data_set_name + '.txt'\n    \n    data_dictionnary[data_set_name] = {\n        'df_train': pd.read_csv(os.path.join(data_directory, file_train), sep=' ', header=-1, names=list_columns),\n        'df_test': pd.read_csv(os.path.join(data_directory, file_test), sep=' ', header=-1, names=list_columns),\n        'RUL_test' :pd.read_csv(os.path.join(data_directory, rul_test), header=-1, names=['RUL']),\n    }\n#data_dictionnary\n#for data_set in data_dictionnary:\n#    print(list(data_dictionnary[data_set]['df_train'].groupby('engine_no')))","metadata":{"_uuid":"80e59990b97dc2a95a4a50f971ff2732ae2b01ef","execution":{"iopub.status.busy":"2022-07-22T14:40:13.160936Z","iopub.execute_input":"2022-07-22T14:40:13.161580Z","iopub.status.idle":"2022-07-22T14:40:14.684728Z","shell.execute_reply.started":"2022-07-22T14:40:13.161530Z","shell.execute_reply":"2022-07-22T14:40:14.683828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add RUL","metadata":{"_uuid":"933aaad717a805839ba2129c93b929c101462b6b"}},{"cell_type":"code","source":"def add_rul(g):\n    g['RUL'] = [max(g['time_in_cycles'])] * len(g)\n    #print([max(g['time_in_cycles'])])\n    g['RUL'] = g['RUL'] - g['time_in_cycles']\n    del g['engine_no']\n    return g.reset_index()\n\nfor data_set in data_dictionnary:\n    #print(data_set)\n    data_dictionnary[data_set]['df_train'] = data_dictionnary[data_set]['df_train']\\\n                        .groupby('engine_no').apply(add_rul).reset_index()\n    del data_dictionnary[data_set]['df_train']['level_1']\n#data_dictionnary","metadata":{"_uuid":"043529ae3a1bf0f979968d22e2d006c556eb3d1d","execution":{"iopub.status.busy":"2022-07-22T14:40:21.802718Z","iopub.execute_input":"2022-07-22T14:40:21.803264Z","iopub.status.idle":"2022-07-22T14:40:24.281803Z","shell.execute_reply.started":"2022-07-22T14:40:21.803133Z","shell.execute_reply":"2022-07-22T14:40:24.281021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Chosing a dataset ","metadata":{"_uuid":"472fca2163a7bc6481a92cfcec17255c0e2bffa2"}},{"cell_type":"code","source":"CHOSEN_DATASET = 'FD003'\n\ndf = data_dictionnary[CHOSEN_DATASET]['df_train'].copy()\n\ndf_eval = data_dictionnary[CHOSEN_DATASET]['df_test'].copy()","metadata":{"_uuid":"9d42850a1b0812292d972fa94339528edae145a8","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data analysis","metadata":{"_uuid":"94b91fd92323500d323a9883afcdd29ce9af7405"}},{"cell_type":"markdown","source":"### Plotting some description of the dataset","metadata":{"_uuid":"bfb14667da13f7b352cc3f1a218a7de674929e4d"}},{"cell_type":"code","source":"dataset_description = df.describe()\ndataset_description","metadata":{"_uuid":"6b9a760f173005f5861b101c84fd4a52d50099f7","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Correlation matrix","metadata":{"_uuid":"5ffd8eb3a653844c9b6e0913c950cc7981130c69"}},{"cell_type":"code","source":"df_plot = df.copy()[features]\ndf_corr = df_plot.corr(method='pearson')\nfig, ax = plt.subplots(figsize=(15,15))\naxes = sns.heatmap(df_corr, linewidths=.2, )","metadata":{"_uuid":"fc0f7277828b7d3af78b07c744c629683ef9b953","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Find columns that can be droped ","metadata":{"_uuid":"a2a203a82cf59c217fd5205caf8e1613c5e9bccf"}},{"cell_type":"code","source":"nan_column = df.columns[df.isna().any()].tolist()\nconst_columns = [c for c in df.columns if len(df[c].drop_duplicates()) <= 2]\nprint('Columns with all nan: \\n' + str(nan_column) + '\\n')\nprint('Columns with all const values: \\n' + str(const_columns) + '\\n')","metadata":{"_uuid":"d4c249914ca0968596c8d8ca1d1b634a5ebf09ca","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plot a temporal vizualisation of the features","metadata":{"_uuid":"30edc28f5b5e6a67d4561cfda7e795b93a4734de"}},{"cell_type":"code","source":"df_plot = df.copy()\ndf_plot = df_plot.sort_values(metadata)\n#print(df_plot)\ngraph = sns.PairGrid(data=df_plot, x_vars=\"RUL\", y_vars=features, hue=\"engine_no\", height=4, aspect=6,)\ngraph = graph.map(plt.plot, alpha=0.5)\ngraph = graph.set(xlim=(df_plot['RUL'].max(),df_plot['RUL'].min()))\ngraph = graph.add_legend()","metadata":{"_uuid":"4df1145629168fd679a1b00faa0ff2923ef278b7","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making a prediction","metadata":{"_uuid":"996ffd94f811d7f9c7350638c20c8c849ed787af"}},{"cell_type":"markdown","source":"### Splitting test / train data","metadata":{"_uuid":"043f3c269a34de934375fe2fbbb4a8455b47b4ca"}},{"cell_type":"code","source":"import random\nnumber_of_engine_no = len(df['engine_no'].drop_duplicates())\n#print(number_of_engine_no,len(df['engine_no']))\nengine_no_val = set(range(1,261)) - set(random.sample(range(1, 261), 208))\nprint(engine_no_val)\nengine_no_train = [x for x in range(number_of_engine_no) if x not in engine_no_val]\n#print(engine_no_train)","metadata":{"_uuid":"3743629d3b5c36af98f74391aa6a86dbaeda5346","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Selecting only relevant features","metadata":{"_uuid":"45c0bb8287e608d843cd17c3c3b89ca4265c2986"}},{"cell_type":"code","source":"selected_features = [x for x in features if x not in nan_column + const_columns]\nselected_features","metadata":{"_uuid":"ce96cf4dac77f05f79c25088c4ab64cdde554037","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Actually making the split","metadata":{"_uuid":"0a0b7417b3ad7ecba6488c1493b02ccedd0ac78c"}},{"cell_type":"code","source":"data_train = df[df['engine_no'].isin(engine_no_train)]\n#print(data_train)\ndata_val = df[df['engine_no'].isin(engine_no_val)]\n\nX_train, y_train = data_train[selected_features], data_train['RUL'] \n#print(X_train,y_train)\nX_val, y_val = data_val[selected_features], data_val['RUL']\n#print(X_val,y_val)\nX_eval = df_eval[selected_features]\n\n\nX_all, y_all = df[selected_features], df['RUL']\nprint(type(X_train),y_train)","metadata":{"_uuid":"58e43303a4c28e0cd07429a681f9f0acf7b5e6d1","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training a random forest","metadata":{"_uuid":"97208bc133eedaf1a920b486515c14de8976b2a5"}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nrf_reg = RandomForestRegressor()\nrf_reg.fit(X_train, y_train)\n\nprint(\"Score on train data : \" + str(rf_reg.score(X_train, y_train)))\nprint(\"Score on test data : \" + str(rf_reg.score(X_val, y_val)))","metadata":{"_uuid":"dcca1e19840410e08c597f3ee42712088bf6a7dc","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plotting the result ","metadata":{"_uuid":"6b8663d52b6ed5485e032e51a7cf7705077aa6a3"}},{"cell_type":"code","source":"df_pred = data_train.copy()\ndf_pred['pred'] = rf_reg.predict(X_train)\n#print(df_pred['pred'])\ndf_pred['error'] = df_pred['pred'] - df_pred['RUL']\ndf_plot = df_pred.copy()\ndf_plot = df_plot.sort_values(['engine_no', 'time_in_cycles'])\ng = sns.PairGrid(data=df_plot, x_vars=\"RUL\", y_vars=['RUL', 'pred', 'error'], hue=\"engine_no\", height=6, aspect=3,)\ng = g.map(plt.plot, alpha=0.5)\ng = g.set(xlim=(df_plot['RUL'].min(),df_plot['RUL'].max()))\ng = g.add_legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_pred = data_val.copy()\ndf_pred['pred'] = rf_reg.predict(X_val)\n#print(df_pred['pred'])\ndf_pred['error'] = df_pred['pred'] - df_pred['RUL']\n\ndf_plot = df_pred.copy()\ndf_plot = df_plot.sort_values(['engine_no', 'time_in_cycles'])\ng = sns.PairGrid(data=df_plot, x_vars=\"RUL\", y_vars=['RUL', 'pred', 'error'], hue=\"engine_no\", height=6, aspect=3,)\ng = g.map(plt.plot, alpha=0.5)\ng = g.set(xlim=(df_plot['RUL'].min(), df_plot['RUL'].max()))\ng = g.add_legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction on test data and Output","metadata":{"_uuid":"25f736f301ddacd883b5c848993dbb30d1ab7c58"}},{"cell_type":"code","source":"df_eval['pred'] = rf_reg.predict(X_eval)\ndf_eval['result'] = df_eval['pred']\ndf_eval['engine_id'] = list(range(len(df_eval)))\ndf_eval[['engine_id','result']].to_csv('submission.csv', index=False)\nprint(df_eval)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}