{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n<span style=\"font-family:Verdana;\">The August 2022 edition of the Tabular Playground Series in an opportunity to help the fictional company Keep It Dry improve its main product Super Soaker. The product is used in factories to absorb spills and leaks.The August 2022 edition of the Tabular Playground Series in an opportunity to help the fictional company Keep It Dry improve its main product Super Soaker. The product is used in factories to absorb spills and leaks.<br>\nFor each product_code you are given a number of product attributes (fixed for the code) as well as a number of measurement values for each individual product, representing various lab testing methods. Each product is used in a simulated real-world environment experiment, and and absorbs a certain amount of fluid (loading) to see whether or not it fails.<br>\nYour task is to use the data to predict individual product failures of new codes with their individual lab test results.For each product_code you are given a number of product attributes (fixed for the code) as well as a number of measurement values for each individual product, representing various lab testing methods. Each product is used in a simulated real-world environment experiment, and and absorbs a certain amount of fluid (loading) to see whether or not it fails.<br>\nTask is to use the data to predict individual product failures of new codes with their individual lab test results.</span>\n\nReference:\n1. https://www.kaggle.com/code/kartushovdanil/tps-aug-22-advanced-eda-catboost\n1. https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\nfrom matplotlib.pyplot import figure\nimport seaborn as sns\n%matplotlib inline\n\nfrom tqdm.notebook import tqdm;\nfrom gc import collect;\n\nfrom sklearn.preprocessing import OneHotEncoder\n\nfrom sklearn.preprocessing import StandardScaler,RobustScaler,PowerTransformer\n\nfrom sklearn.impute import SimpleImputer, KNNImputer\nimport imblearn\nfrom imblearn.under_sampling import NearMiss, ClusterCentroids","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-09T12:19:24.009178Z","iopub.execute_input":"2022-08-09T12:19:24.010138Z","iopub.status.idle":"2022-08-09T12:19:25.980487Z","shell.execute_reply.started":"2022-08-09T12:19:24.010041Z","shell.execute_reply":"2022-08-09T12:19:25.979296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE = '../input/tabular-playground-series-aug-2022'\ntrain_df = pd.read_csv(BASE + '/train.csv')\ntest_df  = pd.read_csv(BASE + '/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:25.983298Z","iopub.execute_input":"2022-08-09T12:19:25.984150Z","iopub.status.idle":"2022-08-09T12:19:26.296552Z","shell.execute_reply.started":"2022-08-09T12:19:25.984104Z","shell.execute_reply":"2022-08-09T12:19:26.295231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:26.298184Z","iopub.execute_input":"2022-08-09T12:19:26.298616Z","iopub.status.idle":"2022-08-09T12:19:26.342474Z","shell.execute_reply.started":"2022-08-09T12:19:26.298558Z","shell.execute_reply":"2022-08-09T12:19:26.341645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.attribute_1.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:26.344743Z","iopub.execute_input":"2022-08-09T12:19:26.345470Z","iopub.status.idle":"2022-08-09T12:19:26.363041Z","shell.execute_reply.started":"2022-08-09T12:19:26.345432Z","shell.execute_reply":"2022-08-09T12:19:26.361687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:26.364555Z","iopub.execute_input":"2022-08-09T12:19:26.365426Z","iopub.status.idle":"2022-08-09T12:19:26.394276Z","shell.execute_reply.started":"2022-08-09T12:19:26.365382Z","shell.execute_reply":"2022-08-09T12:19:26.393026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.iloc[:, 1:-1].describe().T.style.background_gradient()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:26.397096Z","iopub.execute_input":"2022-08-09T12:19:26.397535Z","iopub.status.idle":"2022-08-09T12:19:26.610231Z","shell.execute_reply.started":"2022-08-09T12:19:26.397489Z","shell.execute_reply":"2022-08-09T12:19:26.608702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:26.612055Z","iopub.execute_input":"2022-08-09T12:19:26.612638Z","iopub.status.idle":"2022-08-09T12:19:26.630770Z","shell.execute_reply.started":"2022-08-09T12:19:26.612570Z","shell.execute_reply":"2022-08-09T12:19:26.629574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.nunique(axis=0)  ","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:26.632382Z","iopub.execute_input":"2022-08-09T12:19:26.633112Z","iopub.status.idle":"2022-08-09T12:19:26.667281Z","shell.execute_reply.started":"2022-08-09T12:19:26.633069Z","shell.execute_reply":"2022-08-09T12:19:26.666379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"duplicate = train_df[train_df.duplicated()]\nduplicate.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:26.668958Z","iopub.execute_input":"2022-08-09T12:19:26.669394Z","iopub.status.idle":"2022-08-09T12:19:26.726656Z","shell.execute_reply.started":"2022-08-09T12:19:26.669350Z","shell.execute_reply":"2022-08-09T12:19:26.725343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['product_code','attribute_0', 'attribute_1','attribute_2', 'attribute_3','failure']\ndf = train_df.groupby(by=cols).count()['id'].reset_index()\ndf = df.rename(columns = {'id':'count'})\ndf.style.background_gradient(subset=['count'])","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:26.732461Z","iopub.execute_input":"2022-08-09T12:19:26.733294Z","iopub.status.idle":"2022-08-09T12:19:26.783847Z","shell.execute_reply.started":"2022-08-09T12:19:26.733246Z","shell.execute_reply":"2022-08-09T12:19:26.782651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['product_code','attribute_0', 'attribute_1','attribute_2', 'attribute_3']\ndf_ts = test_df.groupby(by=cols).count()['id'].reset_index()\ndf_ts = df_ts.rename(columns = {'id':'count'})\ndf_ts.style.background_gradient(subset=['count'])","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:26.785892Z","iopub.execute_input":"2022-08-09T12:19:26.786704Z","iopub.status.idle":"2022-08-09T12:19:26.829030Z","shell.execute_reply.started":"2022-08-09T12:19:26.786651Z","shell.execute_reply":"2022-08-09T12:19:26.827629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-size:18px;font-family:Verdana;\">💡Conclusion:</span>\n1. <span style=\"color:brown;font-family:Verdana;\">This is 28k data, so it is the small-medium level of datasets.</span>\n1. <span style=\"color:brown;font-family:Verdana;\">There are some missing records, we can use different imputing methods, which we saw in June 2022 TPS.</span>\n1. <span style=\"color:brown;font-family:Verdana;\">There are no duplicate records.</span>\n1. <span style=\"color:brown;font-family:Verdana;\">Most of the measurements column has mutli-cardinality, as we have lot of unique values.</span>","metadata":{}},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"target_class = pd.DataFrame({'count': train_df.failure.value_counts(),\n                             'percentage': train_df['failure'].value_counts() / train_df.shape[0] * 100})\nplt.figure(figsize=(8, 6))\nplt.pie(target_class.percentage)\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:26.830925Z","iopub.execute_input":"2022-08-09T12:19:26.831754Z","iopub.status.idle":"2022-08-09T12:19:27.000924Z","shell.execute_reply.started":"2022-08-09T12:19:26.831709Z","shell.execute_reply":"2022-08-09T12:19:26.999118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [x for x in train_df.columns if x !='id']","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:19:27.002767Z","iopub.execute_input":"2022-08-09T12:19:27.003363Z","iopub.status.idle":"2022-08-09T12:19:27.012061Z","shell.execute_reply.started":"2022-08-09T12:19:27.003306Z","shell.execute_reply":"2022-08-09T12:19:27.010069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ncols = 6\nfor i, f in enumerate(features):\n    if i % ncols == 0: \n        if i > 0: plt.show()\n        plt.figure(figsize=(25, 2))\n        if i == 0: plt.suptitle('Data Symmetry check', fontsize=20, y=1.02)\n    plt.subplot(1, ncols, i % ncols + 1)\n    plt.hist(train_df[f], bins=50)\n    plt.xlabel(f)\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-09T12:19:27.017319Z","iopub.execute_input":"2022-08-09T12:19:27.021206Z","iopub.status.idle":"2022-08-09T12:19:32.137834Z","shell.execute_reply.started":"2022-08-09T12:19:27.021130Z","shell.execute_reply":"2022-08-09T12:19:32.136393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generating correlation analysis without transforms:-\nfor method in tqdm(['pearson', 'spearman', 'kendall']):\n    fig, ax= plt.subplots(1,1, figsize= (11,7));\n    _corr = train_df.corr(method=method);\n    sns.heatmap(_corr, annot= True, fmt= '.0%', linewidth= 1, center= True, cmap= 'Spectral_r',\n                cbar= False, linecolor= 'white', mask = np.triu(np.ones_like(_corr)),ax= ax);\n    ax.set_title(f\"\\n{method.capitalize()} correlation plot before transforms\\n\", \n                 color= 'tab:blue', fontsize= 8);\n    plt.tight_layout();\n    plt.yticks(rotation= 0);\n    plt.xticks(rotation= 90);\n    plt.show();\n    del _corr;\n    collect();\ncollect();","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-09T12:19:32.139307Z","iopub.execute_input":"2022-08-09T12:19:32.139686Z","iopub.status.idle":"2022-08-09T12:19:40.538794Z","shell.execute_reply.started":"2022-08-09T12:19:32.139651Z","shell.execute_reply":"2022-08-09T12:19:40.537710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 2, figsize=(10, 6))\nfig.suptitle('Attribute data count')\n\nsns.countplot(ax=axes[0, 0], data=train_df, x='attribute_0', hue=\"failure\")\nsns.countplot(ax=axes[0, 1], data=train_df, x='attribute_1', hue=\"failure\")\nsns.countplot(ax=axes[1, 0], data=train_df, x='attribute_2', hue=\"failure\")\nsns.countplot(ax=axes[1, 1], data=train_df, x='attribute_3', hue=\"failure\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-09T12:19:40.540355Z","iopub.execute_input":"2022-08-09T12:19:40.541026Z","iopub.status.idle":"2022-08-09T12:19:41.134956Z","shell.execute_reply.started":"2022-08-09T12:19:40.540990Z","shell.execute_reply":"2022-08-09T12:19:41.133669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=\"product_code\", hue=\"failure\", data=train_df)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-09T12:19:41.136304Z","iopub.execute_input":"2022-08-09T12:19:41.136678Z","iopub.status.idle":"2022-08-09T12:19:41.383346Z","shell.execute_reply.started":"2022-08-09T12:19:41.136643Z","shell.execute_reply":"2022-08-09T12:19:41.382174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"int_cols = ['measurement_0','measurement_1','measurement_2','measurement_3','measurement_4','measurement_5','measurement_6','measurement_7','measurement_8',\n            'measurement_9','measurement_10','measurement_11','measurement_12','measurement_13','measurement_14','measurement_15','measurement_16','measurement_17']\ntrain_int = train_df[int_cols]\nsns.pairplot(train_int)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-09T12:19:41.384543Z","iopub.execute_input":"2022-08-09T12:19:41.384889Z","iopub.status.idle":"2022-08-09T12:21:07.208113Z","shell.execute_reply.started":"2022-08-09T12:19:41.384859Z","shell.execute_reply":"2022-08-09T12:21:07.206632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(4, 3, figsize=(12, 14))\nfig.suptitle('Attribute data count')\n\nsns.boxplot(ax=axes[0, 0], data=train_df, y='attribute_2')\nsns.boxplot(ax=axes[0, 1], data=train_df, y='attribute_3')\nsns.boxplot(ax=axes[0, 2], data=train_df, y='measurement_0')\nsns.boxplot(ax=axes[1, 0], data=train_df, y='measurement_1')\nsns.boxplot(ax=axes[1, 1], data=train_df, y='measurement_2')\nsns.boxplot(ax=axes[1, 2], data=train_df, y='measurement_3')\nsns.boxplot(ax=axes[2, 0], data=train_df, y='measurement_4')\nsns.boxplot(ax=axes[2, 1], data=train_df, y='measurement_5')\nsns.boxplot(ax=axes[2, 2], data=train_df, y='measurement_6')\nsns.boxplot(ax=axes[3, 0], data=train_df, y='measurement_7')\nsns.boxplot(ax=axes[3, 1], data=train_df, y='measurement_8')\nsns.boxplot(ax=axes[3, 2], data=train_df, y='measurement_9')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-09T12:21:07.209877Z","iopub.execute_input":"2022-08-09T12:21:07.210279Z","iopub.status.idle":"2022-08-09T12:21:08.407290Z","shell.execute_reply.started":"2022-08-09T12:21:07.210245Z","shell.execute_reply":"2022-08-09T12:21:08.406079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"font-family:Verdana;font-size:18px;\"> 💡Conclusion</span>\n1. <span style=\"color:brown;font-family:Verdana;\">Our target data set is imbalanced, we can try undersampling or oversampling to improve performance.</span>\n1. <span style=\"color:brown;font-family:Verdana;\">Attribute 0 & Attribute 1 has text data that can be converted to 5/6/7/8 value. We can use getdummies for the same.</span>\n1. <span style=\"color:brown;font-family:Verdana;\">All measurement columns are normally distributed, except measurement 0/1/2.</span>\n1. <span style=\"color:brown;font-family:Verdana;\">As shown in scatter plot, measurement column doesn't have strong relations.</span>\n1. <span style=\"color:brown;font-family:Verdana;\">Target column doesn't have stronger relations with independent columns, except loading column.</span>","metadata":{}},{"cell_type":"markdown","source":"# Feature Engineering\n\n<span style=\"font-family:Verdana;font-size:18px;\"> ⚓Actions taken</span>\n1. <span style=\"color:green;font-family:Verdana;\">We will convert our categorical data to interger with the help of OneHotEncoding.</span>\n1. <span style=\"color:green;font-family:Verdana;\">Missing data will be handled with Data impute technique.</span>\n1. <span style=\"color:green;font-family:Verdana;\">As data is imbalanced, and value with status 1 is low, we will adopt undersampling technique.</span>\n1. <span style=\"color:green;font-family:Verdana;\">We have different scale data, we will transform them.</span>","metadata":{}},{"cell_type":"code","source":"enc = OneHotEncoder(handle_unknown='ignore')\nohe_attributes = ['attribute_0', 'attribute_1']\nohe_output = ['ohe_a_7', 'ohe_a_6', 'ohe_a_8']\nohe = OneHotEncoder(categories=[['material_5', 'material_7'],['material_5', 'material_6', 'material_8']],\n                    drop='first', sparse=False, handle_unknown='ignore')\nohe.fit(train_df[ohe_attributes])\n\ntrain_df[ohe_output] = ohe.transform(train_df[ohe_attributes])\ntest_df[ohe_output] = ohe.transform(test_df[ohe_attributes])\n\ntrain_df = train_df.drop(['attribute_0', 'attribute_1'], axis=1)\ntest_df = test_df.drop(['attribute_0', 'attribute_1'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:21:08.409244Z","iopub.execute_input":"2022-08-09T12:21:08.409787Z","iopub.status.idle":"2022-08-09T12:21:08.485555Z","shell.execute_reply.started":"2022-08-09T12:21:08.409736Z","shell.execute_reply":"2022-08-09T12:21:08.484192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prod_code_encoding = {'A': 1, 'B':2, 'C':3, 'D': 4, 'E': 5, 'F':6, 'G': 7, 'H':8, 'I': 9}\n\ntrain_df.product_code = [prod_code_encoding[val] for val in train_df.product_code]\ntest_df.product_code = [prod_code_encoding[val] for val in test_df.product_code]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:21:08.487267Z","iopub.execute_input":"2022-08-09T12:21:08.487693Z","iopub.status.idle":"2022-08-09T12:21:08.518300Z","shell.execute_reply.started":"2022-08-09T12:21:08.487655Z","shell.execute_reply":"2022-08-09T12:21:08.516868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['loading'] + int_cols\nimputer = SimpleImputer(strategy=\"mean\")\nimputer.fit(train_df[features])\n\ntrain_df[features] = imputer.transform(train_df[features])\ntest_df[features] = imputer.transform(test_df[features])","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:21:08.520158Z","iopub.execute_input":"2022-08-09T12:21:08.520569Z","iopub.status.idle":"2022-08-09T12:21:08.570195Z","shell.execute_reply.started":"2022-08-09T12:21:08.520513Z","shell.execute_reply":"2022-08-09T12:21:08.568927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train_df['failure']\nX = train_df.drop(['id','failure'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:21:08.572753Z","iopub.execute_input":"2022-08-09T12:21:08.573235Z","iopub.status.idle":"2022-08-09T12:21:08.583565Z","shell.execute_reply.started":"2022-08-09T12:21:08.573190Z","shell.execute_reply":"2022-08-09T12:21:08.582572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define the undersampling method\nundersample = NearMiss(n_neighbors=1)\n\n# transform the dataset\nX, y = undersample.fit_resample(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:21:08.584818Z","iopub.execute_input":"2022-08-09T12:21:08.585537Z","iopub.status.idle":"2022-08-09T12:27:29.131931Z","shell.execute_reply.started":"2022-08-09T12:21:08.585503Z","shell.execute_reply":"2022-08-09T12:27:29.130372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Id = test_df['id']\ntest_df = test_df.drop(['id'], axis=1)\ntrain_df = X","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:27:29.133622Z","iopub.execute_input":"2022-08-09T12:27:29.133990Z","iopub.status.idle":"2022-08-09T12:27:29.144888Z","shell.execute_reply.started":"2022-08-09T12:27:29.133950Z","shell.execute_reply":"2022-08-09T12:27:29.143431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = PowerTransformer()\ndata_scaled = scaler.fit_transform(train_df)\ndata_scaled_train_df=pd.DataFrame(data_scaled, index=train_df.index, columns=train_df.columns)\ndata_scaled_train_df.head()\n\ndata_scaled = scaler.fit_transform(test_df)\ndata_scaled_test_df=pd.DataFrame(data_scaled, index=test_df.index, columns=test_df.columns)\ndata_scaled_test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:27:29.146224Z","iopub.execute_input":"2022-08-09T12:27:29.146607Z","iopub.status.idle":"2022-08-09T12:27:30.105379Z","shell.execute_reply.started":"2022-08-09T12:27:29.146555Z","shell.execute_reply":"2022-08-09T12:27:30.104351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_scaled_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:27:30.106686Z","iopub.execute_input":"2022-08-09T12:27:30.107053Z","iopub.status.idle":"2022-08-09T12:27:30.134343Z","shell.execute_reply.started":"2022-08-09T12:27:30.107020Z","shell.execute_reply":"2022-08-09T12:27:30.133120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_scaled_train_df.shape, data_scaled_test_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:27:30.140514Z","iopub.execute_input":"2022-08-09T12:27:30.140951Z","iopub.status.idle":"2022-08-09T12:27:30.148692Z","shell.execute_reply.started":"2022-08-09T12:27:30.140913Z","shell.execute_reply":"2022-08-09T12:27:30.147234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\n\nclf = lgb.LGBMClassifier(objective= 'binary',learning_rate = 0.05, max_depth = 3, n_estimators=500)\nclf.fit(data_scaled_train_df, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:27:30.150825Z","iopub.execute_input":"2022-08-09T12:27:30.151484Z","iopub.status.idle":"2022-08-09T12:27:31.871029Z","shell.execute_reply.started":"2022-08-09T12:27:30.151441Z","shell.execute_reply":"2022-08-09T12:27:31.869909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = clf.predict_proba(data_scaled_test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:27:31.876197Z","iopub.execute_input":"2022-08-09T12:27:31.877420Z","iopub.status.idle":"2022-08-09T12:27:32.041921Z","shell.execute_reply.started":"2022-08-09T12:27:31.877362Z","shell.execute_reply":"2022-08-09T12:27:32.040849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:27:32.046507Z","iopub.execute_input":"2022-08-09T12:27:32.049196Z","iopub.status.idle":"2022-08-09T12:27:32.059845Z","shell.execute_reply.started":"2022-08-09T12:27:32.049148Z","shell.execute_reply":"2022-08-09T12:27:32.058470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create a  DataFrame with the passengers ids and our prediction\nsubmission = pd.DataFrame({'id':Id,'failure':y_pred[:,1]})","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:27:32.061201Z","iopub.execute_input":"2022-08-09T12:27:32.061885Z","iopub.status.idle":"2022-08-09T12:27:32.070059Z","shell.execute_reply.started":"2022-08-09T12:27:32.061839Z","shell.execute_reply":"2022-08-09T12:27:32.068659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:27:32.072723Z","iopub.execute_input":"2022-08-09T12:27:32.073451Z","iopub.status.idle":"2022-08-09T12:27:32.086217Z","shell.execute_reply.started":"2022-08-09T12:27:32.073415Z","shell.execute_reply":"2022-08-09T12:27:32.085233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T12:27:32.087863Z","iopub.execute_input":"2022-08-09T12:27:32.088526Z","iopub.status.idle":"2022-08-09T12:27:32.146880Z","shell.execute_reply.started":"2022-08-09T12:27:32.088492Z","shell.execute_reply":"2022-08-09T12:27:32.145906Z"},"trusted":true},"execution_count":null,"outputs":[]}]}