{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-08T18:42:19.330726Z","iopub.execute_input":"2022-08-08T18:42:19.331227Z","iopub.status.idle":"2022-08-08T18:42:19.368340Z","shell.execute_reply.started":"2022-08-08T18:42:19.331120Z","shell.execute_reply":"2022-08-08T18:42:19.367046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Loading libraries</h3>","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import Normalizer\nfrom sklearn.naive_bayes import GaussianNB, BernoulliNB, ComplementNB\nfrom sklearn.ensemble import RandomForestClassifier, VotingClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import auc,roc_curve, RocCurveDisplay\nfrom lightgbm import LGBMClassifier\nfrom sklearn.neural_network import MLPClassifier\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:42:21.228817Z","iopub.execute_input":"2022-08-08T18:42:21.229244Z","iopub.status.idle":"2022-08-08T18:42:23.348972Z","shell.execute_reply.started":"2022-08-08T18:42:21.229207Z","shell.execute_reply":"2022-08-08T18:42:23.347751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h4 style=\"text-align: center;\">We are going to see which model without any hyperparameter tunning will perform best. So, let's create a dict to keep the scores updated.</h4>","metadata":{}},{"cell_type":"code","source":"models_performance = {}\n\ndef fit_and_draw_roc_auc(model):\n    model.fit(X_train, y_train)\n    predictions = model.predict_proba(X_test)\n    fpr, tpr, thresholds = roc_curve(y_test, predictions[:, 1])\n    roc_auc = auc(fpr, tpr)\n    print(f'roc_auc: {roc_auc}')\n    models_performance[str(model).split('(')[0]] = roc_auc\n    display = RocCurveDisplay(fpr=fpr, tpr=tpr, roc_auc=roc_auc, estimator_name=str(model).split('(')[0])\n    display.plot()\n    plt.show()\n    plt.clf()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:42:57.444240Z","iopub.execute_input":"2022-08-08T18:42:57.444796Z","iopub.status.idle":"2022-08-08T18:42:57.457487Z","shell.execute_reply.started":"2022-08-08T18:42:57.444746Z","shell.execute_reply":"2022-08-08T18:42:57.456196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h4 style=\"text-align: center;\">Loading the data</h4>","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv').set_index('id')\ntest_df = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv').set_index('id')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:43:16.157024Z","iopub.execute_input":"2022-08-08T18:43:16.158369Z","iopub.status.idle":"2022-08-08T18:43:16.467621Z","shell.execute_reply.started":"2022-08-08T18:43:16.158309Z","shell.execute_reply":"2022-08-08T18:43:16.466314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.shape)\nprint(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:44:22.596739Z","iopub.execute_input":"2022-08-08T18:44:22.597198Z","iopub.status.idle":"2022-08-08T18:44:22.602927Z","shell.execute_reply.started":"2022-08-08T18:44:22.597162Z","shell.execute_reply":"2022-08-08T18:44:22.601760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:44:25.344638Z","iopub.execute_input":"2022-08-08T18:44:25.345928Z","iopub.status.idle":"2022-08-08T18:44:25.372633Z","shell.execute_reply.started":"2022-08-08T18:44:25.345876Z","shell.execute_reply":"2022-08-08T18:44:25.371412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Treating features...</h3>","metadata":{}},{"cell_type":"code","source":"attributes = [x for x in train_df.columns if x.startswith('attribute')]\nmeasurements = [x for x in train_df.columns if x.startswith('measurement')]\nattributes, measurements","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:44:27.706236Z","iopub.execute_input":"2022-08-08T18:44:27.707360Z","iopub.status.idle":"2022-08-08T18:44:27.716139Z","shell.execute_reply.started":"2022-08-08T18:44:27.707312Z","shell.execute_reply":"2022-08-08T18:44:27.714856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_cols = train_df.isnull().sum().sort_values(ascending=False)\nnull_cols = list(null_cols[null_cols>1].index)\nnull_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:44:29.819351Z","iopub.execute_input":"2022-08-08T18:44:29.819796Z","iopub.status.idle":"2022-08-08T18:44:29.835178Z","shell.execute_reply.started":"2022-08-08T18:44:29.819757Z","shell.execute_reply":"2022-08-08T18:44:29.834042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"object_cols = list(train_df.select_dtypes(include=['object']))\nnumerics_cols = list(train_df.select_dtypes(exclude=['object']))\nnumerics_cols.remove('failure')\nnumerics_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:44:31.609291Z","iopub.execute_input":"2022-08-08T18:44:31.609685Z","iopub.status.idle":"2022-08-08T18:44:31.625511Z","shell.execute_reply.started":"2022-08-08T18:44:31.609653Z","shell.execute_reply":"2022-08-08T18:44:31.624582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', 30)\ntrain_df\n\ncolumns_chart_1 = ['measurement_3', 'measurement_4', 'measurement_5', 'measurement_6', \n                  'measurement_7', 'measurement_8', 'measurement_9', 'failure']\n\ncolumns_chart_2 = ['measurement_10', 'measurement_11', 'measurement_12', 'measurement_13', 'measurement_14', \n                  'measurement_15', 'measurement_16', 'measurement_17', 'failure']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T19:26:24.162400Z","iopub.execute_input":"2022-08-08T19:26:24.162884Z","iopub.status.idle":"2022-08-08T19:26:24.169569Z","shell.execute_reply.started":"2022-08-08T19:26:24.162844Z","shell.execute_reply":"2022-08-08T19:26:24.168277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(35, 35))\n\nsns.pairplot(train_df[columns_chart_1], hue='failure')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T19:27:58.363311Z","iopub.execute_input":"2022-08-08T19:27:58.363772Z","iopub.status.idle":"2022-08-08T19:29:10.073058Z","shell.execute_reply.started":"2022-08-08T19:27:58.363726Z","shell.execute_reply":"2022-08-08T19:29:10.071535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(35, 35))\n\nsns.pairplot(train_df[columns_chart_2], hue='failure')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T19:29:51.895677Z","iopub.execute_input":"2022-08-08T19:29:51.896123Z","iopub.status.idle":"2022-08-08T19:31:21.025328Z","shell.execute_reply.started":"2022-08-08T19:29:51.896087Z","shell.execute_reply":"2022-08-08T19:31:21.023971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"measurement_cols = [i for i in train_df.columns if i.startswith('measu')]\nattribute_cols = [i for i in train_df.columns if i.startswith('attri')]\nattribute_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-08T18:44:35.532731Z","iopub.execute_input":"2022-08-08T18:44:35.534001Z","iopub.status.idle":"2022-08-08T18:44:35.542168Z","shell.execute_reply.started":"2022-08-08T18:44:35.533951Z","shell.execute_reply":"2022-08-08T18:44:35.540956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['measurement_t1'] = train_df[measurement_cols].sum(axis=1)\ntrain_df['attribute_t1'] = train_df[attribute_cols].sum(axis=1)\n\ntest_df['measurement_t1'] = test_df[measurement_cols].sum(axis=1)\ntest_df['attribute_t1'] = test_df[attribute_cols].sum(axis=1)\n\ntest_df['att2*3'] = test_df['attribute_2']*test_df['attribute_3']\ntrain_df['att2*3'] = train_df['attribute_2']*train_df['attribute_3']","metadata":{"execution":{"iopub.status.busy":"2022-08-08T19:33:56.788638Z","iopub.execute_input":"2022-08-08T19:33:56.789723Z","iopub.status.idle":"2022-08-08T19:33:56.850417Z","shell.execute_reply.started":"2022-08-08T19:33:56.789666Z","shell.execute_reply":"2022-08-08T19:33:56.849193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\n\nsp = SimpleImputer(strategy='mean')\n\nct = ColumnTransformer([(\"imp\", SimpleImputer(), null_cols)])\n    \ntrain_df[null_cols] = ct.fit_transform(train_df[null_cols])\ntest_df[null_cols] = ct.fit_transform(test_df[null_cols])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T19:33:58.731013Z","iopub.execute_input":"2022-08-08T19:33:58.731448Z","iopub.status.idle":"2022-08-08T19:33:58.785392Z","shell.execute_reply.started":"2022-08-08T19:33:58.731407Z","shell.execute_reply":"2022-08-08T19:33:58.784121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', 30)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-08-08T19:34:00.072388Z","iopub.execute_input":"2022-08-08T19:34:00.072788Z","iopub.status.idle":"2022-08-08T19:34:00.123088Z","shell.execute_reply.started":"2022-08-08T19:34:00.072753Z","shell.execute_reply":"2022-08-08T19:34:00.121300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#One Hot Encoding\n\nfull_df_for_dummies = pd.concat([train_df, test_df])\nfull_df_for_dummies = pd.get_dummies(full_df_for_dummies)\ntrain_df = full_df_for_dummies.iloc[0:26570,:]\ntest_df = full_df_for_dummies.iloc[26570:,:]\ntest_df.drop('failure', inplace=True, axis=1)\n\ndel full_df_for_dummies\n\n#train_df.fillna(0, inplace=True)\n#test_df.fillna(0, inplace=True)\n\nX = train_df.copy()\ny = X.pop('failure')\n#test_df.drop('failure', axis=1, inplace=True)\n\nfrom sklearn.model_selection import cross_val_score, train_test_split, GridSearchCV\nX_train, X_test, y_train, y_test = train_test_split(X, y, random_state=23)\n\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler\n\nct = ColumnTransformer(\n    [(\"standS\", StandardScaler(), measurement_cols)])\n     \nX_train[measurement_cols] = ct.fit_transform(X_train[measurement_cols])\nX_test[measurement_cols] = ct.fit_transform(X_test[measurement_cols])\ntest_df[measurement_cols] = ct.fit_transform(test_df[measurement_cols])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T19:34:18.343872Z","iopub.execute_input":"2022-08-08T19:34:18.344322Z","iopub.status.idle":"2022-08-08T19:34:18.436623Z","shell.execute_reply.started":"2022-08-08T19:34:18.344284Z","shell.execute_reply":"2022-08-08T19:34:18.434973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fit_and_draw_roc_auc(LogisticRegression(penalty=\"l2\", solver='liblinear'))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T19:34:23.240143Z","iopub.execute_input":"2022-08-08T19:34:23.240902Z","iopub.status.idle":"2022-08-08T19:34:23.755852Z","shell.execute_reply.started":"2022-08-08T19:34:23.240861Z","shell.execute_reply":"2022-08-08T19:34:23.754901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fit_and_draw_roc_auc(LGBMClassifier())","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:50:01.237572Z","iopub.execute_input":"2022-08-08T11:50:01.237894Z","iopub.status.idle":"2022-08-08T11:50:01.739914Z","shell.execute_reply.started":"2022-08-08T11:50:01.237862Z","shell.execute_reply":"2022-08-08T11:50:01.739096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fit_and_draw_roc_auc(XGBClassifier())","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:50:01.741097Z","iopub.execute_input":"2022-08-08T11:50:01.742387Z","iopub.status.idle":"2022-08-08T11:50:04.564951Z","shell.execute_reply.started":"2022-08-08T11:50:01.742361Z","shell.execute_reply":"2022-08-08T11:50:04.563447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fit_and_draw_roc_auc(RandomForestClassifier())","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:50:04.566362Z","iopub.execute_input":"2022-08-08T11:50:04.566949Z","iopub.status.idle":"2022-08-08T11:50:14.490628Z","shell.execute_reply.started":"2022-08-08T11:50:04.566924Z","shell.execute_reply":"2022-08-08T11:50:14.489639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fit_and_draw_roc_auc(MLPClassifier(max_iter=400))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:50:14.492079Z","iopub.execute_input":"2022-08-08T11:50:14.492365Z","iopub.status.idle":"2022-08-08T11:50:16.247926Z","shell.execute_reply.started":"2022-08-08T11:50:14.492337Z","shell.execute_reply":"2022-08-08T11:50:16.247209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fit_and_draw_roc_auc(BernoulliNB())","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:50:16.249029Z","iopub.execute_input":"2022-08-08T11:50:16.249485Z","iopub.status.idle":"2022-08-08T11:50:16.403148Z","shell.execute_reply.started":"2022-08-08T11:50:16.249457Z","shell.execute_reply":"2022-08-08T11:50:16.402457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fit_and_draw_roc_auc(GaussianNB())","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:50:16.404387Z","iopub.execute_input":"2022-08-08T11:50:16.404883Z","iopub.status.idle":"2022-08-08T11:50:16.557255Z","shell.execute_reply.started":"2022-08-08T11:50:16.404854Z","shell.execute_reply":"2022-08-08T11:50:16.556159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Let's see the two models that performed better and perform a Grid Search on it!</h3>","metadata":{}},{"cell_type":"code","source":"models_performance","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:50:16.558833Z","iopub.execute_input":"2022-08-08T11:50:16.559094Z","iopub.status.idle":"2022-08-08T11:50:16.564567Z","shell.execute_reply.started":"2022-08-08T11:50:16.559068Z","shell.execute_reply":"2022-08-08T11:50:16.563635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"text-align: center;\">Logistic Regression rocks! Grid Search time:</h3>","metadata":{}},{"cell_type":"code","source":"param_grid = {'penalty':('l1', 'l2'), 'C':[0.5, 1, 2, 5, 10]}\nLR = LogisticRegression(solver='liblinear')\nclf = GridSearchCV(LR, param_grid, scoring='roc_auc', n_jobs=None, refit=True, cv=None, verbose=0, pre_dispatch='2*n_jobs', return_train_score=False)\nclf.fit(X, y)\nprint(clf.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:50:16.565848Z","iopub.execute_input":"2022-08-08T11:50:16.566108Z","iopub.status.idle":"2022-08-08T11:51:57.304205Z","shell.execute_reply.started":"2022-08-08T11:50:16.566083Z","shell.execute_reply":"2022-08-08T11:51:57.303346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_selection import RFE\n\nestimator = LogisticRegression(penalty=\"l1\", solver='liblinear', C=1)\nselector = RFE(estimator, n_features_to_select=5, step=1)\nselector = selector.fit(X, y)\nbest_cols = selector.get_feature_names_out()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:51:57.306172Z","iopub.execute_input":"2022-08-08T11:51:57.306534Z","iopub.status.idle":"2022-08-08T11:53:57.868627Z","shell.execute_reply.started":"2022-08-08T11:51:57.306508Z","shell.execute_reply":"2022-08-08T11:53:57.867507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-08T11:53:57.869859Z","iopub.execute_input":"2022-08-08T11:53:57.870078Z","iopub.status.idle":"2022-08-08T11:53:57.875872Z","shell.execute_reply.started":"2022-08-08T11:53:57.870055Z","shell.execute_reply":"2022-08-08T11:53:57.874892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X['sum_best_cols'] = X[best_cols].sum(axis=1)\ntest_df['sum_best_cols'] = test_df[best_cols].sum(axis=1)   \n\nX['mean_best_cols'] = X[best_cols].mean(axis=1)\ntest_df['mean_best_cols'] = test_df[best_cols].mean(axis=1)  \n\nX['std_best_cols'] = X[best_cols].std(axis=1)\ntest_df['std_best_cols'] = test_df[best_cols].std(axis=1) ","metadata":{"execution":{"iopub.status.busy":"2022-08-08T12:03:40.553550Z","iopub.execute_input":"2022-08-08T12:03:40.553958Z","iopub.status.idle":"2022-08-08T12:03:40.578072Z","shell.execute_reply.started":"2022-08-08T12:03:40.553932Z","shell.execute_reply":"2022-08-08T12:03:40.577396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_lr = LogisticRegression(penalty=\"l1\", solver='liblinear', C=1)\nfinal_lr.fit(X, y)\nfinal_predictions = final_lr.predict_proba(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T19:34:43.043323Z","iopub.execute_input":"2022-08-08T19:34:43.043746Z","iopub.status.idle":"2022-08-08T19:34:48.450019Z","shell.execute_reply.started":"2022-08-08T19:34:43.043711Z","shell.execute_reply":"2022-08-08T19:34:48.448027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(index=test_df.index)\nsubmission['failure'] =  final_predictions[:, 1]\nsubmission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-08T19:34:53.825135Z","iopub.execute_input":"2022-08-08T19:34:53.825561Z","iopub.status.idle":"2022-08-08T19:34:53.890310Z","shell.execute_reply.started":"2022-08-08T19:34:53.825524Z","shell.execute_reply":"2022-08-08T19:34:53.889008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"text-align: center;\">Work on progress!</h1>","metadata":{}}]}