{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This is a simple notebook based on [lr is all u need](https://www.kaggle.com/code/heyspaceturtle/logistic-regression-is-all-u-need) \n\nSpecial thanks to Devastator for his [clean notebook ](https://www.kaggle.com/code/thedevastator/tps-aug-simple-baseline)\nand to Lucas See for his [great EDA](https://www.kaggle.com/code/pinstripezebra/eda-baseline-model). Make sure to give them an upvote! \n\nplease upvote for above author","metadata":{}},{"cell_type":"code","source":"# Viz\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Data Processing\nimport numpy as np\nimport pandas as pd\nfrom sklearn import preprocessing\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\n\n# Modeling \nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cluster, accuracy_score, roc_auc_score\nfrom sklearn.model_selection import cross_validate, GridSearchCV, cross_val_score, StratifiedKFold\n\n# Other\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-06T02:50:44.846536Z","iopub.execute_input":"2022-08-06T02:50:44.846931Z","iopub.status.idle":"2022-08-06T02:50:44.855441Z","shell.execute_reply.started":"2022-08-06T02:50:44.846899Z","shell.execute_reply":"2022-08-06T02:50:44.854129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Data\ntrain = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\nsample_submission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:44.920478Z","iopub.execute_input":"2022-08-06T02:50:44.920866Z","iopub.status.idle":"2022-08-06T02:50:45.126533Z","shell.execute_reply.started":"2022-08-06T02:50:44.920836Z","shell.execute_reply":"2022-08-06T02:50:45.125278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Variable lists for easy manipulation\nid_var = ['id']\ntarget= ['failure']\ncat_vars = ['product_code','attribute_0','attribute_1']\nnum_vars = [v for v in test.columns if v not in id_var and v not in cat_vars]\npredictors = cat_vars + num_vars","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:45.128539Z","iopub.execute_input":"2022-08-06T02:50:45.128868Z","iopub.status.idle":"2022-08-06T02:50:45.134070Z","shell.execute_reply.started":"2022-08-06T02:50:45.128839Z","shell.execute_reply":"2022-08-06T02:50:45.133015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = train.corr()\nmask = np.triu(np.ones_like(corr, dtype=bool))\nf, ax = plt.subplots(figsize=(12, 12), facecolor='#EAECEE')\ncmap = sns.color_palette(\"rainbow\", as_cmap=True)\nsns.heatmap(corr, mask=mask, cmap=cmap, vmax=1, vmin=-1., center=0, annot=False,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": 0.75})\n\nax.set_title('Correlation heatmap', fontsize=24, y= 1.05)\ncolorbar = ax.collections[0].colorbar","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:45.135352Z","iopub.execute_input":"2022-08-06T02:50:45.136111Z","iopub.status.idle":"2022-08-06T02:50:45.843986Z","shell.execute_reply.started":"2022-08-06T02:50:45.136060Z","shell.execute_reply":"2022-08-06T02:50:45.842869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pre-Processing ","metadata":{}},{"cell_type":"code","source":"#Imputing missing values \nmulti_imp = IterativeImputer(max_iter = 9, random_state = 42, verbose = 0, skip_complete = True, n_nearest_features = 10, tol = 0.001)\nmulti_imp.fit(train[num_vars])\ntrain[num_vars] = multi_imp.transform(train[num_vars])\ntest[num_vars] = multi_imp.transform(test[num_vars])\n\nprint('Train NA:', train.isna().sum().sum())\nprint('Test NA:', train.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:45.847370Z","iopub.execute_input":"2022-08-06T02:50:45.848612Z","iopub.status.idle":"2022-08-06T02:50:52.183628Z","shell.execute_reply.started":"2022-08-06T02:50:45.848561Z","shell.execute_reply":"2022-08-06T02:50:52.181837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['measurement_0'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:52.185062Z","iopub.execute_input":"2022-08-06T02:50:52.185841Z","iopub.status.idle":"2022-08-06T02:50:52.373842Z","shell.execute_reply.started":"2022-08-06T02:50:52.185796Z","shell.execute_reply":"2022-08-06T02:50:52.372752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['loading'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:52.375169Z","iopub.execute_input":"2022-08-06T02:50:52.375542Z","iopub.status.idle":"2022-08-06T02:50:52.535955Z","shell.execute_reply.started":"2022-08-06T02:50:52.375510Z","shell.execute_reply":"2022-08-06T02:50:52.534822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"float_cols = [col for col in train.columns if train[col].dtypes == 'float64']","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:52.537497Z","iopub.execute_input":"2022-08-06T02:50:52.538475Z","iopub.status.idle":"2022-08-06T02:50:52.545425Z","shell.execute_reply.started":"2022-08-06T02:50:52.538431Z","shell.execute_reply":"2022-08-06T02:50:52.544154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#both train and test loading value is right skewed\nfor i in float_cols:\n    train[i] = train[i].apply(lambda x:np.log(x+1))\n    test[i] = test[i].apply(lambda x:np.log(x+1))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:52.549999Z","iopub.execute_input":"2022-08-06T02:50:52.550504Z","iopub.status.idle":"2022-08-06T02:50:53.806898Z","shell.execute_reply.started":"2022-08-06T02:50:52.550460Z","shell.execute_reply":"2022-08-06T02:50:53.805850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['loading'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:53.808026Z","iopub.execute_input":"2022-08-06T02:50:53.808340Z","iopub.status.idle":"2022-08-06T02:50:53.978567Z","shell.execute_reply.started":"2022-08-06T02:50:53.808298Z","shell.execute_reply":"2022-08-06T02:50:53.977508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalize columns\nattributes = ['attribute_2', 'attribute_3', 'measurement_4', 'measurement_5', 'measurement_6']\ntrain[attributes] = preprocessing.normalize(train[attributes])\ntest[attributes] = preprocessing.normalize(test[attributes])","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:53.980077Z","iopub.execute_input":"2022-08-06T02:50:53.980627Z","iopub.status.idle":"2022-08-06T02:50:53.998201Z","shell.execute_reply.started":"2022-08-06T02:50:53.980573Z","shell.execute_reply":"2022-08-06T02:50:53.997377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop product code\ntest = test.drop(['product_code'], axis = 1)\ntrain = train.drop(['product_code'], axis = 1)\ncat_vars.remove('product_code')","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:53.999336Z","iopub.execute_input":"2022-08-06T02:50:54.000283Z","iopub.status.idle":"2022-08-06T02:50:54.010671Z","shell.execute_reply.started":"2022-08-06T02:50:54.000249Z","shell.execute_reply":"2022-08-06T02:50:54.009676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One Hot Encoding\nfor v in cat_vars:\n    tempdf = pd.get_dummies(train[v], prefix = v)\n    tempdf_test = pd.get_dummies(test[v], prefix = v)\n    train = pd.merge(left = train, right = tempdf, left_index = True, right_index = True)\n    test = pd.merge(left = test, right = tempdf_test, left_index = True, right_index = True)\ntrain = train.drop(cat_vars, axis = 1)\ntest = test.drop(cat_vars, axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:54.012811Z","iopub.execute_input":"2022-08-06T02:50:54.013150Z","iopub.status.idle":"2022-08-06T02:50:54.050817Z","shell.execute_reply.started":"2022-08-06T02:50:54.013096Z","shell.execute_reply":"2022-08-06T02:50:54.049712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Training","metadata":{}},{"cell_type":"code","source":"# update predictor list\npredictors = [v for v in train.columns if v not in id_var and v not in target]\n\n# Train test split\nX_train, X_test, y_train, y_test = train_test_split(train[predictors], train[target], test_size=0.20, random_state=42)\nX_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:54.054177Z","iopub.execute_input":"2022-08-06T02:50:54.055081Z","iopub.status.idle":"2022-08-06T02:50:54.092084Z","shell.execute_reply.started":"2022-08-06T02:50:54.055033Z","shell.execute_reply":"2022-08-06T02:50:54.090985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression ","metadata":{"execution":{"iopub.status.busy":"2022-08-05T10:21:19.770777Z","iopub.execute_input":"2022-08-05T10:21:19.771354Z","iopub.status.idle":"2022-08-05T10:21:19.777448Z","shell.execute_reply.started":"2022-08-05T10:21:19.771312Z","shell.execute_reply":"2022-08-05T10:21:19.775839Z"}}},{"cell_type":"code","source":"# Logistic Regression\nlr = LogisticRegression(random_state = 0)\nlr.fit(X_train, y_train)\nprint('Validation AUC:', round(roc_auc_score(y_test, lr.predict_proba(X_test)[:,1]), 4))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:54.093392Z","iopub.execute_input":"2022-08-06T02:50:54.093700Z","iopub.status.idle":"2022-08-06T02:50:54.552236Z","shell.execute_reply.started":"2022-08-06T02:50:54.093672Z","shell.execute_reply":"2022-08-06T02:50:54.551075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LR Grid Search\nThis is commented out so the notebook runs faster. params are saved on the next cell","metadata":{}},{"cell_type":"code","source":"# Logistic Regression Grid Search\n#space = {\"C\":np.logspace(-4, 4, 50),\"penalty\":[\"l1\", \"l2\"], \"solver\":['liblinear', 'lbfgs', 'newton-cg']}\n#gsLR = GridSearchCV(LogisticRegression(max_iter = 200), space ,scoring='roc_auc', cv = kfold, verbose = 1)\n#gsLR.fit(X_train, y_train.values.squeeze())\n#print(gsLR.best_params_)\n\n#LR_best = gsLR.best_estimator_\n#print(gsLR.best_params_)\n#print('Validation AUC:', round(roc_auc_score(y_test, LR_best.predict_proba(X_test)[:,1]), 4))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-06T02:50:54.553863Z","iopub.execute_input":"2022-08-06T02:50:54.554439Z","iopub.status.idle":"2022-08-06T02:50:54.567601Z","shell.execute_reply.started":"2022-08-06T02:50:54.554394Z","shell.execute_reply":"2022-08-06T02:50:54.565582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Best Params from Grid Search\nlrgs = LogisticRegression(max_iter = 200, C=0.0001, penalty='l2', solver='newton-cg')\nlrgs.fit(X_train, y_train)\n\nprint('Validation AUC:', round(roc_auc_score(y_test, lrgs.predict_proba(X_test)[:,1]), 4))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:54.570278Z","iopub.execute_input":"2022-08-06T02:50:54.578709Z","iopub.status.idle":"2022-08-06T02:50:54.794323Z","shell.execute_reply.started":"2022-08-06T02:50:54.578637Z","shell.execute_reply":"2022-08-06T02:50:54.793163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LDA\nLDA = LinearDiscriminantAnalysis()\nLDA.fit(X_train,y_train.values.squeeze())\nprint('Validation AUC:', round(roc_auc_score(y_test, LDA.predict_proba(X_test)[:,1]), 4))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:54.796071Z","iopub.execute_input":"2022-08-06T02:50:54.796777Z","iopub.status.idle":"2022-08-06T02:50:54.972715Z","shell.execute_reply.started":"2022-08-06T02:50:54.796730Z","shell.execute_reply":"2022-08-06T02:50:54.971499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Public Leaderboard scores: \n- Logistic Regression GS = 0.58786 \n- LDA = 0.58133\n\nLDA drops a lot on the public LB, but tbh everything about this LB is weird. I still need to do some in-depth cross validation to make sure the score is stable and we don't drop fro the LB on private.","metadata":{}},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"# Remove id\ntest = test.drop('id', axis = 1)\n\n# LR GS  0.58786\nsub1 = sample_submission.copy()\nsub1.failure = lr.predict_proba(test)[:,1]\nsub1.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T02:50:54.974479Z","iopub.execute_input":"2022-08-06T02:50:54.975294Z","iopub.status.idle":"2022-08-06T02:50:55.131131Z","shell.execute_reply.started":"2022-08-06T02:50:54.975245Z","shell.execute_reply":"2022-08-06T02:50:55.130199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}