{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nfrom pandas import read_csv\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport os\nimport random\nrandom.seed(108)\nimport datetime","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T04:39:08.130119Z","iopub.execute_input":"2022-08-12T04:39:08.130598Z","iopub.status.idle":"2022-08-12T04:39:08.139909Z","shell.execute_reply.started":"2022-08-12T04:39:08.130559Z","shell.execute_reply":"2022-08-12T04:39:08.138344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.getcwd()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:39:10.534323Z","iopub.execute_input":"2022-08-12T04:39:10.534787Z","iopub.status.idle":"2022-08-12T04:39:10.542770Z","shell.execute_reply.started":"2022-08-12T04:39:10.534751Z","shell.execute_reply":"2022-08-12T04:39:10.541472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv',index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:39:17.279936Z","iopub.execute_input":"2022-08-12T04:39:17.280487Z","iopub.status.idle":"2022-08-12T04:39:17.398187Z","shell.execute_reply.started":"2022-08-12T04:39:17.280439Z","shell.execute_reply":"2022-08-12T04:39:17.396818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:39:20.246383Z","iopub.execute_input":"2022-08-12T04:39:20.247684Z","iopub.status.idle":"2022-08-12T04:39:20.277920Z","shell.execute_reply.started":"2022-08-12T04:39:20.247639Z","shell.execute_reply":"2022-08-12T04:39:20.276489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:39:28.191754Z","iopub.execute_input":"2022-08-12T04:39:28.192844Z","iopub.status.idle":"2022-08-12T04:39:28.208983Z","shell.execute_reply.started":"2022-08-12T04:39:28.192800Z","shell.execute_reply":"2022-08-12T04:39:28.207324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imputing Missing Values\nfrom sklearn.base import TransformerMixin\nclass DataFrameImputer(TransformerMixin):\n\n    def __init__(self):\n        \"\"\"Columns of dtype object are imputed with the most frequent value in column. Columns of other types are imputed with mean of column.\"\"\"\n    def fit(self, X, y=None):\n        self.fill = pd.Series([X[c].value_counts().index[0]\n            if X[c].dtype == np.dtype('O') else X[c].mean() for c in \n            X],index=X.columns)\n        return self\n    def transform(self, X, y=None):\n        return X.fillna(self.fill)\nX = pd.DataFrame(df) \ndf = DataFrameImputer().fit_transform(X)\ndf.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:39:36.501483Z","iopub.execute_input":"2022-08-12T04:39:36.501901Z","iopub.status.idle":"2022-08-12T04:39:36.547804Z","shell.execute_reply.started":"2022-08-12T04:39:36.501867Z","shell.execute_reply":"2022-08-12T04:39:36.546710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Categorical Features vs. Target — Grouped Bar Chart","metadata":{}},{"cell_type":"code","source":"# bar plot \n\ncat_list = ['product_code', 'attribute_0', 'attribute_1']\nfig = plt.figure(figsize = (16,8))\n\nfor i in range(len(cat_list)): \n    column = cat_list[i]\n    sub = fig.add_subplot(2,4, i+1)\n    chart = sns.countplot(data = df, x=column, hue='failure', palette='RdYlBu')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:39:44.531063Z","iopub.execute_input":"2022-08-12T04:39:44.531824Z","iopub.status.idle":"2022-08-12T04:39:45.080283Z","shell.execute_reply.started":"2022-08-12T04:39:44.531781Z","shell.execute_reply":"2022-08-12T04:39:45.078871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# box plot\n\nfrom matplotlib import pyplot as plt\n%matplotlib inline\n\nnum_list = ['loading', 'attribute_2', 'measurement_1','measurement_2', 'measurement_3', 'measurement_4', 'measurement_5', 'measurement_6', 'measurement_7', 'measurement_8', 'measurement_9', 'measurement_10', 'measurement_11', 'measurement_12', 'measurement_13', 'measurement_14', 'measurement_15', 'measurement_16', 'measurement_17']\nfor col in num_list:\n    df.boxplot(column=col, by='failure', figsize=(6,6))\n    plt.title(col)\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:39:56.681421Z","iopub.execute_input":"2022-08-12T04:39:56.681874Z","iopub.status.idle":"2022-08-12T04:40:00.256353Z","shell.execute_reply.started":"2022-08-12T04:39:56.681842Z","shell.execute_reply":"2022-08-12T04:40:00.254726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get all categorical columns\ncat_columns = df.select_dtypes(['object']).columns\n\n#convert all categorical columns to numeric\ndf[cat_columns] = df[cat_columns].apply(lambda x: pd.factorize(x)[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:40:13.859617Z","iopub.execute_input":"2022-08-12T04:40:13.860020Z","iopub.status.idle":"2022-08-12T04:40:13.883521Z","shell.execute_reply.started":"2022-08-12T04:40:13.859988Z","shell.execute_reply":"2022-08-12T04:40:13.882375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"All = df.shape[0]\nfailed = df[df['failure'] == 1]\nnonFailed = df[df['failure'] == 0]\n\nx = len(failed)/All\ny = len(nonFailed)/All\n\nprint('failed :',x*100,'%')\nprint('non Failed :',y*100,'%')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:40:19.530731Z","iopub.execute_input":"2022-08-12T04:40:19.531159Z","iopub.status.idle":"2022-08-12T04:40:19.548232Z","shell.execute_reply.started":"2022-08-12T04:40:19.531123Z","shell.execute_reply":"2022-08-12T04:40:19.547056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Let's plot the Transaction class against the Frequency\nlabels = ['non Failed','failed']\nclasses = pd.value_counts(df['failure'], sort = True)\nclasses.plot(kind = 'bar', rot=0)\nplt.title(\"Transaction failure distribution\")\nplt.xticks(range(2), labels)\nplt.xlabel(\"failure\")\nplt.ylabel(\"Frequency\")","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:40:29.278316Z","iopub.execute_input":"2022-08-12T04:40:29.279386Z","iopub.status.idle":"2022-08-12T04:40:29.491342Z","shell.execute_reply.started":"2022-08-12T04:40:29.279341Z","shell.execute_reply":"2022-08-12T04:40:29.489993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# heat map of correlation of features\ncorrelation_matrix = df.corr()\nfig = plt.figure(figsize=(12,9))\nsns.heatmap(correlation_matrix,vmax=0.8,square = True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:40:37.472728Z","iopub.execute_input":"2022-08-12T04:40:37.473161Z","iopub.status.idle":"2022-08-12T04:40:38.261032Z","shell.execute_reply.started":"2022-08-12T04:40:37.473120Z","shell.execute_reply":"2022-08-12T04:40:38.259362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate features and labels\nfeatures = ['attribute_2', 'loading', 'measurement_0', 'measurement_1','measurement_2', 'measurement_3', \n            'measurement_4', 'measurement_5', 'measurement_6', 'measurement_7', \n            'measurement_8', 'measurement_9', 'measurement_10', 'measurement_11', \n            'measurement_12', 'measurement_13', 'measurement_14', 'measurement_15', \n            'measurement_16', 'measurement_17']\nlabel = 'failure'\n\nX, y = df[features].values, df[label].values","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:40:47.848297Z","iopub.execute_input":"2022-08-12T04:40:47.848724Z","iopub.status.idle":"2022-08-12T04:40:47.858859Z","shell.execute_reply.started":"2022-08-12T04:40:47.848690Z","shell.execute_reply":"2022-08-12T04:40:47.857805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Split data 70%-30% into training set and test set\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.30, random_state=0)\n\nprint ('Training cases: %d\\nTest cases: %d' % (X_train.shape[0], X_test.shape[0]))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:40:56.155666Z","iopub.execute_input":"2022-08-12T04:40:56.156096Z","iopub.status.idle":"2022-08-12T04:40:56.174062Z","shell.execute_reply.started":"2022-08-12T04:40:56.156059Z","shell.execute_reply":"2022-08-12T04:40:56.172640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.linear_model import LogisticRegression\nimport numpy as np\n\n# Define preprocessing for numeric columns (normalize them so they're on the same scale)\nnum_list = [0,1,2,3,4,5,6]\nnumeric_transformer = Pipeline(steps=[\n    ('scaler', StandardScaler())])\n\n# Define preprocessing for categorical features\ncat_list = [2]\ncategorical_transformer = Pipeline(steps=[\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))])\n\n# Combine preprocessing steps\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, num_list),\n        ('cat', categorical_transformer, cat_list)])\n\n\n# Create preprocessing and training pipeline\nreg = 0.01\npipeline = Pipeline(steps=[('preprocessor', preprocessor),\n                           ('logregressor', LogisticRegression(C=1/reg, solver=\"liblinear\"))])\n\n# fit the pipeline to train a random forest model on the training set\nmodel = pipeline.fit(X_train, (y_train))\nprint (model)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:41:02.122840Z","iopub.execute_input":"2022-08-12T04:41:02.123270Z","iopub.status.idle":"2022-08-12T04:41:02.354632Z","shell.execute_reply.started":"2022-08-12T04:41:02.123234Z","shell.execute_reply":"2022-08-12T04:41:02.353378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_score, recall_score\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import classification_report \nfrom sklearn.metrics import confusion_matrix \n\npredictions = model.predict(X_test)\ny_scores = model.predict_proba(X_test)\ncm = confusion_matrix(y_test, predictions)\nprint ('Confusion Matrix:\\n',cm, '\\n')\nprint('Accuracy:', accuracy_score(y_test, predictions))\nprint(\"Overall Precision:\",precision_score(y_test, predictions))\nprint(\"Overall Recall:\",recall_score(y_test, predictions))\nauc = roc_auc_score(y_test,y_scores[:,1])\nprint('\\nAUC: ' + str(auc))\n\n# calculate ROC curve\nfpr, tpr, thresholds = roc_curve(y_test, y_scores[:,1])\n\n# plot ROC curve\nfig = plt.figure(figsize=(6, 6))\n# Plot the diagonal 50% line\nplt.plot([0, 1], [0, 1], 'k--')\n# Plot the FPR and TPR achieved by our model\nplt.plot(fpr, tpr)\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:41:11.706040Z","iopub.execute_input":"2022-08-12T04:41:11.706478Z","iopub.status.idle":"2022-08-12T04:41:11.963857Z","shell.execute_reply.started":"2022-08-12T04:41:11.706442Z","shell.execute_reply":"2022-08-12T04:41:11.962549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv', index_col='id')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:41:22.046334Z","iopub.execute_input":"2022-08-12T04:41:22.046765Z","iopub.status.idle":"2022-08-12T04:41:22.158766Z","shell.execute_reply.started":"2022-08-12T04:41:22.046731Z","shell.execute_reply":"2022-08-12T04:41:22.157904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test.drop(['product_code', 'attribute_0', 'attribute_1', 'attribute_3'], axis=1)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:41:29.089351Z","iopub.execute_input":"2022-08-12T04:41:29.090206Z","iopub.status.idle":"2022-08-12T04:41:29.127064Z","shell.execute_reply.started":"2022-08-12T04:41:29.090150Z","shell.execute_reply":"2022-08-12T04:41:29.126201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imputing Missing Values\nfrom sklearn.base import TransformerMixin\nclass DataFrameImputer(TransformerMixin):\n\n    def __init__(self):\n        \"\"\"Columns of dtype object are imputed with the most frequent value in column. Columns of other types are imputed with mean of column.\"\"\"\n    def fit(self, X, y=None):\n        self.fill = pd.Series([X[c].value_counts().index[0]\n            if X[c].dtype == np.dtype('O') else X[c].mean() for c in \n            X],index=X.columns)\n        return self\n    def transform(self, X, y=None):\n        return X.fillna(self.fill)\nX = pd.DataFrame(test) \ntest = DataFrameImputer().fit_transform(X)\ntest.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:41:37.475261Z","iopub.execute_input":"2022-08-12T04:41:37.475694Z","iopub.status.idle":"2022-08-12T04:41:37.503520Z","shell.execute_reply.started":"2022-08-12T04:41:37.475658Z","shell.execute_reply":"2022-08-12T04:41:37.502277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scale_test = StandardScaler() \nscale_test.fit(test)\ntest_features=scale_test.transform(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:41:54.343338Z","iopub.execute_input":"2022-08-12T04:41:54.344252Z","iopub.status.idle":"2022-08-12T04:41:54.360014Z","shell.execute_reply.started":"2022-08-12T04:41:54.344209Z","shell.execute_reply":"2022-08-12T04:41:54.358762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred = model.predict(test_features)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:47:05.512475Z","iopub.execute_input":"2022-08-12T04:47:05.513335Z","iopub.status.idle":"2022-08-12T04:47:05.531617Z","shell.execute_reply.started":"2022-08-12T04:47:05.513294Z","shell.execute_reply":"2022-08-12T04:47:05.530706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['failure'] = test_pred","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:47:09.089096Z","iopub.execute_input":"2022-08-12T04:47:09.090291Z","iopub.status.idle":"2022-08-12T04:47:09.096441Z","shell.execute_reply.started":"2022-08-12T04:47:09.090245Z","shell.execute_reply":"2022-08-12T04:47:09.095424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'id':test.index, 'failure': test['failure']})\nsubmission.to_csv('subKT1.csv', index=False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-08-12T04:47:39.078610Z","iopub.execute_input":"2022-08-12T04:47:39.079044Z","iopub.status.idle":"2022-08-12T04:47:39.116600Z","shell.execute_reply.started":"2022-08-12T04:47:39.079008Z","shell.execute_reply":"2022-08-12T04:47:39.115335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn. metrics import classification_report\n\nprint(classification_report(y_test, predictions))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:14:56.494801Z","iopub.execute_input":"2022-08-12T02:14:56.495147Z","iopub.status.idle":"2022-08-12T02:14:56.517021Z","shell.execute_reply.started":"2022-08-12T02:14:56.495117Z","shell.execute_reply":"2022-08-12T02:14:56.515720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier \nfrom sklearn.tree import DecisionTreeClassifier \nfrom sklearn.ensemble import RandomForestClassifier \nfrom sklearn.naive_bayes import GaussianNB\n\nmodel_pipeline = []\nmodel_pipeline.append(LogisticRegression(solver='liblinear'))\nmodel_pipeline.append(SVC())\nmodel_pipeline.append(KNeighborsClassifier())\nmodel_pipeline.append(DecisionTreeClassifier())\nmodel_pipeline.append(RandomForestClassifier())\nmodel_pipeline.append(GaussianNB())","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:14:56.518611Z","iopub.execute_input":"2022-08-12T02:14:56.518993Z","iopub.status.idle":"2022-08-12T02:14:56.683909Z","shell.execute_reply.started":"2022-08-12T02:14:56.518958Z","shell.execute_reply":"2022-08-12T02:14:56.682830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics \nfrom sklearn.metrics import classification_report \nfrom sklearn.metrics import confusion_matrix \n\nmodel_list = ['LogisticRegression', 'SVC', 'KNN', 'Decision Tree', 'Random Forest', 'Naive Bayes']\nacc_list = []\nauc_list = []\ncm_list = []\n\nfor model in model_pipeline:\n    model.fit(X_train, y_train)\n    y_pred = model.predict(X_test)\n    acc_list.append(metrics.accuracy_score(y_test, y_pred))\n    fpr, tpr, _thresholds = metrics.roc_curve(y_test, y_pred)\n    auc_list.append(round(metrics.auc(fpr, tpr),2))\n    cm_list.append(confusion_matrix(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:14:56.685635Z","iopub.execute_input":"2022-08-12T02:14:56.686030Z","iopub.status.idle":"2022-08-12T02:15:25.542745Z","shell.execute_reply.started":"2022-08-12T02:14:56.685999Z","shell.execute_reply":"2022-08-12T02:15:25.541344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot confusion matrix \nfig = plt.figure(figsize = (18,10))\nfor i in range(len(cm_list)):\n    cm = cm_list[i]\n    model = model_list[i]\n    sub = fig.add_subplot(2, 3, i+1).set_title(model)\n    cm_plot = sns.heatmap(cm, annot=True, cmap=\"Blues_r\")\n    cm_plot.set_xlabel(\"Predicted Value\")\n    cm_plot.set_ylabel(\"Actual Value\")","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:25.544232Z","iopub.execute_input":"2022-08-12T02:15:25.544635Z","iopub.status.idle":"2022-08-12T02:15:27.011448Z","shell.execute_reply.started":"2022-08-12T02:15:25.544592Z","shell.execute_reply":"2022-08-12T02:15:27.010149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df = pd.DataFrame({\"Model\":model_list, \"Accuracy\":acc_list, \"AUC\":auc_list})\nresult_df","metadata":{"execution":{"iopub.status.busy":"2022-08-12T02:15:27.013141Z","iopub.execute_input":"2022-08-12T02:15:27.014343Z","iopub.status.idle":"2022-08-12T02:15:27.028982Z","shell.execute_reply.started":"2022-08-12T02:15:27.014301Z","shell.execute_reply":"2022-08-12T02:15:27.027895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}