{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-14T03:08:49.296122Z","iopub.execute_input":"2022-08-14T03:08:49.296603Z","iopub.status.idle":"2022-08-14T03:08:49.328532Z","shell.execute_reply.started":"2022-08-14T03:08:49.296504Z","shell.execute_reply":"2022-08-14T03:08:49.327739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reference\nhttps://www.kaggle.com/code/cabaxiom/tps-aug-22-eda-logistic-regression-baseline\n\nhttps://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\n\nfrom scipy import stats\nfrom colorama import Fore, Back, Style\n\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import KNNImputer, SimpleImputer, IterativeImputer\nfrom sklearn.preprocessing import OneHotEncoder, OrdinalEncoder, StandardScaler\nfrom sklearn.pipeline import make_pipeline\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import LogisticRegression\n\nfrom sklearn.model_selection import GroupKFold, GridSearchCV, StratifiedGroupKFold\nfrom sklearn.preprocessing import StandardScaler, PowerTransformer\nfrom sklearn.metrics import roc_auc_score, accuracy_score, roc_curve","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:49.330094Z","iopub.execute_input":"2022-08-14T03:08:49.330965Z","iopub.status.idle":"2022-08-14T03:08:50.852711Z","shell.execute_reply.started":"2022-08-14T03:08:49.330930Z","shell.execute_reply":"2022-08-14T03:08:50.851435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv', index_col='id')\ndf_test = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:50.855615Z","iopub.execute_input":"2022-08-14T03:08:50.856128Z","iopub.status.idle":"2022-08-14T03:08:51.138642Z","shell.execute_reply.started":"2022-08-14T03:08:50.856083Z","shell.execute_reply":"2022-08-14T03:08:51.137146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:51.140852Z","iopub.execute_input":"2022-08-14T03:08:51.141803Z","iopub.status.idle":"2022-08-14T03:08:51.213860Z","shell.execute_reply.started":"2022-08-14T03:08:51.141754Z","shell.execute_reply":"2022-08-14T03:08:51.211902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:51.217732Z","iopub.execute_input":"2022-08-14T03:08:51.218395Z","iopub.status.idle":"2022-08-14T03:08:51.262096Z","shell.execute_reply.started":"2022-08-14T03:08:51.218359Z","shell.execute_reply":"2022-08-14T03:08:51.260764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The samples of train and test are similar","metadata":{}},{"cell_type":"code","source":"df_train.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:51.263731Z","iopub.execute_input":"2022-08-14T03:08:51.264460Z","iopub.status.idle":"2022-08-14T03:08:51.396207Z","shell.execute_reply.started":"2022-08-14T03:08:51.264414Z","shell.execute_reply":"2022-08-14T03:08:51.395050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:51.397977Z","iopub.execute_input":"2022-08-14T03:08:51.398731Z","iopub.status.idle":"2022-08-14T03:08:51.423406Z","shell.execute_reply.started":"2022-08-14T03:08:51.398661Z","shell.execute_reply":"2022-08-14T03:08:51.421949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have *3 category* type,*6 integer* type and others are float type","metadata":{}},{"cell_type":"markdown","source":"# Missing Data","metadata":{}},{"cell_type":"code","source":"df_train_miss = df_train.isna().sum()\ndf_test_miss = df_test.isna().sum()\n\ndf_miss = pd.concat([df_train_miss, df_test_miss], keys=['train', 'test'], axis=0).reset_index()\ndf_miss.columns = ['data', 'columns', 'missing']\n\nplt.figure(figsize=(10, 12))\nsns.barplot(data=df_miss,y='columns', x='missing', hue='data')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:51.427013Z","iopub.execute_input":"2022-08-14T03:08:51.427806Z","iopub.status.idle":"2022-08-14T03:08:52.238221Z","shell.execute_reply.started":"2022-08-14T03:08:51.427757Z","shell.execute_reply":"2022-08-14T03:08:52.236917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"float columns have missing data","metadata":{}},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"## Target columns (failure)","metadata":{}},{"cell_type":"code","source":"def value_count(df,  column):\n    total = df[column].value_counts().sum()\n    count = df[column].value_counts()\n    percent = count / total * 100\n    \n    df_count = pd.concat([count, percent], keys=['value count', 'percentage'], axis=1).reset_index().rename(columns={'index': column})\n    return df_count\n\n\ndef plot_value_count(df, column):\n    df_count = value_count(df, column)\n    display(df_count)\n    \n    df_count.set_index(column).plot.pie(y='value count', figsize=(5,5))\n    \n\ndef plot_and_compare_value_count(train, test, column):\n    df_count1 = value_count(train, column).rename(columns={'value count': 'train value count',\n                                                          'percentage': 'train percentage'})\n    df_count2 = value_count(test, column).rename(columns={'value count': 'test value count',\n                                                         'percentage': 'test percentage'})\n    \n#     df_compare = pd.concat([df_count1, df_count2], ignore_index=True).sort_values(column)\n    df_compare = pd.merge(df_count1, df_count2, on=column, how='outer').sort_values(column)\n    df_compare.reset_index(drop=True, inplace=True)\n\n    display(df_compare)\n\n    fix, axs = plt.subplots(1,2, figsize=(10,10))\n    df_compare.set_index(column).plot.pie(y='train value count', ax=axs[0], title='train', ylabel='', legend=False)\n    df_compare.set_index(column).plot.pie(y='test value count', ax=axs[1], title='test', ylabel='', legend=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:52.239957Z","iopub.execute_input":"2022-08-14T03:08:52.240345Z","iopub.status.idle":"2022-08-14T03:08:52.252930Z","shell.execute_reply.started":"2022-08-14T03:08:52.240312Z","shell.execute_reply":"2022-08-14T03:08:52.251604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_value_count(df_train, 'failure')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:52.254619Z","iopub.execute_input":"2022-08-14T03:08:52.255870Z","iopub.status.idle":"2022-08-14T03:08:52.496144Z","shell.execute_reply.started":"2022-08-14T03:08:52.255819Z","shell.execute_reply":"2022-08-14T03:08:52.494553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- try to apply Stratified fold for the imbalanced dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:55:19.964730Z","iopub.execute_input":"2022-08-06T08:55:19.965714Z","iopub.status.idle":"2022-08-06T08:55:20.158288Z","shell.execute_reply.started":"2022-08-06T08:55:19.965659Z","shell.execute_reply":"2022-08-06T08:55:20.157002Z"}}},{"cell_type":"markdown","source":"## Categorical columns","metadata":{}},{"cell_type":"code","source":"plot_and_compare_value_count(df_train, df_test, 'product_code')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:52.498524Z","iopub.execute_input":"2022-08-14T03:08:52.499437Z","iopub.status.idle":"2022-08-14T03:08:52.785169Z","shell.execute_reply.started":"2022-08-14T03:08:52.499377Z","shell.execute_reply":"2022-08-14T03:08:52.783522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- product codes are disjoint in train and test\n- similar samples for each product code","metadata":{}},{"cell_type":"code","source":"plot_and_compare_value_count(df_train, df_test, 'attribute_0')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:52.792814Z","iopub.execute_input":"2022-08-14T03:08:52.794069Z","iopub.status.idle":"2022-08-14T03:08:53.031315Z","shell.execute_reply.started":"2022-08-14T03:08:52.794000Z","shell.execute_reply":"2022-08-14T03:08:53.029537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_and_compare_value_count(df_train, df_test, 'attribute_1')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:53.033397Z","iopub.execute_input":"2022-08-14T03:08:53.034893Z","iopub.status.idle":"2022-08-14T03:08:53.331923Z","shell.execute_reply.started":"2022-08-14T03:08:53.034829Z","shell.execute_reply":"2022-08-14T03:08:53.330232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- material_7 only exists in test data\n- material_8 only exists in train data\n- material_7 and material_8 are disjoint in train and test","metadata":{}},{"cell_type":"code","source":"plot_and_compare_value_count(df_train, df_test, 'attribute_2')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:53.334345Z","iopub.execute_input":"2022-08-14T03:08:53.335371Z","iopub.status.idle":"2022-08-14T03:08:53.566551Z","shell.execute_reply.started":"2022-08-14T03:08:53.335306Z","shell.execute_reply":"2022-08-14T03:08:53.565036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 5 and 8 are only exists in train data\n- 7 is only exists in test data","metadata":{}},{"cell_type":"code","source":"plot_and_compare_value_count(df_train, df_test, 'attribute_3')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:53.568542Z","iopub.execute_input":"2022-08-14T03:08:53.569260Z","iopub.status.idle":"2022-08-14T03:08:53.816404Z","shell.execute_reply.started":"2022-08-14T03:08:53.569213Z","shell.execute_reply":"2022-08-14T03:08:53.814745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- 4 and 7 are only exits in test data\n- 6 and 8 are only exists in train data","metadata":{}},{"cell_type":"markdown","source":"## Numerical columns","metadata":{}},{"cell_type":"markdown","source":"## Integer columns","metadata":{}},{"cell_type":"code","source":"integer_features = ['measurement_0', 'measurement_1', 'measurement_2']\n\ndf_train['data'] = 'train'\ndf_test['data'] = 'test'\ndf_combine = pd.concat([df_train, df_test]).reset_index()\n\nplt.subplots(figsize=(25,25))\nfor i, col in enumerate(integer_features):\n    plt.subplot(5,5, i + 1)\n    sns.histplot(data=df_combine, x=col, hue='data', kde=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:53.818386Z","iopub.execute_input":"2022-08-14T03:08:53.819130Z","iopub.status.idle":"2022-08-14T03:08:57.240619Z","shell.execute_reply.started":"2022-08-14T03:08:53.819086Z","shell.execute_reply":"2022-08-14T03:08:57.239223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- it seems the distributions are no normal\n- all integer features look like postive skewed","metadata":{}},{"cell_type":"markdown","source":"## Float columns","metadata":{}},{"cell_type":"code","source":"float_features = [col for col in df_train.columns if 'float' in str(type(df_train[col].dtype))]\n\nplt.subplots(figsize=(25,25))\nfor i, col in enumerate(float_features):\n    plt.subplot(5,5, i + 1)\n    sns.histplot(data=df_combine, x=col, hue='data', kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:08:57.242670Z","iopub.execute_input":"2022-08-14T03:08:57.243756Z","iopub.status.idle":"2022-08-14T03:09:14.742075Z","shell.execute_reply.started":"2022-08-14T03:08:57.243687Z","shell.execute_reply":"2022-08-14T03:09:14.741209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- the loading feature look like postive skewd\n- all float features look like normally distributed","metadata":{}},{"cell_type":"markdown","source":"## Fix \"loading\" feature skew","metadata":{}},{"cell_type":"code","source":"df_train['loading'] = np.log(df_train['loading'])\ndf_test['loading'] = np.log(df_test['loading'])\n\nsns.histplot(data=df_train, x='loading')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:09:14.743308Z","iopub.execute_input":"2022-08-14T03:09:14.743841Z","iopub.status.idle":"2022-08-14T03:09:15.047888Z","shell.execute_reply.started":"2022-08-14T03:09:14.743807Z","shell.execute_reply":"2022-08-14T03:09:15.046672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Nomality test","metadata":{}},{"cell_type":"code","source":"for col in (integer_features + float_features):\n    stat, p = stats.shapiro(df_train[col])\n    if p > 0.05:\n        print(f'{col}: follow normal distribution test:' + Fore.GREEN + 'Accepted', Style.RESET_ALL)\n    else:\n        print(f'{col}: not follow normal distribution test:' + Fore.RED + 'Rejected', Style.RESET_ALL)\n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:09:15.049463Z","iopub.execute_input":"2022-08-14T03:09:15.050385Z","iopub.status.idle":"2022-08-14T03:09:15.108550Z","shell.execute_reply.started":"2022-08-14T03:09:15.050348Z","shell.execute_reply":"2022-08-14T03:09:15.107156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlations","metadata":{}},{"cell_type":"code","source":"plt.subplots(figsize=(15,15))\nsns.heatmap(df_train[integer_features + float_features].corr(), annot=True, fmt='.2f')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:09:15.110102Z","iopub.execute_input":"2022-08-14T03:09:15.110440Z","iopub.status.idle":"2022-08-14T03:09:16.879229Z","shell.execute_reply.started":"2022-08-14T03:09:15.110411Z","shell.execute_reply":"2022-08-14T03:09:16.878318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Encode Categorical Features","metadata":{}},{"cell_type":"code","source":"train = df_train.drop(columns=['data'])\ntest = df_test.drop(columns=['data'])\n\ncat_features = ['product_code', 'attribute_0', 'attribute_1']\n\nencoder = OrdinalEncoder()\ntrain[cat_features] = encoder.fit_transform(train[cat_features])\ntest[cat_features] = encoder.fit_transform(test[cat_features])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:09:16.880511Z","iopub.execute_input":"2022-08-14T03:09:16.881059Z","iopub.status.idle":"2022-08-14T03:09:16.940011Z","shell.execute_reply.started":"2022-08-14T03:09:16.881024Z","shell.execute_reply":"2022-08-14T03:09:16.939094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Handle Missing Data","metadata":{}},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=3)\ntrain[float_features] = imputer.fit_transform(train[float_features])\ntest[float_features] = imputer.transform(test[float_features])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:09:16.941377Z","iopub.execute_input":"2022-08-14T03:09:16.942346Z","iopub.status.idle":"2022-08-14T03:10:06.169319Z","shell.execute_reply.started":"2022-08-14T03:09:16.942311Z","shell.execute_reply":"2022-08-14T03:10:06.168129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('train null count:', train.isna().sum().sum())\nprint('test null count:', test.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:10:06.170856Z","iopub.execute_input":"2022-08-14T03:10:06.171189Z","iopub.status.idle":"2022-08-14T03:10:06.185485Z","shell.execute_reply.started":"2022-08-14T03:10:06.171159Z","shell.execute_reply":"2022-08-14T03:10:06.184355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"# from https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense\n\nauc_list = []\ntest_pred_list = []\nimportance_list = []\n\nkf = StratifiedGroupKFold(n_splits=5) # train data have 5 product codes\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(train, train.failure, train.product_code)):\n    X_train = train.iloc[idx_tr][test.columns]\n    X_valid = train.iloc[idx_va][test.columns]\n    X_test = test.copy()\n    y_train = train.iloc[idx_tr].failure\n    y_valid = train.iloc[idx_va].failure\n    \n    cols = [col for col in X_train.columns if col != 'product_code']\n    model = make_pipeline(StandardScaler(),\n                         LogisticRegression(penalty='l1', C=0.011, solver='liblinear', random_state=1))\n    model.fit(X_train[cols], y_train)\n    importance_list.append(model.named_steps['logisticregression'].coef_.ravel())\n    \n    y_valid_pred = model.predict_proba(X_valid[cols])[:, 1]\n    score = roc_auc_score(y_valid, y_valid_pred)\n    print(f\"Fold {fold}: auc = {score:.5f}\")\n    auc_list.append(score)\n    \n    test_pred_list.append(model.predict_proba(X_test[cols])[:,1])\n    \n# Show overall score\nprint(f\"{Fore.GREEN}{Style.BRIGHT}Average auc = {sum(auc_list) / len(auc_list):.5f}{Style.RESET_ALL}\")\n\n# Show feature importances\ndf_importance = pd.DataFrame(np.array(importance_list).T, index=cols)\ndf_importance['mean'] = df_importance.mean(axis=1).abs()\ndf_importance['feature'] = cols\ndf_importance = df_importance.sort_values('mean', ascending=False).reset_index().head(10) # select top 10 features\n\ndisplay(df_importance)\n\nsns.barplot(data=df_importance,y='feature', x='mean', color='lightgreen')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:10:06.187296Z","iopub.execute_input":"2022-08-14T03:10:06.187784Z","iopub.status.idle":"2022-08-14T03:10:07.217851Z","shell.execute_reply.started":"2022-08-14T03:10:06.187741Z","shell.execute_reply":"2022-08-14T03:10:07.216754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Select Most importance features","metadata":{}},{"cell_type":"code","source":"# from https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense\ntop_features = df_importance['feature'].to_list()\n\nkf = StratifiedGroupKFold(n_splits=5) # train data have 5 product codes\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(train, train.failure, train.product_code)):\n    X_train = train.iloc[idx_tr][test.columns]\n    X_valid = train.iloc[idx_va][test.columns]\n    X_test = test.copy()\n    y_train = train.iloc[idx_tr].failure\n    y_valid = train.iloc[idx_va].failure\n    \n    cols = top_features\n    model = make_pipeline(StandardScaler(),\n                         LogisticRegression(penalty='l1', C=0.011, solver='liblinear', random_state=1))\n    model.fit(X_train[cols], y_train)\n    importance_list.append(model.named_steps['logisticregression'].coef_.ravel())\n    \n    y_valid_pred = model.predict_proba(X_valid[cols])[:, 1]\n    score = roc_auc_score(y_valid, y_valid_pred)\n    print(f\"Fold {fold}: auc = {score:.5f}\")\n    auc_list.append(score)\n    \n    test_pred_list.append(model.predict_proba(X_test[cols])[:,1])\n    \n# Show overall score\nprint(f\"{Fore.GREEN}{Style.BRIGHT}Average auc = {sum(auc_list) / len(auc_list):.5f}{Style.RESET_ALL}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:10:07.219307Z","iopub.execute_input":"2022-08-14T03:10:07.219663Z","iopub.status.idle":"2022-08-14T03:10:07.801441Z","shell.execute_reply.started":"2022-08-14T03:10:07.219632Z","shell.execute_reply":"2022-08-14T03:10:07.799880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense\n\nplt.figure(figsize=(5, 5))\nfpr, tpr, _ = roc_curve(y_valid, y_valid_pred)\nplt.plot(fpr, tpr, color='#c00000', lw=3, label=f\"(auc (fold 4) = {roc_auc_score(y_valid, y_valid_pred):.5f})\") # curve\nplt.fill_between(fpr, tpr, color='#ffc0c0') # area under the curve\n\nplt.plot([0, 1], [0, 1], color=\"navy\", lw=1, linestyle=\"--\") # diagonal\nplt.gca().set_aspect('equal')\n\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.0])\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\n\nplt.title(\"Receiver operating characteristic\")\nplt.legend(loc=\"lower right\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:10:07.803855Z","iopub.execute_input":"2022-08-14T03:10:07.805207Z","iopub.status.idle":"2022-08-14T03:10:08.058603Z","shell.execute_reply.started":"2022-08-14T03:10:07.805140Z","shell.execute_reply":"2022-08-14T03:10:08.057759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_submit = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\n\npred = sum(test_pred_list)/5\ndf_submit['failure'] = pred\n\ndf_submit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:10:08.060012Z","iopub.execute_input":"2022-08-14T03:10:08.060585Z","iopub.status.idle":"2022-08-14T03:10:08.126901Z","shell.execute_reply.started":"2022-08-14T03:10:08.060552Z","shell.execute_reply":"2022-08-14T03:10:08.125753Z"},"trusted":true},"execution_count":null,"outputs":[]}]}