{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-14T03:25:17.760969Z","iopub.execute_input":"2022-08-14T03:25:17.762249Z","iopub.status.idle":"2022-08-14T03:25:17.775474Z","shell.execute_reply.started":"2022-08-14T03:25:17.762204Z","shell.execute_reply":"2022-08-14T03:25:17.774257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\n\nfrom scipy import stats\nfrom colorama import Fore, Back, Style\n\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import KNNImputer, SimpleImputer, IterativeImputer\nfrom sklearn.preprocessing import OneHotEncoder, OrdinalEncoder, StandardScaler\nfrom sklearn.pipeline import make_pipeline\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import LogisticRegression\n\nfrom sklearn.model_selection import GroupKFold, GridSearchCV, StratifiedGroupKFold, StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler, PowerTransformer\nfrom sklearn.metrics import roc_auc_score, accuracy_score, roc_curve","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:25:17.777686Z","iopub.execute_input":"2022-08-14T03:25:17.778389Z","iopub.status.idle":"2022-08-14T03:25:17.785872Z","shell.execute_reply.started":"2022-08-14T03:25:17.778353Z","shell.execute_reply":"2022-08-14T03:25:17.784988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For the EDA, you can check my previous [notebook](https://www.kaggle.com/code/hugolearn/tps-aug-practice)","metadata":{}},{"cell_type":"markdown","source":"# Reference\nhttps://www.kaggle.com/code/samuelcortinhas/tps-aug-22-failure-prediction\n\nhttps://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense","metadata":{}},{"cell_type":"markdown","source":"# Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv', index_col='id')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:25:17.787444Z","iopub.execute_input":"2022-08-14T03:25:17.787878Z","iopub.status.idle":"2022-08-14T03:25:17.986043Z","shell.execute_reply.started":"2022-08-14T03:25:17.787847Z","shell.execute_reply":"2022-08-14T03:25:17.984898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from https://www.kaggle.com/code/samuelcortinhas/tps-aug-22-failure-prediction\npd.concat([train,train])[['product_code','attribute_0','attribute_1','attribute_2','attribute_3']].drop_duplicates().set_index('product_code')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:25:17.988295Z","iopub.execute_input":"2022-08-14T03:25:17.988662Z","iopub.status.idle":"2022-08-14T03:25:18.042481Z","shell.execute_reply.started":"2022-08-14T03:25:17.988628Z","shell.execute_reply":"2022-08-14T03:25:18.041518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Thoese data are unique in tran and test dataset\n- We drop them","metadata":{}},{"cell_type":"code","source":"train['loading'] = np.log(train['loading'])\ntest['loading'] = np.log(test['loading'])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:25:18.043831Z","iopub.execute_input":"2022-08-14T03:25:18.044353Z","iopub.status.idle":"2022-08-14T03:25:18.051866Z","shell.execute_reply.started":"2022-08-14T03:25:18.044319Z","shell.execute_reply":"2022-08-14T03:25:18.050785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encoder category value\ncat_features = ['product_code', 'attribute_0', 'attribute_1']\n\nencoder = OrdinalEncoder()\ntrain[cat_features] = encoder.fit_transform(train[cat_features])\ntest[cat_features] = encoder.fit_transform(test[cat_features])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:25:18.053977Z","iopub.execute_input":"2022-08-14T03:25:18.054939Z","iopub.status.idle":"2022-08-14T03:25:18.112564Z","shell.execute_reply.started":"2022-08-14T03:25:18.054902Z","shell.execute_reply":"2022-08-14T03:25:18.111517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fix missing data\nfloat_features = [col for col in train.columns if 'float' in str(type(train[col].dtype))]\n\nimputer = KNNImputer(n_neighbors=3)\ntrain[float_features] = imputer.fit_transform(train[float_features])\ntest[float_features] = imputer.transform(test[float_features])\n\nprint('train null count:', train.isna().sum().sum())\nprint('test null count:', test.isna().sum().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:25:18.113719Z","iopub.execute_input":"2022-08-14T03:25:18.114603Z","iopub.status.idle":"2022-08-14T03:26:13.152098Z","shell.execute_reply.started":"2022-08-14T03:25:18.114567Z","shell.execute_reply":"2022-08-14T03:26:13.150862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\ntrain[float_features] = scaler.fit_transform(train[float_features])\ntest[float_features] = scaler.fit_transform(test[float_features])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:26:13.155045Z","iopub.execute_input":"2022-08-14T03:26:13.155407Z","iopub.status.idle":"2022-08-14T03:26:13.187219Z","shell.execute_reply.started":"2022-08-14T03:26:13.155376Z","shell.execute_reply":"2022-08-14T03:26:13.185555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"train = train.drop(columns=['product_code','attribute_0','attribute_1','attribute_2','attribute_3'])\ntest = test.drop(columns=['product_code','attribute_0','attribute_1','attribute_2','attribute_3'])\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:26:13.189463Z","iopub.execute_input":"2022-08-14T03:26:13.189995Z","iopub.status.idle":"2022-08-14T03:26:13.220752Z","shell.execute_reply.started":"2022-08-14T03:26:13.189944Z","shell.execute_reply":"2022-08-14T03:26:13.219387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense\n\nauc_list = []\ntest_pred_list = []\nimportance_list = []\n\nkf = StratifiedKFold(n_splits=5)\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(train, train.failure)):\n    X_train = train.iloc[idx_tr][test.columns]\n    X_valid = train.iloc[idx_va][test.columns]\n    X_test = test.copy()\n    y_train = train.iloc[idx_tr].failure\n    y_valid = train.iloc[idx_va].failure\n    \n    cols = [col for col in X_train.columns if col != 'product_code']\n    model =  LogisticRegression(penalty='l1', C=0.011, solver='liblinear', random_state=1)\n    model.fit(X_train[cols], y_train)\n    importance_list.append(model.coef_.ravel())\n    \n    y_valid_pred = model.predict_proba(X_valid[cols])[:, 1]\n    score = roc_auc_score(y_valid, y_valid_pred)\n    print(f\"Fold {fold}: auc = {score:.5f}\")\n    auc_list.append(score)\n    \n    test_pred_list.append(model.predict_proba(X_test[cols])[:,1])\n    \n# Show overall score\nprint(f\"{Fore.GREEN}{Style.BRIGHT}Average auc = {sum(auc_list) / len(auc_list):.5f}{Style.RESET_ALL}\")\n\n# Show feature importances\ndf_importance = pd.DataFrame(np.array(importance_list).T, index=cols)\ndf_importance['mean'] = df_importance.mean(axis=1).abs()\ndf_importance['feature'] = cols\ndf_importance = df_importance.sort_values('mean', ascending=False).reset_index().head(10) # select top 10 features\n\ndisplay(df_importance)\n\nsns.barplot(data=df_importance,y='feature', x='mean', color='lightgreen')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:26:13.222231Z","iopub.execute_input":"2022-08-14T03:26:13.222951Z","iopub.status.idle":"2022-08-14T03:26:14.372766Z","shell.execute_reply.started":"2022-08-14T03:26:13.222915Z","shell.execute_reply":"2022-08-14T03:26:14.371936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # from https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense\n# top_features = df_importance['feature'].to_list()\n\n# kf = StratifiedKFold(n_splits=5)\n# for fold, (idx_tr, idx_va) in enumerate(kf.split(train, train.failure)):\n#     X_train = train.iloc[idx_tr][test.columns]\n#     X_valid = train.iloc[idx_va][test.columns]\n#     X_test = test.copy()\n#     y_train = train.iloc[idx_tr].failure\n#     y_valid = train.iloc[idx_va].failure\n    \n#     cols = top_features\n#     model =  LogisticRegression(penalty='l1', C=0.011, solver='liblinear', random_state=1)\n#     model.fit(X_train[cols], y_train)\n#     importance_list.append(model.coef_.ravel())\n    \n#     y_valid_pred = model.predict_proba(X_valid[cols])[:, 1]\n#     score = roc_auc_score(y_valid, y_valid_pred)\n#     print(f\"Fold {fold}: auc = {score:.5f}\")\n#     auc_list.append(score)\n    \n#     test_pred_list.append(model.predict_proba(X_test[cols])[:,1])\n    \n# # Show overall score\n# print(f\"{Fore.GREEN}{Style.BRIGHT}Average auc = {sum(auc_list) / len(auc_list):.5f}{Style.RESET_ALL}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:26:14.373844Z","iopub.execute_input":"2022-08-14T03:26:14.374668Z","iopub.status.idle":"2022-08-14T03:26:14.381188Z","shell.execute_reply.started":"2022-08-14T03:26:14.374632Z","shell.execute_reply":"2022-08-14T03:26:14.379910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from https://www.kaggle.com/code/ambrosm/tpsaug22-eda-which-makes-sense\n\nplt.figure(figsize=(5, 5))\nfpr, tpr, _ = roc_curve(y_valid, y_valid_pred)\nplt.plot(fpr, tpr, color='#c00000', lw=3, label=f\"(auc (fold 4) = {roc_auc_score(y_valid, y_valid_pred):.5f})\") # curve\nplt.fill_between(fpr, tpr, color='#ffc0c0') # area under the curve\n\nplt.plot([0, 1], [0, 1], color=\"navy\", lw=1, linestyle=\"--\") # diagonal\nplt.gca().set_aspect('equal')\n\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.0])\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\n\nplt.title(\"Receiver operating characteristic\")\nplt.legend(loc=\"lower right\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:26:14.383298Z","iopub.execute_input":"2022-08-14T03:26:14.383776Z","iopub.status.idle":"2022-08-14T03:26:14.617987Z","shell.execute_reply.started":"2022-08-14T03:26:14.383731Z","shell.execute_reply":"2022-08-14T03:26:14.616804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_submit = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\n\npred = sum(test_pred_list)/5\ndf_submit['failure'] = pred\n\ndf_submit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T03:26:14.619135Z","iopub.execute_input":"2022-08-14T03:26:14.619707Z","iopub.status.idle":"2022-08-14T03:26:14.685142Z","shell.execute_reply.started":"2022-08-14T03:26:14.619665Z","shell.execute_reply":"2022-08-14T03:26:14.684031Z"},"trusted":true},"execution_count":null,"outputs":[]}]}