{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler\nfrom sklearn.model_selection import train_test_split, GridSearchCV \nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import classification_report\nfrom imblearn.over_sampling import SMOTE\n\nUSE_SMOTE = True\n\npd.set_option('display.max_columns', None) # display all columns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-06T17:24:57.266207Z","iopub.execute_input":"2022-08-06T17:24:57.266703Z","iopub.status.idle":"2022-08-06T17:24:57.916428Z","shell.execute_reply.started":"2022-08-06T17:24:57.266604Z","shell.execute_reply":"2022-08-06T17:24:57.915235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\")\ndf_test = pd.read_csv(\"../input/tabular-playground-series-aug-2022/test.csv\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:24:57.918967Z","iopub.execute_input":"2022-08-06T17:24:57.919425Z","iopub.status.idle":"2022-08-06T17:24:58.129839Z","shell.execute_reply.started":"2022-08-06T17:24:57.919380Z","shell.execute_reply":"2022-08-06T17:24:58.128582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AttributeEncoder(BaseEstimator, TransformerMixin):\n    \n    def __init__(self):\n        self.materials = []\n        \n    def material_encoder(self, material_to_encode):\n        materials = list(self.materials)\n        out = np.zeros(len(materials))\n        idx = materials.index(material_to_encode)\n        out[idx] = 1\n        return pd.Series(out)\n        \n    def fit(self, X, y=None):\n        materials = np.concatenate((X[\"attribute_0\"].unique(), X[\"attribute_1\"].unique()))\n        materials = np.sort(np.unique(materials))\n        print(f\"Unique materials: {materials}\")\n        self.materials = materials\n        return self\n\n    def transform(self, X, y=None):\n        materials = self.materials\n        if \"attribute_0\" not in X.columns or \"attribute_1\" not in X.columns:\n            raise KeyError(\"Dataframe should have \\\"attribute_0\\\" and \\\"attribute_1\\\" as columns\")          \n        else :\n            X[materials] = X[[\"attribute_0\",\"attribute_1\"]].apply( lambda row: self.material_encoder(row[0]) + self.material_encoder(row[1]), axis = 1)\n            X.drop(columns = [\"attribute_0\",\"attribute_1\", \"product_code\", \"id\"], inplace = True)\n        return X   \n    \nclass Scaler(BaseEstimator, TransformerMixin):\n    \n    def __init__(self, scaler_type, columns):\n        self.columns = columns\n        if scaler_type is None or scaler_type == \"StandardScaler\":\n            self.scaler = StandardScaler()\n        elif scaler_type == \"MinMaxScaler\" :\n            self.scaler = MinMaxScaler()\n        else:\n            raise ValueError(\"scaler_type should be 'StandardScaler' or 'MinMaxScaler'\")\n        \n    def fit(self, X, y=None):\n        self.scaler.fit(X[self.columns])\n        return self\n\n    def transform(self, X, y=None):\n        X[self.columns] = self.scaler.transform(X[self.columns])\n        return X\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:24:58.131284Z","iopub.execute_input":"2022-08-06T17:24:58.131929Z","iopub.status.idle":"2022-08-06T17:24:58.149118Z","shell.execute_reply.started":"2022-08-06T17:24:58.131883Z","shell.execute_reply":"2022-08-06T17:24:58.147838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"measurment_col_names = [col_name for col_name in df_train.columns.to_list() if \"measurement\" in col_name]\n\ncolumn_transformer =  ColumnTransformer(\n    [\n        (\"measurment_loading_mean_imputer\", SimpleImputer(missing_values=np.nan, strategy=\"mean\"), measurment_col_names + [\"loading\"]),\n        #(\"measurment_loading_standard_scaler\", MinMaxScaler(), [\"loading\"])\n    ],\n    remainder = 'passthrough'\n)\n\npipeline = Pipeline(\n    [\n        ('attribute encoder', AttributeEncoder()),\n        (\"scaler\", Scaler(scaler_type = \"MinMaxScaler\", columns = measurment_col_names + [\"loading\"] )),\n        (\"numerical transform\", column_transformer),\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:24:58.152382Z","iopub.execute_input":"2022-08-06T17:24:58.152888Z","iopub.status.idle":"2022-08-06T17:24:58.161167Z","shell.execute_reply.started":"2022-08-06T17:24:58.152841Z","shell.execute_reply":"2022-08-06T17:24:58.160064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = df_train.drop(columns = [\"failure\"], inplace = False)\ntrain_labels = df_train[\"failure\"]\n\nX_train, X_test, y_train, y_test = train_test_split(train_data, train_labels, test_size=0.2, stratify = train_labels, shuffle=True)\n\nprint(f\"X_train.shape: {X_train.shape} || y_train.shape: {y_train.shape}\")\nprint(f\"X_test.shape: {X_test.shape} || y_test.shape: {y_test.shape}\")\n\nfig, axs = plt.subplots(nrows=1, ncols=2,figsize=(15, 15))\nfor i, labels in enumerate([y_train, y_test]):\n    axs[i].pie(labels.value_counts(), startangle=90, wedgeprops={'width':0.3})\n    axs[i].text(0, 0, f\"{labels.value_counts()[0] / labels.count() * 100:.2f}%\", ha='center', va='center', fontweight='bold', fontsize=42)\n    axs[i].legend(labels.value_counts().index, ncol=2, loc='lower center', fontsize=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:24:58.162680Z","iopub.execute_input":"2022-08-06T17:24:58.163065Z","iopub.status.idle":"2022-08-06T17:24:58.449978Z","shell.execute_reply.started":"2022-08-06T17:24:58.163031Z","shell.execute_reply":"2022-08-06T17:24:58.446413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_transformed = pipeline.fit_transform(X_train.copy())\nX_test_transformed = pipeline.transform(X_test.copy())","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:24:58.451639Z","iopub.execute_input":"2022-08-06T17:24:58.453553Z","iopub.status.idle":"2022-08-06T17:25:07.451657Z","shell.execute_reply.started":"2022-08-06T17:24:58.453499Z","shell.execute_reply":"2022-08-06T17:25:07.450736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_SMOTE:\n    sm = SMOTE(n_jobs = -1)\n    X_train_transformed, y_train = sm.fit_resample(X_train_transformed, y_train)\n\n    fig, axs = plt.subplots(figsize=(10, 10))\n    plt.pie(y_train.value_counts(), startangle=90, wedgeprops={'width':0.3})\n    plt.text(0, 0, f\"{y_train.value_counts()[0] / y_train.count() * 100:.2f}%\", ha='center', va='center', fontweight='bold', fontsize=42)\n    plt.legend(y_train.value_counts().index, ncol=2, loc='lower center', fontsize=16)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:25:07.453067Z","iopub.execute_input":"2022-08-06T17:25:07.453632Z","iopub.status.idle":"2022-08-06T17:25:08.101486Z","shell.execute_reply.started":"2022-08-06T17:25:07.453597Z","shell.execute_reply":"2022-08-06T17:25:08.100270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nclassifier = RandomForestClassifier(class_weight = 'balanced', n_jobs=-1)\n\nparam_grid = { \n    'n_estimators': [150,200],\n    'max_features': ['sqrt'],\n    'max_depth' : [8,10],\n    'criterion' :['entropy',]\n}\n\nCV_rfc = GridSearchCV(estimator=classifier, param_grid=param_grid, cv=5, n_jobs = -1, verbose = 1, scoring=\"balanced_accuracy\")\nCV_rfc.fit(X_train_transformed, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:25:08.103396Z","iopub.execute_input":"2022-08-06T17:25:08.104167Z","iopub.status.idle":"2022-08-06T17:27:38.824482Z","shell.execute_reply.started":"2022-08-06T17:25:08.104115Z","shell.execute_reply":"2022-08-06T17:27:38.823163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_rfc = CV_rfc.best_estimator_\nprint(CV_rfc.best_estimator_)\nprint(CV_rfc.best_params_)\npredicted_rfc = best_rfc.predict(X_test_transformed)\nprint(classification_report(y_true = y_test, y_pred = predicted_rfc))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:27:38.828273Z","iopub.execute_input":"2022-08-06T17:27:38.828624Z","iopub.status.idle":"2022-08-06T17:27:38.952199Z","shell.execute_reply.started":"2022-08-06T17:27:38.828588Z","shell.execute_reply":"2022-08-06T17:27:38.951093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_grid = {'C': [ 0.05 ,0.1, 0.5, 1],\n                 'penalty':['l2','l1'],\n                 'solver': [ 'liblinear' ]}  \nCV_lr = GridSearchCV(LogisticRegression(max_iter = 500, class_weight=\"balanced\"), param_grid, cv = 5, verbose = 1, scoring=\"balanced_accuracy\", n_jobs=-1) \nCV_lr.fit(X_train_transformed, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:27:38.953437Z","iopub.execute_input":"2022-08-06T17:27:38.953751Z","iopub.status.idle":"2022-08-06T17:28:36.758663Z","shell.execute_reply.started":"2022-08-06T17:27:38.953720Z","shell.execute_reply":"2022-08-06T17:28:36.757511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_lr = CV_lr.best_estimator_\nprint(CV_lr.best_params_)\nprint(CV_lr.best_estimator_)\npredicted_lr = best_lr.predict(X_test_transformed)\nprint(classification_report(y_true = y_test, y_pred = predicted_lr))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:28:36.760259Z","iopub.execute_input":"2022-08-06T17:28:36.760597Z","iopub.status.idle":"2022-08-06T17:28:36.789749Z","shell.execute_reply.started":"2022-08-06T17:28:36.760565Z","shell.execute_reply":"2022-08-06T17:28:36.788518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_proba_lr = best_lr.predict_proba(X_test_transformed)\npredicted_proba_rfc = best_rfc.predict_proba(X_test_transformed)\nprint(predicted_proba_lr[:5], predicted_proba_rfc[:5] )","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:28:36.791554Z","iopub.execute_input":"2022-08-06T17:28:36.792404Z","iopub.status.idle":"2022-08-06T17:28:37.008249Z","shell.execute_reply.started":"2022-08-06T17:28:36.792357Z","shell.execute_reply":"2022-08-06T17:28:37.007334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted_proba = (predicted_proba_lr + predicted_proba_rfc)/2\npredicted = np.argmax(predicted_proba, axis=1)\nprint(classification_report(y_true = y_test, y_pred = predicted))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:28:37.009129Z","iopub.execute_input":"2022-08-06T17:28:37.009449Z","iopub.status.idle":"2022-08-06T17:28:37.032026Z","shell.execute_reply.started":"2022-08-06T17:28:37.009418Z","shell.execute_reply":"2022-08-06T17:28:37.030994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_submission = pipeline.transform(df_test.copy())\nid_submission = df_test[\"id\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:28:37.033800Z","iopub.execute_input":"2022-08-06T17:28:37.036036Z","iopub.status.idle":"2022-08-06T17:28:44.180335Z","shell.execute_reply.started":"2022-08-06T17:28:37.035999Z","shell.execute_reply":"2022-08-06T17:28:44.179395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_predicted_rfc = best_rfc.predict_proba(X_test_submission)\nsub_predicted_lr = best_lr.predict_proba(X_test_submission)\nfailure = (sub_predicted_rfc + sub_predicted_lr)/2","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:28:44.181905Z","iopub.execute_input":"2022-08-06T17:28:44.182385Z","iopub.status.idle":"2022-08-06T17:28:44.394281Z","shell.execute_reply.started":"2022-08-06T17:28:44.182341Z","shell.execute_reply":"2022-08-06T17:28:44.392860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\"id\":id_submission, \"failure\":failure[:,1]})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:28:44.396057Z","iopub.execute_input":"2022-08-06T17:28:44.396648Z","iopub.status.idle":"2022-08-06T17:28:44.497298Z","shell.execute_reply.started":"2022-08-06T17:28:44.396600Z","shell.execute_reply":"2022-08-06T17:28:44.495912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T17:29:32.317466Z","iopub.execute_input":"2022-08-06T17:29:32.318543Z","iopub.status.idle":"2022-08-06T17:29:32.329047Z","shell.execute_reply.started":"2022-08-06T17:29:32.318494Z","shell.execute_reply":"2022-08-06T17:29:32.328253Z"},"trusted":true},"execution_count":null,"outputs":[]}]}