{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport imblearn\nfrom imblearn.under_sampling import NearMiss, ClusterCentroids\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.svm import SVC\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import StandardScaler,RobustScaler,PowerTransformer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.tree import DecisionTreeClassifier\nimport seaborn as sns\nimport warnings\nimport plotly.express as px\nfrom matplotlib import rcParams\nfrom tqdm.notebook import tqdm\nfrom gc import collect\nwarnings.filterwarnings(\"ignore\")\nimport lightgbm as lgb\nimport pandas as pd\nfrom sklearn.metrics import mean_squared_error\nimport xgboost as xgb\nfrom sklearn.linear_model import LinearRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T10:49:05.383768Z","iopub.execute_input":"2022-08-11T10:49:05.384243Z","iopub.status.idle":"2022-08-11T10:49:05.396001Z","shell.execute_reply.started":"2022-08-11T10:49:05.384205Z","shell.execute_reply":"2022-08-11T10:49:05.394543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Import Dataset**","metadata":{}},{"cell_type":"code","source":"train=pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest=pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:05.415154Z","iopub.execute_input":"2022-08-11T10:49:05.415628Z","iopub.status.idle":"2022-08-11T10:49:05.603854Z","shell.execute_reply.started":"2022-08-11T10:49:05.415589Z","shell.execute_reply":"2022-08-11T10:49:05.602599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EDA**","metadata":{}},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:05.606204Z","iopub.execute_input":"2022-08-11T10:49:05.606563Z","iopub.status.idle":"2022-08-11T10:49:05.648163Z","shell.execute_reply.started":"2022-08-11T10:49:05.606532Z","shell.execute_reply":"2022-08-11T10:49:05.646821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:05.651710Z","iopub.execute_input":"2022-08-11T10:49:05.652062Z","iopub.status.idle":"2022-08-11T10:49:05.693465Z","shell.execute_reply.started":"2022-08-11T10:49:05.652029Z","shell.execute_reply":"2022-08-11T10:49:05.692162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nrcParams['figure.figsize'] = 11.7,8.27\nsns.barplot(x='product_code',y='failure',hue='attribute_0',data=train)\nplt.title('Probablity Of failure in material 7 and material 8')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:05.696397Z","iopub.execute_input":"2022-08-11T10:49:05.696893Z","iopub.status.idle":"2022-08-11T10:49:06.447223Z","shell.execute_reply.started":"2022-08-11T10:49:05.696849Z","shell.execute_reply":"2022-08-11T10:49:06.445933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nrcParams['figure.figsize'] = 11.7,8.27\nsns.barplot(x='product_code',y='failure',hue='attribute_1',data=train)\nplt.title('Probablity Of failure in material 8, material 5, and matrial 6')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:06.448931Z","iopub.execute_input":"2022-08-11T10:49:06.449391Z","iopub.status.idle":"2022-08-11T10:49:07.189065Z","shell.execute_reply.started":"2022-08-11T10:49:06.449338Z","shell.execute_reply":"2022-08-11T10:49:07.187954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(25,25)})\nfor i, column in enumerate(list(train.columns), 1):\n    plt.subplot(5,6,i)\n    p=sns.histplot(x=column,data=train.sample(1000),stat='count',kde=True,color='orange')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:07.190644Z","iopub.execute_input":"2022-08-11T10:49:07.191100Z","iopub.status.idle":"2022-08-11T10:49:12.928085Z","shell.execute_reply.started":"2022-08-11T10:49:07.191049Z","shell.execute_reply":"2022-08-11T10:49:12.926888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(25,21)})\nsns.heatmap(train.drop('id',axis=1).corr(),annot=True,fmt='.2f')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:12.929851Z","iopub.execute_input":"2022-08-11T10:49:12.930603Z","iopub.status.idle":"2022-08-11T10:49:15.297321Z","shell.execute_reply.started":"2022-08-11T10:49:12.930563Z","shell.execute_reply":"2022-08-11T10:49:15.296449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"string_cols = [f for f in train.columns if train[f].dtype == object]\n\n_, axs = plt.subplots(1, 3, figsize=(17, 5))\nfor f, ax in zip(string_cols, axs.ravel()):\n    temp1 = train[f].value_counts(dropna=False, normalize=True)\n    temp2 = test[f].value_counts(dropna=False, normalize=True)\n    values = sorted(set(temp1.index).union(temp2.index))\n    temp1 = temp1.reindex(values)\n    temp2 = temp2.reindex(values)\n    ax.bar(range(len(values)), temp1, alpha=0.5, label='train')\n    ax.bar(range(len(values)), temp2, alpha=0.5, label='test')\n    ax.set_xlabel(f)\n    ax.set_ylabel('frequency')\n    ax.set_xticks(range(len(values)), values)\n    \n    temp1 = train.failure.groupby(train[f]).agg(['mean', 'size'])\n    temp1 = temp1.reindex(values)\n    ax2 = ax.twinx()\n    ax2.scatter(range(len(values)), temp1['mean'],\n                color='m', label='failure probability')\n    ax2.tick_params(axis='y', colors='m')\n    ax2.set_ylim(0, 0.5)\n    if ax == axs[0]: ax2.legend(loc='lower right')\n\naxs[0].legend()\nplt.suptitle('Train and test distributions of the string features', fontsize=20, y=0.96)\nplt.tight_layout(w_pad=1)\nplt.show()\ndel temp1, temp2   ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:15.298877Z","iopub.execute_input":"2022-08-11T10:49:15.299528Z","iopub.status.idle":"2022-08-11T10:49:16.188301Z","shell.execute_reply.started":"2022-08-11T10:49:15.299478Z","shell.execute_reply":"2022-08-11T10:49:16.187074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Feature Engineering**","metadata":{}},{"cell_type":"code","source":"for col in train.columns:\n    print(col,train[col].unique(),len(train[col].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:16.190056Z","iopub.execute_input":"2022-08-11T10:49:16.190540Z","iopub.status.idle":"2022-08-11T10:49:16.240829Z","shell.execute_reply.started":"2022-08-11T10:49:16.190473Z","shell.execute_reply":"2022-08-11T10:49:16.239559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our target data set is imbalanced, we can try undersampling or oversampling to improve performance.\n\nAttribute 0 & Attribute 1 has text data that can be converted to 5/6/7/8 value. We can use getdummies for the same.","metadata":{}},{"cell_type":"markdown","source":"## **Dealing with Categorical values**","metadata":{}},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:16.245653Z","iopub.execute_input":"2022-08-11T10:49:16.246024Z","iopub.status.idle":"2022-08-11T10:49:16.255481Z","shell.execute_reply.started":"2022-08-11T10:49:16.245990Z","shell.execute_reply":"2022-08-11T10:49:16.254212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"int_cols = list(train.drop(['id','failure'],axis=1).select_dtypes(exclude=[\"object\"]).columns)\n# train.drop('id',axis=1).select_dtypes(exclude=['object']).columns\nint_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:16.257535Z","iopub.execute_input":"2022-08-11T10:49:16.258343Z","iopub.status.idle":"2022-08-11T10:49:16.273930Z","shell.execute_reply.started":"2022-08-11T10:49:16.258299Z","shell.execute_reply":"2022-08-11T10:49:16.272983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enc = OneHotEncoder(handle_unknown='ignore')\nohe_attributes = ['attribute_0', 'attribute_1']\nohe_output = ['ohe_a_7', 'ohe_a_6', 'ohe_a_8']\nohe = OneHotEncoder(categories=[['material_5', 'material_7'],['material_5', 'material_6', 'material_8']],\n                    drop='first', sparse=False, handle_unknown='ignore')\nohe.fit(train[ohe_attributes])\n\ntrain[ohe_output] = ohe.transform(train[ohe_attributes])\ntest[ohe_output] = ohe.transform(test[ohe_attributes])\n\ntrain = train.drop(['attribute_0', 'attribute_1'], axis=1)\ntest = test.drop(['attribute_0', 'attribute_1'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:16.275162Z","iopub.execute_input":"2022-08-11T10:49:16.275705Z","iopub.status.idle":"2022-08-11T10:49:16.345148Z","shell.execute_reply.started":"2022-08-11T10:49:16.275672Z","shell.execute_reply":"2022-08-11T10:49:16.344010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:16.349233Z","iopub.execute_input":"2022-08-11T10:49:16.349609Z","iopub.status.idle":"2022-08-11T10:49:16.393063Z","shell.execute_reply.started":"2022-08-11T10:49:16.349576Z","shell.execute_reply":"2022-08-11T10:49:16.391932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mannual_encoding  = {'A': 1, 'B':2, 'C':3, 'D': 4, 'E': 5, 'F':6, 'G': 7, 'H':8, 'I': 9}\n\ntrain.product_code = [mannual_encoding[val] for val in train.product_code]\ntest.product_code = [mannual_encoding[val] for val in test.product_code]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:16.394753Z","iopub.execute_input":"2022-08-11T10:49:16.395074Z","iopub.status.idle":"2022-08-11T10:49:16.425486Z","shell.execute_reply.started":"2022-08-11T10:49:16.395045Z","shell.execute_reply":"2022-08-11T10:49:16.423979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Dealing with Null Values**","metadata":{}},{"cell_type":"code","source":"features = int_cols\nimputer = SimpleImputer(strategy=\"mean\")\nimputer.fit(train[features])\n\ntrain[features] = imputer.transform(train[features])\n# features.remove('failure')\ntest[features] = imputer.transform(test[features])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:16.427545Z","iopub.execute_input":"2022-08-11T10:49:16.428279Z","iopub.status.idle":"2022-08-11T10:49:16.475100Z","shell.execute_reply.started":"2022-08-11T10:49:16.428232Z","shell.execute_reply":"2022-08-11T10:49:16.474186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:16.476330Z","iopub.execute_input":"2022-08-11T10:49:16.477111Z","iopub.status.idle":"2022-08-11T10:49:16.508974Z","shell.execute_reply.started":"2022-08-11T10:49:16.477076Z","shell.execute_reply":"2022-08-11T10:49:16.508054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ID=test['id']\ntrain.drop(['id'],axis=1,inplace=True)\ntest.drop(['id'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:16.510258Z","iopub.execute_input":"2022-08-11T10:49:16.510795Z","iopub.status.idle":"2022-08-11T10:49:16.526152Z","shell.execute_reply.started":"2022-08-11T10:49:16.510761Z","shell.execute_reply":"2022-08-11T10:49:16.525240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for method in tqdm(['pearson', 'spearman', 'kendall']):\n    fig, ax= plt.subplots(1,1, figsize= (15,10));\n    _corr = train.corr(method=method);\n    sns.heatmap(_corr, annot= True, fmt= '.0%', linewidth= 1, center= True, cmap= 'Spectral_r',\n                cbar= False, linecolor= 'white', mask = np.triu(np.ones_like(_corr)),ax= ax);\n    ax.set_title(f\"\\n{method.capitalize()} correlation plot before transforms\\n\", \n                 color= 'tab:blue', fontsize= 8);\n    plt.tight_layout();\n    plt.yticks(rotation= 0);\n    plt.xticks(rotation= 90);\n    plt.show();\n    del _corr;\n    collect();\ncollect();","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:16.527422Z","iopub.execute_input":"2022-08-11T10:49:16.527972Z","iopub.status.idle":"2022-08-11T10:49:26.035685Z","shell.execute_reply.started":"2022-08-11T10:49:16.527939Z","shell.execute_reply":"2022-08-11T10:49:26.034462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, axs = plt.subplots(3, 3, figsize=(18, 18))\nfor product, ax in zip(np.unique(train.product_code), axs.ravel()):\n    corr = train.loc[train.product_code == product, [f'measurement_{i}' for i in range(3, 18)]].corr()\n    mask = np.triu(np.ones_like(corr, dtype=bool))\n    sns.heatmap(corr*10, mask=mask, linewidth=0.0, fmt='.0f', \n                annot=True, annot_kws={'size': 8}, \n                cmap='twilight_shifted_r', center=0, ax=ax, cbar=False)\n    ax.set_title(f'product code: {product}')\nplt.tight_layout(w_pad=0.5)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:26.037384Z","iopub.execute_input":"2022-08-11T10:49:26.037885Z","iopub.status.idle":"2022-08-11T10:49:32.845175Z","shell.execute_reply.started":"2022-08-11T10:49:26.037837Z","shell.execute_reply":"2022-08-11T10:49:32.843971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Prediction**","metadata":{}},{"cell_type":"code","source":"x=train.iloc[:,:-1]\ny=train.iloc[:,-1]\n\n\n# define the undersampling method\nundersample = NearMiss(n_neighbors=1)\n\n# transform the dataset\nx, y = undersample.fit_resample(x, y)\n\ntrain=x","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:32.846563Z","iopub.execute_input":"2022-08-11T10:49:32.846900Z","iopub.status.idle":"2022-08-11T10:49:34.878869Z","shell.execute_reply.started":"2022-08-11T10:49:32.846870Z","shell.execute_reply":"2022-08-11T10:49:34.877604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Feature Scaling**","metadata":{}},{"cell_type":"code","source":"scaler = PowerTransformer()\nDataScaled = scaler.fit_transform(train)\ntrain=pd.DataFrame(DataScaled, index=train.index, columns=train.columns)\ntrain.head()\n\nDataScaled = scaler.fit_transform(test)\ntest=pd.DataFrame(DataScaled, index=test.index, columns=test.columns)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:49:34.880195Z","iopub.execute_input":"2022-08-11T10:49:34.880532Z","iopub.status.idle":"2022-08-11T10:49:36.089846Z","shell.execute_reply.started":"2022-08-11T10:49:34.880488Z","shell.execute_reply":"2022-08-11T10:49:36.088639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **PCA**","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(train,y, test_size=0.001)\nprint(X_train.shape, X_test.shape, y_train.shape, y_test.shape,test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T11:00:34.444206Z","iopub.execute_input":"2022-08-11T11:00:34.444834Z","iopub.status.idle":"2022-08-11T11:00:34.463610Z","shell.execute_reply.started":"2022-08-11T11:00:34.444797Z","shell.execute_reply":"2022-08-11T11:00:34.462192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\npca = PCA(n_components = 5)\nX_train = pca.fit_transform(X_train)\nX_test = pca.transform(X_test)\ntest = pca.transform(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T11:01:13.378202Z","iopub.execute_input":"2022-08-11T11:01:13.378631Z","iopub.status.idle":"2022-08-11T11:01:13.536726Z","shell.execute_reply.started":"2022-08-11T11:01:13.378596Z","shell.execute_reply":"2022-08-11T11:01:13.534889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T11:01:48.684792Z","iopub.execute_input":"2022-08-11T11:01:48.685562Z","iopub.status.idle":"2022-08-11T11:01:48.692239Z","shell.execute_reply.started":"2022-08-11T11:01:48.685521Z","shell.execute_reply":"2022-08-11T11:01:48.690766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Ensemble Learning Technique**","metadata":{}},{"cell_type":"code","source":"# importing utility modules\nfrom sklearn.metrics import mean_squared_error\n# importing machine learning models for prediction\nfrom sklearn.ensemble import RandomForestRegressor\nimport xgboost as xgb\nfrom sklearn.linear_model import LinearRegression\n\n# Splitting between train data into training and validation dataset\n\n# initializing all the model objects with default parameters\nmodel_1 = LinearRegression()\nmodel_2 = xgb.XGBRegressor()\nmodel_3 = RandomForestRegressor()\n\n# training all the model on the training dataset\nmodel_1.fit(X_train, y_train)\nmodel_2.fit(X_train, y_train)\nmodel_3.fit(X_train, y_train)\n\n# predicting the output on the validation dataset\npred_1 = model_1.predict(test)\npred_2 = model_2.predict(test)\npred_3 = model_3.predict(test)\n\n# final prediction after averaging on the prediction of all 3 models\npred_final = (pred_1+pred_2+pred_3)/3.0","metadata":{"execution":{"iopub.status.busy":"2022-08-11T11:02:01.673108Z","iopub.execute_input":"2022-08-11T11:02:01.674577Z","iopub.status.idle":"2022-08-11T11:02:11.831369Z","shell.execute_reply.started":"2022-08-11T11:02:01.674528Z","shell.execute_reply":"2022-08-11T11:02:11.830261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission1 = pd.DataFrame({\n        'id': ID,\n        'failure': pred_final\n    })\n\n\nsubmission1['failure'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T11:02:21.139904Z","iopub.execute_input":"2022-08-11T11:02:21.140750Z","iopub.status.idle":"2022-08-11T11:02:21.154346Z","shell.execute_reply.started":"2022-08-11T11:02:21.140705Z","shell.execute_reply":"2022-08-11T11:02:21.152711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission1['failure'].shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T11:02:29.246792Z","iopub.execute_input":"2022-08-11T11:02:29.247556Z","iopub.status.idle":"2022-08-11T11:02:29.255967Z","shell.execute_reply.started":"2022-08-11T11:02:29.247491Z","shell.execute_reply":"2022-08-11T11:02:29.254943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission1.to_csv('submission10.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T11:02:37.082360Z","iopub.execute_input":"2022-08-11T11:02:37.082893Z","iopub.status.idle":"2022-08-11T11:02:37.138798Z","shell.execute_reply.started":"2022-08-11T11:02:37.082845Z","shell.execute_reply":"2022-08-11T11:02:37.137305Z"},"trusted":true},"execution_count":null,"outputs":[]}]}