{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img width='600px' src='https://memegenerator.net/img/instances/65185399.jpg'/>","metadata":{}},{"cell_type":"markdown","source":"# Feature Extraction Guide","metadata":{}},{"cell_type":"code","source":"import numpy as np\nfrom numpy import mean\nfrom numpy import std\nimport pandas as pd\nfrom sklearn.model_selection import RepeatedStratifiedKFold\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.pipeline import FeatureUnion\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.preprocessing import QuantileTransformer\nfrom sklearn.preprocessing import KBinsDiscretizer\nfrom sklearn.feature_selection import RFE\nfrom sklearn.decomposition import PCA\nfrom sklearn.decomposition import TruncatedSVD\n\nfrom sklearn.impute import KNNImputer\nfrom sklearn.linear_model import HuberRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-13T07:22:37.858846Z","iopub.execute_input":"2022-08-13T07:22:37.859271Z","iopub.status.idle":"2022-08-13T07:22:37.871321Z","shell.execute_reply.started":"2022-08-13T07:22:37.859238Z","shell.execute_reply":"2022-08-13T07:22:37.870443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')\nsubmission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\ntarget = train['failure']\ntrain.drop('failure',axis=1, inplace = True)\ndata = pd.concat([train, test])\ntrain.shape,test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:23:00.202857Z","iopub.execute_input":"2022-08-13T07:23:00.203816Z","iopub.status.idle":"2022-08-13T07:23:00.412935Z","shell.execute_reply.started":"2022-08-13T07:23:00.203773Z","shell.execute_reply":"2022-08-13T07:23:00.411808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"! pip install feature_engine\nfrom feature_engine.encoding import WoEEncoder","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-13T07:23:01.912839Z","iopub.execute_input":"2022-08-13T07:23:01.913265Z","iopub.status.idle":"2022-08-13T07:23:12.775193Z","shell.execute_reply.started":"2022-08-13T07:23:01.913229Z","shell.execute_reply":"2022-08-13T07:23:12.773647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['m3_missing'] = data['measurement_3'].isnull().astype(np.int8)\ndata['m5_missing'] = data['measurement_5'].isnull().astype(np.int8)\ndata['area'] = data['attribute_2'] * data['attribute_3']\n\nfeature = [f for f in test.columns if f.startswith('measurement') or f=='loading']\n\n# dictionnary of dictionnaries (for the 11 best correlated measurement columns), \n# we will use the dictionnaries below to select the best correlated columns according to the product code)\n# Only for 'measurement_17' we make a 'manual' selection :\nfull_fill_dict ={}\nfull_fill_dict['measurement_17'] = {\n    'A': ['measurement_5','measurement_6','measurement_8'],\n    'B': ['measurement_4','measurement_5','measurement_7'],\n    'C': ['measurement_5','measurement_7','measurement_8','measurement_9'],\n    'D': ['measurement_5','measurement_6','measurement_7','measurement_8'],\n    'E': ['measurement_4','measurement_5','measurement_6','measurement_8'],\n    'F': ['measurement_4','measurement_5','measurement_6','measurement_7'],\n    'G': ['measurement_4','measurement_6','measurement_8','measurement_9'],\n    'H': ['measurement_4','measurement_5','measurement_7','measurement_8','measurement_9'],\n    'I': ['measurement_3','measurement_7','measurement_8']\n}\n\n# collect the name of the next 10 best measurement columns sorted by correlation (except 17 already done above):\ncol = [col for col in test.columns if 'measurement' not in col]+ ['loading','m3_missing','m5_missing']\na = []\nb =[]\nfor x in range(3,17):\n    corr = np.absolute(data.drop(col, axis=1).corr()[f'measurement_{x}']).sort_values(ascending=False)\n    a.append(np.round(np.sum(corr[1:4]),3)) # we add the 3 first lines of the correlation values to get the \"most correlated\"\n    b.append(f'measurement_{x}')\nc = pd.DataFrame()\nc['Selected columns'] = b\nc['correlation total'] = a\nc = c.sort_values(by = 'correlation total',ascending=False).reset_index(drop = True)\nprint(f'Columns selected by correlation sum of the 3 first rows : ')\ndisplay(c.head(10))\n\nfor i in range(10):\n    measurement_col = 'measurement_' + c.iloc[i,0][12:] # we select the next best correlated column \n    fill_dict ={}\n    for x in data.product_code.unique() : \n        corr = np.absolute(data[data.product_code == x].drop(col, axis=1).corr()[measurement_col]).sort_values(ascending=False)\n        measurement_col_dic = {}\n        measurement_col_dic[measurement_col] = corr[1:5].index.tolist()\n        fill_dict[x] = measurement_col_dic[measurement_col]\n    full_fill_dict[measurement_col] =fill_dict\n    \nfeature = [f for f in data.columns if f.startswith('measurement') or f=='loading']\nnullValue_cols = [col for col in train.columns if train[col].isnull().sum()!=0]\n    \nfor code in data.product_code.unique():\n    total_na_filled_by_linear_model = 0\n    for measurement_col in list(full_fill_dict.keys()):\n        tmp = data[data.product_code==code]\n        column = full_fill_dict[measurement_col][code]\n        tmp_train = tmp[column+[measurement_col]].dropna(how='any')\n        tmp_test = tmp[(tmp[column].isnull().sum(axis=1)==0)&(tmp[measurement_col].isnull())]\n\n        model = HuberRegressor(epsilon=1.9)\n        model.fit(tmp_train[column], tmp_train[measurement_col])\n        data.loc[(data.product_code==code)&(data[column].isnull().sum(axis=1)==0)&(data[measurement_col].isnull()),measurement_col] = model.predict(tmp_test[column])\n        total_na_filled_by_linear_model += len(tmp_test)\n        \n    # others NA columns:\n    NA = data.loc[data[\"product_code\"] == code,nullValue_cols ].isnull().sum().sum()\n    model1 = KNNImputer(n_neighbors=3)\n    data.loc[data.product_code==code, feature] = model1.fit_transform(data.loc[data.product_code==code, feature])\n    \ndata['measurement_avg'] = data[[f'measurement_{i}' for i in range(3, 17)]].mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:23:12.778346Z","iopub.execute_input":"2022-08-13T07:23:12.778868Z","iopub.status.idle":"2022-08-13T07:23:30.302498Z","shell.execute_reply.started":"2022-08-13T07:23:12.778818Z","shell.execute_reply":"2022-08-13T07:23:30.301170Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"code","source":"X = data.iloc[:len(train)].drop(columns=['id', 'product_code', 'attribute_0', 'attribute_1'], axis=0)\ny = target\n\nX = X.astype('float')\ny = LabelEncoder().fit_transform(y.astype('str'))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:26:30.024570Z","iopub.execute_input":"2022-08-13T07:26:30.025025Z","iopub.status.idle":"2022-08-13T07:26:30.061199Z","shell.execute_reply.started":"2022-08-13T07:26:30.024989Z","shell.execute_reply":"2022-08-13T07:26:30.059790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transformers","metadata":{}},{"cell_type":"code","source":"transforms = list()\ntransforms.append(('mms', MinMaxScaler()))\ntransforms.append(('ss', StandardScaler()))\ntransforms.append(('rs', RobustScaler()))\ntransforms.append(('qt', QuantileTransformer(n_quantiles=100, output_distribution='normal')))\ntransforms.append(('kbd', KBinsDiscretizer(n_bins=10, encode='ordinal', strategy='uniform')))\ntransforms.append(('pca', PCA(n_components=7)))\ntransforms.append(('svd', TruncatedSVD(n_components=7)))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:26:31.713086Z","iopub.execute_input":"2022-08-13T07:26:31.714227Z","iopub.status.idle":"2022-08-13T07:26:31.722028Z","shell.execute_reply.started":"2022-08-13T07:26:31.714181Z","shell.execute_reply":"2022-08-13T07:26:31.720991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Feature Selection Pipeline","metadata":{}},{"cell_type":"code","source":"# best model for this dataset\nmodel = LogisticRegression(max_iter=200, C=0.0001, penalty='l2', solver='newton-cg')\n\n# create the feature union\nfu = FeatureUnion(transforms)\n\n# define the feature selection\nrfe = RFE(estimator=model, n_features_to_select=10)\n\n# define the pipeline\nsteps = list()\nsteps.append(('fu', fu))\nsteps.append(('rfe', rfe))\nsteps.append(('m', model))\npipeline = Pipeline(steps=steps)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:26:32.862487Z","iopub.execute_input":"2022-08-13T07:26:32.863226Z","iopub.status.idle":"2022-08-13T07:26:32.870343Z","shell.execute_reply.started":"2022-08-13T07:26:32.863186Z","shell.execute_reply":"2022-08-13T07:26:32.869105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluation","metadata":{}},{"cell_type":"code","source":"# define the cross-validation procedure\ncv = RepeatedStratifiedKFold(n_splits=5, random_state=1)\n\n# evaluate model\nscores = cross_val_score(pipeline, X, y, scoring='accuracy', cv=cv, n_jobs=-1)\n\n# report performance\nprint('Accuracy: %.3f (%.3f)' % (mean(scores), std(scores)))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:29:48.336500Z","iopub.execute_input":"2022-08-13T07:29:48.336969Z"},"trusted":true},"execution_count":null,"outputs":[]}]}