{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## **Feature Engineering**","metadata":{}},{"cell_type":"markdown","source":"Steps in Feature Engineering:\n    \n   1. Missing values\n   2. Temporal variables\n   3. Categorical variables: remove rare labels \n   4. Standarise the values of the variables to the same range ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom pandas.api.types import is_numeric_dtype\n%matplotlib inline\npd.pandas.set_option('display.max_columns', None)\npd.pandas.set_option('display.max_rows', None)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:52.056412Z","iopub.execute_input":"2022-08-03T12:20:52.057181Z","iopub.status.idle":"2022-08-03T12:20:52.066605Z","shell.execute_reply.started":"2022-08-03T12:20:52.057142Z","shell.execute_reply":"2022-08-03T12:20:52.065351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ndataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:52.167273Z","iopub.execute_input":"2022-08-03T12:20:52.167723Z","iopub.status.idle":"2022-08-03T12:20:52.241254Z","shell.execute_reply.started":"2022-08-03T12:20:52.167677Z","shell.execute_reply":"2022-08-03T12:20:52.240070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Always remember there way always be a chance of data Leakage so we need to split the data first and then apply feature\n## Engineering\n\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(dataset, dataset['SalePrice'], test_size=0.1, random_state=0) \nX_train.shape, X_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:52.264055Z","iopub.execute_input":"2022-08-03T12:20:52.264445Z","iopub.status.idle":"2022-08-03T12:20:52.280104Z","shell.execute_reply.started":"2022-08-03T12:20:52.264414Z","shell.execute_reply":"2022-08-03T12:20:52.278797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Missing Values**","metadata":{}},{"cell_type":"code","source":"## Capture all the nan values \n\nfeatures_nan = [feature for feature in dataset.columns if dataset[feature].isnull().sum() > 1 and not is_numeric_dtype(dataset[feature])]\nfor feature in features_nan:\n    print(f\"{feature}: {np.round(dataset[feature].isnull().mean(), 4)}% missing values\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:52.361876Z","iopub.execute_input":"2022-08-03T12:20:52.363077Z","iopub.status.idle":"2022-08-03T12:20:52.395515Z","shell.execute_reply.started":"2022-08-03T12:20:52.363032Z","shell.execute_reply":"2022-08-03T12:20:52.394707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Replace missing value with a new label\n\ndef replace_cat_feature(dataset, features_nan):\n    data = dataset.copy()\n    data[features_nan] = data[features_nan].fillna(\"Missing\")\n    return data\n\ndataset = replace_cat_feature(dataset, features_nan)\ndataset[features_nan].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:52.455375Z","iopub.execute_input":"2022-08-03T12:20:52.456358Z","iopub.status.idle":"2022-08-03T12:20:52.478138Z","shell.execute_reply.started":"2022-08-03T12:20:52.456307Z","shell.execute_reply":"2022-08-03T12:20:52.477041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:52.553023Z","iopub.execute_input":"2022-08-03T12:20:52.553866Z","iopub.status.idle":"2022-08-03T12:20:52.609152Z","shell.execute_reply.started":"2022-08-03T12:20:52.553824Z","shell.execute_reply":"2022-08-03T12:20:52.607862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Check for numerical variables the contains missing values \n\nnumerical_with_nan = [feature for feature in dataset.columns if dataset[feature].isnull().sum() > 1 and is_numeric_dtype(dataset[feature])]\nfor feature in numerical_with_nan:\n    print(f\"{feature}: {(np.round(dataset[feature].isnull().mean(), 4))}% missing values\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:52.647946Z","iopub.execute_input":"2022-08-03T12:20:52.648643Z","iopub.status.idle":"2022-08-03T12:20:52.678238Z","shell.execute_reply.started":"2022-08-03T12:20:52.648606Z","shell.execute_reply":"2022-08-03T12:20:52.676972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Replacing the numerical Missing Values \n\nfor feature in numerical_with_nan:\n    median_value = dataset[feature].median()\n    dataset[feature+'nan'] = np.where(dataset[feature].isnull(),1,0)\n    dataset[feature].fillna(median_value, inplace=True)\n    \ndataset[numerical_with_nan].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:52.740703Z","iopub.execute_input":"2022-08-03T12:20:52.741773Z","iopub.status.idle":"2022-08-03T12:20:52.757728Z","shell.execute_reply.started":"2022-08-03T12:20:52.741721Z","shell.execute_reply":"2022-08-03T12:20:52.756403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head(50)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:52.837651Z","iopub.execute_input":"2022-08-03T12:20:52.838793Z","iopub.status.idle":"2022-08-03T12:20:52.974606Z","shell.execute_reply.started":"2022-08-03T12:20:52.838747Z","shell.execute_reply":"2022-08-03T12:20:52.973399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Temporal Variables (Date Time Variables)\n\nfor feature in ['YearBuilt', 'YearRemodAdd', 'GarageYrBlt']:\n    dataset[feature] = dataset['YrSold'] - dataset[feature]\n    \ndataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:52.976956Z","iopub.execute_input":"2022-08-03T12:20:52.977328Z","iopub.status.idle":"2022-08-03T12:20:53.044419Z","shell.execute_reply.started":"2022-08-03T12:20:52.977295Z","shell.execute_reply":"2022-08-03T12:20:53.043204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset[['YearBuilt', 'YearRemodAdd', 'GarageYrBlt']].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.046748Z","iopub.execute_input":"2022-08-03T12:20:53.047462Z","iopub.status.idle":"2022-08-03T12:20:53.065557Z","shell.execute_reply.started":"2022-08-03T12:20:53.047416Z","shell.execute_reply":"2022-08-03T12:20:53.064439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Numeric Variables** ","metadata":{}},{"cell_type":"markdown","source":"Since the numeric variables are skewed we will perform log normal distribution ","metadata":{}},{"cell_type":"code","source":"num_features = ['LotFrontage', 'LotArea', '1stFlrSF', 'GrLivArea', 'SalePrice']\n\nfor feature in num_features:\n    dataset[feature] = np.log(dataset[feature])\n\ndataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.068278Z","iopub.execute_input":"2022-08-03T12:20:53.068926Z","iopub.status.idle":"2022-08-03T12:20:53.132158Z","shell.execute_reply.started":"2022-08-03T12:20:53.068881Z","shell.execute_reply":"2022-08-03T12:20:53.130981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Handling Rare Categorical Feature**","metadata":{}},{"cell_type":"markdown","source":"Remove categorical variables that are present less than 1% of the observations ","metadata":{}},{"cell_type":"code","source":"categorical_features = [feature for feature in dataset.columns if not is_numeric_dtype(dataset[feature])]\ncategorical_features","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.133539Z","iopub.execute_input":"2022-08-03T12:20:53.134482Z","iopub.status.idle":"2022-08-03T12:20:53.146592Z","shell.execute_reply.started":"2022-08-03T12:20:53.134445Z","shell.execute_reply":"2022-08-03T12:20:53.145443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for feature in categorical_features:\n    temp = dataset.groupby(feature)['SalePrice'].count()/len(dataset)\n    temp_df = temp[temp>0.01].index\n    dataset[feature] = np.where(dataset[feature].isin(temp_df), dataset[feature], 'Rare_var')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.204515Z","iopub.execute_input":"2022-08-03T12:20:53.205772Z","iopub.status.idle":"2022-08-03T12:20:53.277151Z","shell.execute_reply.started":"2022-08-03T12:20:53.205720Z","shell.execute_reply":"2022-08-03T12:20:53.275989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head(100)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.278967Z","iopub.execute_input":"2022-08-03T12:20:53.279335Z","iopub.status.idle":"2022-08-03T12:20:53.513504Z","shell.execute_reply.started":"2022-08-03T12:20:53.279304Z","shell.execute_reply":"2022-08-03T12:20:53.512383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for feature in categorical_features:\n    labels_ordered = dataset.groupby([feature])['SalePrice'].mean().sort_values().index\n    labels_ordered = {k: i for i, k in enumerate(labels_ordered, 0) }\n    dataset[feature] = dataset[feature].map(labels_ordered)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.515458Z","iopub.execute_input":"2022-08-03T12:20:53.515840Z","iopub.status.idle":"2022-08-03T12:20:53.608028Z","shell.execute_reply.started":"2022-08-03T12:20:53.515805Z","shell.execute_reply":"2022-08-03T12:20:53.606898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.609535Z","iopub.execute_input":"2022-08-03T12:20:53.610023Z","iopub.status.idle":"2022-08-03T12:20:53.658869Z","shell.execute_reply.started":"2022-08-03T12:20:53.609989Z","shell.execute_reply":"2022-08-03T12:20:53.657371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Feature Scaling**","metadata":{}},{"cell_type":"code","source":"scaling_feature =  [feature for feature in dataset.columns if feature not in ['Id', 'SalePrice']]\nlen(scaling_feature)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.661902Z","iopub.execute_input":"2022-08-03T12:20:53.662991Z","iopub.status.idle":"2022-08-03T12:20:53.670387Z","shell.execute_reply.started":"2022-08-03T12:20:53.662955Z","shell.execute_reply":"2022-08-03T12:20:53.669081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaling_feature","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.672066Z","iopub.execute_input":"2022-08-03T12:20:53.672598Z","iopub.status.idle":"2022-08-03T12:20:53.685029Z","shell.execute_reply.started":"2022-08-03T12:20:53.672554Z","shell.execute_reply":"2022-08-03T12:20:53.683907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.686452Z","iopub.execute_input":"2022-08-03T12:20:53.686832Z","iopub.status.idle":"2022-08-03T12:20:53.734808Z","shell.execute_reply.started":"2022-08-03T12:20:53.686801Z","shell.execute_reply":"2022-08-03T12:20:53.733686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_scale = [feature for feature in dataset.columns if feature not in ['Id', 'SalePrice']]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.736493Z","iopub.execute_input":"2022-08-03T12:20:53.737048Z","iopub.status.idle":"2022-08-03T12:20:53.743512Z","shell.execute_reply.started":"2022-08-03T12:20:53.737003Z","shell.execute_reply":"2022-08-03T12:20:53.742219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import  MinMaxScaler\n\nscaler = MinMaxScaler()\nscaler.fit(dataset[feature_scale])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.745370Z","iopub.execute_input":"2022-08-03T12:20:53.746171Z","iopub.status.idle":"2022-08-03T12:20:53.766967Z","shell.execute_reply.started":"2022-08-03T12:20:53.746135Z","shell.execute_reply":"2022-08-03T12:20:53.765903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler.transform(dataset[feature_scale])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.770489Z","iopub.execute_input":"2022-08-03T12:20:53.771637Z","iopub.status.idle":"2022-08-03T12:20:53.783658Z","shell.execute_reply.started":"2022-08-03T12:20:53.771600Z","shell.execute_reply":"2022-08-03T12:20:53.782524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transform the train and test set, and add on the Id and SalePrice variables\ndata = pd.concat([dataset[['Id', 'SalePrice']].reset_index(drop=True), pd.DataFrame(scaler.transform(dataset[feature_scale]), columns=feature_scale)], axis=1)\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.785537Z","iopub.execute_input":"2022-08-03T12:20:53.786633Z","iopub.status.idle":"2022-08-03T12:20:53.885103Z","shell.execute_reply.started":"2022-08-03T12:20:53.786585Z","shell.execute_reply":"2022-08-03T12:20:53.883753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.to_csv('X_train.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:53.887095Z","iopub.execute_input":"2022-08-03T12:20:53.887572Z","iopub.status.idle":"2022-08-03T12:20:54.014242Z","shell.execute_reply.started":"2022-08-03T12:20:53.887526Z","shell.execute_reply":"2022-08-03T12:20:54.012935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Feature Selection**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Lasso\nfrom sklearn.feature_selection import SelectFromModel","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:54.015538Z","iopub.execute_input":"2022-08-03T12:20:54.016524Z","iopub.status.idle":"2022-08-03T12:20:54.021550Z","shell.execute_reply.started":"2022-08-03T12:20:54.016484Z","shell.execute_reply":"2022-08-03T12:20:54.020365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = data\ndataset.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:54.024771Z","iopub.execute_input":"2022-08-03T12:20:54.025799Z","iopub.status.idle":"2022-08-03T12:20:54.111497Z","shell.execute_reply.started":"2022-08-03T12:20:54.025760Z","shell.execute_reply":"2022-08-03T12:20:54.110122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Capture the dependent feature \ny_train = dataset[['SalePrice']]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:54.113175Z","iopub.execute_input":"2022-08-03T12:20:54.113544Z","iopub.status.idle":"2022-08-03T12:20:54.120299Z","shell.execute_reply.started":"2022-08-03T12:20:54.113502Z","shell.execute_reply":"2022-08-03T12:20:54.119114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Drop dependent feature from dataset\nX_train = dataset.drop(['Id', 'SalePrice'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:54.122290Z","iopub.execute_input":"2022-08-03T12:20:54.122743Z","iopub.status.idle":"2022-08-03T12:20:54.133483Z","shell.execute_reply.started":"2022-08-03T12:20:54.122704Z","shell.execute_reply":"2022-08-03T12:20:54.132402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Aply Feature Selection\nfeature_sel_model = SelectFromModel(Lasso(alpha=0.005, random_state=0))\nfeature_sel_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:54.135435Z","iopub.execute_input":"2022-08-03T12:20:54.136148Z","iopub.status.idle":"2022-08-03T12:20:54.165080Z","shell.execute_reply.started":"2022-08-03T12:20:54.136086Z","shell.execute_reply":"2022-08-03T12:20:54.163991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_sel_model.get_support()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:54.167259Z","iopub.execute_input":"2022-08-03T12:20:54.167895Z","iopub.status.idle":"2022-08-03T12:20:54.178406Z","shell.execute_reply.started":"2022-08-03T12:20:54.167849Z","shell.execute_reply":"2022-08-03T12:20:54.177141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Print some stats\n\nselected_feat = X_train.columns[(feature_sel_model.get_support())]\nprint(f\"Total features: {X_train.shape[1]}\")\nprint(f\"Selected features: {len(selected_feat)}\")\n#print(f\"Features with coefficients shrank to zero: {np.sum(sel_.estimator_.coef_ == 0)}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:54.180387Z","iopub.execute_input":"2022-08-03T12:20:54.180970Z","iopub.status.idle":"2022-08-03T12:20:54.192096Z","shell.execute_reply.started":"2022-08-03T12:20:54.180921Z","shell.execute_reply":"2022-08-03T12:20:54.190811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_feat","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:54.193849Z","iopub.execute_input":"2022-08-03T12:20:54.194885Z","iopub.status.idle":"2022-08-03T12:20:54.207350Z","shell.execute_reply.started":"2022-08-03T12:20:54.194840Z","shell.execute_reply":"2022-08-03T12:20:54.205915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train[selected_feat]\nX_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:54.209410Z","iopub.execute_input":"2022-08-03T12:20:54.210215Z","iopub.status.idle":"2022-08-03T12:20:54.267645Z","shell.execute_reply.started":"2022-08-03T12:20:54.210165Z","shell.execute_reply":"2022-08-03T12:20:54.266411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.to_csv('submit.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:20:54.269631Z","iopub.execute_input":"2022-08-03T12:20:54.270423Z","iopub.status.idle":"2022-08-03T12:20:54.311392Z","shell.execute_reply.started":"2022-08-03T12:20:54.270375Z","shell.execute_reply":"2022-08-03T12:20:54.310165Z"},"trusted":true},"execution_count":null,"outputs":[]}]}