{"cells":[{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#loading all the required packages\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e79de8bf21eac7cff497197ce64da159f3b5514f"},"cell_type":"code","source":"#loading the train and test dataset\ntrain = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\ntrain_raw = train.copy()\ntest_raw = test.copy()\nprint(\"Train : {}\".format(train.shape))\nprint(\"Test : {}\".format(test.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b99b52b0afa6e6c5eb2fdf5100098438077e989"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c00c6645fe27c6b9474dc314762aec668347c55"},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b4552da85ab43d4590472a1ed13ca15c6069e9e"},"cell_type":"code","source":"#some useful information about the data\ntrain.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4dae336bf6a3ae7407b14766da01819547df7426"},"cell_type":"markdown","source":"There are 38 columns with numeric data and 43 columns with categorical data in the train dataset."},{"metadata":{"trusted":true,"_uuid":"6fcbb9e6eb40cde7f9209da5562242e1d9e71a4f"},"cell_type":"code","source":"test.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5d6e51e6519895664959f2daee16280e8a3a231f"},"cell_type":"markdown","source":"There are 37 columns(*not having SalePrice column*) with numeric data and 43 columns with categorical data in the test dataset."},{"metadata":{"trusted":true,"_uuid":"5937b54169aa56d439bcf70bcc84a58fc8c254a2"},"cell_type":"code","source":"#outliers in grlivarea (indicated in dataset documentation)\nsns.scatterplot(x = train.GrLivArea,y = train.SalePrice)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d313cd8dbecb2cda997af1c95512a00c8c8afb8"},"cell_type":"code","source":"#taking out outliers\ntrain.drop(index = train[(train['GrLivArea'] > 4000) & (train['SalePrice'] < 300000)].index,\n           inplace = True)\nsns.scatterplot(x = train.GrLivArea,y = train.SalePrice)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"07ca20c7477ed65fefdb892dbc20fb8bff17c2b4"},"cell_type":"markdown","source":"# Analysing target variable : SalePrice"},{"metadata":{"trusted":true,"_uuid":"94861d5894d9a604c84b89f4b65f402a01f5efde"},"cell_type":"code","source":"train.SalePrice.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"162c21404a64a69fd334275c4093989efa6dfb49"},"cell_type":"code","source":"#histogram plot\nfrom scipy.stats import norm\nsns.distplot(train.SalePrice,fit = norm)\nprint('Skewness : %f' %train.SalePrice.skew())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e5274014d6370be6e6086da03944b88886e5643"},"cell_type":"code","source":"#log transforming saleprice\ntrain['SalePrice'] = np.log1p(train.SalePrice)\nsns.distplot(train.SalePrice,fit = norm)\nprint('Skewness : %f' %train.SalePrice.skew())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8463e2b56fd8dadcc7f6640b64d3ebab79d37bc8"},"cell_type":"markdown","source":"# Handling missing values"},{"metadata":{"trusted":true,"_uuid":"4fae4f345744d939f6218b4bfc7aacfb8674a446"},"cell_type":"code","source":"#missing values in training set\ntrain.isnull().sum().sort_values(ascending = False)[:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a2b6a35fc2b24e5f11cdca1f43662d8f337b8cda"},"cell_type":"code","source":"#missing values in test set\ntest.isnull().sum().sort_values(ascending = False)[:35]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"16c93eb26fe88324a28b2deae15a9cc47a41999b"},"cell_type":"code","source":"n_train = train.shape[0]\nn_test = test.shape[0]\ny_train = train.SalePrice.values\ntrain.drop(columns = 'SalePrice',inplace = True)\nall_data = pd.concat((train,test)).reset_index(drop = True)\nprint(\"all_data size is : {}\".format(all_data.shape))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"65de20cb39f454f0bdd744a6d30d0d91ab48eef2"},"cell_type":"markdown","source":"**MSZoning** - filling the missing values with mode 'RL'"},{"metadata":{"trusted":true,"_uuid":"7c5152ba36e01b2a847fdd6523db608ae77dda5d"},"cell_type":"code","source":"all_data.MSZoning.fillna(all_data.MSZoning.mode()[0],inplace = True)\n#all_data.MSZoning.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5eb9a1b07b5f0546a1472fc72196195034a74bb9"},"cell_type":"markdown","source":"**LotFrontage** - filling missing values with median LotFrontage of Neighborhood"},{"metadata":{"trusted":true,"_uuid":"4303aa04fb20896e8e8e15672322971e0bf966b3"},"cell_type":"code","source":"all_data['LotFrontage'] = all_data.groupby('Neighborhood')['LotFrontage'].transform(lambda x : \n                                                                                   x.fillna(x.median()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a1ecc7a19ac81a775581dc9683473793da61eb5"},"cell_type":"markdown","source":"**Alley** - NA in Alley means No alley access.Hence filling missing values with 'None'"},{"metadata":{"trusted":true,"_uuid":"887f3e8705f3b16c8e8e56e087746b451d6d5ff1"},"cell_type":"code","source":"all_data.Alley.fillna('None',inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1898c2ec4d0c2cbab7534c254fe259791c34bca7"},"cell_type":"markdown","source":"**PoolQC** - NA means no pool. Hence filling missing values with 'None'"},{"metadata":{"trusted":true,"_uuid":"a88f6047b1f301e379e3ae3c9c980f5a6bec93f6"},"cell_type":"code","source":"all_data.PoolQC.fillna('None',inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d04ef5917995bb11ec8e974e84a8aa5b413e5c97"},"cell_type":"markdown","source":"**MiscFeature** - NA means \"no misc feature\" .Hence filling missing values with 'None'"},{"metadata":{"trusted":true,"_uuid":"f0a193fcf3b91736d293b27dbaa1d53e5121e52a"},"cell_type":"code","source":"all_data.MiscFeature.fillna('None',inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dd790d6b06e3a46df925906011ea09a8823c4e33"},"cell_type":"markdown","source":"**Fence** - NA means \"no fence\""},{"metadata":{"trusted":true,"_uuid":"95231549f70f06bc37199f5c392001020d9cd4cc"},"cell_type":"code","source":"all_data.Fence.fillna('None',inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f1560800cd49e15ea1349f6f2b1420651652c2f6"},"cell_type":"markdown","source":"**FireplaceQu** - NA means \"no fireplace\""},{"metadata":{"trusted":true,"_uuid":"36aab24e598493dec9acbf0b277a78dfc9658f20"},"cell_type":"code","source":"all_data.FireplaceQu.fillna('None',inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f11f452d42a1f8c9b5a73532fd9abb78290ae0d"},"cell_type":"markdown","source":"**Garage Variables** - GarageType, GarageFinish, GarageQual, GarageCond, GarageYrBlt, GarageArea and GarageCars\n\nNA in Garage Variables means no Garage.Hence imputing the numerical features with '0' and imputing the categorical features with 'None'. "},{"metadata":{"trusted":true,"_uuid":"f08085fca2ae832c51c065bbe81c3a295ad5912e"},"cell_type":"code","source":"for feature in ('GarageType','GarageFinish','GarageQual','GarageCond') :\n    all_data[feature].fillna('None',inplace = True)\nfor feature in ('GarageYrBlt','GarageArea','GarageCars') :\n    all_data[feature].fillna(0,inplace = True)    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0ddb9de7ec6de3a46b089fe7732dfe9f5e57708d"},"cell_type":"markdown","source":"**Basement Variables**\n\n* *BsmtFinSF1, BsmtFinSF2, BsmtUnfSF, TotalBsmtSF, BsmtFullBath and BsmtHalfBath* : In these numerical features NA most likely means no basement. Hence we can safely fill the missing values with zero.\n\n* *BsmtQual, BsmtCond, BsmtExposure, BsmtFinType1 and BsmtFinType2* : In these categorical features NA means no basement.Hence imputing the missing values with 'None'. "},{"metadata":{"trusted":true,"_uuid":"c807db692984c74d3976baa3901a9abf80e35773"},"cell_type":"code","source":"for feature in ('BsmtFinSF1', 'BsmtFinSF2', 'BsmtUnfSF','TotalBsmtSF', 'BsmtFullBath', 'BsmtHalfBath'):\n    all_data[feature].fillna(0,inplace = True)\nfor feature in ('BsmtQual', 'BsmtCond', 'BsmtExposure', 'BsmtFinType1', 'BsmtFinType2'):\n    all_data[feature].fillna('None',inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fb8b303dac0e279b18f29061c99247e03f3148a5"},"cell_type":"markdown","source":"**MasVnrArea and MasVnrType** - NA most likely means no masonry veneer for these houses."},{"metadata":{"trusted":true,"_uuid":"c877c7601328db2b22bd5090f4fb565bc7e9dc1f"},"cell_type":"code","source":"all_data[\"MasVnrType\"].fillna(\"None\",inplace = True)\nall_data[\"MasVnrArea\"].fillna(0,inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6700c153c26d623d747aacab323578675eaa852a"},"cell_type":"markdown","source":"**Electrical** - Filling NA with mode."},{"metadata":{"trusted":true,"_uuid":"2da0ed371e4627e57f2765ab119c13c0bebdb4e2"},"cell_type":"code","source":"all_data['Electrical'].fillna(all_data['Electrical'].mode()[0],inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9a65df093191775d83ef02a160435d5f66359ba5"},"cell_type":"markdown","source":"**Utilities** -  Filling NA with mode."},{"metadata":{"trusted":true,"_uuid":"abca232dc9b1c8aea5bae0de7aa6717eeedd250a"},"cell_type":"code","source":"all_data['Utilities'].fillna(all_data['Utilities'].mode()[0],inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"086ec399464152d455b158917b1c51e91d948160"},"cell_type":"markdown","source":"**Functional** - According to data description NA means Typical"},{"metadata":{"trusted":true,"_uuid":"e27f28c3f971daa37c46c8e1284b47a248ab1bda"},"cell_type":"code","source":"all_data['Functional'].fillna('Typ',inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"082e2e8e8a7d413b4851e7f05daba621a7ceec54"},"cell_type":"markdown","source":"**KitchenQual** - Filling NA with mode"},{"metadata":{"trusted":true,"_uuid":"b25d91efae7f96e597e03d9c742e9b2e7ca0c038"},"cell_type":"code","source":"all_data['KitchenQual'].fillna(all_data['KitchenQual'].mode()[0],inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f109506c6e157b931c6c8804adbdc9a8b0822787"},"cell_type":"markdown","source":"**Exterior1st and Exterior2nd** - Filling NA with mode"},{"metadata":{"trusted":true,"_uuid":"70b145b4663cae46a6747bbbe08536a70ab026d6"},"cell_type":"code","source":"all_data['Exterior1st'].fillna(all_data['Exterior1st'].mode()[0],inplace = True)\nall_data['Exterior2nd'].fillna(all_data['Exterior2nd'].mode()[0],inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b8884b63f042151ce30f1cea638cd84cd027f4f5"},"cell_type":"markdown","source":"**SaleType** - Filling NA with mode"},{"metadata":{"trusted":true,"_uuid":"06e0546b73c72c549ce879880e4eb75410acae2d"},"cell_type":"code","source":"all_data['SaleType'].fillna(all_data['SaleType'].mode()[0],inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a62e25523e160a91eaa23dbe838824fe4ec4acec"},"cell_type":"markdown","source":"# **Feature Engineering**"},{"metadata":{"trusted":true,"_uuid":"ce3fa971b110c62eb13fdc1281efd5e3c525dc85"},"cell_type":"code","source":"#some numerical features are actually categorical\nfor feature in ('MSSubClass','MoSold') :\n    all_data[feature] = all_data[feature].astype(str)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ec207af53cda6d7c87d5534d74adf17b75ee261"},"cell_type":"code","source":"#mapping the quality variables to numerical\n#lower quality to higher quality mapped to lower number to higher number\nqual_dict = {'None': 0,'Po': 1,'Fa': 2,'TA': 3,'Gd': 4,'Ex': 5 }\nfor feat in ('ExterQual','ExterCond','BsmtQual','BsmtCond','HeatingQC','KitchenQual','FireplaceQu',\n            'GarageQual','GarageCond','PoolQC') :\n    all_data[feat] = all_data[feat].map(qual_dict).astype(int)\n    \n#further mapping\nall_data[\"BsmtExposure\"] = all_data[\"BsmtExposure\"].map(\n    {\"None\": 0, \"No\": 1, \"Mn\": 2, \"Av\": 3, \"Gd\": 4}).astype(int)\nall_data[\"Street\"] = all_data[\"Street\"].map({'None': 0 , 'Grvl': 1 , 'Pave': 2}).astype(int)\nall_data[\"Alley\"] = all_data[\"Alley\"].map({'None': 0 , 'Grvl': 1 , 'Pave': 2}).astype(int)\n\nbsmtfin_dict = {\"None\": 0, \"Unf\": 1, \"LwQ\": 2, \"Rec\": 3, \"BLQ\": 4, \"ALQ\": 5, \"GLQ\": 6}\nall_data[\"BsmtFinType1\"] = all_data[\"BsmtFinType1\"].map(bsmtfin_dict).astype(int)\nall_data[\"BsmtFinType2\"] = all_data[\"BsmtFinType2\"].map(bsmtfin_dict).astype(int)\n\nall_data[\"CentralAir\"] = all_data[\"CentralAir\"].map({\"N\": 0 , \"Y\": 1})\nall_data[\"Functional\"] = all_data[\"Functional\"].map({\"Sal\": 1, \"Sev\": 1, \"Maj2\": 2, \"Maj1\": 2, \n         \"Mod\": 3, \"Min2\": 3, \"Min1\": 3, \"Typ\": 4}).astype(int)\nall_data[\"GarageFinish\"] = all_data[\"GarageFinish\"].map(\n    {\"None\": 0, \"Unf\": 1, \"RFn\": 2,\"Fin\": 3}).astype(int)\nall_data[\"Fence\"] = all_data[\"Fence\"].map(\n        {\"None\": 0, \"MnWw\": 1, \"GdWo\": 2, \"MnPrv\": 3, \"GdPrv\": 4}).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"95d0b28ffb0d37bc02513c839a7fd01ffe6a7df0"},"cell_type":"code","source":"#Extracting Features\n#LotShape - Distinguishing just between regular and irregular as IR3 and IR2 appears rarely\nall_data['IsRegularLotShape'] = (all_data['LotShape'] == 'Reg') * 1\n#LandContour - Most properties are level\nall_data['IsLandLevel'] = (all_data['LandContour'] == 'Lvl') * 1\n#LandSlope - most land slopes are gentle\nall_data['IsGentleSlope'] = (all_data['LandSlope'] == 'Gtl') * 1\n#Electrical - Most properties have standard circuit breakers\nall_data['IsElectricalSBrkr'] = (all_data['Electrical'] == 'SBrkr') * 1\n#GarageType - Almost all the properties have attached garage\nall_data['IsGargageDetached'] = (all_data['GarageType'] == 'Detchd') * 1\n#MiscFeature - Presence or absence\nall_data['HasMiscFeature'] = (all_data['MiscFeature'] != 'None') * 1\n#PavedDrive - Ispaved drive or not\nall_data['IsPavedDrive'] = (all_data['PavedDrive'] == 'Y') * 1\n#House not completed\nall_data['HouseNotCompleted'] = (all_data['SaleCondition'] == 'Partial') * 1\n\n# complete information is extracted from these two , so dropping these features\ndropcols = ['MiscFeature','PavedDrive']\nall_data.drop(dropcols,axis = 1,inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"47f22d63344495cca99d80cfba9609385bb600c3"},"cell_type":"code","source":"#Creating new features\n#whether the house was remodelled once\nall_data['Remodelled'] = (all_data['YearBuilt'] != all_data['YearRemodAdd']) * 1\n#whether it is a new house\nall_data['NewHouse'] = (all_data['YearBuilt'] == all_data['YrSold']) * 1\n#whether remodelled recently\nall_data['RecentRemodel'] = (all_data['YearRemodAdd'] == all_data['YrSold']) * 1\n#Total sqft for house\nall_data['AllSF'] = all_data['GrLivArea'] + all_data['TotalBsmtSF']\n#Total full bathrooms\nall_data['TotalFullBath'] = all_data['FullBath'] + all_data['BsmtFullBath']\n#Total half bathrooms\nall_data['TotalHalfBath'] = all_data['HalfBath'] + all_data['BsmtHalfBath']\n#All floors sqft\nall_data['AllFlrSF'] = all_data['1stFlrSF'] + all_data['2ndFlrSF']\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"23572a5059a3d9cff57e5e02a9d14e306737d565"},"cell_type":"markdown","source":"'YearBuilt' ,'YearRemodAdd','GarageYrBlt','YrSold'"},{"metadata":{"trusted":true,"_uuid":"be0ee81a99a4787352b5a30728c18782c6d0ac36"},"cell_type":"code","source":"#Handling year variables\nfor feat in ('YearBuilt','YearRemodAdd','GarageYrBlt','YrSold') :\n    all_data[feat] = pd.qcut(all_data[feat],q = 10,duplicates = 'drop')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45a1c59c72bf8c43dcfb21cd8a29d4bcc51b9590"},"cell_type":"code","source":"#splitting up into train and test sets\ntrain = all_data[:n_train]\ntest = all_data[n_train:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa360a7c7eeca0075d4bd88dda9958ed1dbc330b"},"cell_type":"code","source":"#applying log transformation for highly skewed features\nfrom scipy.stats import skew\nnumeric_features = train.select_dtypes(exclude = [object,'category']).drop(columns = ['Id']).columns\nskewness = train[numeric_features].apply(lambda x : skew(x.dropna()))\nskewed_features = skewness[abs(skewness) > 0.75].index\ntrain[skewed_features] = np.log1p(train[skewed_features])\ntest[skewed_features] = np.log1p(test[skewed_features])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2a7cb0151e779bb4193dbc7e0ec39c2d3e791629"},"cell_type":"code","source":"train = pd.get_dummies(train)\ntest = pd.get_dummies(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b11a19e89a024b53b4036082980bad991ef70c86"},"cell_type":"code","source":"print(train.shape)\nprint(test.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0cde7da1e4e59c4d62cbd447f2bffd54113cf432"},"cell_type":"markdown","source":"There exists some features which are only in training set and not in test set.Removing those features so as not to overfit on the training set."},{"metadata":{"trusted":true,"_uuid":"78c240a6262717dcb08095dc4332d627e4623028"},"cell_type":"code","source":"common_features = train.columns & test.columns\ntrain = train[common_features]\ntest = test[common_features]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"971ace5e774086763704436ef4d10895e8228dcd"},"cell_type":"markdown","source":"# **Modelling**"},{"metadata":{"trusted":true,"_uuid":"670e709937eff9976cba66f230d0983f99249035"},"cell_type":"code","source":"#defining our error metric\nfrom sklearn.model_selection import KFold,cross_val_score\ndef rmse(model) :\n    kf = KFold(n_splits = 5,shuffle = True,random_state = 6)\n    rmse = np.sqrt(-1 * cross_val_score(model,X = train,y = y_train,\n                                        scoring = \"neg_mean_squared_error\",cv = kf))\n    return rmse\n    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"37acafb9cb8875c7a43c7c51d51e7dacfa79103e"},"cell_type":"markdown","source":"# Linear Regression"},{"metadata":{"trusted":true,"_uuid":"486b334c87defcab4cf0a91377585d4e22d6e07c"},"cell_type":"code","source":"#Linear Regression\nfrom sklearn.linear_model import LinearRegression\nlr = LinearRegression()\nprint(\"Linear Regression score : {:.4f}\".format(rmse(lr).mean()))\nlr.fit(train,y_train)\npred_lr = np.exp(lr.predict(test))\nsubmission_lr = pd.DataFrame({\"Id\" : test_raw[\"Id\"], \"SalePrice\" : pred_lr })\nsubmission_lr.to_csv(\"linear_regression\",index = False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"874da98ae2039e30e42b14bc6f0f0e2f2e947b81"},"cell_type":"markdown","source":"The above submission scored 0.13847 on public LB."},{"metadata":{"_uuid":"87f97ccdeac13df88cdd8f03d2b3a1014c72d899"},"cell_type":"markdown","source":"# Ridge Regression"},{"metadata":{"trusted":true,"_uuid":"0659550cae8bf00dbc9e370b1e65142b696a6fc6"},"cell_type":"code","source":"#Let's do cross validation to find the best alpha\n# from sklearn.linear_model import Ridge, RidgeCV\n# kfold = KFold(n_splits = 5,shuffle = True,random_state = 6)\n# ridgecv = RidgeCV(alphas = [0.01,0.03,0.06,0.1,0.3,0.6,1,3,6,10,30,60],\n#                   scoring = 'neg_mean_squared_error',cv = kfold)\n# ridgecv.fit(train,y_train)\n# print(\"Best alpha {}\".format(ridgecv.alpha_))\n# #further cross validation with alphas centred around best alpha\n# alphas = [coeff * ridgecv.alpha_ for coeff in np.arange(0.6,1.4,0.05)]\n# ridgecv = RidgeCV(alphas = alphas,scoring = 'neg_mean_squared_error',cv = kfold)\n# ridgecv.fit(train,y_train)\n# print(\"Best alpha after further cross validation {}\".format(ridgecv.alpha_))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60713b08831944dd1030e9bf544b88cf2c6ff899"},"cell_type":"code","source":"#fitting ridge regression\nfrom sklearn.linear_model import Ridge, RidgeCV\nridge = Ridge(alpha = 6.3) #found by cross validation\nprint(\"Ridge Regression score : {:.4f}\".format(rmse(ridge).mean()))\nridge.fit(train,y_train)\npred_rr = np.exp(ridge.predict(test))\nsubmission_rr = pd.DataFrame({\"Id\" : test_raw[\"Id\"], \"SalePrice\" : pred_rr })\nsubmission_rr.to_csv(\"ridge_regression\",index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b49dba7d72b5d2375164e3fe5849d453d9d42d9"},"cell_type":"markdown","source":"The above submission scored 0.12467 on public LB"},{"metadata":{"_uuid":"ef0b5f3c8c514480daccc8966c336f754ce66c20"},"cell_type":"markdown","source":"# LASSO Regression \n**Least Absolute Shrinkage Selector Operator**"},{"metadata":{"trusted":true,"_uuid":"71b82d5d8cc250ebcdf08f80a65a21d7f1236dc6"},"cell_type":"code","source":"#cross validation to find optimum alpha for lasso regression\n# from sklearn.linear_model import LassoCV,Lasso\n# kfold = KFold(n_splits = 5,shuffle = True,random_state = 6)\n# lassocv = LassoCV(alphas = [0.0001,0.0003,0.0006,0.001,0.003,0.006,0.01,0.03,0.06,0.1,0.3,0.6,1,3,6],\n#                  max_iter = 50000,cv = kfold)\n# lassocv.fit(train,y_train)\n# print(\"Best alpha {}\".format(lassocv.alpha_))\n# #further cross validation with alphas centred around best alpha\n# alphas = [coeff * lassocv.alpha_ for coeff in np.arange(0.6,1.4,0.05)]\n# lassocv = LassoCV(alphas = alphas,max_iter = 50000,cv = kfold)\n# lassocv.fit(train,y_train)\n# print(\"Best alpha after further cross validation {}\".format(lassocv.alpha_))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9dff62d2f06c9a1fbc8c88e0f97530633012018c"},"cell_type":"code","source":"#fitting lasso regression\nfrom sklearn.linear_model import LassoCV,Lasso\nlasso = Lasso(alpha = 0.00039,max_iter = 50000) #found by cross validation\nprint(\"Lasso Regression score : {:.4f}\".format(rmse(lasso).mean()))\nlasso.fit(train,y_train)\npred_lasso = np.exp(lasso.predict(test))\nsubmission_lasso = pd.DataFrame({\"Id\" : test_raw[\"Id\"], \"SalePrice\" : pred_lasso })\nsubmission_lasso.to_csv(\"lasso_regression\",index = False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3fbd70e03e055d2c51268ec446bc36fe216a05ee"},"cell_type":"markdown","source":"The above submission scored 0.12424 in public LB"},{"metadata":{"_uuid":"31ded4fa47cb7eee04f9b37470f301583c142a70"},"cell_type":"markdown","source":"# ElasticNet Regression"},{"metadata":{"trusted":true,"_uuid":"c925d792fd034bf1ef36b4fbcc77eecbf638bd34"},"cell_type":"code","source":"#cross validation to find the optimum alpha and l1_ratio\n# from sklearn.linear_model import ElasticNetCV, ElasticNet\n# kfold = KFold(n_splits = 5,shuffle = True,random_state = 6)\n# encv = ElasticNetCV(alphas = [0.0001,0.0003,0.0006,0.001,0.003,0.006,0.01,0.03,0.06,0.1,0.3,0.6,1,3,6],\n#                    l1_ratio = [0.1, 0.3, 0.5, 0.6, 0.7, 0.8, 0.85, 0.9, 0.95, 1],\n#                     max_iter = 50000,cv = kfold)\n# encv.fit(train,y_train)\n# print(\"Best alpha {}\".format(encv.alpha_))\n# print(\"Best l1_ratio {}\".format(encv.l1_ratio_))\n# #further cross validation\n# alphas = [coeff * encv.alpha_ for coeff in np.arange(0.6,1.4,0.05)]\n# l1_ratios = [coeff * encv.l1_ratio_ for coeff in np.arange(0.6,1.4,0.05)]\n# encv = ElasticNetCV(alphas = alphas,l1_ratio = l1_ratios,max_iter = 50000,cv = kfold)\n# encv.fit(train,y_train)\n# print(\"Best alpha after further cross validation {}\".format(encv.alpha_))\n# print(\"Best l1_ratio after further cross validation {}\".format(encv.l1_ratio_))\n# #further narrowing\n# alphas = [coeff * encv.alpha_ for coeff in np.arange(0.6,1.4,0.05)]\n# l1_ratios = [coeff * encv.l1_ratio_ for coeff in np.arange(0.9,1.2,0.05)]\n# encv = ElasticNetCV(alphas = alphas,l1_ratio = l1_ratios,max_iter = 50000,cv = kfold)\n# encv.fit(train,y_train)\n# print(\"Best alpha after further cross validation {}\".format(encv.alpha_))\n# print(\"Best l1_ratio after further cross validation {}\".format(encv.l1_ratio_))\n# #further narrowing (as it is best to keep l1_ratio near to 1 (that is more inclined towards Lasso))\n# alphas = [coeff * encv.alpha_ for coeff in np.arange(0.6,1.4,0.05)]\n# l1_ratios = [coeff * encv.l1_ratio_ for coeff in np.arange(0.9,1.1,0.05)]\n# encv = ElasticNetCV(alphas = alphas,l1_ratio = l1_ratios,max_iter = 50000,cv = kfold)\n# encv.fit(train,y_train)\n# print(\"Best alpha after further cross validation {}\".format(encv.alpha_))\n# print(\"Best l1_ratio after further cross validation {}\".format(encv.l1_ratio_))\n#Both l1_ratio and alpha are converged","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d368f9c1750b6d07dd1d2aa29ce5204c18c5d67f"},"cell_type":"code","source":"#fitting ElasticNet Regression\nfrom sklearn.linear_model import ElasticNetCV, ElasticNet\nen = ElasticNet(alpha = 0.000432,l1_ratio = 0.891,\n               max_iter = 50000) #found by a series of cross validation and narrowing down\nprint(\"ElasticNet Regression score : {:.4f}\".format(rmse(en).mean()))\nen.fit(train,y_train)\npred_en = np.exp(en.predict(test))\nsubmission_en = pd.DataFrame({\"Id\" : test_raw[\"Id\"], \"SalePrice\" : pred_en})\nsubmission_en.to_csv(\"ElasticNet_regression\",index = False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"703e107143bf6ab2a1eab6333810d4efc9021105"},"cell_type":"markdown","source":"The above submission scored 0.12421 in public LB which is not much different from Lasso."}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}