{"cells":[{"metadata":{"_uuid":"1aac485aefe8721f37f0de2a8b64c765376d4e5d"},"cell_type":"markdown","source":"## Some credit here due to https://www.kaggle.com/apapiu/regularized-linear-models/notebook\n\n## **Import and quick view of the data**"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport os\nfrom copy import deepcopy\n\nprint(os.listdir(\"../input\"))\n\nfrom sklearn.linear_model import Ridge\nfrom sklearn.linear_model import Lasso\nfrom sklearn.metrics import mean_squared_error","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')\n\n# test doesn't have 'SalePrice'\n# Note that the 'Id' for train range 1-1460 and test range 1461-2919\nprint('Train Shape: {}, Train ID Range {} to {}'.format(train.shape,train.Id.min(), train.Id.max()))\nprint('Test Shape: {}, Test ID Range {} to {}'.format(test.shape,test.Id.min(), test.Id.max()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"652d4f017c5338efc601dc5bacd502c7d5136f3d"},"cell_type":"code","source":"train.head(3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"787a77292833242539a6cc7031781ae8b6a7ff17"},"cell_type":"markdown","source":"## **Combining train and test for all preprocessing to be done on one dataframe**"},{"metadata":{"trusted":true,"_uuid":"ef92e676e77c0e4815f1c9820f6c64a208e26600"},"cell_type":"code","source":"all_data = train.append(test,sort = False)\nprint(all_data.shape)\nall_data.head(3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f11110891f15432211cab1ca250052007785f2e8"},"cell_type":"markdown","source":"### **Data type counts**"},{"metadata":{"trusted":true,"_uuid":"b70775b9186e1d52a5d2a6d9d1c4ef2586bc63a1"},"cell_type":"code","source":"all_data.dtypes.groupby(all_data.dtypes).count()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"81ef69f7ceb38cc5b2cb0932da5d4cf69f751e01"},"cell_type":"markdown","source":"### **Checking out numeric data types**"},{"metadata":{"trusted":true,"_uuid":"1192e0d6e88fcc0ca47f3e196469b01de9a4d1b2"},"cell_type":"code","source":"numerics = all_data.dtypes[all_data.dtypes != object].index.tolist()\nprint(numerics)\nprint(len(numerics))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2ee5d1c4b7f2a9b95db231eac488e7895e989af1"},"cell_type":"markdown","source":"## Looking for skewed data to log transform\n### Will log-transform those with skew > .75"},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"d2d12a8cf03fa9a4390c5db877ac4ac128911d0e"},"cell_type":"code","source":"skew_calcs = all_data[numerics].skew().sort_values()\n# Note that the skew filter below removes Id from the skewed feature list - we wouldn't have wanted to log-transform Id\nskew_calcs = skew_calcs[skew_calcs > .75]\nprint(skew_calcs)\nskew_feats = skew_calcs.index\n# Also note that SalePrice will be log_transformed","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f46a00df063d8ae21fa4bcf36a709ec00e08baa8"},"cell_type":"markdown","source":"### **Quick Visual of Distribution of skewed features before and after log transformation**"},{"metadata":{"trusted":true,"_uuid":"fad3418a8003aacede2c552b9aa1248389493918"},"cell_type":"code","source":"all_data[skew_feats].hist()\nnp.log1p(all_data[skew_feats]).hist()\nplt.show()\n#plt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e3bd63c922de0b8461c097caf60b9f6a7510b0aa"},"cell_type":"markdown","source":"## **Log Transforming Skewed features**"},{"metadata":{"trusted":true,"_uuid":"4f2274252b257df87da2f3558b492ce95150f8e9"},"cell_type":"code","source":"all_data[skew_feats] = np.log1p(all_data[skew_feats])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7ac505e829c015a6a5ecda9b5111bdc2881f3605"},"cell_type":"markdown","source":"### Checking object data types"},{"metadata":{"trusted":true,"scrolled":false,"_uuid":"4f2274252b257df87da2f3558b492ce95150f8e9"},"cell_type":"code","source":"#creating list of objects and View the unique strings in columns with object datatype\nobjects = train.dtypes[train.dtypes == object].index.tolist()\nprint(objects)\ntrain[objects].apply(lambda x: set(x))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bc9e440c23b28d81cc8ab54f5163d7c700b873e1"},"cell_type":"markdown","source":"### It looks like we can create dummy variables for all of the objects, since there aren't too many to handle..."},{"metadata":{"trusted":true,"_uuid":"2ef0a5b77a74d6773d36acbe7fb94d7e694e6516"},"cell_type":"code","source":"# Create Dummy Variables\nall_data = pd.concat([all_data,pd.get_dummies(all_data[objects])],axis = 1)\nprint(all_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ef0a5b77a74d6773d36acbe7fb94d7e694e6516","collapsed":true},"cell_type":"code","source":"# Remove original features now that dummies have been created\nall_data = all_data.drop(objects,axis = 1)\nprint(all_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"576d54fca986fd7630cfbcd5416b3ca118434f39"},"cell_type":"code","source":"all_data = all_data.fillna(all_data.mean())\n# could also create dummy variable for nulls\nall_data.head(3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bbf8228a9f5f06289d38e111d0aba1932415e8e3"},"cell_type":"markdown","source":"### **Create Train and Test**"},{"metadata":{"trusted":true,"_uuid":"4ec91f44fe9acab6dbb053fc0fa0facaff4c040e"},"cell_type":"code","source":"train_data = all_data[:1460]\n# drop SalePrice from test_data, was created during .fillna(all_data.mean())\ntest_data = all_data[1460:].drop('SalePrice',axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"021174824181f015ab467c766bd72b97f0b4fd8a"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb8fcc2f180f2529686d5e1e3020b0a6acdb14a5"},"cell_type":"code","source":"X = train_data.drop(['SalePrice'],axis = 1)\nprint(X.shape)\ny = train_data['SalePrice']\nprint(y.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"90e47c3fac41f0fc238a553e30cd20c57db8afee"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba57c60231534d6996cc63b7fc3f84d31748c0db"},"cell_type":"code","source":"def get_mse(model,alpha):\n    run_model = model(alpha = alpha)\n    run_model.fit(X_train,y_train)\n    return mean_squared_error(y_pred=run_model.predict(X_test),y_true=y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4733e46784f29a5ec9f0b4b7ec0a594eaf46f021"},"cell_type":"code","source":"alphas = [.01,.05,0.1,0.5,1,2,3,5,10,20,40]\nridge_mses = [get_mse(Ridge,x) for x in alphas]\nplt.plot(alphas,ridge_mses)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dbbc354c641e824b2c0fe08e0b69e551bcf7801a"},"cell_type":"code","source":"lasso_mses = [get_mse(Lasso,x) for x in alphas]\nplt.plot(alphas,lasso_mses)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f3f70cb3ce4b7c400c6a96e776e735b73ece00f7"},"cell_type":"code","source":"# alpha = 5 looks best\npd.DataFrame({'alpha':alphas, 'MSE':ridge_mses}).sort_values('MSE',ascending = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58dd61c8795456213d8341681b9a84fce6e2e3c5"},"cell_type":"code","source":"ridge_model = Ridge(alpha = 5)\nridge_model.fit(X,y)\ncoefs = ridge_model.coef_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54e38fcf338da547bba5e9c440f570e4967a0bc3"},"cell_type":"code","source":"labels_and_weights = pd.DataFrame({'field':X.columns,'weight':coefs})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2678e1c090f81a7d1520db8964fb74eb862c1a98"},"cell_type":"code","source":"top_weight = labels_and_weights['weight'].quantile(.97)\nbot_weight = labels_and_weights['weight'].quantile(.03)\nheavy_weights = labels_and_weights[(labels_and_weights['weight']>top_weight) | (labels_and_weights['weight']<bot_weight)].sort_values('weight')\nplt.barh(heavy_weights['field'],heavy_weights['weight'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b4cbd9d1cad81aab3556edf85fdddba4c30a9ce"},"cell_type":"code","source":"predictions = ridge_model.predict(test_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cd5e972ddb341fb6bf3081d592c0ad1db9d3eb19"},"cell_type":"code","source":"print(predictions.mean())\n# inverse log-transform\npreds = np.expm1(predictions)\nprint(preds.mean())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"75ee4b2842c9963d71dc2a1738edefa89ac3c8ba"},"cell_type":"code","source":"sub = pd.DataFrame()\nsub['Id'] = test_data['Id']\nsub['SalePrice'] = preds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a1648f42f4e506da178dee63496ee533f1f1b172"},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fccf77c50e0d0cff566474fcb27913b9a04136ff"},"cell_type":"code","source":"sub.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}