{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T16:36:31.663663Z","iopub.execute_input":"2022-07-23T16:36:31.664208Z","iopub.status.idle":"2022-07-23T16:36:31.672575Z","shell.execute_reply.started":"2022-07-23T16:36:31.664164Z","shell.execute_reply":"2022-07-23T16:36:31.671561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom scipy.stats import norm\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy import stats\nimport warnings\nwarnings.filterwarnings('ignore')\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:36:31.674249Z","iopub.execute_input":"2022-07-23T16:36:31.675019Z","iopub.status.idle":"2022-07-23T16:36:31.685931Z","shell.execute_reply.started":"2022-07-23T16:36:31.674926Z","shell.execute_reply":"2022-07-23T16:36:31.684546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/house-prices-advanced-regression-techniques/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:36:31.688209Z","iopub.execute_input":"2022-07-23T16:36:31.688874Z","iopub.status.idle":"2022-07-23T16:36:31.728809Z","shell.execute_reply.started":"2022-07-23T16:36:31.688832Z","shell.execute_reply":"2022-07-23T16:36:31.727478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corrmat = train.corr()\nf, ax = plt.subplots(figsize=(12, 9))\nsns.heatmap(corrmat, vmax=.8, square=True);","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:36:31.731521Z","iopub.execute_input":"2022-07-23T16:36:31.731955Z","iopub.status.idle":"2022-07-23T16:36:32.271324Z","shell.execute_reply.started":"2022-07-23T16:36:31.731918Z","shell.execute_reply":"2022-07-23T16:36:32.270044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k = 10 #number of variables for heatmap\ncols = corrmat.nlargest(k, 'SalePrice')['SalePrice'].index\ncm = np.corrcoef(train[cols].values.T)\nsns.set(font_scale=1.25)\nhm = sns.heatmap(cm, cbar=True, annot=True, square=True, fmt='.2f', annot_kws={'size': 10}, yticklabels=cols.values, xticklabels=cols.values)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:44:37.488869Z","iopub.execute_input":"2022-07-23T16:44:37.489434Z","iopub.status.idle":"2022-07-23T16:44:38.216836Z","shell.execute_reply.started":"2022-07-23T16:44:37.489390Z","shell.execute_reply":"2022-07-23T16:44:38.215651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k = 20 #number of variables for heatmap\ncols = corrmat.nlargest(k, 'SalePrice')['SalePrice'].index\ncm = np.corrcoef(train[cols].values.T)\nsns.set(font_scale=1.25)\nhm = sns.heatmap(cm, yticklabels=cols.values, xticklabels=cols.values)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:44:17.321796Z","iopub.execute_input":"2022-07-23T16:44:17.322251Z","iopub.status.idle":"2022-07-23T16:44:17.831704Z","shell.execute_reply.started":"2022-07-23T16:44:17.322215Z","shell.execute_reply":"2022-07-23T16:44:17.830426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scatterplot\nsns.set()\ncols = ['SalePrice', 'OverallQual', 'GrLivArea', 'GarageCars', 'TotalBsmtSF', 'FullBath', 'YearBuilt']\nsns.pairplot(train[cols], size = 2.5)\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:49:31.773614Z","iopub.execute_input":"2022-07-23T16:49:31.774123Z","iopub.status.idle":"2022-07-23T16:49:41.057639Z","shell.execute_reply.started":"2022-07-23T16:49:31.774079Z","shell.execute_reply":"2022-07-23T16:49:41.056438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature=['YearBuilt', 'YearRemodAdd', 'BsmtFinSF1', 'BsmtFinSF2', 'BsmtUnfSF', 'TotalBsmtSF',  '1stFlrSF', '2ndFlrSF',\n       'LowQualFinSF', 'GrLivArea', 'BsmtFullBath', 'BsmtHalfBath', 'FullBath',\n       'HalfBath', 'BedroomAbvGr',  \n        'GarageCars', 'GarageArea', \n     'WoodDeckSF', 'OpenPorchSF',\n        'PoolArea']","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:36:32.972358Z","iopub.execute_input":"2022-07-23T16:36:32.972898Z","iopub.status.idle":"2022-07-23T16:36:32.978760Z","shell.execute_reply.started":"2022-07-23T16:36:32.972852Z","shell.execute_reply":"2022-07-23T16:36:32.978020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=train[feature]","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:36:32.979764Z","iopub.execute_input":"2022-07-23T16:36:32.980685Z","iopub.status.idle":"2022-07-23T16:36:32.992716Z","shell.execute_reply.started":"2022-07-23T16:36:32.980653Z","shell.execute_reply":"2022-07-23T16:36:32.991784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y=train.SalePrice","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:36:32.993767Z","iopub.execute_input":"2022-07-23T16:36:32.994669Z","iopub.status.idle":"2022-07-23T16:36:33.002480Z","shell.execute_reply.started":"2022-07-23T16:36:32.994632Z","shell.execute_reply":"2022-07-23T16:36:33.001372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeRegressor\nmodel=DecisionTreeRegressor()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:36:33.004840Z","iopub.execute_input":"2022-07-23T16:36:33.005317Z","iopub.status.idle":"2022-07-23T16:36:33.222489Z","shell.execute_reply.started":"2022-07-23T16:36:33.005274Z","shell.execute_reply":"2022-07-23T16:36:33.221259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from  sklearn.model_selection import train_test_split\nX_train,X_test,Y_train,Y_test=train_test_split(X,Y,random_state=0)\nmodel.fit(X_train,Y_train)\nY_pred = model.predict(X_test)\nacc_decision_tree = round(model.score(X_train, Y_train) * 100, 2)\nacc_decision_tree","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:36:33.224391Z","iopub.execute_input":"2022-07-23T16:36:33.224846Z","iopub.status.idle":"2022-07-23T16:36:33.256922Z","shell.execute_reply.started":"2022-07-23T16:36:33.224798Z","shell.execute_reply":"2022-07-23T16:36:33.255772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nmodel1=LogisticRegression(random_state=0)\nmodel1.fit(X_train,Y_train)\nY_pred = model1.predict(X_test)\nacc_logistic_regression = round(model1.score(X_train, Y_train) * 100, 2)\nacc_logistic_regression","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:36:33.259149Z","iopub.execute_input":"2022-07-23T16:36:33.259773Z","iopub.status.idle":"2022-07-23T16:36:43.513089Z","shell.execute_reply.started":"2022-07-23T16:36:33.259737Z","shell.execute_reply":"2022-07-23T16:36:43.511867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nmodel2=LinearRegression()\nmodel2.fit(X_train,Y_train)\nY_pred = model2.predict(X_test)\nacc_linear_regression = round(model2.score(X_train, Y_train) * 100, 2)\nacc_linear_regression\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T16:36:43.514917Z","iopub.execute_input":"2022-07-23T16:36:43.515367Z","iopub.status.idle":"2022-07-23T16:36:43.554165Z","shell.execute_reply.started":"2022-07-23T16:36:43.515326Z","shell.execute_reply":"2022-07-23T16:36:43.552989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = pd.DataFrame({\n    'Model': ['Logistic Regression', \n               'Linear Regression', \n              'Decision Tree'],\n    'Score': [acc_linear_regression, acc_logistic_regression, acc_decision_tree, ]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T17:06:40.729470Z","iopub.execute_input":"2022-07-23T17:06:40.730021Z","iopub.status.idle":"2022-07-23T17:06:40.753336Z","shell.execute_reply.started":"2022-07-23T17:06:40.729957Z","shell.execute_reply":"2022-07-23T17:06:40.751841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}