{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T20:50:08.650645Z","iopub.execute_input":"2022-07-11T20:50:08.651386Z","iopub.status.idle":"2022-07-11T20:50:08.663194Z","shell.execute_reply.started":"2022-07-11T20:50:08.651281Z","shell.execute_reply":"2022-07-11T20:50:08.661770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DF3 = pd.read_csv(r'/kaggle/input/house-prices-advanced-regression-techniques/sample_submission.csv')\nDF3","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:08.664738Z","iopub.execute_input":"2022-07-11T20:50:08.665044Z","iopub.status.idle":"2022-07-11T20:50:08.704812Z","shell.execute_reply.started":"2022-07-11T20:50:08.665012Z","shell.execute_reply":"2022-07-11T20:50:08.704057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install lightgbm","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:08.705934Z","iopub.execute_input":"2022-07-11T20:50:08.706337Z","iopub.status.idle":"2022-07-11T20:50:15.913243Z","shell.execute_reply.started":"2022-07-11T20:50:08.706307Z","shell.execute_reply":"2022-07-11T20:50:15.911881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport datetime","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:15.915356Z","iopub.execute_input":"2022-07-11T20:50:15.915703Z","iopub.status.idle":"2022-07-11T20:50:15.920703Z","shell.execute_reply.started":"2022-07-11T20:50:15.915648Z","shell.execute_reply":"2022-07-11T20:50:15.919574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, importing the CSV trained version file called houseprices_data.csv which contains pre-processed information about housing prices.","metadata":{}},{"cell_type":"code","source":"# Reading in the CSV file as a DataFrame\ndf = pd.read_csv(r'/kaggle/input/house-prices-advanced-regression-techniques/train.csv', low_memory=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:17.495894Z","iopub.execute_input":"2022-07-11T20:50:17.496236Z","iopub.status.idle":"2022-07-11T20:50:17.537153Z","shell.execute_reply.started":"2022-07-11T20:50:17.496205Z","shell.execute_reply":"2022-07-11T20:50:17.536273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:20.435495Z","iopub.execute_input":"2022-07-11T20:50:20.435860Z","iopub.status.idle":"2022-07-11T20:50:20.477949Z","shell.execute_reply.started":"2022-07-11T20:50:20.435827Z","shell.execute_reply":"2022-07-11T20:50:20.477071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the shape\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:20.849241Z","iopub.execute_input":"2022-07-11T20:50:20.849904Z","iopub.status.idle":"2022-07-11T20:50:20.856498Z","shell.execute_reply.started":"2022-07-11T20:50:20.849865Z","shell.execute_reply":"2022-07-11T20:50:20.855704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:21.767555Z","iopub.execute_input":"2022-07-11T20:50:21.767929Z","iopub.status.idle":"2022-07-11T20:50:21.794882Z","shell.execute_reply.started":"2022-07-11T20:50:21.767897Z","shell.execute_reply":"2022-07-11T20:50:21.794035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n**Lets check all Null Values**","metadata":{}},{"cell_type":"code","source":"\ndf.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:23.240735Z","iopub.execute_input":"2022-07-11T20:50:23.241365Z","iopub.status.idle":"2022-07-11T20:50:23.256123Z","shell.execute_reply.started":"2022-07-11T20:50:23.241328Z","shell.execute_reply":"2022-07-11T20:50:23.254973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import missingno as mn \nmn.heatmap(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:24.194781Z","iopub.execute_input":"2022-07-11T20:50:24.195427Z","iopub.status.idle":"2022-07-11T20:50:25.924477Z","shell.execute_reply.started":"2022-07-11T20:50:24.195390Z","shell.execute_reply":"2022-07-11T20:50:25.923564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we replace LotFrontage with mode\ndf['LotFrontage'].mode()\ndf['LotFrontage'].fillna(60,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:25.925650Z","iopub.execute_input":"2022-07-11T20:50:25.926067Z","iopub.status.idle":"2022-07-11T20:50:25.933309Z","shell.execute_reply.started":"2022-07-11T20:50:25.926035Z","shell.execute_reply":"2022-07-11T20:50:25.932490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['LotFrontage'].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:25.934550Z","iopub.execute_input":"2022-07-11T20:50:25.934980Z","iopub.status.idle":"2022-07-11T20:50:25.947101Z","shell.execute_reply.started":"2022-07-11T20:50:25.934948Z","shell.execute_reply":"2022-07-11T20:50:25.946155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Alley'].fillna(df['Alley'].mode,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:26.273493Z","iopub.execute_input":"2022-07-11T20:50:26.274041Z","iopub.status.idle":"2022-07-11T20:50:26.278889Z","shell.execute_reply.started":"2022-07-11T20:50:26.274001Z","shell.execute_reply":"2022-07-11T20:50:26.278006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Alley'].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:27.326341Z","iopub.execute_input":"2022-07-11T20:50:27.326728Z","iopub.status.idle":"2022-07-11T20:50:27.333445Z","shell.execute_reply.started":"2022-07-11T20:50:27.326690Z","shell.execute_reply":"2022-07-11T20:50:27.332501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['PoolQC'].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:27.897119Z","iopub.execute_input":"2022-07-11T20:50:27.897527Z","iopub.status.idle":"2022-07-11T20:50:27.905200Z","shell.execute_reply.started":"2022-07-11T20:50:27.897488Z","shell.execute_reply":"2022-07-11T20:50:27.903711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['PoolQC'].fillna(df['PoolQC'].mean,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:28.349479Z","iopub.execute_input":"2022-07-11T20:50:28.350171Z","iopub.status.idle":"2022-07-11T20:50:28.356183Z","shell.execute_reply.started":"2022-07-11T20:50:28.350122Z","shell.execute_reply":"2022-07-11T20:50:28.355138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Fence'].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:28.774077Z","iopub.execute_input":"2022-07-11T20:50:28.774432Z","iopub.status.idle":"2022-07-11T20:50:28.780723Z","shell.execute_reply.started":"2022-07-11T20:50:28.774400Z","shell.execute_reply":"2022-07-11T20:50:28.780042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Fence'].fillna(df['Fence'].mean, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:29.532173Z","iopub.execute_input":"2022-07-11T20:50:29.532720Z","iopub.status.idle":"2022-07-11T20:50:29.538624Z","shell.execute_reply.started":"2022-07-11T20:50:29.532672Z","shell.execute_reply":"2022-07-11T20:50:29.537892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Now that all null values have been cleared based on attribute natures as we want our predictions to e as solid as possible thus using mean, median and mode are the best techniques for doing so\n# Lets first drop all string columns as ML algorithms cant read string attributes¶","metadata":{}},{"cell_type":"code","source":"# lets check all data types\ndf.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:30.902286Z","iopub.execute_input":"2022-07-11T20:50:30.902675Z","iopub.status.idle":"2022-07-11T20:50:30.910301Z","shell.execute_reply.started":"2022-07-11T20:50:30.902634Z","shell.execute_reply":"2022-07-11T20:50:30.909673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['MSZoning'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:31.348805Z","iopub.execute_input":"2022-07-11T20:50:31.349178Z","iopub.status.idle":"2022-07-11T20:50:31.355687Z","shell.execute_reply.started":"2022-07-11T20:50:31.349142Z","shell.execute_reply":"2022-07-11T20:50:31.354634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['Street'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:31.941383Z","iopub.execute_input":"2022-07-11T20:50:31.941758Z","iopub.status.idle":"2022-07-11T20:50:31.950440Z","shell.execute_reply.started":"2022-07-11T20:50:31.941723Z","shell.execute_reply":"2022-07-11T20:50:31.949565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['Alley'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:32.387304Z","iopub.execute_input":"2022-07-11T20:50:32.387682Z","iopub.status.idle":"2022-07-11T20:50:32.393718Z","shell.execute_reply.started":"2022-07-11T20:50:32.387646Z","shell.execute_reply":"2022-07-11T20:50:32.392675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['LotConfig'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:32.749938Z","iopub.execute_input":"2022-07-11T20:50:32.750604Z","iopub.status.idle":"2022-07-11T20:50:32.757215Z","shell.execute_reply.started":"2022-07-11T20:50:32.750551Z","shell.execute_reply":"2022-07-11T20:50:32.756223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['LandSlope'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:33.059135Z","iopub.execute_input":"2022-07-11T20:50:33.059821Z","iopub.status.idle":"2022-07-11T20:50:33.066560Z","shell.execute_reply.started":"2022-07-11T20:50:33.059770Z","shell.execute_reply":"2022-07-11T20:50:33.065851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['Neighborhood'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:33.442350Z","iopub.execute_input":"2022-07-11T20:50:33.443030Z","iopub.status.idle":"2022-07-11T20:50:33.449572Z","shell.execute_reply.started":"2022-07-11T20:50:33.442978Z","shell.execute_reply":"2022-07-11T20:50:33.448756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['Condition1'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:33.850685Z","iopub.execute_input":"2022-07-11T20:50:33.851236Z","iopub.status.idle":"2022-07-11T20:50:33.858121Z","shell.execute_reply.started":"2022-07-11T20:50:33.851178Z","shell.execute_reply":"2022-07-11T20:50:33.857128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['Condition2'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:34.077514Z","iopub.execute_input":"2022-07-11T20:50:34.078192Z","iopub.status.idle":"2022-07-11T20:50:34.087623Z","shell.execute_reply.started":"2022-07-11T20:50:34.078151Z","shell.execute_reply":"2022-07-11T20:50:34.086687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['BldgType'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:34.484664Z","iopub.execute_input":"2022-07-11T20:50:34.485226Z","iopub.status.idle":"2022-07-11T20:50:34.492561Z","shell.execute_reply.started":"2022-07-11T20:50:34.485166Z","shell.execute_reply":"2022-07-11T20:50:34.491590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['PoolQC'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:35.184123Z","iopub.execute_input":"2022-07-11T20:50:35.184892Z","iopub.status.idle":"2022-07-11T20:50:35.193522Z","shell.execute_reply.started":"2022-07-11T20:50:35.184821Z","shell.execute_reply":"2022-07-11T20:50:35.192349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['Fence'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:37.846394Z","iopub.execute_input":"2022-07-11T20:50:37.846759Z","iopub.status.idle":"2022-07-11T20:50:37.852972Z","shell.execute_reply.started":"2022-07-11T20:50:37.846714Z","shell.execute_reply":"2022-07-11T20:50:37.852200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['MiscFeature'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:38.164816Z","iopub.execute_input":"2022-07-11T20:50:38.165324Z","iopub.status.idle":"2022-07-11T20:50:38.171538Z","shell.execute_reply.started":"2022-07-11T20:50:38.165290Z","shell.execute_reply":"2022-07-11T20:50:38.170844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['SaleType'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:38.464467Z","iopub.execute_input":"2022-07-11T20:50:38.465020Z","iopub.status.idle":"2022-07-11T20:50:38.471285Z","shell.execute_reply.started":"2022-07-11T20:50:38.464982Z","shell.execute_reply":"2022-07-11T20:50:38.470580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['SaleCondition'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:38.792472Z","iopub.execute_input":"2022-07-11T20:50:38.793032Z","iopub.status.idle":"2022-07-11T20:50:38.800428Z","shell.execute_reply.started":"2022-07-11T20:50:38.792994Z","shell.execute_reply":"2022-07-11T20:50:38.799207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['HouseStyle'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:39.047336Z","iopub.execute_input":"2022-07-11T20:50:39.047897Z","iopub.status.idle":"2022-07-11T20:50:39.053905Z","shell.execute_reply.started":"2022-07-11T20:50:39.047852Z","shell.execute_reply":"2022-07-11T20:50:39.052866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['RoofStyle'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:39.281195Z","iopub.execute_input":"2022-07-11T20:50:39.281557Z","iopub.status.idle":"2022-07-11T20:50:39.287460Z","shell.execute_reply.started":"2022-07-11T20:50:39.281521Z","shell.execute_reply":"2022-07-11T20:50:39.286567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['RoofMatl'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:39.517908Z","iopub.execute_input":"2022-07-11T20:50:39.518569Z","iopub.status.idle":"2022-07-11T20:50:39.524827Z","shell.execute_reply.started":"2022-07-11T20:50:39.518481Z","shell.execute_reply":"2022-07-11T20:50:39.523605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['Exterior1st'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:39.742986Z","iopub.execute_input":"2022-07-11T20:50:39.743531Z","iopub.status.idle":"2022-07-11T20:50:39.750328Z","shell.execute_reply.started":"2022-07-11T20:50:39.743480Z","shell.execute_reply":"2022-07-11T20:50:39.749345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['Exterior2nd'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:40.122089Z","iopub.execute_input":"2022-07-11T20:50:40.122465Z","iopub.status.idle":"2022-07-11T20:50:40.128875Z","shell.execute_reply.started":"2022-07-11T20:50:40.122431Z","shell.execute_reply":"2022-07-11T20:50:40.127850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['MasVnrType'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:40.357212Z","iopub.execute_input":"2022-07-11T20:50:40.357855Z","iopub.status.idle":"2022-07-11T20:50:40.365196Z","shell.execute_reply.started":"2022-07-11T20:50:40.357804Z","shell.execute_reply":"2022-07-11T20:50:40.364353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(['ExterQual'], axis=1, inplace=True)\ndf.drop(['ExterCond'], axis=1, inplace=True)\ndf.drop(['Foundation'], axis=1, inplace=True)\ndf.drop(['BsmtQual'], axis=1, inplace=True)\ndf.drop(['BsmtCond'], axis=1, inplace=True)\n#BsmtExposure\ndf.drop(['BsmtExposure'], axis=1, inplace=True)\ndf.drop(['BsmtFinType1'], axis=1, inplace=True)\ndf.drop(['BsmtFinSF1'], axis=1, inplace=True)\ndf.drop(['BsmtFinType2'], axis=1, inplace=True)\ndf.drop(['Heating'], axis=1, inplace=True)\ndf.drop(['HeatingQC', 'CentralAir', 'Electrical', 'KitchenQual', 'FireplaceQu',\n'GarageType', 'GarageFinish', 'GarageQual', 'GarageCond', 'PavedDrive', 'MoSold'], axis=1, inplace=True)\ndf.drop(['YrSold'], axis=1, inplace=True)\ndf.drop(['LotFrontage', 'MasVnrArea', 'Functional', 'GarageYrBlt'], axis=1, inplace=True)\ndf.drop(['LotShape', 'LandContour', 'Utilities'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:40.785972Z","iopub.execute_input":"2022-07-11T20:50:40.786470Z","iopub.status.idle":"2022-07-11T20:50:40.810339Z","shell.execute_reply.started":"2022-07-11T20:50:40.786435Z","shell.execute_reply":"2022-07-11T20:50:40.809005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:44.343124Z","iopub.execute_input":"2022-07-11T20:50:44.343658Z","iopub.status.idle":"2022-07-11T20:50:44.364720Z","shell.execute_reply.started":"2022-07-11T20:50:44.343591Z","shell.execute_reply":"2022-07-11T20:50:44.363990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:50:44.620512Z","iopub.execute_input":"2022-07-11T20:50:44.621081Z","iopub.status.idle":"2022-07-11T20:50:44.629140Z","shell.execute_reply.started":"2022-07-11T20:50:44.621040Z","shell.execute_reply":"2022-07-11T20:50:44.628315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# As seen above we only have integer values which will make it accurate for us to make our prediction","metadata":{}},{"cell_type":"markdown","source":"\n# First of all, let us split the dataset based on a 70:20 ratio. 70% of the dataset will be used for training our RandomForest model and 30% of the dataset will be used for evaluating it.\n\n# Next, let us get the target variable (y) and the features (X) from the splitted DataFrames. Please mind that we will be removing some columns since they cannot be used for training the model.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n# Create target object and call it y\ny = df.SalePrice\n# Create X\nfeatures = ['LotArea', 'YearBuilt', '1stFlrSF', '2ndFlrSF', 'FullBath', 'BedroomAbvGr', 'TotRmsAbvGrd', 'WoodDeckSF','EnclosedPorch','3SsnPorch','ScreenPorch','PoolArea','MiscVal', 'BedroomAbvGr']\nX = df[features]\ntrain_X, eval_X, train_y, eval_y = train_test_split(X, y, test_size=0.20, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:11.158103Z","iopub.execute_input":"2022-07-11T20:51:11.158456Z","iopub.status.idle":"2022-07-11T20:51:11.167446Z","shell.execute_reply.started":"2022-07-11T20:51:11.158425Z","shell.execute_reply":"2022-07-11T20:51:11.166711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, confusion_matrix, classification_report\nfrom sklearn.ensemble import RandomForestClassifier, RandomForestRegressor, GradientBoostingClassifier, GradientBoostingRegressor\nfrom sklearn.linear_model import LogisticRegression, LinearRegression","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:12.966921Z","iopub.execute_input":"2022-07-11T20:51:12.967570Z","iopub.status.idle":"2022-07-11T20:51:13.329405Z","shell.execute_reply.started":"2022-07-11T20:51:12.967530Z","shell.execute_reply":"2022-07-11T20:51:13.328407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_RF = LinearRegression()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:13.704114Z","iopub.execute_input":"2022-07-11T20:51:13.704491Z","iopub.status.idle":"2022-07-11T20:51:13.708520Z","shell.execute_reply.started":"2022-07-11T20:51:13.704457Z","shell.execute_reply":"2022-07-11T20:51:13.707447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_RF.fit(train_X, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:15.135700Z","iopub.execute_input":"2022-07-11T20:51:15.136493Z","iopub.status.idle":"2022-07-11T20:51:15.176489Z","shell.execute_reply.started":"2022-07-11T20:51:15.136445Z","shell.execute_reply":"2022-07-11T20:51:15.174668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_RF = model_RF.predict(eval_X)\npredict_RF","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:16.354114Z","iopub.execute_input":"2022-07-11T20:51:16.354901Z","iopub.status.idle":"2022-07-11T20:51:16.368048Z","shell.execute_reply.started":"2022-07-11T20:51:16.354853Z","shell.execute_reply":"2022-07-11T20:51:16.367102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\nmean_absolute_error(predict_RF, eval_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:17.677244Z","iopub.execute_input":"2022-07-11T20:51:17.677969Z","iopub.status.idle":"2022-07-11T20:51:17.685922Z","shell.execute_reply.started":"2022-07-11T20:51:17.677915Z","shell.execute_reply":"2022-07-11T20:51:17.684890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score\nr2_score(predict_RF, eval_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:19.521307Z","iopub.execute_input":"2022-07-11T20:51:19.521717Z","iopub.status.idle":"2022-07-11T20:51:19.530020Z","shell.execute_reply.started":"2022-07-11T20:51:19.521671Z","shell.execute_reply":"2022-07-11T20:51:19.528736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# So we can here see that the random Forest model is trained well as represented by the accuracy score and r_2 score but the cross validation score proves that it still needsmore training and ETL processing before being check on other enviroments as indicated by the cross val score","metadata":{}},{"cell_type":"markdown","source":"# We now make predictions with the test data","metadata":{}},{"cell_type":"code","source":"# path to file you will use for predictions\ntest_data_path = '/kaggle/input/house-prices-advanced-regression-techniques/test.csv'\n\n# read test data file using pandas\ntest_data = pd.read_csv(test_data_path)\ntest_data","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:22.183840Z","iopub.execute_input":"2022-07-11T20:51:22.184535Z","iopub.status.idle":"2022-07-11T20:51:22.251538Z","shell.execute_reply.started":"2022-07-11T20:51:22.184484Z","shell.execute_reply":"2022-07-11T20:51:22.250475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:22.631420Z","iopub.execute_input":"2022-07-11T20:51:22.631788Z","iopub.status.idle":"2022-07-11T20:51:22.639649Z","shell.execute_reply.started":"2022-07-11T20:51:22.631754Z","shell.execute_reply":"2022-07-11T20:51:22.638885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = pd.read_csv(r'/kaggle/input/house-prices-advanced-regression-techniques/sample_submission.csv')\ndf1.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:23.423220Z","iopub.execute_input":"2022-07-11T20:51:23.423758Z","iopub.status.idle":"2022-07-11T20:51:23.437451Z","shell.execute_reply.started":"2022-07-11T20:51:23.423720Z","shell.execute_reply":"2022-07-11T20:51:23.436382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create test_X which comes from test_data but includes only the columns you used for prediction.\n# The list of columns is stored in a variable called features\ntest_X = test_data[features]\n\n# make predictions which we will submit. \ntest_preds = model_RF.predict(test_X)\n\n# The lines below shows how to save predictions in format used for competition scoring\noutput = pd.DataFrame({'Id': test_data.Id,\n                      'SalePrice': test_preds})\noutput.to_csv('submission.csv', index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:51:27.509178Z","iopub.execute_input":"2022-07-11T20:51:27.509538Z","iopub.status.idle":"2022-07-11T20:51:27.538730Z","shell.execute_reply.started":"2022-07-11T20:51:27.509504Z","shell.execute_reply":"2022-07-11T20:51:27.537514Z"},"trusted":true},"execution_count":null,"outputs":[]}]}