{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename == 'train.csv':\n            data = pd.read_csv(os.path.join(dirname, filename))\n        elif filename == 'test.csv':\n            test_data = pd.read_csv(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T11:12:55.987620Z","iopub.execute_input":"2022-07-28T11:12:55.987985Z","iopub.status.idle":"2022-07-28T11:12:56.031952Z","shell.execute_reply.started":"2022-07-28T11:12:55.987954Z","shell.execute_reply":"2022-07-28T11:12:56.031073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T11:12:58.377590Z","iopub.execute_input":"2022-07-28T11:12:58.377972Z","iopub.status.idle":"2022-07-28T11:12:58.400768Z","shell.execute_reply.started":"2022-07-28T11:12:58.377939Z","shell.execute_reply":"2022-07-28T11:12:58.399937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T11:12:59.196827Z","iopub.execute_input":"2022-07-28T11:12:59.197667Z","iopub.status.idle":"2022-07-28T11:12:59.220929Z","shell.execute_reply.started":"2022-07-28T11:12:59.197630Z","shell.execute_reply":"2022-07-28T11:12:59.219884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check high correlations between variables\nfor column in data:\n    dtype = data[column].dtype\n    if dtype in ['int64', 'float64']:\n        for column2 in data:\n            dtype2 = data[column2].dtype\n            if dtype2 == dtype:\n                corr = data[column].corr(data[column2])\n                if corr > 0.8 and corr < 0.999:\n                    print('The correlation between {} and {} is {}!'.format(column, column2, corr))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T11:12:59.890065Z","iopub.execute_input":"2022-07-28T11:12:59.890461Z","iopub.status.idle":"2022-07-28T11:13:00.183848Z","shell.execute_reply.started":"2022-07-28T11:12:59.890428Z","shell.execute_reply":"2022-07-28T11:13:00.182795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove attributes with high correlation to some other attribute (they describe the same thing)\n# and attributes that have high chance of being similar (Exterior2nd in comparison to Exterior1st , Condition2 in comparison to Condition1)\ndata = data.drop(['Id', 'PoolQC', 'MiscFeature', 'Condition2', 'Exterior2nd', '1stFlrSF', 'TotRmsAbvGrd', 'GarageArea', 'Utilities', 'Street', 'Alley'], axis=1)\ntrain_y = data['SalePrice']\ntrain_x = data.drop('SalePrice', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T11:13:00.453763Z","iopub.execute_input":"2022-07-28T11:13:00.454141Z","iopub.status.idle":"2022-07-28T11:13:00.462799Z","shell.execute_reply.started":"2022-07-28T11:13:00.454107Z","shell.execute_reply":"2022-07-28T11:13:00.461953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\n# Helper function to encode categorical labels\ndef encode_labels(dataset):\n    label_encoder = preprocessing.LabelEncoder()\n    for column in dataset:\n        dtype = dataset[column].dtype\n        if dtype not in ['int64', 'float64']:\n            dataset[column] = label_encoder.fit_transform(dataset[column])\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-28T11:13:01.047814Z","iopub.execute_input":"2022-07-28T11:13:01.048401Z","iopub.status.idle":"2022-07-28T11:13:01.053081Z","shell.execute_reply.started":"2022-07-28T11:13:01.048367Z","shell.execute_reply":"2022-07-28T11:13:01.052341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Standardize and normalize the data\nfrom sklearn.preprocessing import StandardScaler\ntrain_x = encode_labels(train_x)\ntest_ids = test_data['Id']\ncommon_cols = [col for col in set(data.columns).intersection(test_data.columns)]\ntest_data = test_data[common_cols]\ntrain_x = train_x[common_cols]\nsc = StandardScaler()\ntrain_x = sc.fit_transform(train_x)\ntest_data = encode_labels(test_data)\ntest_data = sc.transform(test_data)  ","metadata":{"execution":{"iopub.status.busy":"2022-07-28T11:13:01.822181Z","iopub.execute_input":"2022-07-28T11:13:01.823146Z","iopub.status.idle":"2022-07-28T11:13:01.898042Z","shell.execute_reply.started":"2022-07-28T11:13:01.823098Z","shell.execute_reply":"2022-07-28T11:13:01.897193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training with XGBoost using stratified K fold and randomized search cross-validation\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import RandomizedSearchCV, StratifiedKFold\n# split data into train and test sets\nparams = {\n    'n_estimators': [100, 400, 800],\n    'max_depth': [3, 6, 9],\n    'learning_rate': [0.05, 0.1, 0.20],\n    'min_child_weight': [1, 10, 100]\n    }\n\nxgb = XGBRegressor(learning_rate=0.02, objective='reg:squarederror',\n                    silent=True, nthread=1)\nfolds = 3\nparam_comb = 16\n\nskf = StratifiedKFold(n_splits=folds, shuffle = True, random_state = 1001)\n\n\nrandom_search = RandomizedSearchCV(xgb, param_distributions=params, n_iter=param_comb, scoring='roc_auc', n_jobs=8, cv=skf.split(train_x,train_y), verbose=3, random_state=1001 )\nrandom_search.fit(train_x, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T11:13:15.305924Z","iopub.execute_input":"2022-07-28T11:13:15.307106Z","iopub.status.idle":"2022-07-28T11:14:51.188575Z","shell.execute_reply.started":"2022-07-28T11:13:15.307066Z","shell.execute_reply":"2022-07-28T11:14:51.187419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform predictions on test data\nmodel = random_search\npreds = model.predict(test_data)\nsubmission = pd.DataFrame()\nsubmission['Id'] = test_ids\nsubmission['SalePrice'] = preds","metadata":{"execution":{"iopub.status.busy":"2022-07-28T11:15:18.357254Z","iopub.execute_input":"2022-07-28T11:15:18.358227Z","iopub.status.idle":"2022-07-28T11:15:18.422922Z","shell.execute_reply.started":"2022-07-28T11:15:18.358192Z","shell.execute_reply":"2022-07-28T11:15:18.421724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the submission file\nfrom pathlib import Path  \nfilepath = Path('submission.csv')  \nfilepath.parent.mkdir(parents=True, exist_ok=True)\nsubmission.to_csv('submission.csv', sep=',', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T11:15:20.034012Z","iopub.execute_input":"2022-07-28T11:15:20.034419Z","iopub.status.idle":"2022-07-28T11:15:20.044816Z","shell.execute_reply.started":"2022-07-28T11:15:20.034385Z","shell.execute_reply":"2022-07-28T11:15:20.044067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}