{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename == 'train.csv':\n            data = pd.read_csv(os.path.join(dirname, filename))\n        elif filename == 'test.csv':\n            test_data = pd.read_csv(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T16:30:39.679333Z","iopub.execute_input":"2022-07-12T16:30:39.679766Z","iopub.status.idle":"2022-07-12T16:30:39.729372Z","shell.execute_reply.started":"2022-07-12T16:30:39.679727Z","shell.execute_reply":"2022-07-12T16:30:39.728373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check high correlations between variables\nfor column in data:\n    dtype = data[column].dtype\n    if dtype in ['int64', 'float64']:\n        for column2 in data:\n            dtype2 = data[column2].dtype\n            if dtype2 == dtype:\n                corr = data[column].corr(data[column2])\n                if corr > 0.8 and corr < 0.999:\n                    print('The correlation between {} and {} is {}!'.format(column, column2, corr))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove attributes with high correlation to some other attribute (they describe the same thing)\n# and attributes that have high chance of being similar (Exterior2nd in comparison to Exterior1st , Condition2 in comparison to Condition1)\ndata = data.drop(['Id', 'PoolQC', 'MiscFeature', 'Condition2', 'Exterior2nd', '1stFlrSF', 'TotRmsAbvGrd', 'GarageArea', 'Utilities', 'Street', 'Alley'], axis=1)\ntrain_y = data['SalePrice']\ntrain_x = data.drop('SalePrice', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T16:30:39.736449Z","iopub.execute_input":"2022-07-12T16:30:39.736972Z","iopub.status.idle":"2022-07-12T16:30:39.750417Z","shell.execute_reply.started":"2022-07-12T16:30:39.736941Z","shell.execute_reply":"2022-07-12T16:30:39.749443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\n# Helper function to encode categorical labels\ndef encode_labels(dataset):\n    label_encoder = preprocessing.LabelEncoder()\n    for column in dataset:\n        dtype = dataset[column].dtype\n        if dtype not in ['int64', 'float64']:\n            dataset[column] = label_encoder.fit_transform(dataset[column])\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-12T16:30:40.011997Z","iopub.execute_input":"2022-07-12T16:30:40.012398Z","iopub.status.idle":"2022-07-12T16:30:40.023699Z","shell.execute_reply.started":"2022-07-12T16:30:40.012366Z","shell.execute_reply":"2022-07-12T16:30:40.022778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Standardize and normalize the data\nfrom sklearn.preprocessing import StandardScaler\ntrain_x = encode_labels(train_x)\ntest_ids = test_data['Id']\ncommon_cols = [col for col in set(data.columns).intersection(test_data.columns)]\ntest_data = test_data[common_cols]\ntrain_x = train_x[common_cols]\nsc = StandardScaler()\ntrain_x = sc.fit_transform(train_x)\ntest_data = encode_labels(test_data)\ntest_data = sc.transform(test_data)  ","metadata":{"execution":{"iopub.status.busy":"2022-07-12T16:30:40.066011Z","iopub.execute_input":"2022-07-12T16:30:40.066733Z","iopub.status.idle":"2022-07-12T16:30:40.143653Z","shell.execute_reply.started":"2022-07-12T16:30:40.066687Z","shell.execute_reply":"2022-07-12T16:30:40.142475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training with XGBoost using stratified K fold and randomized search cross-validation\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import RandomizedSearchCV, StratifiedKFold\n# split data into train and test sets\nparams = {\n    'n_estimators': [100, 400, 800],\n    'max_depth': [3, 6, 9],\n    'learning_rate': [0.05, 0.1, 0.20],\n    'min_child_weight': [1, 10, 100]\n    }\n\nxgb = XGBRegressor(learning_rate=0.02, objective='reg:squarederror',\n                    silent=True, nthread=1)\nfolds = 3\nparam_comb = 16\n\nskf = StratifiedKFold(n_splits=folds, shuffle = True, random_state = 1001)\n\n\nrandom_search = RandomizedSearchCV(xgb, param_distributions=params, n_iter=param_comb, scoring='roc_auc', n_jobs=8, cv=skf.split(train_x,train_y), verbose=3, random_state=1001 )\nrandom_search.fit(train_x, train_y)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T16:30:40.145947Z","iopub.execute_input":"2022-07-12T16:30:40.147026Z","iopub.status.idle":"2022-07-12T16:31:31.282537Z","shell.execute_reply.started":"2022-07-12T16:30:40.146992Z","shell.execute_reply":"2022-07-12T16:31:31.280954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform predictions on test data\nmodel = random_search\npreds = model.predict(test_data)\nsubmission = pd.DataFrame()\nsubmission['Id'] = test_ids\nsubmission['SalePrice'] = preds","metadata":{"execution":{"iopub.status.busy":"2022-07-12T16:31:31.283482Z","iopub.status.idle":"2022-07-12T16:31:31.283846Z","shell.execute_reply.started":"2022-07-12T16:31:31.283674Z","shell.execute_reply":"2022-07-12T16:31:31.283692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the submission file\nfrom pathlib import Path  \nfilepath = Path('submission.csv')  \nfilepath.parent.mkdir(parents=True, exist_ok=True)\nsubmission.to_csv('submission.csv', sep=',', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T16:31:31.28774Z","iopub.status.idle":"2022-07-12T16:31:31.288767Z","shell.execute_reply.started":"2022-07-12T16:31:31.288533Z","shell.execute_reply":"2022-07-12T16:31:31.288556Z"},"trusted":true},"execution_count":null,"outputs":[]}]}