{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom lightgbm import LGBMRegressor\npd.set_option('display.max_columns', None)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nfrom sklearn.model_selection import KFold\nfrom tqdm.notebook import tqdm\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T17:52:33.055763Z","iopub.execute_input":"2022-07-11T17:52:33.056475Z","iopub.status.idle":"2022-07-11T17:52:33.086123Z","shell.execute_reply.started":"2022-07-11T17:52:33.056326Z","shell.execute_reply":"2022-07-11T17:52:33.085308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/sample_submission.csv')\ntrain = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/train.csv')\ntest = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/test.csv')\n\nobjects = [features for features in train.columns if train[features].dtype.name =='object']\ntrain[objects] = train[objects].astype('category')\ntest[objects] = test[objects].astype('category')\n\ntrain['preds'] = -99","metadata":{"execution":{"iopub.status.busy":"2022-07-11T19:12:59.337467Z","iopub.execute_input":"2022-07-11T19:12:59.338267Z","iopub.status.idle":"2022-07-11T19:12:59.499921Z","shell.execute_reply.started":"2022-07-11T19:12:59.338214Z","shell.execute_reply":"2022-07-11T19:12:59.498689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T19:13:00.096881Z","iopub.execute_input":"2022-07-11T19:13:00.097308Z","iopub.status.idle":"2022-07-11T19:13:00.165245Z","shell.execute_reply.started":"2022-07-11T19:13:00.097274Z","shell.execute_reply":"2022-07-11T19:13:00.164123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"relevant = ['Id', 'MSSubClass', 'MSZoning', 'LotFrontage', 'LotArea', 'Street',\n       'LotShape', 'LandContour', 'Utilities', 'LotConfig',\n       'LandSlope', 'Neighborhood', 'Condition1', 'BldgType',\n       'HouseStyle', 'OverallQual', 'OverallCond', 'YearBuilt', 'YearRemodAdd',\n       'RoofStyle','Exterior1st', 'Exterior2nd', 'MasVnrType',\n       'MasVnrArea', 'ExterQual', 'ExterCond', 'Foundation', 'BsmtQual',\n       'BsmtCond', 'BsmtExposure', 'BsmtFinType1', 'BsmtFinSF1',\n       'BsmtFinType2', 'BsmtFinSF2', 'TotalBsmtSF', 'Heating',\n       'HeatingQC', 'CentralAir', 'Electrical',\n       'LowQualFinSF', 'GrLivArea', 'BsmtFullBath', 'BsmtHalfBath', 'FullBath',\n       'HalfBath', 'BedroomAbvGr', 'KitchenAbvGr', 'KitchenQual',\n       'TotRmsAbvGrd', 'Functional', 'Fireplaces', 'FireplaceQu', 'GarageType',\n       'GarageYrBlt', 'GarageFinish', 'GarageCars' , 'GarageCond', 'PavedDrive', 'WoodDeckSF', 'ScreenPorch', 'PoolArea', 'PoolQC',\n     'YrSold', 'SaleType',\n       'SaleCondition']\nTARGET =  'SalePrice'","metadata":{"execution":{"iopub.status.busy":"2022-07-11T19:13:01.194026Z","iopub.execute_input":"2022-07-11T19:13:01.194809Z","iopub.status.idle":"2022-07-11T19:13:01.203945Z","shell.execute_reply.started":"2022-07-11T19:13:01.194765Z","shell.execute_reply":"2022-07-11T19:13:01.202929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###############################\n#BASELINE\n###############################\nmodels = []\nFEATURES = relevant.copy()\nKFOLD = KFold(n_splits=5, shuffle=True, random_state=42)\nfor idx_train, idx_val in tqdm(KFOLD.split(train)):\n    \n    #Training the model\n    model = LGBMRegressor()\n    model.fit(train.iloc[idx_train][FEATURES], train.iloc[idx_train][TARGET])\n    \n    #Validation\n    pred_idx = [(i, feat) for i,feat in  enumerate(train.columns) if feat =='preds'][0][0]\n    train.iloc[idx_val, pred_idx] = model.predict(train.iloc[idx_val][FEATURES])\n    \n#baseline acc\nprint(f\"ACC: {np.mean(np.square(train['preds'] - train[TARGET]))}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T19:15:53.812887Z","iopub.execute_input":"2022-07-11T19:15:53.813748Z","iopub.status.idle":"2022-07-11T19:15:55.056403Z","shell.execute_reply.started":"2022-07-11T19:15:53.813703Z","shell.execute_reply":"2022-07-11T19:15:55.055358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"867564873 / 10e9","metadata":{"execution":{"iopub.status.busy":"2022-07-11T19:22:09.418097Z","iopub.execute_input":"2022-07-11T19:22:09.418526Z","iopub.status.idle":"2022-07-11T19:22:09.424856Z","shell.execute_reply.started":"2022-07-11T19:22:09.418491Z","shell.execute_reply":"2022-07-11T19:22:09.423808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###############################\n#EXP 1\n###############################\nmodels = []\nFEATURES = [feat for feat in train.columns if feat not in [TARGET, 'Id', 'preds']]\nKFOLD = KFold(n_splits=5, shuffle=True, random_state=42)\nfor idx_train, idx_val in tqdm(KFOLD.split(train)):\n    \n    #Training the model\n    model = LGBMRegressor()\n    model.fit(train.iloc[idx_train][FEATURES], train.iloc[idx_train][TARGET])\n    \n    #Validation\n    pred_idx = [(i, feat) for i,feat in  enumerate(train.columns) if feat =='preds'][0][0]\n    train.iloc[idx_val, pred_idx] = model.predict(train.iloc[idx_val][FEATURES])\n    \n#baseline acc\nprint(f\"ACC: {np.mean(np.square(train['preds'] - train[TARGET])) / 10e9}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T19:31:04.316802Z","iopub.execute_input":"2022-07-11T19:31:04.317258Z","iopub.status.idle":"2022-07-11T19:31:05.746247Z","shell.execute_reply.started":"2022-07-11T19:31:04.317221Z","shell.execute_reply":"2022-07-11T19:31:05.745203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm\nlightgbm.plot_importance(model, figsize=(10,20))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T19:27:38.710536Z","iopub.execute_input":"2022-07-11T19:27:38.710962Z","iopub.status.idle":"2022-07-11T19:27:39.824058Z","shell.execute_reply.started":"2022-07-11T19:27:38.710928Z","shell.execute_reply":"2022-07-11T19:27:39.822956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}