{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Ignore useless warnings\nimport warnings\nwarnings.filterwarnings(action=\"ignore\")\n\n# -----------------------------------\n# 学習データ、テストデータの読み込み\n# -----------------------------------\n# 学習データ、テストデータの読み込み\ntrain = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T05:58:29.061624Z","iopub.execute_input":"2022-07-05T05:58:29.062052Z","iopub.status.idle":"2022-07-05T05:58:29.103268Z","shell.execute_reply.started":"2022-07-05T05:58:29.062015Z","shell.execute_reply":"2022-07-05T05:58:29.102307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colcut = [('GrLivArea',3500),('GarageCars',3),('GarageArea',950),('TotalBsmtSF',2500),('1stFlrSF',2500),('TotRmsAbvGrd',11),('MasVnrArea',1000),('Fireplaces',2),('BsmtFinSF1',1500)]\n\n#上限カットオフを入れて、外れ値をごまかす ⇒ これは上手くいかなかった が 、 Dropするよりはこっちの方が安定する。 \n#いずれにせよ、加工するとtestの結果が悪化するみたい。外れ値じゃないのかもしれない...\n#for (col,cutoff) in colcut:\n#    train[col] = train[col].apply(lambda x: cutoff if x>cutoff else x)\n#    test[col] = test[col].apply(lambda x: cutoff if x>cutoff else x)\n\n#Drop作戦\n#for (col,cutoff) in colcut:\n#    train = train.drop(train[(train[col]>cutoff)].index)\n\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.105306Z","iopub.execute_input":"2022-07-05T05:58:29.105788Z","iopub.status.idle":"2022-07-05T05:58:29.1146Z","shell.execute_reply.started":"2022-07-05T05:58:29.10574Z","shell.execute_reply":"2022-07-05T05:58:29.113646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#trainの行数をとっておく\nntrain = train.shape[0]\n\n# 全体統計\nall_df = pd.concat([train.drop(['SalePrice'],axis=1),test])\n\nall_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.116704Z","iopub.execute_input":"2022-07-05T05:58:29.117604Z","iopub.status.idle":"2022-07-05T05:58:29.143324Z","shell.execute_reply.started":"2022-07-05T05:58:29.117549Z","shell.execute_reply":"2022-07-05T05:58:29.142371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#数値データなのだけど、ラベルとして扱うべきものを文字列変換\nall_df['MSSubClass'] = all_df['MSSubClass'].apply(str)\nall_df['YrSold'] = all_df['YrSold'].astype(str)\nall_df['MoSold'] = all_df['MoSold'].astype(str)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.1449Z","iopub.execute_input":"2022-07-05T05:58:29.145261Z","iopub.status.idle":"2022-07-05T05:58:29.15742Z","shell.execute_reply.started":"2022-07-05T05:58:29.145219Z","shell.execute_reply":"2022-07-05T05:58:29.156024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#欠損データを埋める\nfor c in [\"MasVnrArea\",\"BsmtFinSF1\",\"BsmtFinSF2\",\"BsmtUnfSF\",\"TotalBsmtSF\",\"BsmtFullBath\",\"BsmtHalfBath\",\"GarageYrBlt\",\"GarageCars\",\"GarageArea\"]:\n    all_df[c] = all_df[c].fillna(0)\n\n# 'NA'はすでにあるので、'None'で埋めておく\nfor c in [\"Alley\",\"BsmtQual\",\"BsmtCond\",\"BsmtExposure\",\"BsmtFinType1\",\"BsmtFinType2\",\"FireplaceQu\",\"GarageType\",\"GarageFinish\",\"GarageQual\",\"GarageCond\",\"PoolQC\",\"Fence\",\"MiscFeature\"]:\n    all_df[c] = all_df[c].fillna('None')\n\n# こちらは'None'がすでにあるので、NAで埋めておく\nall_df[\"MasVnrType\"] = all_df[\"MasVnrType\"].fillna('NA')\n    \n# 通りに面した長さ 0はないと思われるので地域の平均を利用\n#\"LotFrontage\"\nall_df[\"LotFrontage\"] = all_df.groupby(\"Neighborhood\")[\"LotFrontage\"].transform(lambda x: x.fillna(x.median()))\n# testだけ存在しないもの : trainの最頻値で埋める\n#\"MSZoning\",\"Utilities\",\"Exterior1st\",\"Exterior2nd\",\"KitchenQual\",\"Functional\",\"SaleType\"\n# 1個だけなのでそれっぽい値でごまかす\n#\"Electrical\"\nfor c in [\"MSZoning\",\"Utilities\",\"Exterior1st\",\"Exterior2nd\",\"KitchenQual\",\"Functional\",\"SaleType\",\"Electrical\"]:\n    all_df[c] = all_df[c].fillna(all_df[c].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.165283Z","iopub.execute_input":"2022-07-05T05:58:29.166112Z","iopub.status.idle":"2022-07-05T05:58:29.205797Z","shell.execute_reply.started":"2022-07-05T05:58:29.166072Z","shell.execute_reply":"2022-07-05T05:58:29.205035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nullEntry = all_df.isnull().sum()\nnullEntry[nullEntry>0]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.207123Z","iopub.execute_input":"2022-07-05T05:58:29.207579Z","iopub.status.idle":"2022-07-05T05:58:29.226632Z","shell.execute_reply.started":"2022-07-05T05:58:29.207546Z","shell.execute_reply":"2022-07-05T05:58:29.22543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 正規分布から大きく外れているデータについて、補正\nfrom scipy.stats import boxcox_normmax\n\n#数値データのみのIndexのリスト作成\nnumeric_feats = all_df.dtypes[all_df.dtypes != object].index\n\n#数値データについて、Skew (正規分布からの外れぐあい:0が正規分布)を計算\n#一括でやってしまっているけど、個別に選んでやったほうがいいかも\nfrom scipy.stats import skew\nskewed_feats = all_df[numeric_feats].apply(lambda x: skew(x.dropna())).sort_values(ascending=False)\nskewness = pd.DataFrame({'Skew' :skewed_feats})\nskewness = skewness[abs(skewness['Skew']) > 0.75]\n\n#Box-Cox変換で歪みを補正\nfrom scipy.special import boxcox1p\nlam = 0.15\nfor feat in skewness.index:\n    all_df[feat] = boxcox1p(all_df[feat], lam)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.315234Z","iopub.execute_input":"2022-07-05T05:58:29.315975Z","iopub.status.idle":"2022-07-05T05:58:29.354553Z","shell.execute_reply.started":"2022-07-05T05:58:29.315909Z","shell.execute_reply":"2022-07-05T05:58:29.353759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skewness.index","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.356978Z","iopub.execute_input":"2022-07-05T05:58:29.357405Z","iopub.status.idle":"2022-07-05T05:58:29.366678Z","shell.execute_reply.started":"2022-07-05T05:58:29.357365Z","shell.execute_reply":"2022-07-05T05:58:29.365414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量を作る (あとで分割するのでall_dfに対して処理している)\n\n# 有意なデータを含まないためDrop\nall_df = all_df.drop(['Id','Utilities', 'Street', 'PoolQC'], axis=1)\n\ncols = ('FireplaceQu', 'BsmtQual', 'BsmtCond', 'GarageQual', 'GarageCond', \n        'ExterQual', 'ExterCond','HeatingQC', 'KitchenQual', 'BsmtFinType1', \n        'BsmtFinType2', 'Functional', 'Fence', 'BsmtExposure', 'GarageFinish', 'LandSlope',\n        'LotShape', 'PavedDrive', 'Alley', 'CentralAir', 'MSSubClass', 'OverallCond', \n        'YrSold', 'MoSold')\n\nfrom sklearn.preprocessing import LabelEncoder\nfor column_name in cols:\n    #文字列だけ処理\n    if(all_df[column_name].dtype == object):\n        le = LabelEncoder()\n        #all_dfつかって全ラベルに\n        le.fit(all_df[column_name].fillna('NA'))\n\n        # 学習データ、テストデータを変換する\n        all_df[column_name] = le.transform(all_df[column_name].fillna('NA'))\n\n#OneHotEncoding        \nall_df = pd.get_dummies(all_df).reset_index(drop=True)\nall_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.368513Z","iopub.execute_input":"2022-07-05T05:58:29.369462Z","iopub.status.idle":"2022-07-05T05:58:29.463961Z","shell.execute_reply.started":"2022-07-05T05:58:29.369414Z","shell.execute_reply":"2022-07-05T05:58:29.462762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column_name in all_df:\n    if(all_df[column_name].dtype == object):\n        print(column_name)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.465389Z","iopub.execute_input":"2022-07-05T05:58:29.465925Z","iopub.status.idle":"2022-07-05T05:58:29.481282Z","shell.execute_reply.started":"2022-07-05T05:58:29.46588Z","shell.execute_reply":"2022-07-05T05:58:29.480433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 追加の特徴量生成\nall_df[\"TotalLiveArea\"]=all_df[\"TotalBsmtSF\"]+all_df[\"GrLivArea\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.482968Z","iopub.execute_input":"2022-07-05T05:58:29.483822Z","iopub.status.idle":"2022-07-05T05:58:29.488934Z","shell.execute_reply.started":"2022-07-05T05:58:29.483785Z","shell.execute_reply":"2022-07-05T05:58:29.488252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# モデル作成 & 推論用のデータに再分割\n\n# データを再度分割\ntrain_x = all_df[:ntrain]\ntrain_y = np.log1p(train['SalePrice'])\ntest_x = all_df[ntrain:]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.490026Z","iopub.execute_input":"2022-07-05T05:58:29.490767Z","iopub.status.idle":"2022-07-05T05:58:29.502309Z","shell.execute_reply.started":"2022-07-05T05:58:29.490726Z","shell.execute_reply":"2022-07-05T05:58:29.501453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# -----------------------------------\n# バリデーション関数を用意 (目的変数はlogとっているので、RMSをそのまま利用できる)\n# -----------------------------------\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import KFold,cross_val_score\n\ndef rmsle_cv(model):\n    kf = KFold(n_splits=5, shuffle=True, random_state=random).split(train_x)\n    rmse= np.sqrt(-cross_val_score(model, train_x, train_y, scoring=\"neg_mean_squared_error\", cv = kf))\n    return(rmse)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.503524Z","iopub.execute_input":"2022-07-05T05:58:29.50393Z","iopub.status.idle":"2022-07-05T05:58:29.513494Z","shell.execute_reply.started":"2022-07-05T05:58:29.503897Z","shell.execute_reply":"2022-07-05T05:58:29.512386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# -----------------------------------\n# モデル作成  XGBoost  (GBDT)\n# -----------------------------------\n# 今回は回帰問題なのでRegressorを使う\n# 結局これを調整するのがいちばん有意な結果がでる...\nfrom xgboost import XGBRegressor\n\nrandom=71\n\n#model_xgb = XGBRegressor(n_estimators=1000, random_state = random)\n\nmodel_xgb = XGBRegressor(colsample_bytree=0.4603, gamma=0.0468, \n                             learning_rate=0.05, max_depth=3, \n                             min_child_weight=1.7817, n_estimators=2200,\n                             reg_alpha=0.4640, reg_lambda=0.8571,\n                             subsample=0.5213,\n                             random_state =7, nthread = -1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.514966Z","iopub.execute_input":"2022-07-05T05:58:29.51535Z","iopub.status.idle":"2022-07-05T05:58:29.523822Z","shell.execute_reply.started":"2022-07-05T05:58:29.515318Z","shell.execute_reply":"2022-07-05T05:58:29.523056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_xgb = rmsle_cv(model_xgb)\nprint(score_xgb.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:58:29.527072Z","iopub.execute_input":"2022-07-05T05:58:29.528023Z","iopub.status.idle":"2022-07-05T05:59:22.479351Z","shell.execute_reply.started":"2022-07-05T05:58:29.52799Z","shell.execute_reply":"2022-07-05T05:59:22.478461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# -----------------------------------\n# モデル作成  LightGBM  (GBDT)\n# -----------------------------------\nfrom lightgbm import LGBMRegressor\nwarnings.filterwarnings(action=\"ignore\")\nmodel_lgb = LGBMRegressor(objective='regression',num_leaves=5,\n                              learning_rate=0.05, n_estimators=720,\n                              max_bin = 55, bagging_fraction = 0.8,\n                              bagging_freq = 5, feature_fraction = 0.2319,\n                              feature_fraction_seed=9, bagging_seed=9,\n                              min_data_in_leaf =6, min_sum_hessian_in_leaf = 11, verbose=-1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:22.480389Z","iopub.execute_input":"2022-07-05T05:59:22.481043Z","iopub.status.idle":"2022-07-05T05:59:22.486296Z","shell.execute_reply.started":"2022-07-05T05:59:22.481006Z","shell.execute_reply":"2022-07-05T05:59:22.48552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_lgb = rmsle_cv(model_lgb)\nprint(score_lgb.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:22.487618Z","iopub.execute_input":"2022-07-05T05:59:22.487981Z","iopub.status.idle":"2022-07-05T05:59:24.732551Z","shell.execute_reply.started":"2022-07-05T05:59:22.487924Z","shell.execute_reply":"2022-07-05T05:59:24.731819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# -----------------------------------\n# 線形 モデル作成 \n# -----------------------------------\nfrom sklearn.linear_model import ElasticNet, Lasso\nfrom sklearn.preprocessing import RobustScaler\nfrom sklearn.pipeline import make_pipeline\n\n# 最小二乗法的 線形回帰\nlasso = make_pipeline(RobustScaler(), Lasso(alpha =0.0005, random_state=1))\n# 最小二乗法的 線形回帰の改良版 (L1 Ratioを0.9にしているので、Lessoとほぼ同じだけど、説明変数が捨てにくくなるらしい)\nENet = make_pipeline(RobustScaler(), ElasticNet(alpha=0.0005, l1_ratio=.9, random_state=3))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:24.735631Z","iopub.execute_input":"2022-07-05T05:59:24.736099Z","iopub.status.idle":"2022-07-05T05:59:24.741982Z","shell.execute_reply.started":"2022-07-05T05:59:24.736063Z","shell.execute_reply":"2022-07-05T05:59:24.741217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_lasso = rmsle_cv(lasso)\nprint(score_lasso.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:24.743Z","iopub.execute_input":"2022-07-05T05:59:24.743772Z","iopub.status.idle":"2022-07-05T05:59:25.90957Z","shell.execute_reply.started":"2022-07-05T05:59:24.743731Z","shell.execute_reply":"2022-07-05T05:59:25.908211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_ENet = rmsle_cv(ENet)\nprint(score_ENet.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:25.915622Z","iopub.execute_input":"2022-07-05T05:59:25.92047Z","iopub.status.idle":"2022-07-05T05:59:27.069647Z","shell.execute_reply.started":"2022-07-05T05:59:25.920402Z","shell.execute_reply":"2022-07-05T05:59:27.068525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# -----------------------------------\n# モデル作成 サポートベクター\n# -----------------------------------\nfrom sklearn.svm import SVR\n\n# Support Vector Regressor\nsvr = make_pipeline(RobustScaler(), SVR(C= 20, epsilon= 0.008, gamma=0.0003))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:27.07576Z","iopub.execute_input":"2022-07-05T05:59:27.080124Z","iopub.status.idle":"2022-07-05T05:59:27.092403Z","shell.execute_reply.started":"2022-07-05T05:59:27.080054Z","shell.execute_reply":"2022-07-05T05:59:27.090762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_svr = rmsle_cv(svr)\nprint(score_svr.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:27.094141Z","iopub.execute_input":"2022-07-05T05:59:27.09521Z","iopub.status.idle":"2022-07-05T05:59:29.115002Z","shell.execute_reply.started":"2022-07-05T05:59:27.095156Z","shell.execute_reply":"2022-07-05T05:59:29.113962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# -----------------------------------\n# モデル作成  勾配ブースト決定木のちがうやつ (GBDT)\n# -----------------------------------\nfrom sklearn.ensemble import GradientBoostingRegressor\nGBoost = GradientBoostingRegressor(n_estimators=3000, learning_rate=0.05,\n                                   max_depth=4, max_features='sqrt',\n                                   min_samples_leaf=15, min_samples_split=10, \n                                   loss='huber', random_state =5)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:29.116478Z","iopub.execute_input":"2022-07-05T05:59:29.116868Z","iopub.status.idle":"2022-07-05T05:59:29.122027Z","shell.execute_reply.started":"2022-07-05T05:59:29.116828Z","shell.execute_reply":"2022-07-05T05:59:29.121149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#結果はいいが遅いのでとりあえず使わない\n#score_GBoost = rmsle_cv(GBoost)\n#print(score_GBoost.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:29.123154Z","iopub.execute_input":"2022-07-05T05:59:29.123845Z","iopub.status.idle":"2022-07-05T05:59:29.136633Z","shell.execute_reply.started":"2022-07-05T05:59:29.123812Z","shell.execute_reply":"2022-07-05T05:59:29.135797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# -----------------------------------\n# モデル作成  ランダムフォレスト\n# -----------------------------------\nfrom sklearn.ensemble import RandomForestRegressor\n# Random Forest Regressor\nrfrst = RandomForestRegressor(n_estimators=1200,\n                          max_depth=15,\n                          min_samples_split=5,\n                          min_samples_leaf=5,\n                          max_features=None,\n                          oob_score=True,\n                          random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:29.138125Z","iopub.execute_input":"2022-07-05T05:59:29.138692Z","iopub.status.idle":"2022-07-05T05:59:29.148728Z","shell.execute_reply.started":"2022-07-05T05:59:29.138652Z","shell.execute_reply":"2022-07-05T05:59:29.148006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#遅いしあまり結果がよくないので使わない\n#score_rfrst = rmsle_cv(rfrst)\n#print(score_rfrst.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:29.14983Z","iopub.execute_input":"2022-07-05T05:59:29.150311Z","iopub.status.idle":"2022-07-05T05:59:29.158808Z","shell.execute_reply.started":"2022-07-05T05:59:29.150279Z","shell.execute_reply":"2022-07-05T05:59:29.158098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#base予測モデルで予測した予測値群を説明変数としてmeta予測モデル構築、meta予測モデルの予測値を予測結果とする\nfrom mlxtend.regressor import StackingCVRegressor\n\n#basemodelは軽めのやつにしておく(metamodelの4倍以上トレーニングが走るため)\nstacked = StackingCVRegressor(regressors = (ENet, lasso, svr,model_lgb),\n                              meta_regressor = model_xgb,\n                              use_features_in_secondary=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:29.159869Z","iopub.execute_input":"2022-07-05T05:59:29.160331Z","iopub.status.idle":"2022-07-05T05:59:29.175905Z","shell.execute_reply.started":"2022-07-05T05:59:29.160299Z","shell.execute_reply":"2022-07-05T05:59:29.174896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score_stacked = rmsle_cv(stacked)\nprint(score_stacked.mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-05T05:59:29.17734Z","iopub.execute_input":"2022-07-05T05:59:29.177706Z","iopub.status.idle":"2022-07-05T06:01:04.64075Z","shell.execute_reply.started":"2022-07-05T05:59:29.177672Z","shell.execute_reply":"2022-07-05T06:01:04.639694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"XGB: \" + str(score_xgb))\nprint(\"LGB: \" + str(score_lgb))\nprint(\"Lasso: \" + str(score_lasso))\nprint(\"ENet: \" + str(score_ENet))\nprint(\"SVR: \" + str(score_svr))\nprint(\"Stacked: \" + str(score_stacked))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T06:01:25.092853Z","iopub.execute_input":"2022-07-05T06:01:25.093384Z","iopub.status.idle":"2022-07-05T06:01:25.101023Z","shell.execute_reply.started":"2022-07-05T06:01:25.093346Z","shell.execute_reply":"2022-07-05T06:01:25.099915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 提出データを作成\nstacked.fit(train_x,train_y)\nmodel_xgb.fit(train_x,train_y)\nmodel_lgb.fit(train_x,train_y)\n\n# テストデータの予測値を出力する\nstacked_pred = np.expm1(stacked.predict(test_x.values))\nxgb_pred = np.expm1(model_xgb.predict(test_x))\nlgb_pred = np.expm1(model_lgb.predict(test_x.values))\npred = stacked_pred*0.70 + xgb_pred*0.15 + lgb_pred*0.15\n\n# 提出用ファイルの作成\nsubmission = pd.DataFrame({'Id': test['Id'], 'SalePrice': pred})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T06:01:28.404687Z","iopub.execute_input":"2022-07-05T06:01:28.405089Z","iopub.status.idle":"2022-07-05T06:02:02.205378Z","shell.execute_reply.started":"2022-07-05T06:01:28.40506Z","shell.execute_reply":"2022-07-05T06:02:02.204552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}