{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\n\n# Ignore useless warnings\nimport warnings\nwarnings.filterwarnings(action=\"ignore\")\n\n# -----------------------------------\n# 学習データ、テストデータの読み込み\n# -----------------------------------\n# 学習データ、テストデータの読み込み\ntrain = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T13:29:06.145810Z","iopub.execute_input":"2022-07-11T13:29:06.146208Z","iopub.status.idle":"2022-07-11T13:29:06.191374Z","shell.execute_reply.started":"2022-07-11T13:29:06.146178Z","shell.execute_reply":"2022-07-11T13:29:06.190396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colcut = [('GrLivArea',3500),('GarageCars',3),('GarageArea',950),('TotalBsmtSF',2500),('1stFlrSF',2500),('TotRmsAbvGrd',11),('MasVnrArea',1000),('Fireplaces',2),('BsmtFinSF1',1500)]\n\n#上限カットオフを入れて、外れ値をごまかす ⇒ これは上手くいかなかった が 、 Dropするよりはこっちの方が安定する。 \n#いずれにせよ、加工するとtestの結果が悪化するみたい。外れ値じゃないのかもしれない...\n#for (col,cutoff) in colcut:\n#    train[col] = train[col].apply(lambda x: cutoff if x>cutoff else x)\n#    test[col] = test[col].apply(lambda x: cutoff if x>cutoff else x)\n\n#Drop作戦\n#for (col,cutoff) in colcut:\n#    train = train.drop(train[(train[col]>cutoff)].index)\n\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:06.227045Z","iopub.execute_input":"2022-07-11T13:29:06.227466Z","iopub.status.idle":"2022-07-11T13:29:06.235770Z","shell.execute_reply.started":"2022-07-11T13:29:06.227434Z","shell.execute_reply":"2022-07-11T13:29:06.234812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#trainの行数をとっておく\nntrain = train.shape[0]\n\n# 全体統計\nall_df = pd.concat([train.drop(['SalePrice'],axis=1),test])\n\nall_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:06.302871Z","iopub.execute_input":"2022-07-11T13:29:06.303483Z","iopub.status.idle":"2022-07-11T13:29:06.324112Z","shell.execute_reply.started":"2022-07-11T13:29:06.303437Z","shell.execute_reply":"2022-07-11T13:29:06.323140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#数値データなのだけど、ラベルとして扱うべきものを文字列変換\nall_df['MSSubClass'] = all_df['MSSubClass'].apply(str)\nall_df['YrSold'] = all_df['YrSold'].astype(str)\nall_df['MoSold'] = all_df['MoSold'].astype(str)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:06.441404Z","iopub.execute_input":"2022-07-11T13:29:06.442016Z","iopub.status.idle":"2022-07-11T13:29:06.454032Z","shell.execute_reply.started":"2022-07-11T13:29:06.441969Z","shell.execute_reply":"2022-07-11T13:29:06.452912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#欠損データを埋める\nfor c in [\"MasVnrArea\",\"BsmtFinSF1\",\"BsmtFinSF2\",\"BsmtUnfSF\",\"TotalBsmtSF\",\"BsmtFullBath\",\"BsmtHalfBath\",\"GarageYrBlt\",\"GarageCars\",\"GarageArea\"]:\n    all_df[c] = all_df[c].fillna(0)\n\n# 'NA'はすでにあるので、'None'で埋めておく\nfor c in [\"Alley\",\"BsmtQual\",\"BsmtCond\",\"BsmtExposure\",\"BsmtFinType1\",\"BsmtFinType2\",\"FireplaceQu\",\"GarageType\",\"GarageFinish\",\"GarageQual\",\"GarageCond\",\"PoolQC\",\"Fence\",\"MiscFeature\"]:\n    all_df[c] = all_df[c].fillna('None')\n\n# こちらは'None'がすでにあるので、NAで埋めておく\nall_df[\"MasVnrType\"] = all_df[\"MasVnrType\"].fillna('NA')\n    \n# 通りに面した長さ 0はないと思われるので地域の平均を利用\n#\"LotFrontage\"\nall_df[\"LotFrontage\"] = all_df.groupby(\"Neighborhood\")[\"LotFrontage\"].transform(lambda x: x.fillna(x.median()))\n# testだけ存在しないもの : trainの最頻値で埋める\n#\"MSZoning\",\"Utilities\",\"Exterior1st\",\"Exterior2nd\",\"KitchenQual\",\"Functional\",\"SaleType\"\n# 1個だけなのでそれっぽい値でごまかす\n#\"Electrical\"\nfor c in [\"MSZoning\",\"Utilities\",\"Exterior1st\",\"Exterior2nd\",\"KitchenQual\",\"Functional\",\"SaleType\",\"Electrical\"]:\n    all_df[c] = all_df[c].fillna(all_df[c].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:06.510502Z","iopub.execute_input":"2022-07-11T13:29:06.511099Z","iopub.status.idle":"2022-07-11T13:29:06.550029Z","shell.execute_reply.started":"2022-07-11T13:29:06.511052Z","shell.execute_reply":"2022-07-11T13:29:06.549063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nullEntry = all_df.isnull().sum()\nnullEntry[nullEntry>0]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:06.578476Z","iopub.execute_input":"2022-07-11T13:29:06.579183Z","iopub.status.idle":"2022-07-11T13:29:06.599776Z","shell.execute_reply.started":"2022-07-11T13:29:06.579141Z","shell.execute_reply":"2022-07-11T13:29:06.598928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 正規分布から大きく外れているデータについて、補正\nfrom scipy.stats import boxcox_normmax\n\n#数値データのみのIndexのリスト作成\nnumeric_feats = all_df.dtypes[all_df.dtypes != object].index\n\n#数値データについて、Skew (正規分布からの外れぐあい:0が正規分布)を計算\n#一括でやってしまっているけど、個別に選んでやったほうがいいかも\nfrom scipy.stats import skew\nskewed_feats = all_df[numeric_feats].apply(lambda x: skew(x.dropna())).sort_values(ascending=False)\nskewness = pd.DataFrame({'Skew' :skewed_feats})\nskewness = skewness[abs(skewness['Skew']) > 0.75]\n\n#Box-Cox変換で歪みを補正\nfrom scipy.special import boxcox1p\nlam = 0.15\nfor feat in skewness.index:\n    all_df[feat] = boxcox1p(all_df[feat], lam)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:06.647397Z","iopub.execute_input":"2022-07-11T13:29:06.647750Z","iopub.status.idle":"2022-07-11T13:29:06.685374Z","shell.execute_reply.started":"2022-07-11T13:29:06.647722Z","shell.execute_reply":"2022-07-11T13:29:06.684409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skewness.index","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:06.714274Z","iopub.execute_input":"2022-07-11T13:29:06.714685Z","iopub.status.idle":"2022-07-11T13:29:06.721463Z","shell.execute_reply.started":"2022-07-11T13:29:06.714652Z","shell.execute_reply":"2022-07-11T13:29:06.720688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量を作る (あとで分割するのでall_dfに対して処理している)\n\n# 有意なデータを含まないためDrop\nall_df = all_df.drop(['Id','Utilities', 'Street', 'PoolQC'], axis=1)\n\n#OneHotEncoding        \nall_df = pd.get_dummies(all_df).reset_index(drop=True)\nall_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:06.830838Z","iopub.execute_input":"2022-07-11T13:29:06.831482Z","iopub.status.idle":"2022-07-11T13:29:06.895834Z","shell.execute_reply.started":"2022-07-11T13:29:06.831437Z","shell.execute_reply":"2022-07-11T13:29:06.894879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column_name in all_df:\n    if(all_df[column_name].dtype == object):\n        print(column_name)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:06.937668Z","iopub.execute_input":"2022-07-11T13:29:06.938336Z","iopub.status.idle":"2022-07-11T13:29:06.955337Z","shell.execute_reply.started":"2022-07-11T13:29:06.938295Z","shell.execute_reply":"2022-07-11T13:29:06.954250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 追加の特徴量生成\nall_df[\"TotalLiveArea\"]=all_df[\"TotalBsmtSF\"]+all_df[\"GrLivArea\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:07.009360Z","iopub.execute_input":"2022-07-11T13:29:07.009762Z","iopub.status.idle":"2022-07-11T13:29:07.015321Z","shell.execute_reply.started":"2022-07-11T13:29:07.009730Z","shell.execute_reply":"2022-07-11T13:29:07.014461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# モデル作成 & 推論用のデータに再分割\n\n# データを再度分割\ntrain_x = all_df[:ntrain]\ntrain_y = np.log1p(train['SalePrice'])\ntest_x = all_df[ntrain:]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:07.077795Z","iopub.execute_input":"2022-07-11T13:29:07.078844Z","iopub.status.idle":"2022-07-11T13:29:07.084608Z","shell.execute_reply.started":"2022-07-11T13:29:07.078790Z","shell.execute_reply":"2022-07-11T13:29:07.083799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:07.174491Z","iopub.execute_input":"2022-07-11T13:29:07.175187Z","iopub.status.idle":"2022-07-11T13:29:07.180491Z","shell.execute_reply.started":"2022-07-11T13:29:07.175144Z","shell.execute_reply":"2022-07-11T13:29:07.179714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# MLP\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\n\ninput_size = train_x.shape[1]\n\nclass MLP(nn.Module):\n    def __init__(self):\n        super(MLP, self).__init__()\n        self.fc1 = nn.Linear(input_size, input_size)\n        self.fc2 = nn.Linear(input_size, 100)\n        self.fc3 = nn.Linear(100, 100)\n        self.fc4 = nn.Linear(100, 50)\n        self.fc5 = nn.Linear(50, 30)\n        self.fc6 = nn.Linear(30, 10)\n        self.fc7 = nn.Linear(10, 1)\n\n    def forward(self, x):\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = F.relu(self.fc3(x))\n        x = F.relu(self.fc4(x))\n        x = F.relu(self.fc5(x))\n        x = F.relu(self.fc6(x))\n        x = self.fc7(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:07.271479Z","iopub.execute_input":"2022-07-11T13:29:07.272016Z","iopub.status.idle":"2022-07-11T13:29:07.280536Z","shell.execute_reply.started":"2022-07-11T13:29:07.271984Z","shell.execute_reply":"2022-07-11T13:29:07.279793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = torch.tensor(train_x.values,dtype = torch.float)\ny = torch.tensor(train_y.values.reshape([len(train_y),1]),dtype = torch.float)\nprint(x.shape)\nprint(y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:07.367873Z","iopub.execute_input":"2022-07-11T13:29:07.368407Z","iopub.status.idle":"2022-07-11T13:29:07.378750Z","shell.execute_reply.started":"2022-07-11T13:29:07.368376Z","shell.execute_reply":"2022-07-11T13:29:07.377950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#MLP使った学習 : パラメータがよくないのか、学習が安定しないので、いい感じの予測モデルができるまで何回かやる\nEPOCH = 5000\nLEARNING_LATE = 0.005\n\nmlp = MLP()\noptimizer = optim.Adam(mlp.parameters(), lr=LEARNING_LATE)\ncriterion = nn.MSELoss()\n\nfor i in range(EPOCH):\n    # 勾配初期化\n    optimizer.zero_grad()\n    output = mlp(x)\n    # loss算出\n    loss = criterion(output, y)\n    loss.backward()\n    optimizer.step()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:29:07.420601Z","iopub.execute_input":"2022-07-11T13:29:07.421041Z","iopub.status.idle":"2022-07-11T13:30:16.661315Z","shell.execute_reply.started":"2022-07-11T13:29:07.421008Z","shell.execute_reply":"2022-07-11T13:30:16.660183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#MLPでの予測 (Trainデータとの整合性確認)\npreds = mlp(x).to('cpu').detach().numpy().copy().reshape([y.shape[0]])\npd.DataFrame({\"act\":train_y, \"pred\" : preds}).plot(figsize=(20,5),kind='line')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:30:16.662981Z","iopub.execute_input":"2022-07-11T13:30:16.663344Z","iopub.status.idle":"2022-07-11T13:30:16.926775Z","shell.execute_reply.started":"2022-07-11T13:30:16.663312Z","shell.execute_reply":"2022-07-11T13:30:16.925992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.DataFrame({\"pred\" : preds, \"act\":train_y})\nsns.scatterplot(x=\"pred\", y='act',data=data)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:30:16.928075Z","iopub.execute_input":"2022-07-11T13:30:16.928605Z","iopub.status.idle":"2022-07-11T13:30:17.123658Z","shell.execute_reply.started":"2022-07-11T13:30:16.928574Z","shell.execute_reply":"2022-07-11T13:30:17.122355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nrmse = np.sqrt(mean_squared_error(train_y, preds))\nprint(rmse)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:30:17.125694Z","iopub.execute_input":"2022-07-11T13:30:17.126323Z","iopub.status.idle":"2022-07-11T13:30:17.133050Z","shell.execute_reply.started":"2022-07-11T13:30:17.126265Z","shell.execute_reply":"2022-07-11T13:30:17.132300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}