{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install japanize-matplotlib ","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:18:25.072451Z","iopub.execute_input":"2022-08-02T07:18:25.073090Z","iopub.status.idle":"2022-08-02T07:18:44.815621Z","shell.execute_reply.started":"2022-08-02T07:18:25.072985Z","shell.execute_reply":"2022-08-02T07:18:44.813502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport japanize_matplotlib # 日本語フォント\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n#最大表示列数の指定（ここでは100列を指定）\npd.set_option('display.max_columns', 100)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:19:17.320090Z","iopub.execute_input":"2022-08-02T07:19:17.320724Z","iopub.status.idle":"2022-08-02T07:19:18.082246Z","shell.execute_reply.started":"2022-08-02T07:19:17.320666Z","shell.execute_reply":"2022-08-02T07:19:18.080770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# データの読込み\npath = '/kaggle/input/house-prices-advanced-regression-techniques/'\n\ndf_train = pd.read_csv(path + 'train.csv')\ndf_test = pd.read_csv(path + 'test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:20:05.259227Z","iopub.execute_input":"2022-08-02T07:20:05.259679Z","iopub.status.idle":"2022-08-02T07:20:05.343435Z","shell.execute_reply.started":"2022-08-02T07:20:05.259646Z","shell.execute_reply":"2022-08-02T07:20:05.342026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習用データの冒頭5行\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:20:16.179472Z","iopub.execute_input":"2022-08-02T07:20:16.179942Z","iopub.status.idle":"2022-08-02T07:20:16.253913Z","shell.execute_reply.started":"2022-08-02T07:20:16.179895Z","shell.execute_reply":"2022-08-02T07:20:16.252505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習用データの末尾5行\ndf_train.tail()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:20:25.313688Z","iopub.execute_input":"2022-08-02T07:20:25.314216Z","iopub.status.idle":"2022-08-02T07:20:25.379403Z","shell.execute_reply.started":"2022-08-02T07:20:25.314180Z","shell.execute_reply":"2022-08-02T07:20:25.378264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 予測用データの冒頭5行\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:20:30.578687Z","iopub.execute_input":"2022-08-02T07:20:30.579099Z","iopub.status.idle":"2022-08-02T07:20:30.649527Z","shell.execute_reply.started":"2022-08-02T07:20:30.579068Z","shell.execute_reply":"2022-08-02T07:20:30.647956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 予測用データの末尾5行\ndf_test.tail()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:20:33.559514Z","iopub.execute_input":"2022-08-02T07:20:33.560539Z","iopub.status.idle":"2022-08-02T07:20:33.635653Z","shell.execute_reply.started":"2022-08-02T07:20:33.560504Z","shell.execute_reply":"2022-08-02T07:20:33.634102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### データの量（行数、列数）、内容（データ型、基本統計量など）を確認","metadata":{}},{"cell_type":"code","source":"# データ形状の確認\nprint(df_train.shape)\nprint(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:21:32.522314Z","iopub.execute_input":"2022-08-02T07:21:32.522783Z","iopub.status.idle":"2022-08-02T07:21:32.530191Z","shell.execute_reply.started":"2022-08-02T07:21:32.522749Z","shell.execute_reply":"2022-08-02T07:21:32.528872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習用データの概要\ndf_train.info()\n\n# 学習用データの基本統計量\ndf_train.describe(include='all') # include='all'オプションで、カテゴリ値の統計量も表示可能","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:22:05.689850Z","iopub.execute_input":"2022-08-02T07:22:05.690265Z","iopub.status.idle":"2022-08-02T07:22:05.983994Z","shell.execute_reply.started":"2022-08-02T07:22:05.690233Z","shell.execute_reply":"2022-08-02T07:22:05.982729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 探索的データ分析（EDA）\n\n可視化して、データの傾向を掴もう<br>","metadata":{}},{"cell_type":"markdown","source":"### 目的変数のヒストグラム\n\ndf_trainのみにある目的変数SalePriceの分布を確認\n\n今回は、\n\n- matplotlib\n- seaborn\n- pandas\n\nの可視化の仕方を比較してみます\n\n冒頭で、japanize_matplotlibをimportしているため日本語でラベル表示されます\n\n[matplotlib公式チュートリアル](https://matplotlib.org/stable/tutorials/introductory/usage.html)<br>\n[matplotlib公式チートシート](https://github.com/matplotlib/cheatsheets)","metadata":{}},{"cell_type":"code","source":"# matplotlibによる可視化\nplt.figure(figsize=(12,7))\n\nplt.hist(df_train[\"SalePrice\"], bins=25)\n\nplt.xlabel(\"住宅価格\", fontsize=15)\nplt.ylabel(\"住宅数\", fontsize=15)\nplt.grid()\nplt.title(\"住宅価格分布\", fontsize=20)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:24:40.099484Z","iopub.execute_input":"2022-08-02T07:24:40.099884Z","iopub.status.idle":"2022-08-02T07:24:40.472732Z","shell.execute_reply.started":"2022-08-02T07:24:40.099852Z","shell.execute_reply":"2022-08-02T07:24:40.471254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[seaborn公式ドキュメント](https://seaborn.pydata.org/)<br>\n[グラフ作成のためのチートシートとPythonによる各種グラフの実装](https://qiita.com/4m1t0/items/76b0033edb545a78cef5#fn3)","metadata":{}},{"cell_type":"code","source":"# seabornによる可視化\nplt.figure(figsize=(12,7))\n\nsns.distplot(df_train[\"SalePrice\"], bins=25)\n\nplt.xlabel(\"住宅価格\", fontsize=15)\nplt.ylabel(\"住宅数\", fontsize=15)\nplt.title(\"住宅価格分布\", fontsize=20)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:25:42.535762Z","iopub.execute_input":"2022-08-02T07:25:42.537528Z","iopub.status.idle":"2022-08-02T07:25:42.830229Z","shell.execute_reply.started":"2022-08-02T07:25:42.537458Z","shell.execute_reply":"2022-08-02T07:25:42.828723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[pandasのプロットメソッド](https://pandas.pydata.org/docs/reference/api/pandas.DataFrame.plot.html)","metadata":{}},{"cell_type":"code","source":"# Pandasによる可視化\nplt.figure(figsize=(12,7))\n\ndf_train[\"SalePrice\"].hist(bins=25)\n\nplt.xlabel(\"住宅価格\", fontsize=15)\nplt.ylabel(\"住宅数\", fontsize=15)\nplt.title(\"住宅価格分布\", fontsize=20)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:26:15.025544Z","iopub.execute_input":"2022-08-02T07:26:15.026012Z","iopub.status.idle":"2022-08-02T07:26:15.304652Z","shell.execute_reply.started":"2022-08-02T07:26:15.025961Z","shell.execute_reply":"2022-08-02T07:26:15.302914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"SalePriceの分布は、正規分布に比べて偏りがあるようです","metadata":{}},{"cell_type":"markdown","source":"### 目的変数と数値データの相関関係\n\ndf_trainの説明変数（数値データ）と説明変数の相関を見てみよう","metadata":{}},{"cell_type":"code","source":"# 数値データのみ表示\n\nnum_features = df_train.select_dtypes(include=[np.number])\nnum_features.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:27:21.471404Z","iopub.execute_input":"2022-08-02T07:27:21.471800Z","iopub.status.idle":"2022-08-02T07:27:21.485038Z","shell.execute_reply.started":"2022-08-02T07:27:21.471768Z","shell.execute_reply":"2022-08-02T07:27:21.483484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 相関係数を算出するメソッド\ncorr = num_features.corr()\n\nplt.figure(figsize=(30, 12))\n# 相関係数に基づいたヒートマップを表示\nsns.heatmap(corr, vmax=0.8, annot_kws={'size': 10}, annot=True, cmap='bone')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:29:23.974366Z","iopub.execute_input":"2022-08-02T07:29:23.975603Z","iopub.status.idle":"2022-08-02T07:29:32.261601Z","shell.execute_reply.started":"2022-08-02T07:29:23.975563Z","shell.execute_reply":"2022-08-02T07:29:32.260220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"cmap='色マップ'を変えてみよう\n\n[color mapの一覧をheatmapで(160個くらい画像があるので注意)](https://pod.hatenablog.com/entry/2018/09/20/212527)","metadata":{}},{"cell_type":"code","source":"# SalePriceと相関の強い説明変数の上位１０個を抽出\n\n# 相関係数の絶対値で比較する\n# インデックス0はSalePrice同士の相関係数なので除外\ntarget_corr = pd.DataFrame(abs(corr['SalePrice']).sort_values(ascending=False)[1:11])\ntarget_corr.columns=['Correlation']\ntarget_corr","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:31:19.864370Z","iopub.execute_input":"2022-08-02T07:31:19.864796Z","iopub.status.idle":"2022-08-02T07:31:19.879532Z","shell.execute_reply.started":"2022-08-02T07:31:19.864765Z","shell.execute_reply":"2022-08-02T07:31:19.878168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"OverallQualとGrLivAreaが、SalePriceと強く相関しています","metadata":{}},{"cell_type":"code","source":"# マルチインデックスのピボットテーブル\npivot_corr = corr.unstack()\n\n# 相関係数の高い説明変数同士を抽出\npd.DataFrame(pivot_corr[(abs(pivot_corr)>0.6) & (abs(pivot_corr)<1)]) # 相関係数の絶対値が0.6より大きく1より小さい条件で絞り込み","metadata":{"execution":{"iopub.status.busy":"2022-08-02T07:33:12.559628Z","iopub.execute_input":"2022-08-02T07:33:12.560089Z","iopub.status.idle":"2022-08-02T07:33:12.581379Z","shell.execute_reply.started":"2022-08-02T07:33:12.560056Z","shell.execute_reply":"2022-08-02T07:33:12.580009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"GarageCars、GarageAreaの相関が強いようです","metadata":{}},{"cell_type":"markdown","source":"### 相関の高い2変量の関係\n\ndf_trainの説明変数と目的変数の相関関係を個別に見てみよう","metadata":{}},{"cell_type":"markdown","source":"#### 連続データ同士の比較\n\nSalePrice（住宅価格）とGrLivArea（上階のリビングエリアの平方フィート）の関係","metadata":{}},{"cell_type":"code","source":"# SalePriceとGrLivAreaの相関\n\nplt.figure(figsize=(12,7))\ndata = pd.concat([df_train['SalePrice'], df_train['GrLivArea']], axis=1)\nsns.scatterplot(data=data, x='GrLivArea', y='SalePrice', alpha=0.5)\nplt.xlabel(\"GrLivArea\", fontsize=15)\nplt.ylabel(\"SalePrice\", fontsize=15)\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"SalePrice（住宅価格）とTotalBsmtSF（地下室の総平方フィート）の関係","metadata":{}},{"cell_type":"code","source":"# SalePriceとTotalBsmtSFの相関\n\nplt.figure(figsize=(12,7))\ndata = pd.concat([df_train['SalePrice'], df_train['TotalBsmtSF']], axis=1)\nsns.scatterplot(data=data, x='TotalBsmtSF', y='SalePrice', alpha=0.5)\nplt.xlabel(\"TotalBsmtSF\", fontsize=15)\nplt.ylabel(\"SalePrice\", fontsize=15)\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 連続データとカテゴリデータの比較\n\nSalePrice（住宅価格）とOverallQual（全体的な素材と仕上げの品質）の関係","metadata":{}},{"cell_type":"code","source":"# 箱ヒゲ図\n\nplt.figure(figsize=(12,7))\ndata = pd.concat([df_train['SalePrice'], df_train['OverallQual']], axis=1)\nsns.boxplot(x=df_train['OverallQual'], y=\"SalePrice\", data=data)\nplt.xlabel(\"OverallQual\", fontsize=15)\nplt.ylabel(\"SalePrice\", fontsize=15)\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1変量の分析\n\n","metadata":{}},{"cell_type":"markdown","source":"カテゴリデータのFireplaces（暖炉の数）の分布","metadata":{}},{"cell_type":"code","source":"# Fireplaces（暖炉の数）をカウントした棒グラフ\n\nplt.figure(figsize=(12,7))\nsns.countplot(df_train[\"Fireplaces\"])\nplt.xlabel(\"Fireplaces\", fontsize=15)\nplt.ylabel(\"Count\", fontsize=15)\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"df_trainの半数以上の住宅に少なくとも1つは暖炉があるようです","metadata":{}},{"cell_type":"markdown","source":"#### df_trainとdf_testの1変量の分布を比較","metadata":{}},{"cell_type":"markdown","source":"YearBuilt（建設された年）をカウントすると、どの年に建設された住宅が多いかが確認できます","metadata":{}},{"cell_type":"code","source":"# 建設された年（df_train）をカウントした棒グラフ\n\ncount_yb = df_train[\"YearBuilt\"].value_counts().reset_index(name='YearBuilt')\ncount_yb.columns = [\"YearBuilt\", \"Count\"]\n\nplt.figure(figsize=(12,7))\nplt.bar(count_yb[\"YearBuilt\"], count_yb[\"Count\"])\nplt.xlabel(\"train YearBuilt\", fontsize=15)\nplt.ylabel(\"Count\", fontsize=15)\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 建設された年（df_test）をカウントした棒グラフ\n\ncount_yb = df_test[\"YearBuilt\"].value_counts().reset_index(name='YearBuilt')\ncount_yb.columns = [\"YearBuilt\", \"Count\"]\n\nplt.figure(figsize=(12,7))\nplt.bar(count_yb[\"YearBuilt\"], count_yb[\"Count\"])\nplt.xlabel(\"test YearBuilt\", fontsize=15)\nplt.ylabel(\"Count\", fontsize=15)\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"データセットが作成された2011年（2006年〜2010年の販売データ）から過去10年に建築された住宅が多いようです","metadata":{}},{"cell_type":"markdown","source":"### 2変量の分析","metadata":{}},{"cell_type":"markdown","source":"### 時系列における住宅価格の変化\n\nYrSold（販売年）ごとのSalePrice（住宅価格）の平均価格を見てみよう","metadata":{}},{"cell_type":"code","source":"# 折れ線グラフ\n\n# YrSold（販売年）ごとのSalePrice（住宅価格）の平均価格を計算\ngroup_meanprice = df_train[[\"YrSold\",\"SalePrice\"]].groupby([\"YrSold\"]).mean().reset_index()\n\nplt.figure(figsize=(12,7))\nsns.lineplot(x=\"YrSold\", y=\"SalePrice\", data=group_meanprice)\nplt.xlabel(\"YrSold\", fontsize=15)\nplt.ylabel(\"Mean SalePrice\", fontsize=15)\nplt.title(\"住宅が売れた年の平均販売価格\", fontsize=20)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2008年は[サブプライム住宅ローン危機](https://ja.wikipedia.org/wiki/%E3%82%B5%E3%83%96%E3%83%97%E3%83%A9%E3%82%A4%E3%83%A0%E4%BD%8F%E5%AE%85%E3%83%AD%E3%83%BC%E3%83%B3%E5%8D%B1%E6%A9%9F)の年でした","metadata":{}},{"cell_type":"markdown","source":"YearBuilt（建築された年）ごとのSalePrice（住宅価格）を見てみよう","metadata":{}},{"cell_type":"code","source":"# 折れ線グラフ\n\nplt.figure(figsize=(12,7))\nsns.lineplot(x=\"YearBuilt\", y=\"SalePrice\", ci=\"sd\", data=df_train) # ci='sd'で標準偏差（バラツキ度合い）を表示\nplt.xlabel(\"YearBuilt\", fontsize=15)\nplt.ylabel(\"SalePrice\", fontsize=15)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"建設された年（df_train）をカウントした棒グラフでは1900年以前の住宅数は少ないことがわかりましたが、この折れ線グラフから価格は高価であることがわかります","metadata":{}}]}