{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# The Net for House Prices - Advanced Regression Techniques\n> Here is the code copied from the book <DIVE INTO DEEP LEARNING> written by Mu Li","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"%matplotlib inline\nimport numpy as np\nimport pandas as pd\nimport torch\nfrom torch import nn\n\ntrain_data = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest_data = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')\nprint(train_data.shape)\nprint(test_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:58:23.161921Z","iopub.execute_input":"2022-07-08T10:58:23.162323Z","iopub.status.idle":"2022-07-08T10:58:23.208531Z","shell.execute_reply.started":"2022-07-08T10:58:23.162289Z","shell.execute_reply":"2022-07-08T10:58:23.207576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_features = pd.concat((train_data.iloc[:, 1:-1], test_data.iloc[:, 1:]))\n\n# 若无法获得测试数据，则可根据训练数据计算均值和标准差\nnumeric_features = all_features.dtypes[all_features.dtypes != 'object'].index\nall_features[numeric_features] = all_features[numeric_features].apply(\n    lambda x: (x - x.mean()) / (x.std()))\n# 在标准化数据之后，所有均值消失，因此我们可以将缺失值设置为0\nall_features[numeric_features] = all_features[numeric_features].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:58:23.209851Z","iopub.execute_input":"2022-07-08T10:58:23.210685Z","iopub.status.idle":"2022-07-08T10:58:23.275389Z","shell.execute_reply.started":"2022-07-08T10:58:23.210652Z","shell.execute_reply":"2022-07-08T10:58:23.274021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# “Dummy_na=True”将“na”（缺失值）视为有效的特征值，并为其创建指示符特征\nall_features = pd.get_dummies(all_features, dummy_na=True)\nall_features.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:58:23.277694Z","iopub.execute_input":"2022-07-08T10:58:23.278286Z","iopub.status.idle":"2022-07-08T10:58:23.324907Z","shell.execute_reply.started":"2022-07-08T10:58:23.278239Z","shell.execute_reply":"2022-07-08T10:58:23.324304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_train = train_data.shape[0]\ntrain_features = torch.tensor(all_features[:n_train].values, dtype=torch.float32)\ntest_features = torch.tensor(all_features[n_train:].values, dtype=torch.float32)\ntrain_labels = torch.tensor(\n    train_data.SalePrice.values.reshape(-1, 1), dtype=torch.float32)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:58:23.325766Z","iopub.execute_input":"2022-07-08T10:58:23.326579Z","iopub.status.idle":"2022-07-08T10:58:23.337360Z","shell.execute_reply.started":"2022-07-08T10:58:23.326548Z","shell.execute_reply":"2022-07-08T10:58:23.336489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = nn.MSELoss()\nin_features = train_features.shape[1]\n\ndef get_net():\n    net = nn.Sequential(nn.Linear(in_features,1))\n    return net\n\ndef log_rmse(net, features, labels):\n    # 为了在取对数时进一步稳定该值，将小于1的值设置为1\n    clipped_preds = torch.clamp(net(features), 1, float('inf'))\n    rmse = torch.sqrt(loss(torch.log(clipped_preds),\n                           torch.log(labels)))\n    return rmse.item()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:58:23.339349Z","iopub.execute_input":"2022-07-08T10:58:23.340217Z","iopub.status.idle":"2022-07-08T10:58:23.347472Z","shell.execute_reply.started":"2022-07-08T10:58:23.340186Z","shell.execute_reply":"2022-07-08T10:58:23.346616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils import data\ndef load_array(data_arrays, batch_size, is_train=True):\n    \"\"\"Construct a PyTorch data iterator.\n\n    Defined in :numref:`sec_linear_concise`\"\"\"\n    dataset = data.TensorDataset(*data_arrays)\n    return data.DataLoader(dataset, batch_size, shuffle=is_train)\n\ndef train(net, train_features, train_labels, test_features, test_labels,\n          num_epochs, learning_rate, weight_decay, batch_size):\n    train_ls, test_ls = [], []\n    train_iter = load_array((train_features, train_labels), batch_size)\n    # 这里使用的是Adam优化算法\n    optimizer = torch.optim.Adam(net.parameters(),\n                                 lr = learning_rate,\n                                 weight_decay = weight_decay)\n    for epoch in range(num_epochs):\n        for X, y in train_iter:\n            optimizer.zero_grad()\n            l = loss(net(X), y)\n            l.backward()\n            optimizer.step()\n        train_ls.append(log_rmse(net, train_features, train_labels))\n        if test_labels is not None:\n            test_ls.append(log_rmse(net, test_features, test_labels))\n    return train_ls, test_ls","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:58:23.348463Z","iopub.execute_input":"2022-07-08T10:58:23.348681Z","iopub.status.idle":"2022-07-08T10:58:23.360464Z","shell.execute_reply.started":"2022-07-08T10:58:23.348660Z","shell.execute_reply":"2022-07-08T10:58:23.359601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nfrom matplotlib_inline import backend_inline\n\ndef use_svg_display():\n    \"\"\"Use the svg format to display a plot in Jupyter.\n\n    Defined in :numref:`sec_calculus`\"\"\"\n    backend_inline.set_matplotlib_formats('svg')\n    \ndef set_figsize(figsize=(3.5, 2.5)):\n    \"\"\"Set the figure size for matplotlib.\n\n    Defined in :numref:`sec_calculus`\"\"\"\n    use_svg_display()\n    plt.rcParams['figure.figsize'] = figsize\n    \ndef set_axes(axes, xlabel, ylabel, xlim, ylim, xscale, yscale, legend):\n    \"\"\"Set the axes for matplotlib.\n\n    Defined in :numref:`sec_calculus`\"\"\"\n    axes.set_xlabel(xlabel)\n    axes.set_ylabel(ylabel)\n    axes.set_xscale(xscale)\n    axes.set_yscale(yscale)\n    axes.set_xlim(xlim)\n    axes.set_ylim(ylim)\n    if legend:\n        axes.legend(legend)\n    axes.grid()\n    \ndef plot(X, Y=None, xlabel=None, ylabel=None, legend=None, xlim=None,\n         ylim=None, xscale='linear', yscale='linear',\n         fmts=('-', 'm--', 'g-.', 'r:'), figsize=(3.5, 2.5), axes=None):\n    \"\"\"Plot data points.\n\n    Defined in :numref:`sec_calculus`\"\"\"\n    if legend is None:\n        legend = []\n\n    set_figsize(figsize)\n    axes = axes if axes else plt.gca()\n\n    # Return True if `X` (tensor or list) has 1 axis\n    def has_one_axis(X):\n        return (hasattr(X, \"ndim\") and X.ndim == 1 or isinstance(X, list)\n                and not hasattr(X[0], \"__len__\"))\n\n    if has_one_axis(X):\n        X = [X]\n    if Y is None:\n        X, Y = [[]] * len(X), X\n    elif has_one_axis(Y):\n        Y = [Y]\n    if len(X) != len(Y):\n        X = X * len(Y)\n    axes.cla()\n    for x, y, fmt in zip(X, Y, fmts):\n        if len(x):\n            axes.plot(x, y, fmt)\n        else:\n            axes.plot(y, fmt)\n    set_axes(axes, xlabel, ylabel, xlim, ylim, xscale, yscale, legend)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:58:23.361706Z","iopub.execute_input":"2022-07-08T10:58:23.362305Z","iopub.status.idle":"2022-07-08T10:58:23.380217Z","shell.execute_reply.started":"2022-07-08T10:58:23.362274Z","shell.execute_reply":"2022-07-08T10:58:23.379394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_k_fold_data(k, i, X, y):\n    assert k > 1\n    fold_size = X.shape[0] // k\n    X_train, y_train = None, None\n    for j in range(k):\n        idx = slice(j * fold_size, (j + 1) * fold_size)\n        X_part, y_part = X[idx, :], y[idx]\n        if j == i:\n            X_valid, y_valid = X_part, y_part\n        elif X_train is None:\n            X_train, y_train = X_part, y_part\n        else:\n            X_train = torch.cat([X_train, X_part], 0)\n            y_train = torch.cat([y_train, y_part], 0)\n    return X_train, y_train, X_valid, y_valid\n\ndef k_fold(k, X_train, y_train, num_epochs, learning_rate, weight_decay,\n           batch_size):\n    train_l_sum, valid_l_sum = 0, 0\n    for i in range(k):\n        data = get_k_fold_data(k, i, X_train, y_train)\n        net = get_net()\n        train_ls, valid_ls = train(net, *data, num_epochs, learning_rate,\n                                   weight_decay, batch_size)\n        train_l_sum += train_ls[-1]\n        valid_l_sum += valid_ls[-1]\n        if i == 0:\n            plot(list(range(1, num_epochs + 1)), [train_ls, valid_ls],\n                     xlabel='epoch', ylabel='rmse', xlim=[1, num_epochs],\n                     legend=['train', 'valid'], yscale='log')\n        print(f'折{i + 1}，训练log rmse{float(train_ls[-1]):f}, '\n              f'验证log rmse{float(valid_ls[-1]):f}')\n    return train_l_sum / k, valid_l_sum / k","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:58:23.382254Z","iopub.execute_input":"2022-07-08T10:58:23.382843Z","iopub.status.idle":"2022-07-08T10:58:23.395618Z","shell.execute_reply.started":"2022-07-08T10:58:23.382813Z","shell.execute_reply":"2022-07-08T10:58:23.394745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k, num_epochs, lr, weight_decay, batch_size = 5, 100, 5, 0, 64\ntrain_l, valid_l = k_fold(k, train_features, train_labels, num_epochs, lr,\n                          weight_decay, batch_size)\nprint(f'{k}-折验证: 平均训练log rmse: {float(train_l):f}, '\n      f'平均验证log rmse: {float(valid_l):f}')","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:58:23.397021Z","iopub.execute_input":"2022-07-08T10:58:23.397504Z","iopub.status.idle":"2022-07-08T10:58:30.480841Z","shell.execute_reply.started":"2022-07-08T10:58:23.397470Z","shell.execute_reply":"2022-07-08T10:58:30.479779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_and_pred(train_features, test_features, train_labels, test_data,\n                   num_epochs, lr, weight_decay, batch_size):\n    net = get_net()\n    train_ls, _ = train(net, train_features, train_labels, None, None,\n                        num_epochs, lr, weight_decay, batch_size)\n    plot(np.arange(1, num_epochs + 1), [train_ls], xlabel='epoch',\n             ylabel='log rmse', xlim=[1, num_epochs], yscale='log')\n    print(f'训练log rmse：{float(train_ls[-1]):f}')\n    # 将网络应用于测试集。\n    preds = net(test_features).detach().numpy()\n    # 将其重新格式化以导出到Kaggle\n    test_data['SalePrice'] = pd.Series(preds.reshape(1, -1)[0])\n    submission = pd.concat([test_data['Id'], test_data['SalePrice']], axis=1)\n    submission.to_csv('submission.csv', index=False)\n    \ntrain_and_pred(train_features, test_features, train_labels, test_data,\n               num_epochs, lr, weight_decay, batch_size)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T10:58:46.975009Z","iopub.execute_input":"2022-07-08T10:58:46.975348Z","iopub.status.idle":"2022-07-08T10:58:49.533094Z","shell.execute_reply.started":"2022-07-08T10:58:46.975321Z","shell.execute_reply":"2022-07-08T10:58:49.531982Z"},"trusted":true},"execution_count":null,"outputs":[]}]}