{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-15T04:55:10.743407Z","iopub.execute_input":"2022-07-15T04:55:10.743814Z","iopub.status.idle":"2022-07-15T04:55:10.760293Z","shell.execute_reply.started":"2022-07-15T04:55:10.743783Z","shell.execute_reply":"2022-07-15T04:55:10.759260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nimport pandas as pd\nimport numpy as np\nimport os\nimport sys\nimport torch\n\n\n\n\n\nloss = torch.nn.MSELoss()\n\ndef get_net(feature_num):\n    net = nn.Linear(feature_num, 1)\n    for param in net.parameters():\n        nn.init.normal_(param, mean=0, std=0.01)\n    return net\n\n\n# In[11]:\n\n\ndef log_rmse(net, features, labels):\n    with torch.no_grad():\n        # 将小于1的值设成1，使得取对数时数值更稳定\n        clipped_preds = torch.max(net(features), torch.tensor(1.0))\n        rmse = torch.sqrt(loss(clipped_preds.log(), labels.log()))\n        # print(f'loss:{loss(clipped_preds,labels)}')\n\n    return rmse.item()\n\n\n# In[12]:\n\n\ndef train(net, train_features, train_labels, test_features, test_labels,\n          num_epochs, learning_rate, weight_decay, batch_size):\n    train_ls, test_ls = [], []\n    dataset = torch.utils.data.TensorDataset(train_features, train_labels)\n    train_iter = torch.utils.data.DataLoader(dataset, batch_size, shuffle=True)\n    # 这里使用了Adam优化算法\n    optimizer = torch.optim.Adam(params=net.parameters(), lr=learning_rate, weight_decay=weight_decay)\n    net = net.float()\n    for epoch in range(num_epochs):\n        for X, y in train_iter:\n            l = loss(net(X.float()), y.float())\n            optimizer.zero_grad()\n            l.backward()\n            optimizer.step()\n        train_ls.append(log_rmse(net, train_features, train_labels))\n        if test_labels is not None:\n            test_ls.append(log_rmse(net, test_features, test_labels))\n    return train_ls, test_ls\n\n\n# ## 3.16.5 $K$折交叉验证\n\n# In[13]:\n\n\ndef get_k_fold_data(k, i, X, y):\n    # 返回第i折交叉验证时所需要的训练和验证数据\n    assert k > 1\n    fold_size = X.shape[0] // k\n    X_train, y_train = None, None\n    for j in range(k):\n        idx = slice(j * fold_size, (j + 1) * fold_size)\n        X_part, y_part = X[idx, :], y[idx]\n        if j == i:\n            X_valid, y_valid = X_part, y_part\n        elif X_train is None:\n            X_train, y_train = X_part, y_part\n        else:\n            X_train = torch.cat((X_train, X_part), dim=0)\n            y_train = torch.cat((y_train, y_part), dim=0)\n    return X_train, y_train, X_valid, y_valid\n\n\n# In[14]:\n\n\ndef k_fold(k, X_train, y_train, num_epochs,\n           learning_rate, weight_decay, batch_size):\n    train_l_sum, valid_l_sum = 0, 0\n    for i in range(k):\n        data = get_k_fold_data(k, i, X_train, y_train)\n        net = get_net(X_train.shape[1])\n        train_ls, valid_ls = train(net, *data, num_epochs, learning_rate,\n                                   weight_decay, batch_size)\n        train_l_sum += train_ls[-1]\n        valid_l_sum += valid_ls[-1]\n        print('fold %d, train rmse %f, valid rmse %f' % (i, train_ls[-1], valid_ls[-1]))\n\n    return train_l_sum / k, valid_l_sum / k\n\n\n# ## 3.16.6 模型选择\n\n# In[15]:\n\n\n\n\n\n# ## 3.16.7 预测并在Kaggle提交结果\n\n# In[16]:\n\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T04:06:14.342281Z","iopub.execute_input":"2022-07-15T04:06:14.342702Z","iopub.status.idle":"2022-07-15T04:06:14.368750Z","shell.execute_reply.started":"2022-07-15T04:06:14.342659Z","shell.execute_reply":"2022-07-15T04:06:14.367222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_data=pd.read_csv('/kaggle/input/california-house-prices/train.csv')\ntest_data=pd.read_csv('/kaggle/input/california-house-prices/test.csv')\nall_features=pd.concat((train_data.iloc[:,train_data.columns!='Sold Price'],test_data))\nall_features=all_features.iloc[:,1:]\nnumeric_features = all_features.dtypes[all_features.dtypes != 'object'].index\nall_features[numeric_features] = all_features[numeric_features].apply(lambda x: (x - x.mean()) / (x.std()))\n# 标准化后，每个数值特征的均值变为0，所以可以直接用0来替换缺失值\nall_features[numeric_features] = all_features[numeric_features].fillna(0)\nall_features=all_features[numeric_features]\nn_train = train_data.shape[0]\ntrain_features = torch.tensor(all_features[:n_train].values, dtype=torch.float)\ntest_features = torch.tensor(all_features[n_train:].values, dtype=torch.float)\ntrain_labels = torch.tensor(train_data['Sold Price'].values, dtype=torch.float).view(-1, 1)\n\nk=5\nnum_epochs=150\n# 100 2.344\n# 110 2.23\n# 120 2.17\nlr=20\nweight_decay=0.5\nbatch_size=64\n","metadata":{"execution":{"iopub.status.busy":"2022-07-15T04:07:59.374116Z","iopub.execute_input":"2022-07-15T04:07:59.374524Z","iopub.status.idle":"2022-07-15T04:08:02.001253Z","shell.execute_reply.started":"2022-07-15T04:07:59.374491Z","shell.execute_reply":"2022-07-15T04:08:02.000114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_l, valid_l = k_fold(k, train_features, train_labels, num_epochs, lr, weight_decay, batch_size)\nprint('%d-fold validation: avg train rmse %f, avg valid rmse %f' % (k, train_l, valid_l))","metadata":{"execution":{"iopub.status.busy":"2022-07-15T04:08:25.443277Z","iopub.execute_input":"2022-07-15T04:08:25.443689Z","iopub.status.idle":"2022-07-15T04:15:20.890356Z","shell.execute_reply.started":"2022-07-15T04:08:25.443656Z","shell.execute_reply":"2022-07-15T04:15:20.889008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_and_pred(train_features, test_features, train_labels, test_data,\n                   num_epochs, lr, weight_decay, batch_size):\n    net = get_net(train_features.shape[1])\n    train_ls, _ = train(net, train_features, train_labels, None, None,\n                        num_epochs, lr, weight_decay, batch_size)\n    print('train rmse %f' % train_ls[-1])\n    preds = net(test_features).detach().numpy()\n    test_data['Sold Price'] = pd.Series(preds.reshape(1, -1)[0])\n    submission = pd.concat([test_data['Id'], test_data['Sold Price']], axis=1)\n    submission.to_csv('./submission.csv', index=False)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T04:17:09.807858Z","iopub.execute_input":"2022-07-15T04:17:09.808976Z","iopub.status.idle":"2022-07-15T04:17:09.818262Z","shell.execute_reply.started":"2022-07-15T04:17:09.808925Z","shell.execute_reply":"2022-07-15T04:17:09.817213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_and_pred(train_features, test_features, train_labels, test_data,\n               num_epochs, lr, weight_decay, batch_size)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T04:18:31.331801Z","iopub.execute_input":"2022-07-15T04:18:31.332216Z","iopub.status.idle":"2022-07-15T04:20:19.786328Z","shell.execute_reply.started":"2022-07-15T04:18:31.332184Z","shell.execute_reply":"2022-07-15T04:20:19.784963Z"},"trusted":true},"execution_count":null,"outputs":[]}]}