{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T07:34:15.125995Z","iopub.execute_input":"2022-07-28T07:34:15.126999Z","iopub.status.idle":"2022-07-28T07:34:15.135627Z","shell.execute_reply.started":"2022-07-28T07:34:15.126961Z","shell.execute_reply":"2022-07-28T07:34:15.134423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T07:34:15.138090Z","iopub.execute_input":"2022-07-28T07:34:15.138923Z","iopub.status.idle":"2022-07-28T07:34:15.197044Z","shell.execute_reply.started":"2022-07-28T07:34:15.138877Z","shell.execute_reply":"2022-07-28T07:34:15.195887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T07:34:15.198784Z","iopub.execute_input":"2022-07-28T07:34:15.199403Z","iopub.status.idle":"2022-07-28T07:34:15.240372Z","shell.execute_reply.started":"2022-07-28T07:34:15.199370Z","shell.execute_reply":"2022-07-28T07:34:15.239183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"women = train_data.loc[train_data.Sex == 'female'][\"Survived\"]\nrate_women = sum(women)/len(women)\n\nprint(\"% of women who survived:\", rate_women)\n\n\nmen = train_data.loc[train_data.Sex == 'male'][\"Survived\"]\nrate_men = sum(men)/len(men)\n\nprint(\"% of men who survived:\", rate_men)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T07:34:15.242894Z","iopub.execute_input":"2022-07-28T07:34:15.243647Z","iopub.status.idle":"2022-07-28T07:34:15.264426Z","shell.execute_reply.started":"2022-07-28T07:34:15.243604Z","shell.execute_reply":"2022-07-28T07:34:15.263146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom torch.nn import Linear\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\n#from torch.utils.data import SubsetRandomSampler\nfrom torch.utils.data import random_split\n#class torch.utils.data.SubsetRandomSampler(indices)：无放回地按照给定的索引列表采样样本元素。\n# prepare dataset\n\nclass TitanicDataset(Dataset):\n    def __init__(self, filepath):\n        xy = pd.read_csv(filepath, dtype=np.float32)\n        #xy = np.loadtxt(filepath, delimiter=',', dtype=np.float32)\n\n        self.len = xy.shape[0]           # shape(多少行，多少列)\n        self.x_data = torch.from_numpy(np.array(xy)[:, :-1])\n        self.y_data = torch.from_numpy(np.array(xy)[:, [-1]])      #(xy[:, [-1]]) [-1]后转化为二维矩阵，否则为一维\n\n    def __getitem__(self, index):\n        return self.x_data[index], self.y_data[index]\n\n    def __len__(self):\n        return self.len\n\n\ndataset = TitanicDataset('/kaggle/input/train-data/train_data.csv')\n\nbatch_size = 32\nvalidation_split = 0.2\nshuffle_dataset = True\n#random_seed = 42\nepochs = 3000\nlearn_rating = 0.001\n\n#random_split 划分训练集与测试集(交叉验证集)\ntrain_split = 0.8\ntrain_size = int(train_split * len(dataset))\ntest_size = len(dataset) - train_size\ntrain_dataset, test_dataset = random_split(dataset, [train_size, test_size])\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nvalidation_loader = DataLoader(test_dataset, batch_size=test_size, shuffle=False)\n\n\n# design model using class\n\nclass Model(torch.nn.Module):\n    linear3: Linear\n\n    def __init__(self):\n        super(Model, self).__init__()\n        self.linear1 = torch.nn.Linear(9, 6)\n        self.linear2 = torch.nn.Linear(6, 4)\n        self.linear3 = torch.nn.Linear(4, 1)\n        self.activate = torch.nn.ReLU()\n        self.sigmoid = torch.nn.Sigmoid()\n\n    def forward(self, x):\n        x = self.activate(self.linear1(x))\n        x = self.activate(self.linear2(x))\n        x = self.sigmoid(self.linear3(x))\n        return x\n\n\nmodel = Model()\n\n# construct loss and optimizer\ncriterion = torch.nn.BCELoss()       #默认reduction='mean'\noptimizer = torch.optim.SGD(model.parameters(), lr=learn_rating)\n\n\ndef train(epoch):\n    for batch_idx, data in enumerate(train_loader, 0):\n        inputs, target = data\n        optimizer.zero_grad()\n        # forward + backward + update\n        outputs = model(inputs)\n        loss = criterion(outputs, target)\n        loss.backward()\n        optimizer.step()\n\n    with torch.no_grad():\n        y_pred = model(train_dataset[:][0])  #train_dataset[:]是包含train_dataset对应的x_data和y_data的元组\n                                            # 然后[0]为第一个tensor，即train_dataset对应的x_data\n        train_loss = criterion(y_pred, train_dataset[:][1])\n\n        y_pred_label = torch.where(y_pred >= 0.5, torch.tensor([1.0]), torch.tensor([0.0]))\n        train_acc = torch.eq(y_pred_label, train_dataset[:][1]).sum().item() / len(y_pred)  # eq将两个tensor进行对比，相同为true\n\n\n\n    return train_loss, train_acc\n\n\ndef test():\n    with torch.no_grad():\n        for inputs, target in validation_loader:\n            y_pred = model(inputs)\n            y_pred_label = torch.where(y_pred >= 0.5, torch.tensor([1.0]), torch.tensor([0.0]))\n            test_acc = torch.eq(y_pred_label, target).sum().item() / len(y_pred)      # eq将两个tensor进行对比，相同为true\n\n    return test_acc\n\n# training cycle forward, backward, update\n\n\nif __name__ == '__main__':\n\n    train_loss_list = []\n    train_acc =[]\n    test_acc = []\n\n    for epoch in range(epochs):\n\n        train_loss_list.append(train(epoch)[0])\n        train_acc.append(train(epoch)[1])\n        test_acc.append(test())\n\n        print(f'epoch {epoch + 1}, train loss {float(train_loss_list[epoch]):f}')\n        print('Accuracy on train set: %d %%' % (100 * train_acc[epoch]),\n              '\\nAccuracy on test set: %d %%' % (100 * test_acc[epoch]))\n\n    #可视化\n    plt.rcParams['font.family'] = ['SimHei']  # 解决不能输出中文的问题。不区分大小写，即SimHei’效果等价于‘simhei’，中括号可以不要\n    plt.rcParams['figure.autolayout'] = True   # 解决不能完整显示的问题（比如因为饼图太大，显示窗口太小）\n    plt.rcParams['axes.unicode_minus'] = False\n    fig, axs = plt.subplots(1, 2, figsize=(12,8))\n    axs[0].plot(range(1, epochs+1), train_loss_list,c = \"b\", label = \"train loss\")\n    axs[1].plot(range(1, epochs+1), train_acc,c = \"r\", label = \"Accuracy on train set\")\n    axs[1].plot(range(1, epochs+1), test_acc, c = \"b\", label = \"Accuracy on test set\")\n\n    axs[0].set(xlabel = \"epoch\", ylabel = \"train loss\")\n    axs[1].set(xlabel = \"epoch\", ylabel = \"Accuracy\")\n\n    for ax in axs:\n        ax.legend()\n    plt.show()\n    \n    test_ = pd.read_csv('../input/test-data/test_data.csv', dtype=np.float32)\n    test_ = torch.from_numpy(np.array(test_))\n    with torch.no_grad():\n        Survived = model(test_)\n        outs = torch.where(Survived >= 0.5, torch.tensor([1.0]), torch.tensor([0.0]))\n        submission = pd.DataFrame({'PassengerId': range(892,1310), 'Survived': outs.detach().numpy().flatten()})\n        submission.Survived = submission.Survived.astype('int64')\n        submission.to_csv('submission.csv', index=False)\n    \n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T08:07:34.733799Z","iopub.execute_input":"2022-07-28T08:07:34.734212Z","iopub.status.idle":"2022-07-28T08:09:37.951194Z","shell.execute_reply.started":"2022-07-28T08:07:34.734178Z","shell.execute_reply":"2022-07-28T08:09:37.949899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T07:52:45.378072Z","iopub.execute_input":"2022-07-28T07:52:45.378484Z","iopub.status.idle":"2022-07-28T07:52:45.396162Z","shell.execute_reply.started":"2022-07-28T07:52:45.378452Z","shell.execute_reply":"2022-07-28T07:52:45.395125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.Survived = submission.Survived.astype('int64')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T07:53:43.898015Z","iopub.execute_input":"2022-07-28T07:53:43.898380Z","iopub.status.idle":"2022-07-28T07:53:43.904247Z","shell.execute_reply.started":"2022-07-28T07:53:43.898351Z","shell.execute_reply":"2022-07-28T07:53:43.902952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T07:53:46.074970Z","iopub.execute_input":"2022-07-28T07:53:46.075647Z","iopub.status.idle":"2022-07-28T07:53:46.087670Z","shell.execute_reply.started":"2022-07-28T07:53:46.075610Z","shell.execute_reply":"2022-07-28T07:53:46.086591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T07:54:32.697366Z","iopub.execute_input":"2022-07-28T07:54:32.697758Z","iopub.status.idle":"2022-07-28T07:54:32.704874Z","shell.execute_reply.started":"2022-07-28T07:54:32.697725Z","shell.execute_reply":"2022-07-28T07:54:32.703921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rr = pd.read_csv('./submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T07:55:44.193459Z","iopub.execute_input":"2022-07-28T07:55:44.193846Z","iopub.status.idle":"2022-07-28T07:55:44.201669Z","shell.execute_reply.started":"2022-07-28T07:55:44.193795Z","shell.execute_reply":"2022-07-28T07:55:44.200848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rr.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T07:55:51.670507Z","iopub.execute_input":"2022-07-28T07:55:51.670942Z","iopub.status.idle":"2022-07-28T07:55:51.683312Z","shell.execute_reply.started":"2022-07-28T07:55:51.670906Z","shell.execute_reply":"2022-07-28T07:55:51.681980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}