{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T23:01:32.570066Z","iopub.execute_input":"2022-07-13T23:01:32.571242Z","iopub.status.idle":"2022-07-13T23:01:32.581701Z","shell.execute_reply.started":"2022-07-13T23:01:32.571193Z","shell.execute_reply":"2022-07-13T23:01:32.580244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pre-Processing","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import StandardScaler\n\ntrain_fname = '/kaggle/input/titanic/train.csv'\ntest_fname = '/kaggle/input/titanic/test.csv'\n\ndf = pd.read_csv(train_fname)\ndf = df.fillna(0)\ndf = df[['PassengerId', 'Pclass', 'Sex', 'Age', 'SibSp', 'Parch', 'Fare', 'Survived']]\nsurvived = df['Survived']\ndf['Sex'] = LabelEncoder().fit_transform(df['Sex'])\nscaled_features = StandardScaler().fit_transform(df[df.columns[:-1]])\ndf = pd.DataFrame(scaled_features, index=df.index, columns=df.columns[:-1])\ndf['Survived'] = survived\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:32.644466Z","iopub.execute_input":"2022-07-13T23:01:32.645331Z","iopub.status.idle":"2022-07-13T23:01:32.678033Z","shell.execute_reply.started":"2022-07-13T23:01:32.645277Z","shell.execute_reply":"2022-07-13T23:01:32.677207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Correlations & Feature Extraction","metadata":{}},{"cell_type":"code","source":"corr = abs(df.corr()['Survived'].drop('Survived')).sort_values(ascending=False)\nfeatures = list(corr[:-1].keys())\nprint(corr)\nprint(features)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:32.679684Z","iopub.execute_input":"2022-07-13T23:01:32.680255Z","iopub.status.idle":"2022-07-13T23:01:32.689090Z","shell.execute_reply.started":"2022-07-13T23:01:32.680224Z","shell.execute_reply":"2022-07-13T23:01:32.687897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SKLearn Models","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = np.array(df[features])\ny = np.array(df['Survived'])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:32.690946Z","iopub.execute_input":"2022-07-13T23:01:32.691241Z","iopub.status.idle":"2022-07-13T23:01:32.700120Z","shell.execute_reply.started":"2022-07-13T23:01:32.691215Z","shell.execute_reply":"2022-07-13T23:01:32.699040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression, Perceptron\nfrom sklearn.svm import SVC\nfrom sklearn.neural_network import MLPClassifier\n\n\nlr = MLPClassifier()\n\nprint('Training Model')\n\nlr.fit(X, y)\n\n#print('Making Predictions')\n\n#preds = lr.predict(X_test)\n\n#print('Scoring Predictions')\n\n#lr.score(X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:32.720457Z","iopub.execute_input":"2022-07-13T23:01:32.720833Z","iopub.status.idle":"2022-07-13T23:01:34.042198Z","shell.execute_reply.started":"2022-07-13T23:01:32.720804Z","shell.execute_reply":"2022-07-13T23:01:34.040996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from sklearn.metrics import f1_score\n\n#f1_score(y_test, preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:34.044506Z","iopub.execute_input":"2022-07-13T23:01:34.045305Z","iopub.status.idle":"2022-07-13T23:01:34.057420Z","shell.execute_reply.started":"2022-07-13T23:01:34.045260Z","shell.execute_reply":"2022-07-13T23:01:34.055809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## PyTorch Models","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torch import nn\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import datasets\nfrom torchvision.transforms import ToTensor","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:34.063020Z","iopub.execute_input":"2022-07-13T23:01:34.064269Z","iopub.status.idle":"2022-07-13T23:01:34.078319Z","shell.execute_reply.started":"2022-07-13T23:01:34.064195Z","shell.execute_reply":"2022-07-13T23:01:34.076800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TitanicDataset(Dataset):\n    def __init__(self, df, transform=None, target_transform=None):\n        self.df = df\n        \n        x = df.iloc[:, :-1].values\n        y = df.iloc[:, -1].values\n        \n        self.transform = transform\n        self.target_transform = target_transform\n        self.x=torch.tensor(x,dtype=torch.float32)\n        self.y=torch.tensor(y,dtype=torch.float32)\n        \n    def __len__(self):\n        return len(self.df)\n        \n    def __getitem__(self, idx):\n        data = self.x[idx]\n        label = self.y[idx]\n        \n        if self.transform:\n            data = self.transform(data)\n        if self.target_transform:\n            label = self.target_transform(label)\n        return self.x[idx], self.y[idx]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:34.084663Z","iopub.execute_input":"2022-07-13T23:01:34.094228Z","iopub.status.idle":"2022-07-13T23:01:34.111881Z","shell.execute_reply.started":"2022-07-13T23:01:34.086952Z","shell.execute_reply":"2022-07-13T23:01:34.110233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch_features = features[:]\ntorch_features.append('Survived')\ntrain_data = TitanicDataset(df[torch_features].sample(frac=0.9))\ntrain_loader = DataLoader(train_data, batch_size=32, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:34.117801Z","iopub.execute_input":"2022-07-13T23:01:34.118793Z","iopub.status.idle":"2022-07-13T23:01:34.138027Z","shell.execute_reply.started":"2022-07-13T23:01:34.118732Z","shell.execute_reply":"2022-07-13T23:01:34.136076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display image and label.\ntrain_features, train_labels = next(iter(train_loader))\nprint(f\"Feature batch shape: {train_features.size()}\")\nprint(f\"Labels batch shape: {train_labels.size()}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:34.140585Z","iopub.execute_input":"2022-07-13T23:01:34.141637Z","iopub.status.idle":"2022-07-13T23:01:34.152528Z","shell.execute_reply.started":"2022-07-13T23:01:34.141544Z","shell.execute_reply":"2022-07-13T23:01:34.151015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn.functional as F\n\nclass MLP(nn.Module):\n    def __init__(self):\n        super(MLP, self).__init__()\n        self.layers = nn.Sequential(\n            nn.Linear(6, 128),\n            nn.ReLU(),\n            nn.Linear(128, 64),\n            nn.ReLU(),\n            nn.Linear(64, 32),\n            nn.ReLU(),\n            nn.Linear(32, 8),\n            nn.ReLU(),\n            nn.Linear(8, 1)\n            )\n\n    def forward(self, x):\n        x = self.layers(x)\n        x = F.sigmoid(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:34.154593Z","iopub.execute_input":"2022-07-13T23:01:34.155515Z","iopub.status.idle":"2022-07-13T23:01:34.164212Z","shell.execute_reply.started":"2022-07-13T23:01:34.155425Z","shell.execute_reply":"2022-07-13T23:01:34.163312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = MLP()\nprint(model)\n\n# Define the loss function and optimizer\nloss_fn = nn.BCELoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=1e-2)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:34.165856Z","iopub.execute_input":"2022-07-13T23:01:34.166545Z","iopub.status.idle":"2022-07-13T23:01:34.182907Z","shell.execute_reply.started":"2022-07-13T23:01:34.166502Z","shell.execute_reply":"2022-07-13T23:01:34.181759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def accuracy(outputs, targets):\n    num_correct = 0\n    num_samples = 0\n    num_correct += (outputs == targets).sum()\n    num_samples += outputs.size(0)\n    return num_correct / num_samples","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:34.184744Z","iopub.execute_input":"2022-07-13T23:01:34.185112Z","iopub.status.idle":"2022-07-13T23:01:34.192269Z","shell.execute_reply.started":"2022-07-13T23:01:34.185082Z","shell.execute_reply":"2022-07-13T23:01:34.191425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 300\n\nmetrics = {}\nmetrics['loss'] = []\nmetrics['accuracy'] = []\n\nprint('Beginning Training')\nfor epoch in range(epochs): # iterate through each epoch\n    epoch_loss = 0\n    epoch_accuracy = 0\n    for idx, batch in enumerate(train_loader):\n        inputs, labels = batch\n        labels = torch.unsqueeze(labels, 1)\n        \n        \n        optimizer.zero_grad() # zero the gradients each batch\n        \n        outputs = model(inputs)# return class probabilities?\n        \n        predictions = (outputs > 0.5).float()\n        \n        loss = loss_fn(outputs, labels) # calculate loss\n        \n        loss.backward() # backpropagation?\n        \n        optimizer.step() # update learning weights\n        \n        epoch_loss += loss.item()\n        epoch_accuracy += accuracy(predictions, labels)\n    if epoch % 10 == 0:\n        print(f'Epoch {epoch + 1} -> Loss: {epoch_loss / (idx + 1)} Accuracy: {epoch_accuracy / (idx + 1)}')\n    metrics['accuracy'].append(epoch_accuracy / (idx + 1))\n    metrics['loss'].append(epoch_loss / (idx + 1))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:34.195902Z","iopub.execute_input":"2022-07-13T23:01:34.196522Z","iopub.status.idle":"2022-07-13T23:01:46.715457Z","shell.execute_reply.started":"2022-07-13T23:01:34.196479Z","shell.execute_reply":"2022-07-13T23:01:46.714532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.plot(np.linspace(0, epochs, epochs), metrics['accuracy'])\nplt.title('Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.show()\n\nplt.plot(np.linspace(0, epochs, epochs), metrics['loss'])\nplt.title('Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:46.717068Z","iopub.execute_input":"2022-07-13T23:01:46.718185Z","iopub.status.idle":"2022-07-13T23:01:47.095262Z","shell.execute_reply.started":"2022-07-13T23:01:46.718137Z","shell.execute_reply":"2022-07-13T23:01:47.094453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(test_fname)\nid = df_test['PassengerId']\ndf_test = df_test.fillna(0)\ndf_test = df_test[['PassengerId', 'Pclass', 'Sex', 'Age', 'SibSp', 'Parch', 'Fare']]\ndf_test['Sex'] = LabelEncoder().fit_transform(df_test['Sex'])\nscaled_features = StandardScaler().fit_transform(df_test)\ndf_test = pd.DataFrame(scaled_features, index=df_test.index, columns=df_test.columns)\ndf_test['PassengerId'] = id\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:08:40.713745Z","iopub.execute_input":"2022-07-13T23:08:40.715304Z","iopub.status.idle":"2022-07-13T23:08:40.747002Z","shell.execute_reply.started":"2022-07-13T23:08:40.715221Z","shell.execute_reply":"2022-07-13T23:08:40.745923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(df_test[features], dtype=np.float32)\n\ninputs = torch.tensor(X)\n\noutputs = model(inputs)# return class probabilities?\n        \npredictions = (outputs > 0.5).int()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:47.120281Z","iopub.execute_input":"2022-07-13T23:01:47.120608Z","iopub.status.idle":"2022-07-13T23:01:47.131034Z","shell.execute_reply.started":"2022-07-13T23:01:47.120578Z","shell.execute_reply":"2022-07-13T23:01:47.129719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = {'PassengerId':df_test['PassengerId'], 'Survived':predictions.reshape((418,))}\nsubmission = pd.DataFrame(submission)\nsubmission = submission.to_csv('submission.csv', index=False)\n#submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T23:01:47.132910Z","iopub.execute_input":"2022-07-13T23:01:47.133663Z","iopub.status.idle":"2022-07-13T23:01:47.144468Z","shell.execute_reply.started":"2022-07-13T23:01:47.133616Z","shell.execute_reply":"2022-07-13T23:01:47.143387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}