{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-20T03:05:50.846899Z","iopub.execute_input":"2022-07-20T03:05:50.847263Z","iopub.status.idle":"2022-07-20T03:05:50.858884Z","shell.execute_reply.started":"2022-07-20T03:05:50.847212Z","shell.execute_reply":"2022-07-20T03:05:50.857842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.tree import DecisionTreeClassifier\nfrom xgboost import XGBClassifier\n","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:05:50.861616Z","iopub.execute_input":"2022-07-20T03:05:50.862385Z","iopub.status.idle":"2022-07-20T03:05:52.180809Z","shell.execute_reply.started":"2022-07-20T03:05:50.862323Z","shell.execute_reply":"2022-07-20T03:05:52.179772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-may-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-may-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:05:52.182674Z","iopub.execute_input":"2022-07-20T03:05:52.182942Z","iopub.status.idle":"2022-07-20T03:06:08.883372Z","shell.execute_reply.started":"2022-07-20T03:05:52.182909Z","shell.execute_reply":"2022-07-20T03:06:08.882260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train.drop(['id', 'target'], axis=1)\ntarget = train.target\n\ntest_df = test.drop(['id'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:08.885004Z","iopub.execute_input":"2022-07-20T03:06:08.885346Z","iopub.status.idle":"2022-07-20T03:06:09.089009Z","shell.execute_reply.started":"2022-07-20T03:06:08.885299Z","shell.execute_reply":"2022-07-20T03:06:09.087678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:09.091843Z","iopub.execute_input":"2022-07-20T03:06:09.092239Z","iopub.status.idle":"2022-07-20T03:06:09.096621Z","shell.execute_reply.started":"2022-07-20T03:06:09.092172Z","shell.execute_reply":"2022-07-20T03:06:09.095761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enc = OrdinalEncoder()\ndef feature_eng(df):\n    df=df.copy()\n    df['char_unique']=df['f_27'].apply(lambda x: len(set(x)))\n    for i in range(df.f_27.str.len().max()):\n        df['f_27_char{}'.format(i+1)]=enc.fit_transform(df['f_27'].str.get(i).values.reshape(-1,1))\n    return df.drop(['f_27'],axis=1)\n\ntrain_df=feature_eng(df=train_df)\ntest_df=feature_eng(df=test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:09.098277Z","iopub.execute_input":"2022-07-20T03:06:09.099408Z","iopub.status.idle":"2022-07-20T03:06:29.381517Z","shell.execute_reply.started":"2022-07-20T03:06:09.099358Z","shell.execute_reply":"2022-07-20T03:06:29.380551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.f_27.str.len().max()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:29.383477Z","iopub.execute_input":"2022-07-20T03:06:29.383849Z","iopub.status.idle":"2022-07-20T03:06:30.037736Z","shell.execute_reply.started":"2022-07-20T03:06:29.383797Z","shell.execute_reply":"2022-07-20T03:06:30.036683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:30.038948Z","iopub.execute_input":"2022-07-20T03:06:30.039178Z","iopub.status.idle":"2022-07-20T03:06:30.088170Z","shell.execute_reply.started":"2022-07-20T03:06:30.039149Z","shell.execute_reply":"2022-07-20T03:06:30.087280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nscaler.fit(train_df)\ntrain_df = scaler.transform(train_df)\nx_test = scaler.transform(test_df)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:30.089882Z","iopub.execute_input":"2022-07-20T03:06:30.090413Z","iopub.status.idle":"2022-07-20T03:06:31.160146Z","shell.execute_reply.started":"2022-07-20T03:06:30.090365Z","shell.execute_reply":"2022-07-20T03:06:31.159043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_val, y_train, y_val = train_test_split(train_df, target)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:31.161664Z","iopub.execute_input":"2022-07-20T03:06:31.161988Z","iopub.status.idle":"2022-07-20T03:06:31.914872Z","shell.execute_reply.started":"2022-07-20T03:06:31.161947Z","shell.execute_reply":"2022-07-20T03:06:31.913707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\nfrom torch.optim import SGD\nfrom torch.nn import CrossEntropyLoss\n\nfrom torch.utils.data import TensorDataset\nfrom torch.utils.data import DataLoader","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:31.918483Z","iopub.execute_input":"2022-07-20T03:06:31.918881Z","iopub.status.idle":"2022-07-20T03:06:33.416079Z","shell.execute_reply.started":"2022-07-20T03:06:31.918829Z","shell.execute_reply":"2022-07-20T03:06:33.415064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn.functional as F","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:33.417857Z","iopub.execute_input":"2022-07-20T03:06:33.418274Z","iopub.status.idle":"2022-07-20T03:06:33.423837Z","shell.execute_reply.started":"2022-07-20T03:06:33.418222Z","shell.execute_reply":"2022-07-20T03:06:33.422817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self):\n        super().__init__()\n    \n        self.fc1 = nn.Linear(41, 100)\n        self.fc2 = nn.Linear(100, 50)\n        self.fc3 = nn.Linear(50, 1)\n        \n    def forward(self, x):\n        x = F.relu(self.fc1(x))\n        x = F.relu(self.fc2(x))\n        x = F.sigmoid(self.fc3(x))\n        \n        return x\n","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:33.424996Z","iopub.execute_input":"2022-07-20T03:06:33.425380Z","iopub.status.idle":"2022-07-20T03:06:33.436246Z","shell.execute_reply.started":"2022-07-20T03:06:33.425347Z","shell.execute_reply":"2022-07-20T03:06:33.435163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Net()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:33.437618Z","iopub.execute_input":"2022-07-20T03:06:33.437841Z","iopub.status.idle":"2022-07-20T03:06:33.479055Z","shell.execute_reply.started":"2022-07-20T03:06:33.437815Z","shell.execute_reply":"2022-07-20T03:06:33.478285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.optim import Adam","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:33.480559Z","iopub.execute_input":"2022-07-20T03:06:33.481629Z","iopub.status.idle":"2022-07-20T03:06:33.485715Z","shell.execute_reply.started":"2022-07-20T03:06:33.481574Z","shell.execute_reply":"2022-07-20T03:06:33.485048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = nn.BCELoss()\n\noptimizer = Adam(model.parameters(), lr=0.01)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:33.487034Z","iopub.execute_input":"2022-07-20T03:06:33.487529Z","iopub.status.idle":"2022-07-20T03:06:33.499177Z","shell.execute_reply.started":"2022-07-20T03:06:33.487484Z","shell.execute_reply":"2022-07-20T03:06:33.497995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = torch.tensor(train_df)\ntarget = torch.tensor(target.values).float()\n\ntest_df = torch.tensor(test_df.values)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:33.501093Z","iopub.execute_input":"2022-07-20T03:06:33.502137Z","iopub.status.idle":"2022-07-20T03:06:33.813230Z","shell.execute_reply.started":"2022-07-20T03:06:33.502081Z","shell.execute_reply":"2022-07-20T03:06:33.812552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.float()\ntest_df = test_df.float()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:33.814345Z","iopub.execute_input":"2022-07-20T03:06:33.815075Z","iopub.status.idle":"2022-07-20T03:06:33.960826Z","shell.execute_reply.started":"2022-07-20T03:06:33.815037Z","shell.execute_reply":"2022-07-20T03:06:33.959900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(train_df, target)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:33.962051Z","iopub.execute_input":"2022-07-20T03:06:33.962330Z","iopub.status.idle":"2022-07-20T03:06:34.517539Z","shell.execute_reply.started":"2022-07-20T03:06:33.962297Z","shell.execute_reply":"2022-07-20T03:06:34.516648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datasets = TensorDataset(x_train, y_train)\ntrain_dataloader = DataLoader(train_datasets, batch_size=256, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:34.518975Z","iopub.execute_input":"2022-07-20T03:06:34.519340Z","iopub.status.idle":"2022-07-20T03:06:34.524748Z","shell.execute_reply.started":"2022-07-20T03:06:34.519292Z","shell.execute_reply":"2022-07-20T03:06:34.523786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(300):\n    loss_sum = 0\n    for x, target in train_dataloader:\n        loss = criterion(model(x).view((-1,)), target)\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        \n        loss_sum += loss.item()\n        \n    loss_average = loss_sum / len(train_dataloader)\n    print(f'{i} loss_average {loss_average}')\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:34.526406Z","iopub.execute_input":"2022-07-20T03:06:34.526919Z","iopub.status.idle":"2022-07-20T03:06:43.466248Z","shell.execute_reply.started":"2022-07-20T03:06:34.526872Z","shell.execute_reply":"2022-07-20T03:06:43.465271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_pred = model(x_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:43.467742Z","iopub.execute_input":"2022-07-20T03:06:43.468602Z","iopub.status.idle":"2022-07-20T03:06:43.940247Z","shell.execute_reply.started":"2022-07-20T03:06:43.468541Z","shell.execute_reply":"2022-07-20T03:06:43.939341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\nroc_auc_score(y_train.detach().numpy(), y_train_pred.detach().numpy())","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:43.941808Z","iopub.execute_input":"2022-07-20T03:06:43.942356Z","iopub.status.idle":"2022-07-20T03:06:44.280366Z","shell.execute_reply.started":"2022-07-20T03:06:43.942309Z","shell.execute_reply":"2022-07-20T03:06:44.279766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_val_pred = model(x_val)\n\nroc_auc_score(y_val.detach().numpy(), y_val_pred.detach().numpy())","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:44.281747Z","iopub.execute_input":"2022-07-20T03:06:44.282422Z","iopub.status.idle":"2022-07-20T03:06:44.541161Z","shell.execute_reply.started":"2022-07-20T03:06:44.282389Z","shell.execute_reply":"2022-07-20T03:06:44.540277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"((y_val_pred > 0.5).int().view((-1,)) == y_val).sum() / y_val.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:06:44.542977Z","iopub.execute_input":"2022-07-20T03:06:44.543259Z","iopub.status.idle":"2022-07-20T03:06:44.570867Z","shell.execute_reply.started":"2022-07-20T03:06:44.543227Z","shell.execute_reply":"2022-07-20T03:06:44.570243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = torch.tensor(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:03:47.421821Z","iopub.execute_input":"2022-07-20T03:03:47.422702Z","iopub.status.idle":"2022-07-20T03:03:47.649644Z","shell.execute_reply.started":"2022-07-20T03:03:47.422657Z","shell.execute_reply":"2022-07-20T03:03:47.648761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = model(x_test.float())","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:03:47.651099Z","iopub.execute_input":"2022-07-20T03:03:47.651568Z","iopub.status.idle":"2022-07-20T03:03:48.729622Z","shell.execute_reply.started":"2022-07-20T03:03:47.651522Z","shell.execute_reply":"2022-07-20T03:03:48.728571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/tabular-playground-series-may-2022/sample_submission.csv')\nsubmission.target = y_test.detach().numpy()\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:03:48.730980Z","iopub.execute_input":"2022-07-20T03:03:48.731757Z","iopub.status.idle":"2022-07-20T03:03:50.916037Z","shell.execute_reply.started":"2022-07-20T03:03:48.731720Z","shell.execute_reply":"2022-07-20T03:03:50.915278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-07-20T03:03:50.919155Z","iopub.execute_input":"2022-07-20T03:03:50.919880Z","iopub.status.idle":"2022-07-20T03:03:50.933190Z","shell.execute_reply.started":"2022-07-20T03:03:50.919836Z","shell.execute_reply":"2022-07-20T03:03:50.932276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}