{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T12:54:50.361573Z","iopub.execute_input":"2022-08-03T12:54:50.362013Z","iopub.status.idle":"2022-08-03T12:54:50.370879Z","shell.execute_reply.started":"2022-08-03T12:54:50.361978Z","shell.execute_reply":"2022-08-03T12:54:50.369771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"thanks https://www.kaggle.com/code/cv13j0/tps-aug22-binary-classification","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.390320Z","iopub.execute_input":"2022-08-03T12:54:50.391084Z","iopub.status.idle":"2022-08-03T12:54:50.396759Z","shell.execute_reply.started":"2022-08-03T12:54:50.391012Z","shell.execute_reply":"2022-08-03T12:54:50.395574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('../input/tabular-playground-series-aug-2022/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.418341Z","iopub.execute_input":"2022-08-03T12:54:50.418963Z","iopub.status.idle":"2022-08-03T12:54:50.616137Z","shell.execute_reply.started":"2022-08-03T12:54:50.418928Z","shell.execute_reply":"2022-08-03T12:54:50.614335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.isnull().sum())\nprint('\\r\\n', '='*30, '\\r\\n')\ndisplay(test.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.618085Z","iopub.execute_input":"2022-08-03T12:54:50.618475Z","iopub.status.idle":"2022-08-03T12:54:50.643114Z","shell.execute_reply.started":"2022-08-03T12:54:50.618443Z","shell.execute_reply":"2022-08-03T12:54:50.641946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fill_missing(df):\n    numerics = ['int16', 'int32', 'int64','float16','float32', 'float64']\n    for col in df.select_dtypes(include=numerics):\n        df[col] = df[col].fillna(value = df[col].mean())\n    \n    for col in df.select_dtypes(exclude=numerics):\n        df[col] = df[col].fillna(value = df[col].mode())\n    return df\n\ntrain = fill_missing(train)\ntest = fill_missing(test)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.644473Z","iopub.execute_input":"2022-08-03T12:54:50.644817Z","iopub.status.idle":"2022-08-03T12:54:50.714959Z","shell.execute_reply.started":"2022-08-03T12:54:50.644786Z","shell.execute_reply":"2022-08-03T12:54:50.713961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def extract_num_code(df, feat = ['attribute_0', 'attribute_1']):\n#     for col in df[feat].columns:\n#         df[col] = df[col].str.split('_', 1).str[1].astype('int')\n#     return df\n\n# train = extract_num_code(train)\n# test = extract_num_code(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.717003Z","iopub.execute_input":"2022-08-03T12:54:50.717824Z","iopub.status.idle":"2022-08-03T12:54:50.721761Z","shell.execute_reply.started":"2022-08-03T12:54:50.717789Z","shell.execute_reply":"2022-08-03T12:54:50.720896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.preprocessing import LabelEncoder\n\n# def encode_labels(df, text_features = ['product_code']):\n#     for categ_col in df[text_features].columns:\n#         encoder = LabelEncoder()\n#         df[categ_col] = encoder.fit_transform(df[categ_col])\n#     return df\n\n# train = encode_labels(train)\n# test = encode_labels(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.722947Z","iopub.execute_input":"2022-08-03T12:54:50.723477Z","iopub.status.idle":"2022-08-03T12:54:50.739411Z","shell.execute_reply.started":"2022-08-03T12:54:50.723446Z","shell.execute_reply":"2022-08-03T12:54:50.738049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop(['product_code', 'attribute_0', 'attribute_1', 'attribute_2', 'attribute_3'], axis=1)\ntest = test.drop(['product_code', 'attribute_0', 'attribute_1', 'attribute_2', 'attribute_3'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.741089Z","iopub.execute_input":"2022-08-03T12:54:50.741570Z","iopub.status.idle":"2022-08-03T12:54:50.754998Z","shell.execute_reply.started":"2022-08-03T12:54:50.741527Z","shell.execute_reply":"2022-08-03T12:54:50.753971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weight = 1. / (train.failure.value_counts().values / train.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.756364Z","iopub.execute_input":"2022-08-03T12:54:50.756893Z","iopub.status.idle":"2022-08-03T12:54:50.766906Z","shell.execute_reply.started":"2022-08-03T12:54:50.756848Z","shell.execute_reply":"2022-08-03T12:54:50.765836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weight","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.771210Z","iopub.execute_input":"2022-08-03T12:54:50.772255Z","iopub.status.idle":"2022-08-03T12:54:50.781831Z","shell.execute_reply.started":"2022-08-03T12:54:50.772216Z","shell.execute_reply":"2022-08-03T12:54:50.780371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_data = train.drop(['id', 'failure'], axis=1)\ny_data = train.failure\n\nx_test = test.drop('id', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.783200Z","iopub.execute_input":"2022-08-03T12:54:50.783653Z","iopub.status.idle":"2022-08-03T12:54:50.796916Z","shell.execute_reply.started":"2022-08-03T12:54:50.783611Z","shell.execute_reply":"2022-08-03T12:54:50.795608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nx_data = scaler.fit_transform(x_data)\nx_test = scaler.transform(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.800203Z","iopub.execute_input":"2022-08-03T12:54:50.800563Z","iopub.status.idle":"2022-08-03T12:54:50.827359Z","shell.execute_reply.started":"2022-08-03T12:54:50.800532Z","shell.execute_reply":"2022-08-03T12:54:50.826010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\npca = PCA(n_components=9)\nx_data = pca.fit_transform(x_data)\nx_test = pca.transform(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:50.829299Z","iopub.execute_input":"2022-08-03T12:54:50.829698Z","iopub.status.idle":"2022-08-03T12:54:51.069318Z","shell.execute_reply.started":"2022-08-03T12:54:50.829663Z","shell.execute_reply":"2022-08-03T12:54:51.067170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train, x_val, y_train, y_val = train_test_split(x_data, y_data, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.071809Z","iopub.execute_input":"2022-08-03T12:54:51.074346Z","iopub.status.idle":"2022-08-03T12:54:51.106558Z","shell.execute_reply.started":"2022-08-03T12:54:51.074277Z","shell.execute_reply":"2022-08-03T12:54:51.104614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.utils.data import TensorDataset\nfrom torch.utils.data import DataLoader\nfrom torch.optim import Adam\nfrom torch.nn import BCELoss\n\nimport torch.nn.functional as F","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.115502Z","iopub.execute_input":"2022-08-03T12:54:51.122469Z","iopub.status.idle":"2022-08-03T12:54:51.145521Z","shell.execute_reply.started":"2022-08-03T12:54:51.122382Z","shell.execute_reply":"2022-08-03T12:54:51.143754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = torch.tensor(x_train, dtype=torch.float32)\ny_train = torch.tensor(y_train.values, dtype=torch.float32)\n\nx_val = torch.tensor(x_val, dtype=torch.float32)\ny_val = torch.tensor(y_val.values, dtype=torch.float32)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.149011Z","iopub.execute_input":"2022-08-03T12:54:51.156354Z","iopub.status.idle":"2022-08-03T12:54:51.168252Z","shell.execute_reply.started":"2022-08-03T12:54:51.156264Z","shell.execute_reply":"2022-08-03T12:54:51.167084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import WeightedRandomSampler\n\nsample_weight = weight[y_train.detach().numpy().astype('int')]\nsampler = WeightedRandomSampler(weights=sample_weight, num_samples=len(sample_weight), replacement=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.172739Z","iopub.execute_input":"2022-08-03T12:54:51.173577Z","iopub.status.idle":"2022-08-03T12:54:51.183935Z","shell.execute_reply.started":"2022-08-03T12:54:51.173535Z","shell.execute_reply":"2022-08-03T12:54:51.182650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datasets = TensorDataset(x_train, y_train)\ntrain_dataloader = DataLoader(train_datasets, batch_size=100, sampler=sampler)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.189100Z","iopub.execute_input":"2022-08-03T12:54:51.189907Z","iopub.status.idle":"2022-08-03T12:54:51.202325Z","shell.execute_reply.started":"2022-08-03T12:54:51.189857Z","shell.execute_reply":"2022-08-03T12:54:51.200989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self):\n        super(Net, self).__init__()\n        \n        self.fc1 = nn.Linear(9, 100)\n        self.fc2 = nn.Linear(100, 50)\n        self.fc3 = nn.Linear(50, 1)\n        \n        self.bn1 = nn.BatchNorm1d(100)\n        self.bn2 = nn.BatchNorm1d(50)\n        \n    def forward(self, x):\n        x = F.relu(self.bn1(self.fc1(x)))\n        x = F.relu(self.bn2(self.fc2(x)))\n        x = F.sigmoid(self.fc3(x))\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.203699Z","iopub.execute_input":"2022-08-03T12:54:51.204828Z","iopub.status.idle":"2022-08-03T12:54:51.218749Z","shell.execute_reply.started":"2022-08-03T12:54:51.204777Z","shell.execute_reply":"2022-08-03T12:54:51.217183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Net()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.221161Z","iopub.execute_input":"2022-08-03T12:54:51.222079Z","iopub.status.idle":"2022-08-03T12:54:51.234086Z","shell.execute_reply.started":"2022-08-03T12:54:51.222030Z","shell.execute_reply":"2022-08-03T12:54:51.232339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = Adam(params=model.parameters(), lr=0.01, weight_decay=0.0001)\nloss_func = BCELoss()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.236227Z","iopub.execute_input":"2022-08-03T12:54:51.236992Z","iopub.status.idle":"2022-08-03T12:54:51.246701Z","shell.execute_reply.started":"2022-08-03T12:54:51.236954Z","shell.execute_reply":"2022-08-03T12:54:51.245245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def weight_init(m):\n    if type(m) == nn.Linear:\n        torch.nn.init.xavier_uniform_(m.weight)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.248267Z","iopub.execute_input":"2022-08-03T12:54:51.248805Z","iopub.status.idle":"2022-08-03T12:54:51.261296Z","shell.execute_reply.started":"2022-08-03T12:54:51.248771Z","shell.execute_reply":"2022-08-03T12:54:51.260113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.apply(weight_init)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.263184Z","iopub.execute_input":"2022-08-03T12:54:51.263594Z","iopub.status.idle":"2022-08-03T12:54:51.276152Z","shell.execute_reply.started":"2022-08-03T12:54:51.263558Z","shell.execute_reply":"2022-08-03T12:54:51.275139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(model, train_data, epochs):\n    for i in range(epochs):\n        loss_sum = 0\n        for x, label in train_dataloader:\n            loss = loss_func(model(x), label.view(-1,1))\n            optimizer.zero_grad()\n            loss.backward()\n            optimizer.step()\n\n            loss_sum += loss.item()\n\n        if i % 10 == 0:\n            print('loss :', loss_sum / len(train_dataloader))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.277793Z","iopub.execute_input":"2022-08-03T12:54:51.278450Z","iopub.status.idle":"2022-08-03T12:54:51.285890Z","shell.execute_reply.started":"2022-08-03T12:54:51.278415Z","shell.execute_reply":"2022-08-03T12:54:51.284879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train(model, train_dataloader, 1000)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:54:51.287717Z","iopub.execute_input":"2022-08-03T12:54:51.288344Z","iopub.status.idle":"2022-08-03T12:56:30.174566Z","shell.execute_reply.started":"2022-08-03T12:54:51.288309Z","shell.execute_reply":"2022-08-03T12:56:30.173470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train_pred = model(x_train).detach().numpy()\ny_train = y_train.detach().numpy()\n\ny_val_pred = model(x_val).detach().numpy()\ny_val = y_val.detach().numpy()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:56:30.176083Z","iopub.execute_input":"2022-08-03T12:56:30.176747Z","iopub.status.idle":"2022-08-03T12:56:30.614008Z","shell.execute_reply.started":"2022-08-03T12:56:30.176709Z","shell.execute_reply":"2022-08-03T12:56:30.613058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\ntrain_score = roc_auc_score(y_train, y_train_pred)\nval_score = roc_auc_score(y_val, y_val_pred)\n\nprint('train_score:', train_score)\nprint('val_score', val_score)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:56:30.615576Z","iopub.execute_input":"2022-08-03T12:56:30.616218Z","iopub.status.idle":"2022-08-03T12:56:30.635975Z","shell.execute_reply.started":"2022-08-03T12:56:30.616174Z","shell.execute_reply":"2022-08-03T12:56:30.635113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_data = torch.tensor(x_data, dtype=torch.float32)\ny_data = torch.tensor(y_data.values, dtype=torch.float32)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:56:30.637529Z","iopub.execute_input":"2022-08-03T12:56:30.638189Z","iopub.status.idle":"2022-08-03T12:56:30.643739Z","shell.execute_reply.started":"2022-08-03T12:56:30.638152Z","shell.execute_reply":"2022-08-03T12:56:30.642701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_weight = weight[y_data.detach().numpy().astype('int')]\nsampler = WeightedRandomSampler(weights=sample_weight, num_samples=len(sample_weight), replacement=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:56:30.645391Z","iopub.execute_input":"2022-08-03T12:56:30.645864Z","iopub.status.idle":"2022-08-03T12:56:30.656854Z","shell.execute_reply.started":"2022-08-03T12:56:30.645827Z","shell.execute_reply":"2022-08-03T12:56:30.655704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datasets = TensorDataset(x_data, y_data)\ntrain_dataloader = DataLoader(train_datasets, batch_size=100, sampler=sampler)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:56:30.661406Z","iopub.execute_input":"2022-08-03T12:56:30.662244Z","iopub.status.idle":"2022-08-03T12:56:30.671158Z","shell.execute_reply.started":"2022-08-03T12:56:30.662195Z","shell.execute_reply":"2022-08-03T12:56:30.670089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Net()\noptimizer = Adam(params=model.parameters(), lr=0.01, weight_decay=0.00003)\nloss_func = BCELoss()\nmodel.apply(weight_init)\ntrain(model, train_dataloader, 100)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T12:56:30.673128Z","iopub.execute_input":"2022-08-03T12:56:30.673975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_data_pred = model(x_data).detach().numpy()\ny_data = y_data.detach().numpy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_score = roc_auc_score(y_data, y_data_pred)\n\nprint('train_score:', train_score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_val_pred = model(x_val).detach().numpy()\nroc_auc_score(y_val, y_val_pred)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = torch.tensor(x_test, dtype=torch.float32)\ny_test = model(x_test).detach().numpy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/tabular-playground-series-aug-2022/sample_submission.csv')\nsubmission.failure = y_test\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}