{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T02:24:30.942717Z","iopub.execute_input":"2022-07-28T02:24:30.943341Z","iopub.status.idle":"2022-07-28T02:24:30.988714Z","shell.execute_reply.started":"2022-07-28T02:24:30.943221Z","shell.execute_reply":"2022-07-28T02:24:30.987665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We will try to solve this problem using a MultiLayer Perceptron in this notebook.","metadata":{}},{"cell_type":"code","source":"import torch\nimport random\nimport numpy as np\n\ntorch.manual_seed(0)\nrandom.seed(0)\nnp.random.seed(0)\ntorch.use_deterministic_algorithms(True)\n\n# import libraries\nfrom matplotlib import pyplot as plt\nfrom torchvision import datasets\nimport torchvision.transforms as transforms\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nprint(\"All libraries are loaded\")","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:24:31.181578Z","iopub.execute_input":"2022-07-28T02:24:31.182277Z","iopub.status.idle":"2022-07-28T02:24:33.431163Z","shell.execute_reply.started":"2022-07-28T02:24:31.182233Z","shell.execute_reply":"2022-07-28T02:24:33.429750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if torch.cuda.is_available():       \n    device = torch.device(\"cuda\")\n    print(f'There are {torch.cuda.device_count()} GPU(s) available.')\n    print('Device name:', torch.cuda.get_device_name(0))\n\nelse:\n    print('No GPU available, using the CPU instead.')\n    device = torch.device(\"cpu\")","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:24:33.433594Z","iopub.execute_input":"2022-07-28T02:24:33.434879Z","iopub.status.idle":"2022-07-28T02:24:33.443380Z","shell.execute_reply.started":"2022-07-28T02:24:33.434828Z","shell.execute_reply":"2022-07-28T02:24:33.441952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We will group our data basis customer ID and use the lst month data for model building.\n\n### We will be dropping all categorical variable and using just the numerical variables to check the predictive power of just the numerical columns.\n\n### We will drop customer ID, S_2 and target columns. We will also drop columns with more than 20 % Null data.","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\ndf = pd.read_feather('../input/amexfeather/train_data.ftr')\n\n#df.info(verbose=True,show_counts=True)\n\ndf_train =  (df\n            .groupby('customer_ID')\n            .tail(1))\n\ncat_f = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_68'] \n\n\n#drop columns with less than 20 % data\nperc = 20.0 # Like N %\nmin_count =  int(((100-perc)/100)*df_train.shape[0] + 1)\n\ndf_train = df_train.dropna( axis=1,thresh=min_count)\n\nall_f = list(df_train.columns)\n\nall_f.remove(\"customer_ID\")\nall_f.remove(\"S_2\")\nall_f.remove(\"target\")\n\nnum_f = list(set(all_f) - set(cat_f))\n\ndfnum = df_train[num_f]\ndfcat = df_train[cat_f]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:24:33.444612Z","iopub.execute_input":"2022-07-28T02:24:33.444959Z","iopub.status.idle":"2022-07-28T02:24:57.617823Z","shell.execute_reply.started":"2022-07-28T02:24:33.444929Z","shell.execute_reply":"2022-07-28T02:24:57.616406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We have imputed the missing data in the dataset with mode of the columns after dropping the Null data.\n\n### Below is our y variable which needs to be predicted.\n\n### EDA for categorical variables","metadata":{}},{"cell_type":"code","source":"import seaborn as sn\nfor col in cat_f:\n    \n    plt.figure(figsize=(7,5))\n\n    ax = sn.countplot(x=col, data=dfcat)\n    for p in ax.patches:#displaying % as annotations\n        ax.annotate(str(round(100*p.get_height()/len(dfcat),2))+\"%\", (p.get_x()+0.2, p.get_height()+2))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:24:57.621154Z","iopub.execute_input":"2022-07-28T02:24:57.621572Z","iopub.status.idle":"2022-07-28T02:25:00.308813Z","shell.execute_reply.started":"2022-07-28T02:24:57.621530Z","shell.execute_reply":"2022-07-28T02:25:00.307789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We observe some \"\" in the column D_64 , since we are not sure about what to impute , we will fill unknown in place of \"\"","metadata":{}},{"cell_type":"code","source":"dfcat['D_64'] = dfcat['D_64'].replace([\"\"],\"unknown\")","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:00.310148Z","iopub.execute_input":"2022-07-28T02:25:00.310464Z","iopub.status.idle":"2022-07-28T02:25:00.318044Z","shell.execute_reply.started":"2022-07-28T02:25:00.310435Z","shell.execute_reply":"2022-07-28T02:25:00.316594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create two DataFrames, one for each data type\n\ntraindata_num = dfnum\ntraindata_cat = dfcat","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:00.319917Z","iopub.execute_input":"2022-07-28T02:25:00.320687Z","iopub.status.idle":"2022-07-28T02:25:00.330883Z","shell.execute_reply.started":"2022-07-28T02:25:00.320639Z","shell.execute_reply":"2022-07-28T02:25:00.330132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(traindata_num.isnull().sum())\nprint(traindata_cat.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:00.331918Z","iopub.execute_input":"2022-07-28T02:25:00.332645Z","iopub.status.idle":"2022-07-28T02:25:00.681534Z","shell.execute_reply.started":"2022-07-28T02:25:00.332614Z","shell.execute_reply":"2022-07-28T02:25:00.680158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in cat_f:\n    traindata_cat[col] = traindata_cat[col].fillna(traindata_cat[col].value_counts().idxmax())\nprint(traindata_cat.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:00.683483Z","iopub.execute_input":"2022-07-28T02:25:00.684315Z","iopub.status.idle":"2022-07-28T02:25:00.748703Z","shell.execute_reply.started":"2022-07-28T02:25:00.684265Z","shell.execute_reply":"2022-07-28T02:25:00.747472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = dfnum.dropna()\n\nfor col in num_f:\n    traindata_num[col] = traindata_num[col].fillna(test[col].mean())\n    traindata_num[col] = traindata_num[col].fillna(test[col].median())\nprint(traindata_num.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:00.750362Z","iopub.execute_input":"2022-07-28T02:25:00.750823Z","iopub.status.idle":"2022-07-28T02:25:04.638383Z","shell.execute_reply.started":"2022-07-28T02:25:00.750770Z","shell.execute_reply":"2022-07-28T02:25:04.637052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputed_train = pd.concat([traindata_cat, traindata_num], axis=1, join='inner')\nimputed_train","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:04.642605Z","iopub.execute_input":"2022-07-28T02:25:04.643067Z","iopub.status.idle":"2022-07-28T02:25:04.810169Z","shell.execute_reply.started":"2022-07-28T02:25:04.643026Z","shell.execute_reply":"2022-07-28T02:25:04.808905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputed_train['target'] = df_train.target","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:04.811598Z","iopub.execute_input":"2022-07-28T02:25:04.811935Z","iopub.status.idle":"2022-07-28T02:25:04.819823Z","shell.execute_reply.started":"2022-07-28T02:25:04.811907Z","shell.execute_reply":"2022-07-28T02:25:04.818523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Using SMOTE to balance the dataset","metadata":{}},{"cell_type":"code","source":"## Importing resample from *sklearn.utils* package.\nfrom sklearn.utils import resample\n\n# Separate the case of yes-subscribes and no-subscribes\ndf_0 = imputed_train[imputed_train.target == 0]\ndf_1 = imputed_train[imputed_train.target == 1]\n\n##Upsample the yes-subscribed cases.\ndf_minority_upsampled = resample(df_1, \n                                 replace=True,     # sample with replacement\n                                 n_samples=330000) \n\n# Combine majority class with upsampled minority class\nnew_df = pd.concat([df_0, df_minority_upsampled])\n\n\nfrom sklearn.utils import shuffle\nnew_df = shuffle(new_df)\nnew_df\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:04.821936Z","iopub.execute_input":"2022-07-28T02:25:04.822889Z","iopub.status.idle":"2022-07-28T02:25:07.938177Z","shell.execute_reply.started":"2022-07-28T02:25:04.822838Z","shell.execute_reply":"2022-07-28T02:25:07.937216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualising the distribution of target in the new dataset","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(5,5))\n\nax = sn.countplot(x=\"target\", data=new_df)\nfor p in ax.patches:#displaying % as annotations\n        ax.annotate(str(round(100*p.get_height()/len(new_df),2))+\"%\", (p.get_x()+0.3, p.get_height()+2))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:07.939882Z","iopub.execute_input":"2022-07-28T02:25:07.940366Z","iopub.status.idle":"2022-07-28T02:25:08.222462Z","shell.execute_reply.started":"2022-07-28T02:25:07.940327Z","shell.execute_reply":"2022-07-28T02:25:08.221300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_df = pd.get_dummies( new_df, \n                                        columns = cat_f,\n                                        drop_first = True )\n\nencoded_df","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:08.224391Z","iopub.execute_input":"2022-07-28T02:25:08.224777Z","iopub.status.idle":"2022-07-28T02:25:09.095615Z","shell.execute_reply.started":"2022-07-28T02:25:08.224745Z","shell.execute_reply":"2022-07-28T02:25:09.094231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### For some weird reason D_64 shows -1 and D_68 shows 1.0 as columns, we need to drop it.","metadata":{}},{"cell_type":"code","source":"encoded_df = encoded_df.drop(['D_64_-1', 'D_68_1.0','target'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:09.097248Z","iopub.execute_input":"2022-07-28T02:25:09.097749Z","iopub.status.idle":"2022-07-28T02:25:09.625067Z","shell.execute_reply.started":"2022-07-28T02:25:09.097712Z","shell.execute_reply":"2022-07-28T02:25:09.623693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = new_df.target.values\n\nX = encoded_df.values","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:09.626720Z","iopub.execute_input":"2022-07-28T02:25:09.627251Z","iopub.status.idle":"2022-07-28T02:25:09.959516Z","shell.execute_reply.started":"2022-07-28T02:25:09.627200Z","shell.execute_reply":"2022-07-28T02:25:09.957945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X,y,test_size=0.05)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:09.962218Z","iopub.execute_input":"2022-07-28T02:25:09.962656Z","iopub.status.idle":"2022-07-28T02:25:12.155262Z","shell.execute_reply.started":"2022-07-28T02:25:09.962621Z","shell.execute_reply":"2022-07-28T02:25:12.154046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:12.157219Z","iopub.execute_input":"2022-07-28T02:25:12.157729Z","iopub.status.idle":"2022-07-28T02:25:12.165025Z","shell.execute_reply.started":"2022-07-28T02:25:12.157677Z","shell.execute_reply":"2022-07-28T02:25:12.164139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = torch.FloatTensor(X_train)\nX_test = torch.FloatTensor(X_test)\ny_train = torch.LongTensor(y_train)\ny_test = torch.LongTensor(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:12.166468Z","iopub.execute_input":"2022-07-28T02:25:12.166857Z","iopub.status.idle":"2022-07-28T02:25:12.384822Z","shell.execute_reply.started":"2022-07-28T02:25:12.166793Z","shell.execute_reply":"2022-07-28T02:25:12.383808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:12.386214Z","iopub.execute_input":"2022-07-28T02:25:12.386762Z","iopub.status.idle":"2022-07-28T02:25:12.398906Z","shell.execute_reply.started":"2022-07-28T02:25:12.386729Z","shell.execute_reply":"2022-07-28T02:25:12.397605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We will create a custom 2 layer NN using PyTorch.","metadata":{}},{"cell_type":"code","source":"\nclass AmexNet(nn.Module):\n    \n    def __init__(self, input_features=176,hidden_layer1=50,hidden_layer2=3,output_features=2):\n        super().__init__()\n        self.fc1 = nn.Linear(input_features,hidden_layer1)\n        self.fc2 = nn.Linear(hidden_layer1,hidden_layer2)\n        self.out = nn.Linear(hidden_layer2,output_features)    \n        #self.dropout = nn.Dropout(0.5)\n        \n    def forward(self, x):\n        \n        x = F.relu(self.fc1(x))\n        #x = self.dropout(x)\n        x = F.relu(self.fc2(x))\n        x = self.out(x)\n        return x\n       \n\nmodel = AmexNet()\n    \nmodel\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:12.400930Z","iopub.execute_input":"2022-07-28T02:25:12.402142Z","iopub.status.idle":"2022-07-28T02:25:12.420858Z","shell.execute_reply.started":"2022-07-28T02:25:12.402083Z","shell.execute_reply":"2022-07-28T02:25:12.419834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.SGD(model.parameters(), lr=0.4, weight_decay=1e-5)\nprint(criterion, optimizer)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:12.422448Z","iopub.execute_input":"2022-07-28T02:25:12.423067Z","iopub.status.idle":"2022-07-28T02:25:12.432927Z","shell.execute_reply.started":"2022-07-28T02:25:12.423020Z","shell.execute_reply":"2022-07-28T02:25:12.431929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 300\nlosses = []\n\nfor i in range(epochs):\n    y_pred = model.forward(X_train)\n    loss = criterion(y_pred, y_train)\n    losses.append(loss)\n    print(f'epoch: {i:2}  loss: {loss.item():10.8f}')\n    \n    optimizer.zero_grad()\n    loss.backward()\n    optimizer.step()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:25:12.434898Z","iopub.execute_input":"2022-07-28T02:25:12.435651Z","iopub.status.idle":"2022-07-28T02:28:32.517688Z","shell.execute_reply.started":"2022-07-28T02:25:12.435611Z","shell.execute_reply":"2022-07-28T02:28:32.516469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ndel df\ndel df_train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:32.519625Z","iopub.execute_input":"2022-07-28T02:28:32.520697Z","iopub.status.idle":"2022-07-28T02:28:32.783224Z","shell.execute_reply.started":"2022-07-28T02:28:32.520643Z","shell.execute_reply":"2022-07-28T02:28:32.781851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X\ndel y\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:32.785365Z","iopub.execute_input":"2022-07-28T02:28:32.786833Z","iopub.status.idle":"2022-07-28T02:28:32.974895Z","shell.execute_reply.started":"2022-07-28T02:28:32.786781Z","shell.execute_reply":"2022-07-28T02:28:32.973558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nwith torch.no_grad():\n    for val in X_test:\n        y_hat = model.forward(val)\n        preds.append(y_hat.argmax().item())","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:32.976922Z","iopub.execute_input":"2022-07-28T02:28:32.977334Z","iopub.status.idle":"2022-07-28T02:28:35.418677Z","shell.execute_reply.started":"2022-07-28T02:28:32.977302Z","shell.execute_reply":"2022-07-28T02:28:35.417391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:35.420778Z","iopub.execute_input":"2022-07-28T02:28:35.421439Z","iopub.status.idle":"2022-07-28T02:28:35.429012Z","shell.execute_reply.started":"2022-07-28T02:28:35.421387Z","shell.execute_reply":"2022-07-28T02:28:35.428126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train\ndel test\ndel dfnum\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:35.436105Z","iopub.execute_input":"2022-07-28T02:28:35.437031Z","iopub.status.idle":"2022-07-28T02:28:35.625457Z","shell.execute_reply.started":"2022-07-28T02:28:35.436977Z","shell.execute_reply":"2022-07-28T02:28:35.624343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_test\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:35.627082Z","iopub.execute_input":"2022-07-28T02:28:35.627402Z","iopub.status.idle":"2022-07-28T02:28:35.813667Z","shell.execute_reply.started":"2022-07-28T02:28:35.627374Z","shell.execute_reply":"2022-07-28T02:28:35.812413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame({'Y': y_test, 'YHat': preds})\ndf['Correct'] = [1 if corr == pred else 0 for corr, pred in zip(df['Y'], df['YHat'])]\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:35.816300Z","iopub.execute_input":"2022-07-28T02:28:35.817555Z","iopub.status.idle":"2022-07-28T02:28:35.876163Z","shell.execute_reply.started":"2022-07-28T02:28:35.817505Z","shell.execute_reply":"2022-07-28T02:28:35.875179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Accuracy on the validation set","metadata":{}},{"cell_type":"code","source":"df['Correct'].sum() / len(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:35.877666Z","iopub.execute_input":"2022-07-28T02:28:35.877978Z","iopub.status.idle":"2022-07-28T02:28:35.885543Z","shell.execute_reply.started":"2022-07-28T02:28:35.877949Z","shell.execute_reply":"2022-07-28T02:28:35.884419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_feather('/kaggle/input/amexfeather/test_data.ftr')\n\ndf_test =  (test_df\n            .groupby('customer_ID')\n            .tail(1))\nall_f.append(\"customer_ID\")\n\ndfn_test = df_test[all_f]\n\ndel test_df,df_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:28:35.887168Z","iopub.execute_input":"2022-07-28T02:28:35.887966Z","iopub.status.idle":"2022-07-28T02:29:24.917423Z","shell.execute_reply.started":"2022-07-28T02:28:35.887929Z","shell.execute_reply":"2022-07-28T02:29:24.916173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfn_test[cat_f]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:24.919131Z","iopub.execute_input":"2022-07-28T02:29:24.919502Z","iopub.status.idle":"2022-07-28T02:29:25.131585Z","shell.execute_reply.started":"2022-07-28T02:29:24.919470Z","shell.execute_reply":"2022-07-28T02:29:25.130429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cid = dfn_test.customer_ID\nall_f.remove(\"customer_ID\")\n\ndfn_test = dfn_test[all_f]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:25.133707Z","iopub.execute_input":"2022-07-28T02:29:25.134646Z","iopub.status.idle":"2022-07-28T02:29:25.766206Z","shell.execute_reply.started":"2022-07-28T02:29:25.134594Z","shell.execute_reply":"2022-07-28T02:29:25.765084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create two DataFrames, one for each data type\ndata_numeric = dfn_test[num_f]\ndata_categorical = pd.DataFrame(dfn_test[cat_f])","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:25.767630Z","iopub.execute_input":"2022-07-28T02:29:25.767967Z","iopub.status.idle":"2022-07-28T02:29:26.401885Z","shell.execute_reply.started":"2022-07-28T02:29:25.767937Z","shell.execute_reply":"2022-07-28T02:29:26.400707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = dfn_test.dropna()\n\nfor col in num_f:\n    data_numeric[col] = data_numeric[col].fillna(test[col].median())\ndel test\ngc.collect()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:26.403582Z","iopub.execute_input":"2022-07-28T02:29:26.403933Z","iopub.status.idle":"2022-07-28T02:29:31.911005Z","shell.execute_reply.started":"2022-07-28T02:29:26.403903Z","shell.execute_reply":"2022-07-28T02:29:31.909708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in cat_f:\n    data_categorical[col] = data_categorical[col].fillna(data_categorical[col].value_counts().idxmax())\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:31.912534Z","iopub.execute_input":"2022-07-28T02:29:31.913070Z","iopub.status.idle":"2022-07-28T02:29:31.999877Z","shell.execute_reply.started":"2022-07-28T02:29:31.913036Z","shell.execute_reply":"2022-07-28T02:29:31.998755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data_categorical.isnull().sum())\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:32.001761Z","iopub.execute_input":"2022-07-28T02:29:32.002147Z","iopub.status.idle":"2022-07-28T02:29:32.024537Z","shell.execute_reply.started":"2022-07-28T02:29:32.002111Z","shell.execute_reply":"2022-07-28T02:29:32.023433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = pd.concat([data_categorical, data_numeric], axis=1, join='inner')\nresult","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:32.025927Z","iopub.execute_input":"2022-07-28T02:29:32.026823Z","iopub.status.idle":"2022-07-28T02:29:32.297100Z","shell.execute_reply.started":"2022-07-28T02:29:32.026786Z","shell.execute_reply":"2022-07-28T02:29:32.295874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_dft = pd.get_dummies( result, \n                                    columns = cat_f,\n                                    drop_first = True )\n\nencoded_dft","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:32.298697Z","iopub.execute_input":"2022-07-28T02:29:32.299062Z","iopub.status.idle":"2022-07-28T02:29:33.472507Z","shell.execute_reply.started":"2022-07-28T02:29:32.299022Z","shell.execute_reply":"2022-07-28T02:29:33.471273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_pred = torch.FloatTensor(encoded_dft.values)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:33.474658Z","iopub.execute_input":"2022-07-28T02:29:33.475133Z","iopub.status.idle":"2022-07-28T02:29:34.243260Z","shell.execute_reply.started":"2022-07-28T02:29:33.475088Z","shell.execute_reply":"2022-07-28T02:29:34.241824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nwith torch.no_grad():\n    for val in X_pred:\n        y_pred = model.forward(val)\n        preds.append(y_pred.argmax().item())\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:29:34.245013Z","iopub.execute_input":"2022-07-28T02:29:34.245515Z","iopub.status.idle":"2022-07-28T02:30:49.248623Z","shell.execute_reply.started":"2022-07-28T02:29:34.245467Z","shell.execute_reply":"2022-07-28T02:30:49.247184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:30:49.250786Z","iopub.execute_input":"2022-07-28T02:30:49.251168Z","iopub.status.idle":"2022-07-28T02:30:49.258974Z","shell.execute_reply.started":"2022-07-28T02:30:49.251135Z","shell.execute_reply":"2022-07-28T02:30:49.257813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(cid)\nsubmission['prediction'] = preds","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:30:49.260821Z","iopub.execute_input":"2022-07-28T02:30:49.261203Z","iopub.status.idle":"2022-07-28T02:30:49.692595Z","shell.execute_reply.started":"2022-07-28T02:30:49.261169Z","shell.execute_reply":"2022-07-28T02:30:49.691263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:30:49.695186Z","iopub.execute_input":"2022-07-28T02:30:49.696372Z","iopub.status.idle":"2022-07-28T02:30:49.712751Z","shell.execute_reply.started":"2022-07-28T02:30:49.696318Z","shell.execute_reply":"2022-07-28T02:30:49.711313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.prediction.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:30:49.714326Z","iopub.execute_input":"2022-07-28T02:30:49.714701Z","iopub.status.idle":"2022-07-28T02:30:49.734105Z","shell.execute_reply.started":"2022-07-28T02:30:49.714669Z","shell.execute_reply":"2022-07-28T02:30:49.732895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T02:30:49.735556Z","iopub.execute_input":"2022-07-28T02:30:49.736148Z","iopub.status.idle":"2022-07-28T02:30:53.451775Z","shell.execute_reply.started":"2022-07-28T02:30:49.736115Z","shell.execute_reply":"2022-07-28T02:30:53.450416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}