{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"##### Pytorch using only 5 column - submission","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-28T08:57:03.193294Z","iopub.execute_input":"2023-02-28T08:57:03.194073Z","iopub.status.idle":"2023-02-28T08:57:03.225552Z","shell.execute_reply.started":"2023-02-28T08:57:03.194019Z","shell.execute_reply":"2023-02-28T08:57:03.22457Z"}}},{"cell_type":"markdown","source":"This is the POC of [this notebook](https://www.kaggle.com/code/daeseung/pytorch-using-5-column).\n\noriginal notebook is [this notebok](https://www.kaggle.com/code/chrisqiu/pytorch-using-only-1-column-submission)\nI couldn't edit well, so submission is failed...","metadata":{}},{"cell_type":"code","source":"THRESHOLD = 0.50\nWPATH = '/kaggle/input/d/daeseung/jowilder-1col-weights'\nFUSE = True","metadata":{"execution":{"iopub.status.busy":"2023-04-18T14:24:40.815590Z","iopub.execute_input":"2023-04-18T14:24:40.816159Z","iopub.status.idle":"2023-04-18T14:24:40.827067Z","shell.execute_reply.started":"2023-04-18T14:24:40.816090Z","shell.execute_reply":"2023-04-18T14:24:40.825806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch \n\ntorch.manual_seed(101)\nnp.random.seed(101)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T14:24:40.828334Z","iopub.execute_input":"2023-04-18T14:24:40.828781Z","iopub.status.idle":"2023-04-18T14:24:41.768543Z","shell.execute_reply.started":"2023-04-18T14:24:40.828728Z","shell.execute_reply":"2023-04-18T14:24:41.767494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport copy\n\n# trim input data\n\ndef trim(X, trim_steps):\n    data_steps = X.shape[0]\n    if data_steps == trim_steps: return X\n    \n    if data_steps < trim_steps:\n        shortage = trim_steps - data_steps\n        return np.pad(X, ((0,shortage),(0,0)),'constant') \n    \n    start = int(np.random.random() * (data_steps - trim_steps))    \n    return X[start:start+trim_steps]  ","metadata":{"execution":{"iopub.status.busy":"2023-04-18T14:24:41.770181Z","iopub.execute_input":"2023-04-18T14:24:41.771046Z","iopub.status.idle":"2023-04-18T14:24:41.779034Z","shell.execute_reply.started":"2023-04-18T14:24:41.771004Z","shell.execute_reply":"2023-04-18T14:24:41.777859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n\nclass Encoder(nn.Module):\n    def __init__(self, nembed, nfeatures = 128):\n        \n        super(Encoder, self).__init__()\n        \n        self.nfeatures = nfeatures\n        \n        self.convs = nn.Sequential(\n            nn.Conv1d(5, 64,  kernel_size = 5),\n            nn.BatchNorm1d(64),\n            nn.LeakyReLU(inplace = True),\n            nn.MaxPool1d(2),\n            \n            nn.Conv1d(64, 128,  kernel_size = 5),\n            nn.BatchNorm1d(128),\n            nn.LeakyReLU(inplace = True),\n            nn.MaxPool1d(2),\n            \n            nn.Conv1d(128, self.nfeatures,  kernel_size = 3),\n            nn.BatchNorm1d(self.nfeatures),\n            nn.LeakyReLU(inplace = True),\n        ) \n        \n\n        self.fcs = nn.Sequential(\n                # 2* because in the pooling layer we take mean AND the std.\n                nn.Linear(2*self.nfeatures, nembed),\n            )\n\n    \n    def forward(self,x):        \n        x = torch.transpose(x, 1,2)\n        x = self.convs(x)        \n        std = torch.std(x, dim = 2)\n        mean = torch.mean(x, dim = 2)\n        x = torch.cat([std, mean], dim = 1)\n        x = self.fcs(x)\n        return x\n    \nclass Model(nn.Module):\n    def __init__(self,nout, nembed, nfeatures):\n        super(Model, self).__init__()\n        self.encoder = Encoder(nembed, nfeatures)\n        self.clf = nn.Sequential(\n            \n            nn.LeakyReLU(inplace = True),\n            nn.Dropout(0.2),\n            \n            nn.Linear(nembed, nembed // 2),\n            nn.LeakyReLU(inplace = True),\n            nn.Dropout(0.2),\n            \n            nn.Linear(nembed // 2, nout),\n\n        ) \n        \n        self.nout = nout\n        self.nembed = nembed\n        self.nfeatures = nfeatures\n                \n    \n    def forward(self, x):\n        x = self.encoder(x)\n        x = self.clf(x)\n        return x\n    \n# assembler\nclass GroupModel(nn.Module):\n    def __init__(self, model, weights, fuse = False):\n        super(GroupModel, self).__init__()\n        \n        self.fuse = fuse\n        self.models = [copy.deepcopy(model) for _ in range(len(weights))]\n        self.weights = weights\n        \n        for i in range(len(self.weights)):\n            self.models[i].load_state_dict(torch.load(weights[i], map_location = 'cpu'))\n            self.models[i].eval()\n\n    def forward(self,x):\n        \n        if self.fuse:\n            with torch.no_grad():\n                preds = [model(x) for model in self.models]\n                \n            preds = torch.mean(torch.cat(preds, dim = 0), dim = 0, keepdims = True)\n        \n        else:\n            with torch.no_grad():\n                preds = torch.sigmoid(self.models[0](x))\n                \n        return preds","metadata":{"execution":{"iopub.status.busy":"2023-04-18T14:24:41.785438Z","iopub.execute_input":"2023-04-18T14:24:41.785813Z","iopub.status.idle":"2023-04-18T14:24:41.804123Z","shell.execute_reply.started":"2023-04-18T14:24:41.785768Z","shell.execute_reply":"2023-04-18T14:24:41.802996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FEATURES = ['event_duration', \"room_coor_x\", \"room_coor_y\", \"screen_coor_x\", \"screen_coor_y\"]","metadata":{"execution":{"iopub.status.busy":"2023-04-18T14:24:41.807951Z","iopub.execute_input":"2023-04-18T14:24:41.808261Z","iopub.status.idle":"2023-04-18T14:24:41.817048Z","shell.execute_reply.started":"2023-04-18T14:24:41.808229Z","shell.execute_reply":"2023-04-18T14:24:41.815935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\n\nweights = [w for w in glob.glob(f'{WPATH}/model-*') if 'encoder' not in w]\n\nGP1 = '0-4'\nGP2 = '5-12'\nGP3 = '13-22'\n\nmodel_params = {}\nmodel_params[GP1] = dict(\n    nout = 3,\n    nembed = 128,\n    nfeatures = 128,\n)\nmodel_params[GP2] = dict(\n    nout = 10,\n    nembed = 128,\n    nfeatures = 128,\n)\nmodel_params[GP3] = dict(\n    nout = 5,\n    nembed = 128,\n    nfeatures = 128,\n)\n\n\nmodels = {\n    GP: GroupModel(\n        model = Model(**params),\n        weights = [w for w in weights if GP in w],\n        fuse = FUSE,\n    ) for GP,params in model_params.items()\n}\n\nfor GP in model_params:    \n    model = models[GP]\n    print(f'group {GP}')\n    print(f'output shape: {model(torch.ones(5, 1000, len(FEATURES))).shape}')\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T14:24:41.818562Z","iopub.execute_input":"2023-04-18T14:24:41.819627Z","iopub.status.idle":"2023-04-18T14:24:42.462160Z","shell.execute_reply.started":"2023-04-18T14:24:41.819583Z","shell.execute_reply":"2023-04-18T14:24:42.460946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T14:24:42.463757Z","iopub.execute_input":"2023-04-18T14:24:42.464410Z","iopub.status.idle":"2023-04-18T14:24:42.473270Z","shell.execute_reply.started":"2023-04-18T14:24:42.464369Z","shell.execute_reply":"2023-04-18T14:24:42.472276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"questions = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\n\nfor (test, sample_submission) in iter_test:\n\n    # test info\n    level_group = test.level_group.unique()[0]    \n    session_id = test.session_id.unique()[0]\n    \n    # model\n    model = models[level_group]\n    \n    # data\n    event_duration = test.elapsed_time.fillna(-1).diff().fillna(0).clip(0, 3.6e6)\n    test['event_duration'] = event_duration\n    \n    \n    X = trim(test[FEATURES].values, 1000)\n\n    X = torch.tensor(X, dtype = torch.float32)\n\n    # predict\n    preds = model(X.unsqueeze(0)).detach().numpy()\n\n\n    # generate question label\n    mask = sample_submission.session_id.str.contains(f'{session_id}')\n    print(mask)\n    print(preds)\n    sample_submission.loc[mask, 'correct'] = preds.reshape(-1)\n    \n    \n    \n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T14:24:42.475075Z","iopub.execute_input":"2023-04-18T14:24:42.475458Z","iopub.status.idle":"2023-04-18T14:24:42.594691Z","shell.execute_reply.started":"2023-04-18T14:24:42.475420Z","shell.execute_reply":"2023-04-18T14:24:42.593485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res = pd.read_csv('/kaggle/working/submission.csv')\nres.correct.value_counts(dropna = False)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T14:24:42.596200Z","iopub.execute_input":"2023-04-18T14:24:42.596524Z","iopub.status.idle":"2023-04-18T14:24:42.610801Z","shell.execute_reply.started":"2023-04-18T14:24:42.596495Z","shell.execute_reply":"2023-04-18T14:24:42.609505Z"},"trusted":true},"execution_count":null,"outputs":[]}]}