{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":10052600,"sourceType":"datasetVersion","datasetId":6194018},{"sourceId":182947,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":155939,"modelId":178390}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Step 1.1\nTo start working with the House Prices dataset, you will need to import the required libraries, and read the data into a pandas DataFrame.\n\n- Import the following libraries using import statements.\n  - **numpy** (for multidimensional array computation) with the alias **np**\n  - **pandas** (for data manipulation) with the alias **pd**\n  - **matplotlib.pyplot** (for data visualization) with the alias **plt**\n\nNote: Run a code cell by clicking on the cell and using the keyboard shortcut &lt;Shift&gt; + &lt;Enter&gt;.","metadata":{}},{"cell_type":"code","source":"# Put your code here\nimport numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import PercentFormatter\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.nn.utils.rnn import pad_sequence\nimport torch.optim as optim\nfrom sklearn.impute import KNNImputer\nfrom sklearn.neighbors import NearestNeighbors\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T17:56:49.522909Z","iopub.execute_input":"2024-11-29T17:56:49.523245Z","iopub.status.idle":"2024-11-29T17:56:54.687204Z","shell.execute_reply.started":"2024-11-29T17:56:49.523215Z","shell.execute_reply":"2024-11-29T17:56:54.686137Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 1.2\n- Activate CoW by setting `pd.options.mode.copy_on_write` as `True`","metadata":{}},{"cell_type":"code","source":"# Put your code here\npd.options.mode.copy_on_write = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:01:18.568951Z","iopub.execute_input":"2024-11-29T18:01:18.569414Z","iopub.status.idle":"2024-11-29T18:01:18.575174Z","shell.execute_reply.started":"2024-11-29T18:01:18.569372Z","shell.execute_reply":"2024-11-29T18:01:18.573564Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 1.3\n- Read the csv file 'house-train.csv' using Pandas' **read_csv** function (<a href=\"https://pandas.pydata.org/pandas-docs/stable/generated/pandas.read_csv.html\">pandas.read_csv</a>) to create a DataFrame called **df** with default settings (i.e. the only argument is the file name).\n","metadata":{}},{"cell_type":"code","source":"def load_data(data_path, csv_file, mode='train'):\n    csv_path = os.path.join(data_path, csv_file)\n    df = pd.read_csv(csv_path)\n\n    df['Fitness_Endurance-Time'] = df['Fitness_Endurance-Time_Mins'] * 60 + df['Fitness_Endurance-Time_Sec']\n    df['PAQ_Total'] = df['PAQ_A-PAQ_A_Total'].fillna(df['PAQ_C-PAQ_C_Total'])\n\n    columns_to_drop = ['Basic_Demos-Enroll_Season', 'Physical-BMI',\n                       'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                       'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season',\n                       'Fitness_Endurance-Max_Stage', 'FGC-Season', 'FGC-FGC_CU_Zone',\n                       'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU_Zone',\n                       'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                       'BIA-BIA_Activity_Level_num', 'BIA-BIA_Frame_num', 'PAQ_A-Season',\n                       'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total',\n                       'SDS-Season', 'PreInt_EduHx-Season']\n    if mode == 'train':\n        row_to_drop = df[(~df['PAQ_A-PAQ_A_Total'].isna()) & (~df['PAQ_C-PAQ_C_Total'].isna())].index\n        df = df.drop(row_to_drop)\n        df = df.fillna({'sii': 4})\n        columns_to_drop += ['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03',\n                                'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07',\n                                'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11',\n                                'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n                                'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19',\n                                'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total']\n    df = df.drop(columns=columns_to_drop)\n\n    df['Basic_Demos-Sex'] = df['Basic_Demos-Sex'].astype('category')\n    df['PreInt_EduHx-computerinternet_hoursday'] = df['PreInt_EduHx-computerinternet_hoursday'].astype('category')\n    df['PreInt_EduHx-computerinternet_hoursday'] = df['PreInt_EduHx-computerinternet_hoursday'].cat.set_categories([0, 1, 2, 3], ordered=True)\n    df = df.reset_index(drop=True)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:01:20.168889Z","iopub.execute_input":"2024-11-29T18:01:20.169919Z","iopub.status.idle":"2024-11-29T18:01:20.180505Z","shell.execute_reply.started":"2024-11-29T18:01:20.169872Z","shell.execute_reply":"2024-11-29T18:01:20.178835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = load_data(data_path = '/kaggle/input/child-mind-institute-problematic-internet-use', csv_file='train.csv', mode='train')\ndf_test = load_data(data_path = '/kaggle/input/child-mind-institute-problematic-internet-use', csv_file='test.csv', mode='test')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:01:27.869315Z","iopub.execute_input":"2024-11-29T18:01:27.869817Z","iopub.status.idle":"2024-11-29T18:01:27.978154Z","shell.execute_reply.started":"2024-11-29T18:01:27.869745Z","shell.execute_reply":"2024-11-29T18:01:27.977067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def impute(df, mode='train'):\n    if mode == 'train':\n        df_num = df.drop(['id', 'sii'], axis=1)\n    else:\n        df_num = df.drop('id', axis=1)\n    knn_imputer = KNNImputer(n_neighbors=5)\n    df_num = knn_imputer.fit_transform(df_num)\n    return df_num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:01:42.555610Z","iopub.execute_input":"2024-11-29T18:01:42.556557Z","iopub.status.idle":"2024-11-29T18:01:42.562450Z","shell.execute_reply.started":"2024-11-29T18:01:42.556518Z","shell.execute_reply":"2024-11-29T18:01:42.561145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_num = impute(df, mode='train')\ndf[[col for col in df.columns if col not in ['id', 'sii']]] = df_num\n\ndf_test_num = impute(df_test, mode='test')\ndf_test[[col for col in df_test.columns if col != 'id']] = df_test_num","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:01:54.148949Z","iopub.execute_input":"2024-11-29T18:01:54.149367Z","iopub.status.idle":"2024-11-29T18:01:58.502176Z","shell.execute_reply.started":"2024-11-29T18:01:54.149335Z","shell.execute_reply":"2024-11-29T18:01:58.501131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def construct_feature_tensors(df, mode='train'):\n    df_num = df.drop(['id', 'Basic_Demos-Sex', 'PreInt_EduHx-computerinternet_hoursday'], axis=1)\n    if mode == 'train':\n        df_num = df_num.drop(['sii'], axis=1)\n    num_tensors = torch.tensor(df_num.to_numpy())\n\n    df_sex = df['Basic_Demos-Sex']\n    sex_tensors = torch.tensor(df_sex.to_numpy())\n    sex_tensors = sex_tensors.to(torch.int64)\n    sex_tensors = F.one_hot(sex_tensors)\n\n    df_pech = df['PreInt_EduHx-computerinternet_hoursday']\n    pech_tensors = torch.tensor(df_pech.to_numpy())\n    pech_tensors = pech_tensors.round().to(torch.int64)\n    pech_tensors = F.one_hot(pech_tensors)\n\n    feature_tensors = torch.cat([num_tensors, sex_tensors, pech_tensors], dim=1)\n\n    if mode == 'train':\n        feature_tensors_train, feature_tensors_valid = feature_tensors[:3166], feature_tensors[3166:]\n        feature_tensors_train_num = feature_tensors_train[:, :-6]\n        feature_tensors_train_cat = feature_tensors_train[:, -6:]\n        feature_tensors_valid_num = feature_tensors_valid[:, :-6]\n        feature_tensors_valid_cat = feature_tensors_valid[:, -6:]\n\n        mean_train = feature_tensors_train_num.mean(axis=0)\n        std_train = feature_tensors_train_num.std(axis=0)\n        feature_tensors_train_num = (feature_tensors_train_num - mean_train) / std_train\n        feature_tensors_train = torch.cat([feature_tensors_train_num, feature_tensors_train_cat], dim=1)\n\n        mean_valid = feature_tensors_valid_num.mean(axis=0)\n        std_valid = feature_tensors_valid_num.std(axis=0)\n        feature_tensors_valid_num = (feature_tensors_valid_num - mean_valid) / std_valid\n        feature_tensors_valid = torch.cat([feature_tensors_valid_num, feature_tensors_valid_cat], dim=1)\n\n        return feature_tensors_train, feature_tensors_valid\n    \n    else:\n        feature_tensors_num = feature_tensors[:, :-6]\n        feature_tensors_cat = feature_tensors[:, -6:]\n        mean = feature_tensors_num.mean(axis=0)\n        std = feature_tensors_num.std(axis=0)\n        feature_tensors_num = (feature_tensors_num - mean) / std\n        feature_tensors = torch.cat([feature_tensors_num, feature_tensors_cat], dim=1)\n    return feature_tensors","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:02:11.307020Z","iopub.execute_input":"2024-11-29T18:02:11.308013Z","iopub.status.idle":"2024-11-29T18:02:11.320172Z","shell.execute_reply.started":"2024-11-29T18:02:11.307973Z","shell.execute_reply":"2024-11-29T18:02:11.318920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_or_process(data_path, parquet_path, tensor_path, id_path, df, length_limit=16384, mode='train'): # 2 ** 14 = 16384\n    tensor_file = os.path.join(data_path, tensor_path)\n    id_file = os.path.join(data_path, id_path)\n    if os.path.exists(tensor_file) and os.path.exists(id_file):\n        enmo_tensors = torch.load(tensor_file)\n        enmo_tensors = enmo_tensors.to_dense()\n        df_ids = pd.read_csv(id_file)\n    else:\n        ids = []\n        enmo_tensors = []\n        for id in df['id']:\n            # print(id)\n            try:\n                parquet_file = os.path.join(data_path, parquet_path, f'id={id}', 'part-0.parquet')\n                parquet = pd.read_parquet(parquet_file)\n                enmo_tensor = torch.tensor(parquet.loc[parquet['non-wear_flag'] == 0, 'enmo'].iloc[:length_limit])\n                enmo_tensor = (enmo_tensor - enmo_tensor.mean()) / enmo_tensor.std()\n                enmo_tensors.append(enmo_tensor)\n                ids.append(id)\n            except:\n                enmo_tensors.append(torch.tensor([0.]))\n\n        enmo_tensors =  pad_sequence(enmo_tensors, batch_first=True)\n        neigh = NearestNeighbors(n_neighbors=5)\n        if mode == 'train':\n            df_num = df.drop(['id', 'sii'], axis=1).to_numpy()\n        else:\n            df_num = df.drop('id', axis=1).to_numpy()\n        neigh.fit(df_num)\n        negihbor_graph = neigh.kneighbors_graph(df_num)\n\n        for i, tensor in enumerate(enmo_tensors):\n            if tensor.sum() == 0:\n                neighbors = negihbor_graph[i].nonzero()[1]\n                neighbor_tensors = enmo_tensors[neighbors]\n                neighbor_tensors_sum = neighbor_tensors.sum(axis=0)\n                neighbor_non_zero = max(neighbor_tensors.count_nonzero(axis=0).max(), 1)\n                mean_tensor = neighbor_tensors_sum / neighbor_non_zero\n                enmo_tensors[i] = mean_tensor\n\n        # torch.save(enmo_tensors.to_sparse(), tensor_file)\n        df_ids = pd.DataFrame({'id': ids})\n        # df_ids.to_csv(id_file, index=False)\n    return df_ids, enmo_tensors","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:04:18.346337Z","iopub.execute_input":"2024-11-29T18:04:18.346880Z","iopub.status.idle":"2024-11-29T18:04:18.365791Z","shell.execute_reply.started":"2024-11-29T18:04:18.346842Z","shell.execute_reply":"2024-11-29T18:04:18.364251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_tensors_train, feature_tensors_valid = construct_feature_tensors(df, mode='train')\ndf_ids, enmo_tensors = load_or_process(data_path='/kaggle/input/child-mind-institute-problematic-internet-use',\n                                   parquet_path='series_train.parquet',\n                                   tensor_path='enmo_tensor_sparse.pt',\n                                   id_path='enmo_ids.csv',\n                                   df=df)\nenmo_tensors = enmo_tensors.reshape([enmo_tensors.size()[0], 1, -1])\nenmo_tensors_train, enmo_tensors_valid = enmo_tensors[:3166], enmo_tensors[3166:]\n\nlabel_percentage = df.loc[df['sii'] != 4, 'sii'].value_counts(normalize=True)\nlabel_percentage\n\nweights = 1 / label_percentage\nweights = weights / weights.sum()\nweights = df['sii'].map(weights)\nweights = weights.fillna(weights.mean()).to_numpy()\nweights = torch.tensor(weights).type(torch.float32)\nweights = weights.reshape([weights.size()[0], 1, -1])\nweights_train, weights_valid = weights[:3166], weights[3166:]\n\nweights_enmo_train = weights_train[enmo_tensors_train.sum(axis=[1,2]) != 0]\nweights_enmo_valid = weights_valid[enmo_tensors_valid.sum(axis=[1,2]) != 0]\nenmo_tensors_train = enmo_tensors_train[enmo_tensors_train.sum(axis=[1,2]) != 0]\nenmo_tensors_valid = enmo_tensors_valid[enmo_tensors_valid.sum(axis=[1,2]) != 0]\n\n\nfeature_tensors_test = construct_feature_tensors(df_test, mode='test')\ndf_test_ids, enmo_tensors_test = load_or_process(data_path='/kaggle/input/child-mind-institute-problematic-internet-use',\n                                   parquet_path='series_test.parquet',\n                                   id_path='enmo_test_ids.csv',\n                                   tensor_path='enmo_tensor_test_sparse.pt',\n                                   df=df_test,\n                                   mode='test')\nenmo_tensors_test = enmo_tensors_test.reshape([enmo_tensors_test.size()[0], 1, -1])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:14:07.594432Z","iopub.execute_input":"2024-11-29T18:14:07.595475Z","iopub.status.idle":"2024-11-29T18:15:21.081641Z","shell.execute_reply.started":"2024-11-29T18:14:07.595430Z","shell.execute_reply":"2024-11-29T18:15:21.080703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 4 Construct the Model","metadata":{}},{"cell_type":"markdown","source":"### Step 4.1 Conv1d for the time-series","metadata":{}},{"cell_type":"markdown","source":"#### Step 4.1.1 Dataloader for the time series","metadata":{}},{"cell_type":"code","source":"from torch.utils.data import TensorDataset, DataLoader, RandomSampler, SequentialSampler\n\n#define a batch size\nbatch_size = 32\n\n# wrap tensors\ntrain_data = TensorDataset(enmo_tensors_train, weights_enmo_train)\n# sampler for sampling the data during training\ntrain_sampler = RandomSampler(train_data)\n# dataLoader for train set\ntrain_dataloader = DataLoader(train_data, sampler=train_sampler, batch_size=batch_size)\n\n# wrap tensors\nval_data = TensorDataset(enmo_tensors_valid, weights_enmo_valid)\n# sampler for sampling the data during training\nval_sampler = SequentialSampler(val_data)\n# dataLoader for validation set\nval_dataloader = DataLoader(val_data, sampler = val_sampler, batch_size=batch_size)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:15:34.634690Z","iopub.execute_input":"2024-11-29T18:15:34.635149Z","iopub.status.idle":"2024-11-29T18:15:34.643214Z","shell.execute_reply.started":"2024-11-29T18:15:34.635114Z","shell.execute_reply":"2024-11-29T18:15:34.642094Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Step 4.1.2 Variational Auto-Encoder for the time series","metadata":{}},{"cell_type":"code","source":"class Conv1dEncoder(nn.Module):\n    def __init__(self, input_size=16384, out_channels=64, kernel_size=12, stride=6, hidden_size=64, latent_size=32, dropout=0.5):\n        super(Conv1dEncoder, self).__init__()\n\n        self.input_size = input_size\n        self.out_channels = out_channels\n        self.kernel_size = kernel_size\n        self.stride= stride\n        self.hidden_size = hidden_size\n        self.dropout = dropout\n\n        self.model = nn.Sequential(\n            nn.Conv1d(in_channels=1, out_channels=self.out_channels, kernel_size=self.kernel_size, stride=self.stride), # (N=batch_size, C_out=64, L_out=2729)\n            nn.ReLU(),\n            nn.Conv1d(in_channels=self.out_channels, out_channels=self.out_channels*2, kernel_size=self.kernel_size, stride=self.stride), # (N=batch_size, C_out=128, L_out=453)\n            nn.ReLU(),\n            nn.Conv1d(in_channels=self.out_channels*2, out_channels=self.out_channels*2, kernel_size=self.kernel_size, stride=self.stride), # (N=batch_size, C_out=128, L_out=74)\n            nn.ReLU(),\n            nn.Flatten(start_dim=1), # (N=batch_size, out_channels*2*L_out=128*74)\n            nn.Linear(self.out_channels * 2 * 74, self.hidden_size), # (N=batch_size, hidden_size=64)\n            nn.ReLU(),\n            # nn.Dropout(self.dropout),\n            # nn.Linear(self.hidden_size, self.hidden_size), # (N=batch_size, hidden_size=64)\n        )\n\n        self.mu = nn.Linear(hidden_size, latent_size) # (N=batch_size, latent_size=32)\n        self.var = nn.Linear(hidden_size, latent_size) # (N=batch_size, latent_size=32)\n\n    def forward(self, input): # (batch_size, 1, input_size=16384)\n        hidden = self.model(input) # (batch_size, hidden_size=64)\n        z_mu = self.mu(hidden) # (batch_size, latent_size=32)\n        z_var = self.var(hidden) # (batch_size, latent_size=32)\n\n        return z_mu, z_var # (batch_size, latent_size=32)\n    \nclass Conv1dBlock(nn.Module):\n    def __init__(self, scale_factor, in_channels, out_channels, kernel_size, num_features):\n        super(Conv1dBlock, self).__init__()\n        self.scale_factor = scale_factor\n        self.in_channels = in_channels\n        self.out_channels = out_channels\n        self.kernel_size = kernel_size\n        self.num_features = num_features\n\n        self.model = nn.Sequential(\n            nn.Upsample(scale_factor=self.scale_factor),\n            nn.Conv1d(in_channels=self.in_channels, out_channels=self.out_channels, kernel_size=self.kernel_size, padding='same'),\n            nn.BatchNorm1d(num_features=self.num_features),\n            nn.LeakyReLU()\n        )\n\n    def forward(self, input):\n        output = self.model(input)\n        return output\n    \nclass Conv1dDecoder(nn.Module):\n    def __init__(self, latent_size=32, hidden_size=128, output_size=16384):\n        super(Conv1dDecoder, self).__init__()\n\n        self.latent_size = latent_size\n        self.hidden_size = hidden_size\n        self.output_size = output_size\n        self.linear_1 = nn.Linear(self.latent_size, self.hidden_size)\n\n        # inner_layers = nn.ModuleList()\n        # kernel_size_list = [3, 6, 12, 12, 12, 12]\n        # for layer, kernel_size in enumerate(kernel_size_list):\n        #     if layer == 0:\n        #         in_channels = 1\n        #     else:\n        #         in_channels = 64\n        #     inner_layers.append(Conv1dBlock(scale_factor=2, in_channels=in_channels, out_channels=64, kernel_size=kernel_size, num_features=64))\n\n        # self.inner_layers = nn.Sequential(\n        #     *inner_layers,\n        #     nn.Upsample(scale_factor=2)\n        # )\n\n        # self.conv1d_1 = nn.Conv1d(in_channels=64, out_channels=64, kernel_size=15, padding='same')\n        # self.conv1d_2 = nn.Conv1d(in_channels=64, out_channels=64, kernel_size=7, padding='same')\n        # self.conv1d_3 = nn.Conv1d(in_channels=64, out_channels=64, kernel_size=4, padding='same')\n        # self.batchnorm1d = nn.BatchNorm1d(num_features=64)\n        # self.conv1d_4 = nn.Conv1d(in_channels=64, out_channels=1, kernel_size=1, padding='same')\n        self.conv1d = nn.Conv1d(in_channels=1, out_channels=1, kernel_size=12, padding='same')\n        self.linear_2 = nn.Linear(self.hidden_size, self.output_size)\n\n    def forward(self, input): # (batch_size, latent_size=32)\n        # output_1 = self.linear_1(input) # (batch_size, 128)\n        # output_1 = output_1.unsqueeze(1) # (batch_size, 1, 128)\n\n        # latent = self.inner_layers(output_1)\n        # branch_1 = self.conv1d_1(latent)\n        # branch_2 = self.conv1d_2(latent)\n        # branch_3 = self.conv1d_3(latent)\n        # output_2 = branch_1 + branch_2 + branch_3\n        # output_2 = self.batchnorm1d(output_2)\n        # output_2 = nn.LeakyReLU()(output_2)\n        # output_2 = self.conv1d_4(output_2)\n        # output_2 = nn.Tanh()(output_2) # (batch_size, 1, 16384)\n\n        # output_1 = self.linear_2(output_1) # (batch_size, 1, 16384)\n        # output_1 = nn.Tanh()(output_1) # (batch_size, 1, 16384)\n\n        # output = output_1 * output_2 # (batch_size, 1, 16384)\n        # return output # (batch_size, 1, 16384)\n\n        output_1 = self.linear_1(input) # (batch_size, 128)\n        output_1 = output_1.unsqueeze(1) # (batch_size, 1, 128)\n        output_1 = self.linear_2(output_1) # (batch_size, 1, 16384)\n\n        output_2 = self.conv1d(output_1) # (batch_size, 1, 16384)\n        output_2 = nn.LeakyReLU()(output_2)\n        output_2 = nn.Tanh()(output_2)\n\n        output_1 = nn.Tanh()(output_1) # (batch_size, 1, 16384)\n\n        output = output_1 * output_2 # (batch_size, 1, 16384)\n        return output # (batch_size, 1, 16384)\n    \nclass Conv1dVAE(nn.Module):\n    def __init__(self, enc, dec):\n        ''' This the VAE, which takes a encoder and decoder.\n        '''\n        super(Conv1dVAE, self).__init__()\n\n        self.enc = enc\n        self.dec = dec\n\n    def forward(self, x):\n        # encode\n        z_mu, z_var = self.enc(x)\n\n        # sample from the distribution having latent parameters z_mu, z_var\n        # reparameterize\n        std = torch.exp(z_var / 2)\n        eps = torch.randn_like(std)\n        x_sample = eps.mul(std).add_(z_mu)\n\n        # decode\n        predicted = self.dec(x_sample)\n        return predicted, z_mu, z_var","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:15:36.866651Z","iopub.execute_input":"2024-11-29T18:15:36.867102Z","iopub.status.idle":"2024-11-29T18:15:36.887115Z","shell.execute_reply.started":"2024-11-29T18:15:36.867066Z","shell.execute_reply":"2024-11-29T18:15:36.885683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_epochs = 100\ninput_size = 16384\nhidden_size_encoder = 64\nhidden_size_decoder = 128\nlatent_size = 16\nlr = 1e-3\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n# encoder\nencoder = Conv1dEncoder(input_size=input_size, hidden_size=hidden_size_encoder, latent_size=latent_size)\n\n# decoder\ndecoder = Conv1dDecoder(latent_size=latent_size, hidden_size=hidden_size_decoder, output_size=input_size)\n\n# vae\nmodel = Conv1dVAE(encoder, decoder).to(device)\n\n# optimizer\noptimizer = optim.Adam(model.parameters(), lr=lr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:15:37.975349Z","iopub.execute_input":"2024-11-29T18:15:37.975816Z","iopub.status.idle":"2024-11-29T18:15:39.385316Z","shell.execute_reply.started":"2024-11-29T18:15:37.975743Z","shell.execute_reply":"2024-11-29T18:15:39.384181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train(reshape_weight=False):\n    # set the train mode\n    model.train()\n\n    # loss of the epoch\n    train_loss = 0\n\n    for i, (x, weight) in enumerate(train_dataloader):\n        x = x.to(device)\n        # if reshape_weight:\n        #     weight = weight.reshape([weight.size()[0], 1, -1])\n        weight = weight.to(device)\n        # print(f'size of x: {x.size()}')\n        # print(f'size of weight: {weight.size()}')\n\n        # update the gradients to zero\n        optimizer.zero_grad()\n\n        # forward pass\n        x_sample, z_mu, z_var = model(x)\n\n        # reconstruction loss\n        # recon_loss = F.mse_loss(x_sample, x, reduction='sum')\n        recon_loss = torch.sum((x_sample - x) ** 2 * weight)\n\n        # kl divergence loss\n        # kl_loss = 0.5 * torch.sum(torch.exp(z_var) + z_mu**2 - 1.0 - z_var)\n        kl_loss = 0.5 * torch.sum((torch.exp(z_var) + (z_mu)**2 - 1.0 - z_var)  * weight)\n\n        # total loss\n        loss = recon_loss + kl_loss\n\n        # backward pass\n        loss.backward()\n        train_loss += loss.item()\n\n        # update the weights\n        optimizer.step()\n\n    return train_loss\n\n\ndef test(reshape_weight=False):\n    # set the evaluation mode\n    model.eval()\n\n    # test loss for the data\n    test_loss = 0\n\n    # we don't need to track the gradients, since we are not updating the parameters during evaluation / testing\n    with torch.no_grad():\n        for i, (x, weight) in enumerate(val_dataloader):\n            x = x.to(device)\n            # if reshape_weight:\n            #     weight = weight.reshape([weight.size()[0], 1, -1])\n            weight = weight.to(device)\n\n            # forward pass\n            x_sample, z_mu, z_var = model(x)\n\n            # reconstruction loss\n            # recon_loss = F.mse_loss(x_sample, x, reduction='sum')\n            recon_loss = torch.sum((x_sample - x) ** 2 * weight)\n\n            # kl divergence loss\n            # kl_loss = 0.5 * torch.sum(torch.exp(z_var) + z_mu**2 - 1.0 - z_var)\n            kl_loss = 0.5 * torch.sum((torch.exp(z_var) + (z_mu)**2 - 1.0 - z_var)  * weight)\n\n            # total loss\n            loss = recon_loss + kl_loss\n            test_loss += loss.item()\n\n    return test_loss\n\nbest_test_loss = float('inf')\n\n# for e in range(n_epochs):\n#     train_loss = train(reshape_weight=True)\n#     test_loss = test(reshape_weight=True)\n#     train_loss /= len(train_dataloader)\n#     test_loss /= len(val_dataloader)\n#     print(f'Epoch {e}, Train Loss: {train_loss:.2f}, Test Loss: {test_loss:.2f}')\n#     # if best_test_loss > test_loss:\n#     #     best_test_loss = test_loss\n#     #     patience_counter = 1\n#     # else:\n#     #     patience_counter += 1\n#     # if patience_counter > 3:\n#     #     break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:15:41.803547Z","iopub.execute_input":"2024-11-29T18:15:41.804840Z","iopub.status.idle":"2024-11-29T18:15:41.816499Z","shell.execute_reply.started":"2024-11-29T18:15:41.804750Z","shell.execute_reply":"2024-11-29T18:15:41.815424Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Step 4.1.4 Save Conv1dVAE","metadata":{}},{"cell_type":"code","source":"# torch.save(model.state_dict(), './model/Conv1dVAE.pt')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:15:44.292503Z","iopub.execute_input":"2024-11-29T18:15:44.292982Z","iopub.status.idle":"2024-11-29T18:15:44.299192Z","shell.execute_reply.started":"2024-11-29T18:15:44.292944Z","shell.execute_reply":"2024-11-29T18:15:44.297643Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 4.2 Construct training and validation datasets","metadata":{}},{"cell_type":"code","source":"encoder = Conv1dEncoder(input_size=input_size, hidden_size=hidden_size_encoder, latent_size=latent_size)\ndecoder = Conv1dDecoder(latent_size=latent_size, hidden_size=hidden_size_decoder, output_size=input_size)\nconv1dvae = Conv1dVAE(encoder, decoder)\nconv1dvae.load_state_dict(torch.load('/kaggle/input/model_6/pytorch/default/1/Conv1dVAE.pt', map_location=torch.device('cpu')))\nconv1dvae.eval()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:15:46.272268Z","iopub.execute_input":"2024-11-29T18:15:46.272733Z","iopub.status.idle":"2024-11-29T18:15:46.635553Z","shell.execute_reply.started":"2024-11-29T18:15:46.272696Z","shell.execute_reply":"2024-11-29T18:15:46.634788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with torch.no_grad():\n    z_mu, z_var = conv1dvae.enc(enmo_tensors)\n    std = torch.exp(z_var / 2)\n    eps = torch.randn_like(std)\n    enmo_tensors_latent = eps.mul(std).add_(z_mu)\n\nenmo_tensors_train_latent, enmo_tensors_valid_latent = enmo_tensors_latent[:3166], enmo_tensors_latent[3166:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:15:55.586021Z","iopub.execute_input":"2024-11-29T18:15:55.586411Z","iopub.status.idle":"2024-11-29T18:16:10.562085Z","shell.execute_reply.started":"2024-11-29T18:15:55.586378Z","shell.execute_reply":"2024-11-29T18:16:10.561164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tensors_train = torch.cat([feature_tensors_train, enmo_tensors_train_latent], axis=1).to(torch.float32)\ntensors_valid = torch.cat([feature_tensors_valid, enmo_tensors_valid_latent], axis=1).to(torch.float32)\ntensors_valid.size()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:12.298190Z","iopub.execute_input":"2024-11-29T18:16:12.298606Z","iopub.status.idle":"2024-11-29T18:16:12.309708Z","shell.execute_reply.started":"2024-11-29T18:16:12.298574Z","shell.execute_reply":"2024-11-29T18:16:12.308543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#define a batch size\nbatch_size = 32\n\n# wrap tensors\ntrain_data = TensorDataset(tensors_train, weights_train)\n# sampler for sampling the data during training\ntrain_sampler = RandomSampler(train_data)\n# dataLoader for train set\ntrain_dataloader = DataLoader(train_data, sampler=train_sampler, batch_size=batch_size)\n\n# wrap tensors\nval_data = TensorDataset(tensors_valid, weights_valid)\n# sampler for sampling the data during training\nval_sampler = SequentialSampler(val_data)\n# dataLoader for validation set\nval_dataloader = DataLoader(val_data, sampler = val_sampler, batch_size=batch_size)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:13.350221Z","iopub.execute_input":"2024-11-29T18:16:13.350692Z","iopub.status.idle":"2024-11-29T18:16:13.358403Z","shell.execute_reply.started":"2024-11-29T18:16:13.350656Z","shell.execute_reply":"2024-11-29T18:16:13.357144Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 4.3 VAE for the concatenated tensors","metadata":{}},{"cell_type":"code","source":"class Encoder(nn.Module):\n    ''' This the encoder part of VAE\n    '''\n    def __init__(self, input_size=39, hidden_size_1=256, hidden_size_2=128, hidden_size_3=64, hidden_size_4=16):\n        super(Encoder, self).__init__()\n\n        self.linear_1 = nn.Linear(input_size, hidden_size_1)\n        self.linear_2 = nn.Linear(hidden_size_1, hidden_size_2)\n        self.linear_3 = nn.Linear(hidden_size_2, hidden_size_3)\n        # self.batchnorm_1 = nn.B\n        self.mu = nn.Linear(hidden_size_3, hidden_size_4)\n        self.var = nn.Linear(hidden_size_3, hidden_size_4)\n\n    def forward(self, x):\n        # x is of shape [batch_size, input_size]\n\n        hidden = nn.ReLU()(self.linear_1(x[:, :-16]))\n        hidden = nn.ReLU()(self.linear_2(hidden))\n        hidden = nn.ReLU()(self.linear_3(hidden))\n        # hidden is of shape [batch_size, hidden_size]\n        z_mu = self.mu(hidden)\n        z_mu = torch.concat([z_mu, x[:, -16:]], axis=1)\n        # z_mu is of shape [batch_size, latent_size]\n        z_var = self.var(hidden)\n        z_var = torch.concat([z_var, x[:, -16:]], axis=1)\n        # z_var is of shape [batch_size, latent_size]\n\n        return z_mu, z_var\n\nclass Decoder(nn.Module):\n    ''' This the decoder part of VAE\n    '''\n    def __init__(self, latent_size, hidden_size, output_size):\n        super(Decoder, self).__init__()\n\n        self.linear = nn.Linear(latent_size, hidden_size)\n        self.out = nn.Linear(hidden_size, output_size)\n\n    def forward(self, x):\n        # x is of shape [batch_size, latent_size]\n\n        hidden = nn.ReLU()(self.linear(x))\n        # hidden is of shape [batch_size, hidden_size]\n\n        predicted = torch.sigmoid(self.out(hidden))\n        # predicted is of shape [batch_size, output_size]\n        return predicted\n\n\nclass VAE(nn.Module):\n    def __init__(self, enc, dec):\n        ''' This the VAE, which takes a encoder and decoder.\n        '''\n        super(VAE, self).__init__()\n\n        self.enc = enc\n        self.dec = dec\n\n    def forward(self, x):\n        # encode\n        z_mu, z_var = self.enc(x)\n\n        # sample from the distribution having latent parameters z_mu, z_var\n        # reparameterize\n        std = torch.exp(z_var / 2)\n        eps = torch.randn_like(std)\n        x_sample = eps.mul(std).add_(z_mu)\n\n        # decode\n        predicted = self.dec(x_sample)\n        return predicted, z_mu, z_var","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:15.903225Z","iopub.execute_input":"2024-11-29T18:16:15.903618Z","iopub.status.idle":"2024-11-29T18:16:15.916638Z","shell.execute_reply.started":"2024-11-29T18:16:15.903584Z","shell.execute_reply":"2024-11-29T18:16:15.915297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_epochs = 100\ninput_size = 39\noutput_size = 55\nhidden_size_1 = 256\nhidden_size_2 = 128\nhidden_size_3 = 64\nhidden_size_4 = 16\nhidden_size_decoder = 256\nlatent_size = 32\nlr = 1e-3\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n# encoder\n# encoder = Encoder(input_size=input_size, hidden_size_1=hidden_size_1, hidden_size_2=hidden_size_2, hidden_size_3=hidden_size_3, latent_size=latent_size)\nencoder = Encoder(input_size=input_size, hidden_size_1=hidden_size_1, hidden_size_2=hidden_size_2, hidden_size_3=hidden_size_3, hidden_size_4=hidden_size_4)\n\n# decoder\n# decoder = Decoder(latent_size=latent_size, hidden_size=hidden_size_decoder, output_size=input_size)\ndecoder = Decoder(latent_size=latent_size, hidden_size=hidden_size_decoder, output_size=output_size)\n\n# vae\nmodel = VAE(encoder, decoder).to(device)\n\n# optimizer\noptimizer = optim.Adam(model.parameters(), lr=lr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:17.039698Z","iopub.execute_input":"2024-11-29T18:16:17.040171Z","iopub.status.idle":"2024-11-29T18:16:17.050705Z","shell.execute_reply.started":"2024-11-29T18:16:17.040133Z","shell.execute_reply":"2024-11-29T18:16:17.049544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best_test_loss = float('inf')\n\n# for e in range(n_epochs):\n#     train_loss = train()\n#     test_loss = test()\n#     train_loss /= len(train_dataloader)\n#     test_loss /= len(val_dataloader)\n#     print(f'Epoch {e}, Train Loss: {train_loss:.2f}, Test Loss: {test_loss:.2f}')\n#     if best_test_loss > test_loss:\n#         best_test_loss = test_loss\n#         patience_counter = 1\n#     else:\n#         patience_counter += 1\n#     if patience_counter > 3:\n#         break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:20.211432Z","iopub.execute_input":"2024-11-29T18:16:20.211912Z","iopub.status.idle":"2024-11-29T18:16:20.217720Z","shell.execute_reply.started":"2024-11-29T18:16:20.211870Z","shell.execute_reply":"2024-11-29T18:16:20.216413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# torch.save(model.state_dict(), './model/VAE.pt')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:21.402284Z","iopub.execute_input":"2024-11-29T18:16:21.402695Z","iopub.status.idle":"2024-11-29T18:16:21.408280Z","shell.execute_reply.started":"2024-11-29T18:16:21.402660Z","shell.execute_reply":"2024-11-29T18:16:21.406999Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 4.4 Plot latent vectors","metadata":{}},{"cell_type":"code","source":"from sklearn.manifold import TSNE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:23.098333Z","iopub.execute_input":"2024-11-29T18:16:23.098806Z","iopub.status.idle":"2024-11-29T18:16:23.170066Z","shell.execute_reply.started":"2024-11-29T18:16:23.098739Z","shell.execute_reply":"2024-11-29T18:16:23.168834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encoder = Encoder(input_size=input_size, hidden_size_1=hidden_size_1, hidden_size_2=hidden_size_2, hidden_size_3=hidden_size_3, hidden_size_4=hidden_size_4)\ndecoder = Decoder(latent_size=latent_size, hidden_size=hidden_size_decoder, output_size=output_size)\nvae = VAE(encoder, decoder)\nvae.load_state_dict(torch.load('/kaggle/input/model_6/pytorch/default/1/VAE.pt', map_location=torch.device('cpu')))\nvae.eval()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:25.475323Z","iopub.execute_input":"2024-11-29T18:16:25.475684Z","iopub.status.idle":"2024-11-29T18:16:25.531265Z","shell.execute_reply.started":"2024-11-29T18:16:25.475655Z","shell.execute_reply":"2024-11-29T18:16:25.530120Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tensors_train_valid = torch.cat([tensors_train, tensors_valid])\n\nwith torch.no_grad():\n    z_mu, z_var = vae.enc(tensors_train_valid)\n    std = torch.exp(z_var / 2)\n    eps = torch.randn_like(std)\n    tensors_train_valid_latent = eps.mul(std).add_(z_mu)\n\n#TSNE for dimension reduction\nreduced_latent_vectors = TSNE(n_components=2, learning_rate='auto', init='random', perplexity=5).fit_transform(np.array(tensors_train_valid_latent))\n\ndf_latent = pd.DataFrame({'X': reduced_latent_vectors[:, 0],\n                                'Y': reduced_latent_vectors[:, 1],\n                                'Z': df['sii']})\nprint(df_latent.head())\ndf_latent.plot.scatter(x='X', y='Y', c='Z')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:28.138656Z","iopub.execute_input":"2024-11-29T18:16:28.139147Z","iopub.status.idle":"2024-11-29T18:16:46.269490Z","shell.execute_reply.started":"2024-11-29T18:16:28.139107Z","shell.execute_reply":"2024-11-29T18:16:46.268296Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Step 4.5 K-Means to predict labels","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nimport pickle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:51.824533Z","iopub.execute_input":"2024-11-29T18:16:51.824956Z","iopub.status.idle":"2024-11-29T18:16:51.887461Z","shell.execute_reply.started":"2024-11-29T18:16:51.824922Z","shell.execute_reply":"2024-11-29T18:16:51.886328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# kmeans = KMeans(n_clusters=4, init='random', random_state=42, n_init=100).fit(tensors_train_valid_latent.numpy())\nwith open(\"/kaggle/input/model_6/pytorch/default/1/kmeans.pkl\", \"rb\") as f:\n    kmeans = pickle.load(f)\ndf_latent['Z_pred'] = kmeans.labels_\ndf_latent","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:53.202947Z","iopub.execute_input":"2024-11-29T18:16:53.203343Z","iopub.status.idle":"2024-11-29T18:16:53.233489Z","shell.execute_reply.started":"2024-11-29T18:16:53.203311Z","shell.execute_reply":"2024-11-29T18:16:53.232130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:16:58.949097Z","iopub.execute_input":"2024-11-29T18:16:58.949452Z","iopub.status.idle":"2024-11-29T18:16:58.954827Z","shell.execute_reply.started":"2024-11-29T18:16:58.949422Z","shell.execute_reply":"2024-11-29T18:16:58.953366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_latent_label = df_latent[df_latent['Z'] != 4]\ncm = confusion_matrix(df_latent_label['Z'], df_latent_label['Z_pred'])\ncm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:17:00.063997Z","iopub.execute_input":"2024-11-29T18:17:00.064357Z","iopub.status.idle":"2024-11-29T18:17:00.079579Z","shell.execute_reply.started":"2024-11-29T18:17:00.064328Z","shell.execute_reply":"2024-11-29T18:17:00.078403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"map_dict = {0: 0, 1: 2, 2: 1, 3: 3}\ndf_latent['Z_pred_map'] = df_latent['Z_pred'].map(map_dict)\ndf_latent","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:17:05.053587Z","iopub.execute_input":"2024-11-29T18:17:05.053972Z","iopub.status.idle":"2024-11-29T18:17:05.072108Z","shell.execute_reply.started":"2024-11-29T18:17:05.053941Z","shell.execute_reply":"2024-11-29T18:17:05.070747Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 5 Predict test dataset","metadata":{}},{"cell_type":"code","source":"with torch.no_grad():\n    z_mu, z_var = conv1dvae.enc(enmo_tensors_test)\n    std = torch.exp(z_var / 2)\n    eps = torch.randn_like(std)\n    enmo_tensors_test_latent = eps.mul(std).add_(z_mu)\n\ntensors_test = torch.cat([feature_tensors_test, enmo_tensors_test_latent], axis=1).to(torch.float32)\nprint(tensors_test.size())\n\nwith torch.no_grad():\n    z_mu, z_var = vae.enc(tensors_test)\n    std = torch.exp(z_var / 2)\n    eps = torch.randn_like(std)\n    tensors_test_latent = eps.mul(std).add_(z_mu)\n\nprint(tensors_test_latent.size())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:17:09.747713Z","iopub.execute_input":"2024-11-29T18:17:09.748142Z","iopub.status.idle":"2024-11-29T18:17:09.811918Z","shell.execute_reply.started":"2024-11-29T18:17:09.748110Z","shell.execute_reply":"2024-11-29T18:17:09.810867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:17:11.159148Z","iopub.execute_input":"2024-11-29T18:17:11.159529Z","iopub.status.idle":"2024-11-29T18:17:11.165106Z","shell.execute_reply.started":"2024-11-29T18:17:11.159498Z","shell.execute_reply":"2024-11-29T18:17:11.163665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(\"/kaggle/input/model_6/pytorch/default/1/kmeans.pkl\", \"rb\") as f:\n    kmeans = pickle.load(f)\n\nlabels_pred = kmeans.predict(tensors_test_latent.numpy())\nmap_dict = {0: 0, 1: 2, 2: 1, 3: 3}\ndf_submit = pd.DataFrame({'id': df_test['id'], 'sii': labels_pred})\ndf_submit['sii'] = df_submit['sii'].map(map_dict)\ndf_submit.to_csv('submission.csv', index=False)\ndf_submit","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T18:17:12.822867Z","iopub.execute_input":"2024-11-29T18:17:12.823529Z","iopub.status.idle":"2024-11-29T18:17:12.848825Z","shell.execute_reply.started":"2024-11-29T18:17:12.823482Z","shell.execute_reply":"2024-11-29T18:17:12.847530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}