{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Check this for [submission](http://www.kaggle.com/code/jerometam/js-pytorch-submission)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nimport math\nimport random\n\nimport os\nimport re\nfrom tqdm import tqdm\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\n\nfrom sklearn.model_selection import train_test_split\n\nimport kaggle_evaluation.jane_street_inference_server\n\ndef seed_everything(seed):\n    np.random.seed(seed)\n    random.seed(seed)\nseed_everything(seed=2025)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-25T08:53:05.471967Z","iopub.execute_input":"2024-10-25T08:53:05.472319Z","iopub.status.idle":"2024-10-25T08:53:14.474856Z","shell.execute_reply.started":"2024-10-25T08:53:05.472286Z","shell.execute_reply":"2024-10-25T08:53:14.473903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"< read parquet >\")\ndatas=[]\nfor i in range(6,10):\n    train=pl.read_parquet(f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet\")\n    train=train.to_pandas().sample(frac=0.97, random_state=2025)\n    datas.append(train)\ntrain=pd.concat(datas)\ndel datas\ngc.collect()\nprint(f\"train.shape:{train.shape}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T08:53:17.143063Z","iopub.execute_input":"2024-10-25T08:53:17.143957Z","iopub.status.idle":"2024-10-25T08:54:17.925782Z","shell.execute_reply.started":"2024-10-25T08:53:17.143914Z","shell.execute_reply":"2024-10-25T08:54:17.924662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.fillna(3)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T08:54:17.927714Z","iopub.execute_input":"2024-10-25T08:54:17.928059Z","iopub.status.idle":"2024-10-25T08:54:32.069684Z","shell.execute_reply.started":"2024-10-25T08:54:17.928027Z","shell.execute_reply":"2024-10-25T08:54:32.068897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#featuresCols = [\"symbol_id\"] + [col for col in train.columns if 'feature_' in col]\nfeaturesCols = [col for col in train.columns if 'feature_' in col]\ntargetsCols = [\"responder_6\"]\nweightCols = [\"weight\"]","metadata":{"execution":{"iopub.status.busy":"2024-10-25T08:54:32.070797Z","iopub.execute_input":"2024-10-25T08:54:32.071117Z","iopub.status.idle":"2024-10-25T08:54:32.076196Z","shell.execute_reply.started":"2024-10-25T08:54:32.071084Z","shell.execute_reply":"2024-10-25T08:54:32.075244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_divisors(N):\n    divisors = []\n    for i in range(1, int(N**0.5) + 1):\n        if N % i == 0:  \n            divisors.append(i)\n            if i != N // i:  \n                divisors.append(N // i)\n    return sorted(divisors)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T08:54:32.078466Z","iopub.execute_input":"2024-10-25T08:54:32.078739Z","iopub.status.idle":"2024-10-25T08:54:32.086194Z","shell.execute_reply.started":"2024-10-25T08:54:32.078709Z","shell.execute_reply":"2024-10-25T08:54:32.085358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"divisors = find_divisors(len(train))\ndiv_idx = int(len(divisors)/3)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T08:54:32.087108Z","iopub.execute_input":"2024-10-25T08:54:32.087389Z","iopub.status.idle":"2024-10-25T08:54:32.095617Z","shell.execute_reply.started":"2024-10-25T08:54:32.087353Z","shell.execute_reply":"2024-10-25T08:54:32.094918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_list = [train.iloc[0:divisors[div_idx]][featuresCols].to_numpy()]\ntarget_list = [train.iloc[0:divisors[div_idx]][targetsCols].to_numpy()]\nweigth_tgt_list = [train.iloc[0:divisors[div_idx]][weightCols].to_numpy()]\n\nfor i in tqdm(range(divisors[div_idx],len(train)-divisors[div_idx],divisors[div_idx])):\n    \n    feat_list.append(train.iloc[i:i+divisors[div_idx]][featuresCols].to_numpy())\n    \n    target_list.append(train.iloc[i:i+divisors[div_idx]][targetsCols].to_numpy())\n    \n    weigth_tgt_list.append(train.iloc[i:i+divisors[div_idx]][weightCols].to_numpy())   \n    \ndel train\ndel divisors\ndel div_idx","metadata":{"execution":{"iopub.status.busy":"2024-10-25T08:54:32.096690Z","iopub.execute_input":"2024-10-25T08:54:32.096965Z","iopub.status.idle":"2024-10-25T09:02:06.834083Z","shell.execute_reply.started":"2024-10-25T08:54:32.096935Z","shell.execute_reply":"2024-10-25T09:02:06.833149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arr_feat = np.stack(feat_list)\ndel feat_list","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:02:06.835384Z","iopub.execute_input":"2024-10-25T09:02:06.835691Z","iopub.status.idle":"2024-10-25T09:02:09.752607Z","shell.execute_reply.started":"2024-10-25T09:02:06.835659Z","shell.execute_reply":"2024-10-25T09:02:09.751587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arr_tgt = np.stack(target_list)\ndel target_list","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:02:09.767856Z","iopub.execute_input":"2024-10-25T09:02:09.768187Z","iopub.status.idle":"2024-10-25T09:02:10.612987Z","shell.execute_reply.started":"2024-10-25T09:02:09.768154Z","shell.execute_reply":"2024-10-25T09:02:10.612190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arr_weight = np.stack(weigth_tgt_list)\ndel weigth_tgt_list","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:02:10.631876Z","iopub.execute_input":"2024-10-25T09:02:10.632172Z","iopub.status.idle":"2024-10-25T09:02:11.592602Z","shell.execute_reply.started":"2024-10-25T09:02:10.632142Z","shell.execute_reply":"2024-10-25T09:02:11.591806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(arr_feat.shape)\nprint(arr_tgt.shape)\nprint(arr_weight.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:02:11.595315Z","iopub.execute_input":"2024-10-25T09:02:11.595624Z","iopub.status.idle":"2024-10-25T09:02:11.600904Z","shell.execute_reply.started":"2024-10-25T09:02:11.595591Z","shell.execute_reply":"2024-10-25T09:02:11.599900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split = int(arr_feat.shape[0] * 0.65)\n\n\nX_train = arr_feat[:split]\nX_test = arr_feat[split:]\ndel arr_feat\n\ny_train = arr_tgt[:split]\ny_test = arr_tgt[split:]\ndel arr_tgt\n\nweight_train = arr_weight[:split]\nweight_test = arr_weight[split:]\ndel arr_weight\n\n\n\n\nprint(\"X_train shape:\", X_train.shape)\nprint(\"X_test shape:\", X_test.shape)\nprint(\"y_train shape:\", y_train.shape)\nprint(\"y_test shape:\", y_test.shape)\nprint(\"weight_train shape:\", weight_train.shape)\nprint(\"weight_test shape:\", weight_test.shape)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:02:11.602064Z","iopub.execute_input":"2024-10-25T09:02:11.602363Z","iopub.status.idle":"2024-10-25T09:02:11.612415Z","shell.execute_reply.started":"2024-10-25T09:02:11.602310Z","shell.execute_reply":"2024-10-25T09:02:11.611527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train_tensor = torch.FloatTensor(X_train_scaled_reshaped)\nX_train_tensor = torch.FloatTensor(X_train)\ndel X_train\n\ny_train_tensor = torch.FloatTensor(y_train)\ndel y_train\n\nweight_train_tensor = torch.FloatTensor(weight_train)\ndel weight_train\n\n#X_test_tensor = torch.FloatTensor(X_test_scaled_reshaped)\nX_test_tensor = torch.FloatTensor(X_test)\ndel X_test\ny_test_tensor = torch.FloatTensor(y_test)\ndel y_test\n\nweight_test_tensor = torch.FloatTensor(weight_test)\ndel weight_test\n\n\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:02:11.613465Z","iopub.execute_input":"2024-10-25T09:02:11.613747Z","iopub.status.idle":"2024-10-25T09:02:11.657257Z","shell.execute_reply.started":"2024-10-25T09:02:11.613717Z","shell.execute_reply":"2024-10-25T09:02:11.656576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 64\ndataset_train = torch.utils.data.TensorDataset(X_train_tensor, y_train_tensor,weight_train_tensor)\ndataloader_train = torch.utils.data.DataLoader(dataset_train, batch_size=batch_size, shuffle=True)\n\ndataset_test = torch.utils.data.TensorDataset(X_test_tensor, y_test_tensor,weight_test_tensor)\ndataloader_test = torch.utils.data.DataLoader(dataset_test, batch_size=batch_size, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:02:11.658173Z","iopub.execute_input":"2024-10-25T09:02:11.658419Z","iopub.status.idle":"2024-10-25T09:02:11.672316Z","shell.execute_reply.started":"2024-10-25T09:02:11.658390Z","shell.execute_reply":"2024-10-25T09:02:11.671285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for batch in dataloader_train:\n    x,y,z = batch\n    break","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:02:11.673386Z","iopub.execute_input":"2024-10-25T09:02:11.673670Z","iopub.status.idle":"2024-10-25T09:02:11.855161Z","shell.execute_reply.started":"2024-10-25T09:02:11.673639Z","shell.execute_reply":"2024-10-25T09:02:11.854382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Encoder(nn.Module):\n    def __init__(self):\n        super(Encoder, self).__init__()\n    \n        self.attn = torch.nn.MultiheadAttention(embed_dim = 64*2,\n                                                num_heads=4,\n                                                dropout=0.1,\n                                                batch_first=True)\n        \n        self.linear1 = nn.Linear(in_features = 64*2,out_features = 128*2)\n        self.linear2 = nn.Linear(in_features = 128*2,out_features = 64*2)\n        \n        self.norm1 = nn.LayerNorm(normalized_shape = 64*2)\n        self.norm2 = nn.LayerNorm(normalized_shape = 64*2)\n        \n        self.dropout1 = torch.nn.Dropout(0.2)\n        self.dropout2 = torch.nn.Dropout(0.2)\n        self.dropout3 = torch.nn.Dropout(0.2)\n        \n    def forward(self,x,attn_mask=None,key_padding_mask=None):\n\n        x_tmp = self.norm1(x)\n        \n        x_tmp, _ = self.attn(query=x_tmp,\n                             key=x_tmp,\n                             value=x_tmp,\n                             attn_mask=attn_mask,\n                             key_padding_mask=key_padding_mask)\n\n            \n        x_tmp = self.dropout1(x_tmp)\n        x = x + x_tmp    \n        x_tmp = self.norm2(x) \n        x_tmp = self.linear1(x_tmp)\n        x_tmp = F.relu(x_tmp)\n            \n        x_tmp = self.dropout2(x_tmp)\n        x_tmp = self.linear2(x_tmp)\n        x_tmp = self.dropout3(x_tmp)\n        x = x + x_tmp\n            \n        return x\n    \n    \nclass PredTimeSeries(nn.Module):\n    def __init__(self,feat_dim,tgt_dim,latent_dim = 64*2):\n        super(PredTimeSeries, self).__init__()\n        \n        self.prenet = nn.Linear(in_features = feat_dim,out_features = 64*2)\n        \n        self.positional_encoding = nn.Parameter(torch.randn(1, 100, 64*2))\n        self.encoder1 = Encoder()\n        self.encoder2 = Encoder()\n\n        \n        self.fc1 = nn.Linear(64*2,32*2)\n        self.fc2 = nn.Linear(32*2,16*2)\n        self.fc3 = nn.Linear(16*2,tgt_dim)\n        \n        self.norm1 = nn.LayerNorm(32*2)\n        self.norm2 = nn.LayerNorm(16*2)\n\n        \n        self.drop1 = nn.Dropout(0.2)\n        self.drop2 = nn.Dropout(0.2)\n\n        \n        \n    def forward(self,x):\n        _,seq_len,feat_d = x.shape\n        \n        x = self.prenet(x)\n\n        x = x + self.positional_encoding[:, :seq_len, :]\n        \n        x = self.encoder1(x)\n        x = self.encoder2(x)\n\n        x = self.fc1(x)\n        x = self.norm1(x)\n        x = F.relu(x)\n        x = self.drop1(x)\n        \n        x = self.fc2(x)\n        x = self.norm2(x)\n        x = F.relu(x)\n        x = self.drop2(x)\n        \n        \n        x = self.fc3(x)\n\n            \n        return x\n\n    \n\nclass CustomWeightedR2Loss(nn.Module):\n    def __init__(self):\n        super(CustomWeightedR2Loss, self).__init__()\n\n    def forward(self, y_pred, y_true, weight):\n    \n        y_true_flat = y_true.flatten()\n        y_pred_flat = y_pred.flatten()\n        weight_flat = weight.flatten()\n\n \n        numerator = torch.sum(weight_flat * (y_true_flat - y_pred_flat) ** 2)\n        denominator = torch.sum(weight_flat * (y_true_flat) ** 2)\n\n        weighted_r2 = 1 - numerator / denominator\n        \n        return 1 - weighted_r2\n    \n    \nclass CustomWeightedR2Loss(nn.Module):\n    def __init__(self):\n        super(CustomWeightedR2Loss, self).__init__()\n\n    def forward(self, y_pred, y_true, weight):\n    \n        y_true_flat = y_true.flatten()\n        y_pred_flat = y_pred.flatten()\n        weight_flat = weight.flatten()\n\n \n        numerator = torch.sum(weight_flat * (y_true_flat - y_pred_flat) ** 2)\n        denominator = torch.sum(weight_flat * (y_true_flat) ** 2)\n\n        weighted_r2 = 1 - numerator / denominator\n        \n        return 1 - weighted_r2","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ ,seq_lgt, feat_dim = x.shape\ntgt_dim = y.shape[-1]\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = PredTimeSeries( feat_dim, tgt_dim)\ncriterion = CustomWeightedR2Loss().to(device)\noptimizer = optim.AdamW(model.parameters(), lr=0.0001)\nepochs = 100","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:02:11.880154Z","iopub.execute_input":"2024-10-25T09:02:11.880607Z","iopub.status.idle":"2024-10-25T09:02:13.971493Z","shell.execute_reply.started":"2024-10-25T09:02:11.880572Z","shell.execute_reply":"2024-10-25T09:02:13.970718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.to(device)  \n\nfor epoch in range(epochs):\n    \n    model.train()\n    train_loss = 0.0\n    for i, batch in enumerate(dataloader_train):\n        \n        optimizer.zero_grad()\n        \n        inputs, tgt, weight = batch\n        inputs, tgt, weight  = inputs.to(device), tgt.to(device),weight.to(device)\n        with torch.amp.autocast(device_type='cuda'):\n            outputs = model(inputs)\n            loss = criterion(outputs, tgt,weight)\n         \n        \n        loss.backward()\n        optimizer.step()\n        \n        train_loss += loss.item()\n    \n    train_loss = train_loss / len(dataloader_train)\n    \n    if (epoch + 1) % 1 == 0:\n        model.eval()\n        test_loss = 0.0\n        \n        with torch.no_grad():  \n            for i, batch in enumerate(dataloader_test):\n                inputs, tgt, weight = batch\n                inputs, tgt, weight  = inputs.to(device), tgt.to(device),weight.to(device)\n                \n                outputs = model(inputs)\n                loss = criterion(outputs, tgt, weight)\n        \n                \n                test_loss += loss.item()\n        \n    \n        test_loss = test_loss / len(dataloader_test)\n        \n        print(f'Epoch [{epoch + 1}/{epochs}], Train_loss: {train_loss:.4f}, Test_loss: {test_loss:.4f}')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:02:13.972725Z","iopub.execute_input":"2024-10-25T09:02:13.973287Z","iopub.status.idle":"2024-10-25T09:19:15.588903Z","shell.execute_reply.started":"2024-10-25T09:02:13.973242Z","shell.execute_reply":"2024-10-25T09:19:15.587454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.save(model.state_dict(), 'model_final.pth')","metadata":{"execution":{"iopub.status.busy":"2024-10-25T09:19:23.103262Z","iopub.execute_input":"2024-10-25T09:19:23.103956Z","iopub.status.idle":"2024-10-25T09:19:23.128636Z","shell.execute_reply.started":"2024-10-25T09:19:23.103918Z","shell.execute_reply":"2024-10-25T09:19:23.127891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_val = torch.utils.data.TensorDataset(X_test_tensor, y_test_tensor)\ndataloader_val = torch.utils.data.DataLoader(dataset_val, batch_size=1, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T10:15:35.642773Z","iopub.execute_input":"2024-10-24T10:15:35.643634Z","iopub.status.idle":"2024-10-24T10:15:35.648209Z","shell.execute_reply.started":"2024-10-24T10:15:35.643595Z","shell.execute_reply":"2024-10-24T10:15:35.647303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = []\nmodel.eval()\nwith torch.no_grad():  \n    for i, batch in enumerate(dataloader_val):\n        inputs, tgt = batch\n        inputs, tgt = inputs.to(device), tgt.to(device)  \n        \n        pred.append(model(inputs).cpu().numpy()[-1])\n \n       \n        \npred = np.stack(pred)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T10:15:39.047159Z","iopub.execute_input":"2024-10-24T10:15:39.047942Z","iopub.status.idle":"2024-10-24T10:17:50.383098Z","shell.execute_reply.started":"2024-10-24T10:15:39.047902Z","shell.execute_reply":"2024-10-24T10:17:50.382251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(test,lags):\n    cols= [\"symbol_id\"] + [f'feature_0{i}' if i<10 else f'feature_{i}' for i in range(79)]\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    df = test[cols].to_pandas()\n\n    df = df.fillna(df.mean())\n\n\n    df_tensor = torch.FloatTensor(df.values).to(device)\n\n    model.eval()\n    test_preds=model(df_tensor.unsqueeze(1)).cpu().detach().numpy().flatten()\n    \n    predictions = predictions.with_columns(pl.Series('responder_6', test_preds.ravel()))\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-10-23T10:15:15.202920Z","iopub.execute_input":"2024-10-23T10:15:15.203816Z","iopub.status.idle":"2024-10-23T10:15:15.210485Z","shell.execute_reply.started":"2024-10-23T10:15:15.203766Z","shell.execute_reply":"2024-10-23T10:15:15.209388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"execution":{"iopub.status.busy":"2024-10-23T10:15:16.339555Z","iopub.execute_input":"2024-10-23T10:15:16.339985Z","iopub.status.idle":"2024-10-23T10:15:16.400529Z","shell.execute_reply.started":"2024-10-23T10:15:16.339947Z","shell.execute_reply":"2024-10-23T10:15:16.399590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}