{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.11"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":194.4788,"end_time":"2025-05-17T09:40:52.636218","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-05-17T09:37:38.157418","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# DRW MLP Regressor Lightning w/Pearson\n","metadata":{"papermill":{"duration":0.004048,"end_time":"2025-05-17T09:37:43.80935","exception":false,"start_time":"2025-05-17T09:37:43.805302","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install lightning","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-05-26T04:30:22.99009Z","iopub.execute_input":"2025-05-26T04:30:22.990413Z","iopub.status.idle":"2025-05-26T04:31:54.666662Z","shell.execute_reply.started":"2025-05-26T04:30:22.990383Z","shell.execute_reply":"2025-05-26T04:31:54.665426Z"},"papermill":{"duration":108.653378,"end_time":"2025-05-17T09:39:32.466846","exception":false,"start_time":"2025-05-17T09:37:43.813468","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport os\nimport random\n\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.datasets import make_classification\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_squared_log_error\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Input, BatchNormalization, Activation\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.optimizers import Adam\n\nfrom lightning.pytorch import LightningDataModule\nfrom lightning.pytorch import LightningModule\nfrom lightning.pytorch import Trainer\nimport lightning.pytorch as L\nprint(L.__version__)\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, TensorDataset\nimport matplotlib.pyplot as plt\nfrom scipy.stats import pearsonr","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2025-05-26T04:31:54.669065Z","iopub.execute_input":"2025-05-26T04:31:54.669383Z","iopub.status.idle":"2025-05-26T04:32:32.111021Z","shell.execute_reply.started":"2025-05-26T04:31:54.669354Z","shell.execute_reply":"2025-05-26T04:32:32.110117Z"},"papermill":{"duration":44.937571,"end_time":"2025-05-17T09:40:17.433698","exception":false,"start_time":"2025-05-17T09:39:32.496127","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\nTEST=pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2025-05-26T04:32:32.119735Z","iopub.execute_input":"2025-05-26T04:32:32.120112Z","iopub.status.idle":"2025-05-26T04:33:19.205385Z","shell.execute_reply.started":"2025-05-26T04:32:32.120077Z","shell.execute_reply":"2025-05-26T04:33:19.204307Z"},"papermill":{"duration":0.092358,"end_time":"2025-05-17T09:40:17.55524","exception":false,"start_time":"2025-05-17T09:40:17.462882","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"https://www.kaggle.com/code/stpeteishii/drw-label-distribution","metadata":{}},{"cell_type":"code","source":"def func(x):\n    return np.sign(x) * np.log1p(np.abs(x))\n    \ndef inverse_func(y):\n    return np.sign(y) * (np.expm1(np.abs(y)))  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T04:32:32.112558Z","iopub.execute_input":"2025-05-26T04:32:32.113376Z","iopub.status.idle":"2025-05-26T04:32:32.118498Z","shell.execute_reply.started":"2025-05-26T04:32:32.113334Z","shell.execute_reply":"2025-05-26T04:32:32.117291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def pearson_metric(y_true, y_pred):\n    pearson_r, _ = pearsonr(y_true, y_pred)\n    return 'pearson', pearson_r, True","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['label']=train['label'].apply(func)\n\nprint(len(train))\nN = list(range(len(train)))\nrandom.shuffle(N) \ntrain=train.iloc[N[0:400000]]\n\ntrain=train.reset_index(drop=True)\ntrainX=train.drop('label',axis=1)\ntrainY=train['label']\nTESTX=TEST.drop('label',axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T04:33:19.206474Z","iopub.execute_input":"2025-05-26T04:33:19.206833Z","iopub.status.idle":"2025-05-26T04:33:23.585688Z","shell.execute_reply.started":"2025-05-26T04:33:19.206803Z","shell.execute_reply":"2025-05-26T04:33:23.584631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainX = np.where(trainX == -np.inf, -10000, trainX)\nTESTX = np.where(TESTX == -np.inf, -10000, TESTX)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\ntrainX = scaler.fit_transform(trainX)\nTESTX = scaler.transform(TESTX)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_splits = 5\nkf = KFold(n_splits=n_splits, shuffle=True, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:33:23.622141Z","iopub.status.idle":"2025-05-26T04:33:23.622719Z","shell.execute_reply.started":"2025-05-26T04:33:23.622419Z","shell.execute_reply":"2025-05-26T04:33:23.622447Z"},"papermill":{"duration":0.036497,"end_time":"2025-05-17T09:40:17.975756","exception":false,"start_time":"2025-05-17T09:40:17.939259","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Lightning Module definition\nclass MLP(LightningModule):\n    def __init__(self, input_dim):\n        super().__init__()\n        self.model = nn.Sequential(\n            nn.Linear(input_dim, 200),\n            nn.BatchNorm1d(200),\n            nn.ReLU(),\n            nn.Linear(200, 100),\n            nn.BatchNorm1d(100),\n            nn.ReLU(),\n            nn.Linear(100, 50),\n            nn.BatchNorm1d(50),\n            nn.ReLU(),\n            nn.Linear(50, 1)\n        )\n        \n    def forward(self, x):\n        return self.model(x)\n        \n    def training_step(self, batch, batch_idx):\n        x, y = batch\n        y_hat = self(x)\n        loss = F.mse_loss(y_hat, y.unsqueeze(1))\n        self.log(\"train_loss\", loss)\n        return loss\n        \n    def validation_step(self, batch, batch_idx):\n        x, y = batch\n        y_hat = self(x)\n        loss = F.mse_loss(y_hat, y.unsqueeze(1))\n        self.log(\"val_loss\", loss)\n        return loss\n        \n    def configure_optimizers(self):\n        return torch.optim.Adam(self.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:33:23.625029Z","iopub.status.idle":"2025-05-26T04:33:23.625584Z","shell.execute_reply.started":"2025-05-26T04:33:23.625332Z","shell.execute_reply":"2025-05-26T04:33:23.625353Z"},"papermill":{"duration":0.051005,"end_time":"2025-05-17T09:40:18.054964","exception":false,"start_time":"2025-05-17T09:40:18.003959","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def safe_mean_squared_error(y_true, y_pred):\n    if hasattr(y_pred, 'numpy'):\n        y_pred = y_pred.numpy()\n    if hasattr(y_true, 'numpy'):\n        y_true = y_true.numpy()\n        \n    mask = ~(np.isnan(y_pred) | np.isnan(y_true))\n    if not np.any(mask):\n        return float('inf')\n    return mean_squared_error(y_true[mask], y_pred[mask])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cross-validation loop\ndef run_cv(trainX, trainY, testX, n_splits=5):\n    # Convert inputs to numpy arrays if they're DataFrames\n    if hasattr(trainX, 'values'):\n        trainX = trainX.values\n    if hasattr(trainY, 'values'):\n        trainY = trainY.values\n    if hasattr(testX, 'values'):\n        testX = testX.values\n    \n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=42)\n    train_oof = np.zeros(len(trainX))\n    test_preds = np.zeros(len(testX))\n    \n    for fold_num, (train_index, val_index) in enumerate(kf.split(trainX)):\n        print(f\"Fitting fold {fold_num+1}/{n_splits}\")\n\n        # Split data\n        train_features = trainX[train_index]\n        train_target = trainY[train_index]\n        val_features = trainX[val_index]\n        val_target = trainY[val_index]\n        \n        # Convert to PyTorch tensors\n        train_features = torch.FloatTensor(train_features)\n        train_target = torch.FloatTensor(train_target)\n        val_features = torch.FloatTensor(val_features)\n        val_target = torch.FloatTensor(val_target)\n        \n        # Create datasets and dataloaders\n        train_dataset = TensorDataset(train_features, train_target)\n        val_dataset = TensorDataset(val_features, val_target)\n        \n        train_loader = DataLoader(train_dataset, batch_size=256, shuffle=True)\n        val_loader = DataLoader(val_dataset, batch_size=256)\n        \n        # Initialize model\n        model = MLP(input_dim=train_features.shape[1])\n        \n        # Trainer with early stopping\n        trainer = L.Trainer(\n            max_epochs=20000,\n            callbacks=[\n                L.callbacks.EarlyStopping(\n                    monitor=\"val_loss\",\n                    patience=10,\n                    mode=\"min\"\n                )\n            ],\n            enable_progress_bar=False,\n            enable_model_summary=False\n        )\n        \n        # Train the model\n        trainer.fit(model, train_loader, val_loader)\n        \n        # Predict on validation set\n        model.eval()\n        with torch.no_grad():\n            val_pred = model(val_features).squeeze().numpy()\n        train_oof[val_index] = val_pred\n        \n        # Metrics\n        pearson_score, _ = pearsonr(val_target, val_pred)\n        #smse = safe_mean_squared_error(val_target, val_pred)\n        print(pearson_score)\n        \n        # Predict on test set (average predictions across folds)\n        test_tensor = torch.FloatTensor(testX)  # Now safe because we converted earlier\n        with torch.no_grad():\n            test_pred_fold = model(test_tensor).squeeze().numpy()\n        test_preds += test_pred_fold / n_splits\n        \n        # Memory cleanup\n        del train_features, train_target, val_features, val_target, model\n        gc.collect()\n    \n    return train_oof, test_preds","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:33:23.625029Z","iopub.status.idle":"2025-05-26T04:33:23.625584Z","shell.execute_reply.started":"2025-05-26T04:33:23.625332Z","shell.execute_reply":"2025-05-26T04:33:23.625353Z"},"papermill":{"duration":0.051005,"end_time":"2025-05-17T09:40:18.054964","exception":false,"start_time":"2025-05-17T09:40:18.003959","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_oof, test_preds = run_cv(trainX, trainY, TESTX, n_splits=5)","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:33:23.62725Z","iopub.status.idle":"2025-05-26T04:33:23.627688Z","shell.execute_reply.started":"2025-05-26T04:33:23.627488Z","shell.execute_reply":"2025-05-26T04:33:23.627505Z"},"papermill":{"duration":31.117963,"end_time":"2025-05-17T09:40:49.202305","exception":false,"start_time":"2025-05-17T09:40:18.084342","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.save('oof.npy',train_oof) #######################","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_true = trainY\ny_pred = train_oof\n\nplt.figure(figsize=(6,6))\nplt.scatter(y_true, y_pred, alpha=0.1, color='blue')\nplt.xlabel(\"True Values\")\nplt.ylabel(\"Predicted Values\")\nplt.title(\"Scatter Plot of True vs Predicted (label)\")\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-05-26T04:33:23.629098Z","iopub.status.idle":"2025-05-26T04:33:23.629574Z","shell.execute_reply.started":"2025-05-26T04:33:23.629332Z","shell.execute_reply":"2025-05-26T04:33:23.629354Z"},"papermill":{"duration":0.359217,"end_time":"2025-05-17T09:40:49.591644","exception":false,"start_time":"2025-05-17T09:40:49.232427","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit=pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_preds2=inverse_func(test_preds)\n\nsubmit.iloc[:,1]= test_preds2\nsubmit.to_csv('submission.csv',index=False)\ndisplay(submit)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T04:33:23.631531Z","iopub.status.idle":"2025-05-26T04:33:23.631935Z","shell.execute_reply.started":"2025-05-26T04:33:23.631737Z","shell.execute_reply":"2025-05-26T04:33:23.631754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}