{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-21T02:05:59.308530Z","iopub.execute_input":"2025-07-21T02:05:59.308775Z","iopub.status.idle":"2025-07-21T02:05:59.592105Z","shell.execute_reply.started":"2025-07-21T02:05:59.308755Z","shell.execute_reply":"2025-07-21T02:05:59.591361Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Environment Setup and Library Imports","metadata":{}},{"cell_type":"code","source":"import gc\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler, QuantileTransformer\nfrom scipy.stats import pearsonr\nimport cudf # For GPU-accelerated dataframes\nimport cupy as cp # For GPU-accelerated numpy operations\nimport lightgbm as lgb # A strong baseline/ensemble candidate, even without explicit time features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T02:06:03.675760Z","iopub.execute_input":"2025-07-21T02:06:03.676681Z","iopub.status.idle":"2025-07-21T02:06:23.998581Z","shell.execute_reply.started":"2025-07-21T02:06:03.676647Z","shell.execute_reply":"2025-07-21T02:06:23.997738Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Data Loading and Preprocessing (Memory Efficient) ","metadata":{}},{"cell_type":"code","source":"# Function to reduce memory usage of a DataFrame (for pandas if you choose to use it for some steps)\ndef reduce_mem_usage(df):\n    start_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage of dataframe is {start_mem:.2f} MB')\n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage after optimization is: {end_mem:.2f} MB')\n    print(f'Decreased by {100 * (start_mem - end_mem) / start_mem:.2f}%')\n    return df\n\n# Load data with Polars first for efficiency, then convert to cuDF\n# Or directly with cuDF if the file is not too large for direct cuDF read_parquet\ntry:\n    train_df_pl = pl.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\n    test_df_pl = pl.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")\n    \n    # Convert Polars DataFrames to cuDF DataFrames for GPU operations\n    train_df = cudf.DataFrame(train_df_pl.to_pandas())\n    test_df = cudf.DataFrame(test_df_pl.to_pandas())\n\n    del train_df_pl, test_df_pl\n    gc.collect()\n\n    print(\"Data loaded into cuDF successfully!\")\n\nexcept Exception as e:\n    print(f\"Error loading data with cuDF/Polars, trying pandas: {e}\")\n    # Fallback to pandas with memory reduction if cuDF direct read or conversion fails due to size\n    train_df = pd.read_parquet(\"train.parquet\")\n    test_df = pd.read_parquet(\"test.parquet\")\n    train_df = reduce_mem_usage(train_df)\n    test_df = reduce_mem_usage(test_df)\n    print(\"Data loaded into pandas with memory reduction.\")\n    # If using pandas, move to CuPy/PyTorch tensors when needed to leverage GPU\n\n# Define features and target\nfeatures = [col for col in train_df.columns if 'X' in col]\ntarget = 'label'\n\n# Handle potential NaN values (e.g., fill with median or mean for numerical features)\n# For cuDF, use .fillna()\nfor col in features:\n    if col in train_df.columns:\n        train_df[col] = train_df[col].fillna(train_df[col].median())\n    if col in test_df.columns:\n        test_df[col] = test_df[col].fillna(test_df[col].median())\n    \n# Convert target to float32\nif target in train_df.columns:\n    train_df[target] = train_df[target].astype(np.float32)\n\nprint(f\"Shape of train_df: {train_df.shape}\")\nprint(f\"Shape of test_df: {test_df.shape}\")\n\n# Drop 'timestamp' from train_df if present, as it's not useful for direct time-series in test\nif 'timestamp' in train_df.columns:\n    train_df = train_df.drop('timestamp', axis=1)\nif 'timestamp' in test_df.columns: # test timestamp is masked anyway\n    test_df = test_df.drop('timestamp', axis=1)\n\n# Ensure all features are float32 for GPU efficiency\nfor col in features:\n    train_df[col] = train_df[col].astype(np.float32)\n    test_df[col] = test_df[col].astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T02:07:35.605312Z","iopub.execute_input":"2025-07-21T02:07:35.605672Z","iopub.status.idle":"2025-07-21T02:08:22.147569Z","shell.execute_reply.started":"2025-07-21T02:07:35.605648Z","shell.execute_reply":"2025-07-21T02:08:22.146805Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Feature Engineering","metadata":{}},{"cell_type":"code","source":"# Simple feature engineering examples (can be expanded)\n# Note: cuDF handles many pandas-like operations\n# If you have specific domain knowledge about these 'X' features, you can create more tailored ones.\n\n# Example: Mean and Std Dev across related features (if X1-X10 represent a group)\n# This is a general example and might need adjustment based on feature relationships\n# For instance, if X1-X5 are bid-related, X6-X10 are ask-related.\n# For simplicity, let's create a few interaction terms and basic statistics.\n\n# Note: These operations can be memory intensive, be selective.\n# Using CuPy for complex numpy-like operations on GPU arrays can be beneficial.\n\ndef feature_engineer(df, features_list):\n    # Convert to CuPy array for faster operations if df is cuDF\n    if isinstance(df, cudf.DataFrame):\n        df_cp = df[features_list].to_cupy()\n    else: # If using pandas\n        df_cp = cp.asarray(df[features_list].values)\n\n    # Example: Simple sum and mean of all X features\n    df['sum_X'] = cp.sum(df_cp, axis=1)\n    df['mean_X'] = cp.mean(df_cp, axis=1)\n    df['std_X'] = cp.std(df_cp, axis=1)\n    \n    # You can add more complex interactions here if RAM permits\n    # e.g., ratios, differences, polynomial features for selected important features\n    # df['X1_div_X2'] = df['X1'] / (df['X2'] + 1e-6) # Avoid division by zero\n\n    # Convert back to cuDF if originally cuDF, or to pandas if originally pandas\n    if isinstance(df, cudf.DataFrame):\n        df['sum_X'] = df['sum_X'].astype(np.float32)\n        df['mean_X'] = df['mean_X'].astype(np.float32)\n        df['std_X'] = df['std_X'].astype(np.float32)\n    else:\n        df['sum_X'] = df['sum_X'].get().astype(np.float32) # .get() to move from CuPy to NumPy\n        df['mean_X'] = df['mean_X'].get().astype(np.float32)\n        df['std_X'] = df['std_X'].get().astype(np.float32)\n\n    return df\n\ntrain_df = feature_engineer(train_df, features)\ntest_df = feature_engineer(test_df, features)\n\n# Update features list to include new engineered features\nfeatures.extend(['sum_X', 'mean_X', 'std_X'])\n\nprint(\"Feature Engineering complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T02:08:47.045594Z","iopub.execute_input":"2025-07-21T02:08:47.045933Z","iopub.status.idle":"2025-07-21T02:08:48.448704Z","shell.execute_reply.started":"2025-07-21T02:08:47.045910Z","shell.execute_reply":"2025-07-21T02:08:48.448085Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Scaling and Transformation","metadata":{}},{"cell_type":"code","source":"# Use StandardScaler for normalization. QuantileTransformer can also be effective.\n# For large datasets, it's common to fit the scaler on a subset or in batches\n# if RAM is very limited, but for numerical features, standard scaling is usually fine.\n# Note: scikit-learn's scalers are CPU-based. If using cuDF, convert to NumPy then scale, or use RAPIDS equivalents.\n\nX_train = train_df[features].to_pandas().values if isinstance(train_df, cudf.DataFrame) else train_df[features].values\ny_train = train_df[target].to_pandas().values if isinstance(train_df, cudf.DataFrame) else train_df[target].values\nX_test = test_df[features].to_pandas().values if isinstance(test_df, cudf.DataFrame) else test_df[features].values\n\n# Free up memory\ndel train_df, test_df\ngc.collect()\n\nprint(\"Data converted to NumPy arrays.\")\n\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n\n# Convert back to CuPy arrays for GPU processing with PyTorch\nX_train_scaled = cp.asarray(X_train_scaled, dtype=cp.float32)\nX_test_scaled = cp.asarray(X_test_scaled, dtype=cp.float32)\ny_train = cp.asarray(y_train, dtype=cp.float32)\n\nprint(\"Data scaled and converted to CuPy arrays.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T02:09:01.389834Z","iopub.execute_input":"2025-07-21T02:09:01.390155Z","iopub.status.idle":"2025-07-21T02:09:23.589304Z","shell.execute_reply.started":"2025-07-21T02:09:01.390123Z","shell.execute_reply":"2025-07-21T02:09:23.588542Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Advanced Model: Deep Neural Network (DNN) with PyTorch","metadata":{}},{"cell_type":"code","source":"# Check for GPU availability\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# Custom Dataset for efficient loading\nclass CryptoDataset(Dataset):\n    def __init__(self, features, labels=None):\n        # Convert CuPy arrays to PyTorch tensors and move to device\n        self.features = torch.from_numpy(cp.asnumpy(features)).to(device)\n        self.labels = torch.from_numpy(cp.asnumpy(labels)).to(device) if labels is not None else None\n\n    def __len__(self):\n        return len(self.features)\n\n    def __getitem__(self, idx):\n        if self.labels is not None:\n            return self.features[idx], self.labels[idx]\n        return self.features[idx]\n\n# Define the Neural Network Architecture\nclass CryptoPredictor(nn.Module):\n    def __init__(self, input_dim, hidden_dims, output_dim=1):\n        super(CryptoPredictor, self).__init__()\n        layers = []\n        # Input layer\n        layers.append(nn.Linear(input_dim, hidden_dims[0]))\n        layers.append(nn.ReLU())\n        layers.append(nn.BatchNorm1d(hidden_dims[0]))\n        layers.append(nn.Dropout(0.2)) # Regularization\n\n        # Hidden layers\n        for i in range(len(hidden_dims) - 1):\n            layers.append(nn.Linear(hidden_dims[i], hidden_dims[i+1]))\n            layers.append(nn.ReLU())\n            layers.append(nn.BatchNorm1d(hidden_dims[i+1]))\n            layers.append(nn.Dropout(0.2)) # Regularization\n\n        # Output layer\n        layers.append(nn.Linear(hidden_dims[-1], output_dim))\n        \n        self.network = nn.Sequential(*layers)\n\n    def forward(self, x):\n        return self.network(x)\n\n# Model Training Function\ndef train_model(model, train_loader, val_loader, criterion, optimizer, num_epochs=10):\n    model.train()\n    best_val_corr = -1.0 # Pearson correlation as evaluation metric\n    \n    for epoch in range(num_epochs):\n        for inputs, labels in train_loader:\n            inputs = inputs.to(device)\n            labels = labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(inputs).squeeze()\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n\n        # Validation phase\n        model.eval()\n        val_preds = []\n        val_true = []\n        with torch.no_grad():\n            for inputs, labels in val_loader:\n                inputs = inputs.to(device)\n                labels = labels.to(device)\n                outputs = model(inputs).squeeze()\n                val_preds.extend(outputs.cpu().numpy())\n                val_true.extend(labels.cpu().numpy())\n        \n        # Calculate Pearson correlation\n        val_corr, _ = pearsonr(val_true, val_preds)\n        print(f\"Epoch {epoch+1}/{num_epochs}, Loss: {loss.item():.4f}, Val Pearson Corr: {val_corr:.4f}\")\n\n        # Save best model\n        if val_corr > best_val_corr:\n            best_val_corr = val_corr\n            torch.save(model.state_dict(), \"best_model.pth\")\n            print(f\"New best model saved with correlation: {best_val_corr:.4f}\")\n        \n        model.train()\n    return model\n\n# Hyperparameters\ninput_dim = X_train_scaled.shape[1]\nhidden_dims = [512, 256, 128] # Example: Adjust based on experimentation and RAM\noutput_dim = 1\nlearning_rate = 0.001\nbatch_size = 2048 # Larger batch size leverages GPU more but consumes more RAM\nnum_epochs = 15 # Adjust based on convergence\n\n# K-Fold Cross-Validation for robust evaluation and prediction\nN_SPLITS = 5\nkf = KFold(n_splits=N_SPLITS, shuffle=True, random_state=42)\n\noof_predictions = cp.zeros(len(y_train), dtype=cp.float32)\ntest_predictions = cp.zeros((N_SPLITS, len(X_test_scaled)), dtype=cp.float32)\n\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_train_scaled)):\n    print(f\"\\n--- Fold {fold+1}/{N_SPLITS} ---\")\n    \n    X_train_fold, X_val_fold = X_train_scaled[train_idx], X_train_scaled[val_idx]\n    y_train_fold, y_val_fold = y_train[train_idx], y_train[val_idx]\n\n    train_dataset = CryptoDataset(X_train_fold, y_train_fold)\n    val_dataset = CryptoDataset(X_val_fold, y_val_fold)\n    test_dataset = CryptoDataset(X_test_scaled) # No labels for test set\n\n    train_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=0) # num_workers=0 for Kaggle GPU\n    val_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False, num_workers=0)\n    test_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=0)\n\n    model = CryptoPredictor(input_dim, hidden_dims).to(device)\n    # Optionally use float16 for reduced memory usage (requires compatible GPU and PyTorch version)\n    # model.half() \n    \n    criterion = nn.MSELoss() # Or HuberLoss for robustness to outliers\n    optimizer = optim.Adam(model.parameters(), lr=learning_rate)\n    \n    trained_model = train_model(model, train_loader, val_loader, criterion, optimizer, num_epochs)\n\n    # Load the best model weights for prediction\n    trained_model.load_state_dict(torch.load(\"best_model.pth\"))\n    trained_model.eval() # Set to evaluation mode\n\n    fold_test_preds = []\n    with torch.no_grad():\n        for inputs in test_loader:\n            inputs = inputs.to(device)\n            # if model.network[0].weight.dtype == torch.float16:\n            #    inputs = inputs.half() # Match input type if model is half()\n            outputs = trained_model(inputs).squeeze()\n            fold_test_preds.extend(outputs.cpu().numpy())\n    \n    test_predictions[fold] = cp.asarray(fold_test_preds, dtype=cp.float32)\n\n    # OOF predictions\n    fold_oof_preds = []\n    with torch.no_grad():\n        for inputs, _ in val_loader: # Use val_loader for OOF predictions\n            inputs = inputs.to(device)\n            # if model.network[0].weight.dtype == torch.float16:\n            #    inputs = inputs.half()\n            outputs = trained_model(inputs).squeeze()\n            fold_oof_preds.extend(outputs.cpu().numpy())\n    oof_predictions[val_idx] = cp.asarray(fold_oof_preds, dtype=cp.float32)\n\nprint(\"\\n--- Training Complete ---\")\n\n# Aggregate OOF predictions\noof_corr, _ = pearsonr(cp.asnumpy(y_train), cp.asnumpy(oof_predictions))\nprint(f\"Overall OOF Pearson Correlation: {oof_corr:.4f}\")\n\n# Average test predictions across folds for final submission\nfinal_test_predictions = cp.mean(test_predictions, axis=0)\n\nprint(\"Final test predictions aggregated.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T02:09:35.718310Z","iopub.execute_input":"2025-07-21T02:09:35.718610Z","iopub.status.idle":"2025-07-21T02:17:06.315134Z","shell.execute_reply.started":"2025-07-21T02:09:35.718577Z","shell.execute_reply":"2025-07-21T02:17:06.314407Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Submission","metadata":{}},{"cell_type":"code","source":"# Load sample submission to get the correct format\nsample_submission_df = pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")\n\n# Ensure test_predictions is a NumPy array for pandas DataFrame creation\nfinal_test_predictions_np = cp.asnumpy(final_test_predictions)\n\n# Create submission DataFrame\nsubmission_df = pd.DataFrame({'ID': sample_submission_df['ID'], 'prediction': final_test_predictions_np})\n\n# Ensure the label column has the correct data type (float32 if required)\nsubmission_df['prediction'] = submission_df['prediction'].astype(np.float32)\n\n# Save the submission file\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint(\"Submission file 'submission.csv' created successfully!\")\nprint(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T02:20:05.874924Z","iopub.execute_input":"2025-07-21T02:20:05.875258Z","iopub.status.idle":"2025-07-21T02:20:06.857440Z","shell.execute_reply.started":"2025-07-21T02:20:05.875226Z","shell.execute_reply":"2025-07-21T02:20:06.856297Z"}},"outputs":[],"execution_count":null}]}