{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":12451511,"sourceType":"datasetVersion","datasetId":7854590}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import Ridge, Lasso, ElasticNet\nimport lightgbm as lgb\nimport xgboost as xgb\nimport os\nimport gc\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import mean_squared_error\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\nimport torch\nprint(f\"CUDA available: {torch.cuda.is_available()}\")\nprint(f\"CUDA device count: {torch.cuda.device_count()}\")\nif torch.cuda.is_available():\n    print(f\"Current CUDA device: {torch.cuda.current_device()}\")\n    print(f\"CUDA device name: {torch.cuda.get_device_name()}\")\n\n\n\n\ndef optimize_memory(df, verbose=True):\n    \"\"\"\n    Optimize memory usage by downcasting numeric types where possible.\n    \"\"\"\n    if verbose:\n        start_mem = df.memory_usage().sum() / 1024**2\n        print(f'Memory usage before optimization: {start_mem:.2f} MB')\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != 'object':\n            c_min = df[col].min()\n            c_max = df[col].max()\n            \n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n    \n    if verbose:\n        end_mem = df.memory_usage().sum() / 1024**2\n        print(f'Memory usage after optimization: {end_mem:.2f} MB')\n        print(f'Decreased by {100 * (start_mem - end_mem) / start_mem:.1f}%')\n    \n    return df\n\ndef preprocess_data_chunked(raw_df, chunk_size=10):\n    \"\"\"\n    Preprocess data with memory-efficient chunked lag creation.\n    \"\"\"\n    assert len(raw_df.shape) == 2\n\n    y = raw_df['label'].to_numpy().astype(np.float32) if 'label' in raw_df.columns else None\n\n    # Original features (keeping your existing selection)\n    cols = [\n        'X363', 'X405', 'X321',\n        'X175', 'X179', 'X137', 'X197', 'X22', 'X40', 'X181',\n        'X28', 'X169', 'X198', 'X173',\n        'X338', 'X288', 'X385', 'X344', 'X427', 'X587', 'X450',\n        'X97', 'X52', 'X444',\n        'X598', 'X379', 'X696', 'X297', 'X138',\n        'X572', 'X343', 'X586', 'X466', 'X438', 'X452', 'X459',\n        'X435', 'X386', 'X55', 'X341', 'X683', 'X428', 'X605',\n        'X445', 'X272', 'X180', 'X593', 'X680',\n        'X686', 'X692', 'X695',\n        \"X603\", \"X674\", \"X421\", \"X333\",\n        \"X415\", \"X345\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\",\n        'X298', 'X45', 'X46', 'X39', 'X752', 'X759', 'X41', 'X42',\n        \"buy_qty\", \"sell_qty\", \"volume\",\n        \"bid_qty\", \"ask_qty\",\n        'X758', 'X296', 'X611', 'X780', 'X451', 'X25', 'X591',\n    ]\n    \n    # Remove duplicates while preserving order\n    cols = list(dict.fromkeys(cols))\n    available_cols = [col for col in cols if col in raw_df.columns]\n    \n    print(f\"Using {len(available_cols)} features\")\n\n    # Select and optimize base features\n    df = raw_df[available_cols].copy()\n    df = optimize_memory(df, verbose=True)\n    assert df.isna().sum().sum() == 0\n\n    # Extended lag features (keeping your existing lags)\n    lag_periods = [1, 3, 5, 6, 7, 8, 9, 12, 15, 18, 20, 30, 120, 150, 365]\n    \n    print(\"Creating lagged features in chunks...\")\n    result_df = df.copy()\n    \n    for i in range(0, len(lag_periods), chunk_size):\n        chunk_lags = lag_periods[i:i+chunk_size]\n        print(f\"  Processing lags: {chunk_lags}\")\n        \n        chunk_dfs = []\n        for lag in chunk_lags:\n            lagged = df.shift(-lag).add_suffix(f'_lead_{lag}')\n            lagged = lagged.fillna(0.0).astype(np.float32)\n            chunk_dfs.append(lagged)\n        \n        if chunk_dfs:\n            chunk_combined = pd.concat(chunk_dfs, axis=1)\n            result_df = pd.concat([result_df, chunk_combined], axis=1)\n            del chunk_dfs, chunk_combined\n            gc.collect()\n    \n    result_df = optimize_memory(result_df, verbose=True)\n    \n    if y is not None:\n        assert 'label' not in result_df.columns\n        assert raw_df.shape[0] == result_df.shape[0]\n        assert result_df.isna().sum().sum() == 0\n        assert result_df.shape[0] == y.shape[0]\n    \n    print(f\"Final feature count: {result_df.shape[1]}\")\n    \n    return result_df, y\n\nclass FastEnsemble:\n    def __init__(self):\n        self.models = {}\n        self.weights = {}\n        self.cv_scores = {}\n        \n    def fit(self, X, y, cv_folds=3):\n        print(\"Training Fast Ensemble with GPU acceleration...\")\n        \n        # GPU-optimized models\n        self.models = {\n            # 'ridge': Ridge(alpha=0.5, copy_X=False, solver='lsqr'),\n            # 'ridge_heavy': Ridge(alpha=5.0, copy_X=False, solver='lsqr'),\n            # 'lasso': Lasso(alpha=0.05, copy_X=False, max_iter=500),\n            # 'elastic': ElasticNet(alpha=0.3, l1_ratio=0.7, copy_X=False, max_iter=500),\n            'lgb_1': lgb.LGBMRegressor(\n                n_estimators=100,\n                learning_rate=0.1,\n                max_depth=6,\n                num_leaves=31,\n                subsample=0.85,\n                colsample_bytree=0.85,\n                random_state=42,\n                verbosity=-1,\n                n_jobs=-1,\n                device='gpu',\n                gpu_platform_id=0,\n                gpu_device_id=0,\n                reg_alpha=0.1,\n                reg_lambda=0.1\n            ),\n            'lgb_2': lgb.LGBMRegressor(\n                n_estimators=80,\n                learning_rate=0.15,\n                max_depth=5,\n                num_leaves=20,\n                subsample=0.8,\n                colsample_bytree=0.9,\n                random_state=123,\n                verbosity=-1,\n                n_jobs=-1,\n                device='gpu',\n                gpu_platform_id=0,\n                gpu_device_id=0,\n                reg_alpha=0.05,\n                reg_lambda=0.05\n            ),\n            'xgb_1': xgb.XGBRegressor(\n                n_estimators=100,\n                learning_rate=0.1,\n                max_depth=6,\n                subsample=0.85,\n                colsample_bytree=0.85,\n                random_state=42,\n                verbosity=0,\n                n_jobs=-1,\n                tree_method='gpu_hist',\n                gpu_id=0,\n                eval_metric='rmse',\n                reg_alpha=0.1,\n                reg_lambda=0.1\n            ),\n            'xgb_2': xgb.XGBRegressor(\n                n_estimators=80,\n                learning_rate=0.15,\n                max_depth=5,\n                subsample=0.8,\n                colsample_bytree=0.9,\n                random_state=123,\n                verbosity=0,\n                n_jobs=-1,\n                tree_method='gpu_hist',\n                gpu_id=0,\n                eval_metric='rmse',\n                reg_alpha=0.05,\n                reg_lambda=0.05\n            )\n        \n        }\n        \n        # Cross-validation for model evaluation\n        kf = KFold(n_splits=cv_folds, shuffle=True, random_state=42)\n        oof_predictions = np.zeros((len(X), len(self.models)))\n        \n        for fold, (train_idx, val_idx) in enumerate(kf.split(X)):\n            print(f\"  Fold {fold + 1}/{cv_folds}\")\n            \n            X_train_fold = X.iloc[train_idx]\n            y_train_fold = y[train_idx]\n            X_val_fold = X.iloc[val_idx]\n            y_val_fold = y[val_idx]\n            \n            for model_idx, (name, model) in enumerate(self.models.items()):\n                print(f\"    Training {name}...\")\n                \n                # Clone model for this fold\n                if name == 'lgb':\n                    fold_model = lgb.LGBMRegressor(**model.get_params())\n                elif name == 'xgb':\n                    fold_model = xgb.XGBRegressor(**model.get_params())\n                else:\n                    fold_model = type(model)(**model.get_params())\n                \n                # Fit and predict\n                fold_model.fit(X_train_fold, y_train_fold)\n                oof_predictions[val_idx, model_idx] = fold_model.predict(X_val_fold)\n        \n        # Calculate CV scores for each model\n        for model_idx, name in enumerate(self.models.keys()):\n            score = mean_squared_error(y, oof_predictions[:, model_idx])\n            self.cv_scores[name] = score\n            print(f\"  {name} CV MSE: {score:.6f}\")\n        \n        # Calculate optimal weights using inverse MSE\n        mse_values = np.array(list(self.cv_scores.values()))\n        inverse_mse = 1.0 / (mse_values + 1e-8)  # Add small epsilon to avoid division by zero\n        self.weights = inverse_mse / inverse_mse.sum()\n        \n        print(\"  Optimal weights:\")\n        for i, (name, weight) in enumerate(zip(self.models.keys(), self.weights)):\n            print(f\"    {name}: {weight:.4f}\")\n        \n        # Fit final models on full data\n        print(\"  Fitting final models on full data...\")\n        for name, model in self.models.items():\n            print(f\"    Fitting {name}...\")\n            model.fit(X, y)\n        \n        # Calculate ensemble CV score\n        ensemble_pred = np.average(oof_predictions, weights=self.weights, axis=1)\n        ensemble_score = mean_squared_error(y, ensemble_pred)\n        print(f\"  Ensemble CV MSE: {ensemble_score:.6f}\")\n        \n        return self\n    \n    def predict(self, X, batch_size=50000):\n        \"\"\"\n        Make predictions with the ensemble.\n        \"\"\"\n        print(\"Making ensemble predictions...\")\n        n_samples = X.shape[0]\n        predictions = np.zeros((n_samples, len(self.models)))\n        \n        # Process in batches to save memory\n        for i in range(0, n_samples, batch_size):\n            end_idx = min(i + batch_size, n_samples)\n            X_batch = X.iloc[i:end_idx]\n            \n            for model_idx, (name, model) in enumerate(self.models.items()):\n                predictions[i:end_idx, model_idx] = model.predict(X_batch)\n            \n            if i // batch_size % 10 == 0:  # Print every 10 batches\n                print(f\"  Processed {end_idx}/{n_samples} samples\")\n        \n        # Weighted average\n        final_predictions = np.average(predictions, weights=self.weights, axis=1)\n        return final_predictions.astype(np.float32)\n\n\n# Set memory-efficient options for pandas\npd.options.mode.chained_assignment = None\npd.options.display.max_columns = None\n\n# Load and preprocess training data\nprint(\"Loading training data...\")\ntrain_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\n\nprint(f\"\\nTotal columns in training data: {len(train_df.columns)}\")\nprint(f\"Sample columns: {list(train_df.columns[:20])}\")\n\nprint(\"\\nOptimizing memory for raw training data...\")\ntrain_df = optimize_memory(train_df, verbose=True)\n\nX_train, y_train = preprocess_data_chunked(train_df, chunk_size=10)\n\ndel train_df\ngc.collect()\n\nprint(f\"\\nTraining data shape: X={X_train.shape}, y={y_train.shape}\")\n\n# Train the ensemble model\nensemble = FastEnsemble()\nensemble.fit(X_train, y_train, cv_folds=3)\n\n# Feature importance from best performing model\nbest_model_name = min(ensemble.cv_scores, key=ensemble.cv_scores.get)\nbest_model = ensemble.models[best_model_name]\n\nprint(f\"\\nBest single model: {best_model_name}\")\n\n# Plot feature importance for interpretable models\nif hasattr(best_model, 'coef_'):\n    print(\"\\nPlotting feature importance...\")\n    coef_series = pd.Series(best_model.coef_, index=X_train.columns).abs().sort_values(ascending=False)\n    top_features = coef_series.head(50)  # Reduced for faster plotting\n    \n    plt.figure(figsize=(12, 15))\n    top_features.sort_values().plot(kind='barh')\n    plt.title(f'Top 50 Feature Coefficients ({best_model_name})')\n    plt.xlabel('Coefficient Magnitude')\n    plt.tight_layout()\n    plt.show()\n    \n    print(\"\\nTop 20 most important features:\")\n    for i, (feat, coef) in enumerate(coef_series.head(20).items(), 1):\n        print(f\"{i:2d}. {feat:30s} {coef:.6f}\")\n\n# Clean up training data\ndel X_train, y_train\ngc.collect()\n\n# Load and process test data\nprint(\"\\nLoading test data...\")\ntest_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\nprint(\"\\nOptimizing memory for raw test data...\")\ntest_df = optimize_memory(test_df, verbose=True)\n\n# Timestamp reconstruction (keeping your existing logic)\ntimestamp_recon_path = '/kaggle/input/closest-rows/closest_rows.csv'\nuse_timestamp_reconstruction = os.path.exists(timestamp_recon_path)\n\nif use_timestamp_reconstruction:\n    print(\"Found timestamp reconstruction file, loading...\")\n    t = pd.Series(pd.read_csv(timestamp_recon_path)['0'].to_numpy())\n    assert t.shape == (test_df.shape[0],)\n    print('Reconstructed timestamps share:', len(t[t >= 0]) / len(t))\n\n    # Process timestamp reconstruction\n    t -= 10080\n    t[t < 0] = 538149\n    t = t.sort_values()\n    t[t <= len(t)] = np.arange(t[t <= len(t)].shape[0])\n    t = t.sort_index()\n    t = pd.Series(np.arange(538150), index=t.to_numpy()).sort_index()\n\n    # Sort test dataset\n    test_df = test_df.iloc[t.to_numpy()]\nelse:\n    print(\"WARNING: Timestamp reconstruction file not found!\")\n    t = pd.Series(np.arange(len(test_df)))\n\n# Preprocess test data\nprint(\"\\nPreprocessing test data...\")\nX_test, _ = preprocess_data_chunked(test_df, chunk_size=10)\n\ndel test_df\ngc.collect()\n\nprint(f\"Test data shape: {X_test.shape}\")\n\n# Make ensemble predictions\ny_pred = ensemble.predict(X_test, batch_size=50000)\n\ndel X_test\ngc.collect()\n\n# Display prediction statistics\nprint(\"\\nPrediction statistics:\")\nprint(pd.Series(y_pred).describe())\n\n# Plot results (reduced plotting for speed)\nplt.figure(figsize=(16, 8))\n\nplt.subplot(2, 2, 1)\nplt.plot(np.cumsum(y_pred))\nplt.title('Cumulative Predictions')\nplt.xlabel('Sample Index')\nplt.ylabel('Cumulative Sum')\nplt.grid(True, alpha=0.3)\n\nplt.subplot(2, 2, 2)\nplt.hist(y_pred, bins=50, alpha=0.7, edgecolor='black')\nplt.title('Prediction Distribution')\nplt.xlabel('Predicted Value')\nplt.ylabel('Frequency')\n\nplt.subplot(2, 2, 3)\nplt.plot(y_pred[:1000])\nplt.title('First 1000 Predictions')\nplt.xlabel('Sample Index')\nplt.ylabel('Predicted Value')\n\nplt.subplot(2, 2, 4)\n# Plot model weights\nnames = list(ensemble.models.keys())\nweights = ensemble.weights\nplt.bar(names, weights)\nplt.title('Ensemble Model Weights')\nplt.ylabel('Weight')\nplt.xticks(rotation=45)\n\nplt.tight_layout()\nplt.show()\n\n# Prepare submission\nprint(\"\\nPreparing submission...\")\nsubmission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n\nif use_timestamp_reconstruction:\n    submission = submission.iloc[t.to_numpy()]\n    submission['prediction'] = y_pred\n    submission = submission.sort_index()\nelse:\n    submission['prediction'] = y_pred\n\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission saved to 'submission.csv'\")\n\nprint(\"\\nSubmission preview:\")\nprint(submission.head())\nprint(f\"\\nSubmission shape: {submission.shape}\")\nprint(f\"Prediction range: [{submission['prediction'].min():.6f}, {submission['prediction'].max():.6f}]\")\n\n# Print ensemble summary\nprint(\"\\n\" + \"=\"*50)\nprint(\"ENSEMBLE SUMMARY\")\nprint(\"=\"*50)\nfor name, score in ensemble.cv_scores.items():\n    weight = ensemble.weights[list(ensemble.models.keys()).index(name)]\n    print(f\"{name:12s}: CV MSE = {score:.6f}, Weight = {weight:.4f}\")\n\ngc.collect()\nprint(\"\\nDone!\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-13T07:46:02.772177Z","iopub.execute_input":"2025-07-13T07:46:02.772932Z","iopub.status.idle":"2025-07-13T08:06:17.544306Z","shell.execute_reply.started":"2025-07-13T07:46:02.772904Z","shell.execute_reply":"2025-07-13T08:06:17.543296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load test data\nprint(\"\\nLoading test data...\")\ntest_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\n# Optimize memory for test data\nprint(\"\\nOptimizing memory for raw test data...\")\ntest_df = optimize_memory_gpu(test_df, verbose=True)\n\n# Try to load precomputed timestamp reconstruction data\ntimestamp_recon_path = '/kaggle/input/closest-rows/closest_rows.csv'\nuse_timestamp_reconstruction = os.path.exists(timestamp_recon_path)\n\nif use_timestamp_reconstruction:\n    print(\"Found timestamp reconstruction file, loading...\")\n    \n    # Load precomputed timestamp reconstruction data\n    t = pd.Series(pd.read_csv(timestamp_recon_path)['0'].to_numpy())\n    assert t.shape == (test_df.shape[0],)\n    print('Reconstructed timestamps share:', len(t[t >= 0]) / len(t))\n\n    # Visualize the reconstructed timestamps\n    plt.figure(figsize=(16, 4))\n    plt.plot(t.sort_values().to_numpy())\n    plt.title('Sorted Reconstructed Timestamps')\n    plt.show()\n\n    plt.figure(figsize=(16, 4))\n    plt.plot(t[t >= 0].sort_values().iloc[:1000].to_numpy())\n    plt.axhline(10080, color='r', linestyle='--')\n    plt.title('First 1000 Valid Reconstructed Timestamps')\n    plt.show()\n\n    # Process timestamp reconstruction\n    t -= 10080\n    t[t < 0] = 538149\n\n    t = t.sort_values()\n    t[t <= len(t)] = np.arange(t[t <= len(t)].shape[0])\n    t = t.sort_index()\n\n    t = pd.Series(np.arange(538150), index=t.to_numpy()).sort_index()\n\n    # Visualize test data before sorting\n    if 'X656' in test_df.columns:\n        plt.figure(figsize=(16, 4))\n        plt.plot(test_df['X656'].to_numpy())\n        plt.title('Test Data Feature X656 - Before Sorting')\n        plt.show()\n\n    # Sort test dataset by reconstructed time order\n    test_df = test_df.iloc[t.to_numpy()]\n\n    # Visualize test data after sorting\n    if 'X656' in test_df.columns:\n        plt.figure(figsize=(16, 4))\n        plt.plot(test_df['X656'].to_numpy())\n        plt.title('Test Data Feature X656 - After Sorting')\n        plt.show()\nelse:\n    print(\"WARNING: Timestamp reconstruction file not found!\")\n    print(f\"Expected path: {timestamp_recon_path}\")\n    print(\"Proceeding without timestamp reconstruction...\")\n    print(\"This may significantly impact model performance since lagged features assume temporal order.\")\n    \n    t = pd.Series(np.arange(len(test_df)))\n\n# Preprocess test data\n# ...existing code for loading and timestamp reconstruction...\n\n# Preprocess test data with enhanced features\nprint(\"\\nPreprocessing test data with enhanced features...\")\nX_test, _ = preprocess_data_chunked(test_df, chunk_size=10, create_advanced=True, top_features=top_features, enhanced=True)\n\n# Apply same feature selection\nX_test_selected = X_test[selected_features_list]\n\n# Clean up test dataframe\ndel test_df, X_test\ngpu_memory_cleanup()\ngc.collect()\n\nprint(f\"Test data shape after processing: {X_test_selected.shape}\")\n\n# Make predictions in smaller batches\nprint(\"\\nMaking predictions...\")\nbatch_size = 40000  # Reduced from 80000\nn_samples = X_test_selected.shape[0]\ny_pred = np.zeros(n_samples, dtype=np.float32)\n\nfor i in range(0, n_samples, batch_size):\n    end_idx = min(i + batch_size, n_samples)\n    print(f\"  Predicting batch {i//batch_size + 1}/{(n_samples + batch_size - 1)//batch_size}\")\n    \n    batch_data = X_test_selected.iloc[i:end_idx]\n    y_pred[i:end_idx] = final_model.predict(batch_data).astype(np.float32)\n    \n    # Cleanup every few batches\n    if (i // batch_size + 1) % 3 == 0:\n        aggressive_memory_cleanup()\n\n# Clean up test features\ndel X_test_selected\naggressive_memory_cleanup()\n\n# Display prediction statistics\nprint(\"\\nPrediction statistics:\")\nprint(pd.Series(y_pred).describe())\n\n# Plot cumulative predictions\nplt.figure(figsize=(16, 4))\nplt.plot(np.cumsum(y_pred))\nplt.title('Cumulative Predictions')\nplt.xlabel('Sample Index')\nplt.ylabel('Cumulative Sum')\nplt.grid(True, alpha=0.3)\nplt.show()\n\n# Plot prediction distribution\nplt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nplt.hist(y_pred, bins=100, alpha=0.7, edgecolor='black')\nplt.title('Prediction Distribution')\nplt.xlabel('Predicted Value')\nplt.ylabel('Frequency')\n\nplt.subplot(1, 2, 2)\nplt.plot(y_pred[:1000])\nplt.title('First 1000 Predictions')\nplt.xlabel('Sample Index')\nplt.ylabel('Predicted Value')\nplt.tight_layout()\nplt.show()\n\n# Prepare submission\nprint(\"\\nPreparing submission...\")\nsubmission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n\nif use_timestamp_reconstruction:\n    # Reorder submission to match original test order\n    submission = submission.iloc[t.to_numpy()]\n    submission['prediction'] = y_pred\n    submission = submission.sort_index()\nelse:\n    # If no timestamp reconstruction, just use predictions in order\n    submission['prediction'] = y_pred\n\n# Save submission\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission saved to 'submission.csv'\")\n\n# Display submission\nprint(\"\\nSubmission preview:\")\nprint(submission.head())\nprint(f\"\\nSubmission shape: {submission.shape}\")\nprint(f\"Prediction range: [{submission['prediction'].min():.6f}, {submission['prediction'].max():.6f}]\")\n\n# Final memory cleanup\ngc.collect()\nprint(\"\\nDone!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:20:28.964399Z","iopub.execute_input":"2025-07-12T18:20:28.964987Z","iopub.status.idle":"2025-07-12T18:21:09.067449Z","shell.execute_reply.started":"2025-07-12T18:20:28.964961Z","shell.execute_reply":"2025-07-12T18:21:09.066656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}