{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-31T16:00:44.857084Z","iopub.execute_input":"2025-05-31T16:00:44.857713Z","iopub.status.idle":"2025-05-31T16:00:46.638530Z","shell.execute_reply.started":"2025-05-31T16:00:44.857686Z","shell.execute_reply":"2025-05-31T16:00:46.637779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.metrics import mean_squared_error\nfrom lightgbm import LGBMRegressor\nimport gc # Import garbage collection module\n\n# Function to reduce memory usage by downcasting numerical columns\ndef reduce_mem_usage(df, verbose=True):\n    \"\"\"\n    Iterates through all numerical columns of a dataframe and modifies the data type\n    to reduce memory usage.\n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    if verbose:\n        print(f'Memory usage of dataframe is {start_mem:.2f} MB')\n\n    for col in df.columns:\n        col_type = df[col].dtype\n\n        if col_type != object: # Only process numerical columns\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else: # float\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64) # Keep as float64 if it doesn't fit in float32\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose:\n        print(f'Memory usage after optimization is: {end_mem:.2f} MB')\n        print(f'Decreased by {100 * (start_mem - end_mem) / start_mem:.1f}%')\n    return df\n\n# Load and explore data\n# Using dummy paths for demonstration; replace with actual paths if running locally\ntry:\n    train_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\n    test_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nexcept FileNotFoundError:\n    print(\"Dataset files not found. Creating dummy dataframes for demonstration.\")\n    # Create dummy dataframes if files are not found (for local testing without the dataset)\n    n_train_rows = 10000 # Increased dummy data size for better memory testing\n    n_test_rows = 5000\n    num_features = 890\n\n    data_train = {f'X{i}': np.random.rand(n_train_rows) for i in range(1, num_features + 1)}\n    data_train.update({\n        'timestamp': pd.to_datetime(pd.date_range(start='2023-03-01', periods=n_train_rows, freq='min')),\n        'bid_qty': np.random.rand(n_train_rows) * 100,\n        'ask_qty': np.random.rand(n_train_rows) * 100,\n        'buy_qty': np.random.rand(n_train_rows) * 200,\n        'sell_qty': np.random.rand(n_train_rows) * 200,\n        'volume': np.random.rand(n_train_rows) * 500,\n        'label': np.random.randn(n_train_rows)\n    })\n    train_df = pd.DataFrame(data_train).set_index('timestamp')\n\n    data_test = {f'X{i}': np.random.rand(n_test_rows) for i in range(1, num_features + 1)}\n    data_test.update({\n        'timestamp': pd.to_datetime(pd.date_range(start=train_df.index[-1] + pd.Timedelta(minutes=1), periods=n_test_rows, freq='min')),\n        'bid_qty': np.random.rand(n_test_rows) * 100,\n        'ask_qty': np.random.rand(n_test_rows) * 100,\n        'buy_qty': np.random.rand(n_test_rows) * 200,\n        'sell_qty': np.random.rand(n_test_rows) * 200,\n        'volume': np.random.rand(n_test_rows) * 500,\n        'label': np.random.randn(n_test_rows) # Test set might have a 'label' column, but it's usually not used for prediction\n    })\n    test_df = pd.DataFrame(data_test).set_index('timestamp')\n\n    # Introduce some NaNs and infs for testing the imputation logic\n    train_df.iloc[10:20, 5] = np.nan\n    train_df.iloc[30, 10] = np.inf\n    train_df.iloc[40, 15] = -np.inf\n    train_df['all_nan_col'] = np.nan # Introduce an entirely NaN column to simulate the error condition\n    test_df['all_nan_col'] = np.nan # Ensure test_df also has this column for consistency\n\n\n# Apply memory reduction\nprint(\"Optimizing train_df memory usage...\")\ntrain_df = reduce_mem_usage(train_df)\nprint(\"Optimizing test_df memory usage...\")\ntest_df = reduce_mem_usage(test_df)\ngc.collect() # Explicitly free up memory after loading and optimizing\n\n# Make column names unique for train_df\nif not train_df.columns.is_unique:\n    train_df.columns = [f\"{col}_{i}\" if train_df.columns.duplicated()[i] else col\n                        for i, col in enumerate(train_df.columns)]\n    print(\"Duplicate columns in train_df renamed.\")\n\n# Make column names unique for test_df\nif not test_df.columns.is_unique:\n    test_df.columns = [f\"{col}_{i}\" if test_df.columns.duplicated()[i] else col\n                       for i, col in enumerate(test_df.columns)]\n    print(\"Duplicate columns in test_df renamed.\")\n\n# Basic data exploration\nprint(\"\\nData Head:\\n\", train_df.head())\nprint(\"Missing Values (before inf handling):\\n\", train_df.isnull().sum().head()) # Print head to avoid excessive output\nprint(\"Label Distribution:\\n\", train_df['label'].describe())\n\n# Identify numerical columns for handling inf values and imputation\n# This needs to be done carefully to ensure 'timestamp' and 'label' are excluded\n# and that the same set of columns is used consistently.\nall_numerical_cols = train_df.select_dtypes(include=[np.number]).columns\n\n# Handle infinite values in numerical columns for both train and test\n# This step is still necessary even after downcasting, as inf/nan are distinct concepts.\ntrain_df[all_numerical_cols] = train_df[all_numerical_cols].replace([np.inf, -np.inf], np.nan)\ntest_df[all_numerical_cols] = test_df[all_numerical_cols].replace([np.inf, -np.inf], np.nan)\ngc.collect() # Collect garbage after replacing inf values\n\n# Check target 'label' for issues\nprint(\"Missing values in label:\", train_df['label'].isnull().sum())\nprint(\"Infinite values in label:\", np.isinf(train_df['label']).sum())\n\n# Define initial feature columns (excluding 'timestamp' and 'label')\n# Note: 'timestamp' is the index, so it won't be in columns unless reset_index() is called.\n# We explicitly exclude 'label' here.\ninitial_feature_cols = [col for col in all_numerical_cols if col not in ['label']]\n\n# Identify columns that are entirely NaN after handling inf values in the initial feature set.\n# These are the columns that SimpleImputer will drop if strategy is 'mean', 'median', 'most_frequent'.\nall_nan_in_features = train_df[initial_feature_cols].columns[train_df[initial_feature_cols].isnull().all()].tolist()\nprint(f\"Columns that are entirely NaN and will be dropped by imputer: {all_nan_in_features}\")\n\n# Filter feature_cols to exclude these all-NaN columns.\n# This list will be the actual columns processed by the imputer and used for training.\nimputer_feature_cols = [col for col in initial_feature_cols if col not in all_nan_in_features]\n\n# Debugging: Check shapes before imputation\nprint(\"Number of initial_feature_cols:\", len(initial_feature_cols))\nprint(\"Number of imputer_feature_cols (after dropping all-NaNs):\", len(imputer_feature_cols))\nprint(\"Shape of train_df[imputer_feature_cols]:\", train_df[imputer_feature_cols].shape)\n\n# Preprocessing: Impute missing values in feature columns\nimputer = SimpleImputer(strategy='mean')\nimputed_array = imputer.fit_transform(train_df[imputer_feature_cols])\n\n# Debugging: Check shape after imputation\nprint(\"Shape of imputed_array:\", imputed_array.shape)\n\n# Assign imputed values back to DataFrame using the filtered list of columns\ntrain_df[imputer_feature_cols] = imputed_array\ngc.collect() # Collect garbage after imputation\n\n# Feature engineering for train_df\ntrain_df['bid_ask_spread'] = train_df['ask_qty'] - train_df['bid_qty']\ntrain_df['buy_sell_ratio'] = train_df['buy_qty'] / (train_df['sell_qty'] + 1e-6)\n\n# Define the final set of features for training (X)\n# This includes the imputed numerical features and the newly engineered features.\nfinal_train_features = imputer_feature_cols + ['bid_ask_spread', 'buy_sell_ratio']\n\n# Split features and target\nX = train_df[final_train_features]\ny = train_df['label']\n\n# Delete the original train_df to free up memory before training the model\ndel train_df\ngc.collect()\n\n# Training and validation with TimeSeriesSplit\ntscv = TimeSeriesSplit(n_splits=5)\nfor fold, (train_idx, val_idx) in enumerate(tscv.split(X)):\n    X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n    \n    print(f\"\\n--- Fold {fold+1} ---\")\n    print(f\"Train data shape: {X_train.shape}, Validation data shape: {X_val.shape}\")\n\n    # Train LightGBM with GPU support\n    # Added 'device=\"gpu\"' to enable GPU training\n    model = LGBMRegressor(n_estimators=100, learning_rate=0.1, num_leaves=31, random_state=42, n_jobs=-1, device=\"gpu\")\n    model.fit(X_train, y_train)\n    \n    # Validate\n    y_pred = model.predict(X_val)\n    mse = mean_squared_error(y_val, y_pred)\n    print(f\"Validation MSE: {mse:.4f}\")\n\n    # Delete fold-specific data to free memory\n    del X_train, X_val, y_train, y_val\n    gc.collect()\n\n# Train on full data for final model\nprint(\"\\nTraining final model on full dataset...\")\n# Ensure GPU is used for the final training as well\nmodel = LGBMRegressor(n_estimators=100, learning_rate=0.1, num_leaves=31, random_state=42, n_jobs=-1, device=\"gpu\")\nmodel.fit(X, y)\nprint(\"Final model trained.\")\n\n# Delete X and y to free up memory before processing test set\ndel X, y\ngc.collect()\n\n# Process test set - apply the same preprocessing steps as train_df\n# Note: all_numerical_cols for test_df was already handled above.\n\n# Apply imputation to test_df using the *fitted* imputer and the same feature columns\n# The imputer was fitted on imputer_feature_cols from train_df.\ntest_df[imputer_feature_cols] = imputer.transform(test_df[imputer_feature_cols])\ngc.collect() # Collect garbage after test imputation\n\n# Feature engineering for test_df\ntest_df['bid_ask_spread'] = test_df['ask_qty'] - test_df['bid_qty']\ntest_df['buy_sell_ratio'] = test_df['buy_qty'] / (test_df['sell_qty'] + 1e-6)\n\n# Prepare X_test with the exact same columns as X (final_train_features)\nX_test = test_df[final_train_features]\n\n# Predict on test set\nprint(\"Predicting on test set...\")\ny_test_pred = model.predict(X_test)\nprint(\"Prediction complete.\")\n\n# Create submission\n# Assuming 'timestamp' is the index in test_df and needs to be a column in submission.\nsubmission = pd.DataFrame({'id': test_df.index, 'label': y_test_pred})\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created: submission.csv\")\n\n# Delete test_df and X_test to free up memory\ndel test_df, X_test\ngc.collect()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T16:26:17.027892Z","iopub.execute_input":"2025-05-31T16:26:17.028188Z","iopub.status.idle":"2025-05-31T16:31:55.856500Z","shell.execute_reply.started":"2025-05-31T16:26:17.028168Z","shell.execute_reply":"2025-05-31T16:31:55.855723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"################ create changes in new cells  #############","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T16:44:16.695744Z","iopub.execute_input":"2025-05-31T16:44:16.696513Z","iopub.status.idle":"2025-05-31T16:44:16.700913Z","shell.execute_reply.started":"2025-05-31T16:44:16.696475Z","shell.execute_reply":"2025-05-31T16:44:16.699976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}