{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.metrics import mean_squared_error\nimport tensorflow as tf\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout, Layer, Input\nfrom tensorflow.keras import Model\nfrom lightgbm import LGBMRegressor\nimport pyarrow.parquet as pq\nimport gc\nimport os\n\n# Configure TensorFlow GPU settings\nphysical_devices = tf.config.list_physical_devices('GPU')\nif physical_devices:\n    try:\n        for gpu in physical_devices:\n            tf.config.experimental.set_memory_growth(gpu, True)\n        print(f\"Enabled memory growth for {len(physical_devices)} GPU(s)\")\n    except RuntimeError as e:\n        print(f\"Error setting GPU memory growth: {e}\")\nelse:\n    print(\"No GPU detected, using CPU\")\n\n# Custom attention layer\nclass AttentionLayer(Layer):\n    def __init__(self):\n        super(AttentionLayer, self).__init__()\n    \n    def build(self, input_shape):\n        self.W = self.add_weight(name='attention_weight', shape=(input_shape[-1], 1), initializer='random_normal', trainable=True)\n        self.b = self.add_weight(name='attention_bias', shape=(1,), initializer='zeros', trainable=True)\n        super(AttentionLayer, self).build(input_shape)\n    \n    def call(self, inputs):\n        e = tf.keras.backend.tanh(tf.keras.backend.dot(inputs, self.W) + self.b)\n        alpha = tf.keras.backend.softmax(e, axis=1)\n        context = inputs * alpha\n        return tf.keras.backend.sum(context, axis=1)\n\n# Memory optimization function\ndef reduce_mem_usage(df, verbose=True):\n    start_mem = df.memory_usage().sum() / 1024**2\n    if verbose:\n        print(f'Memory usage of dataframe is {start_mem:.2f} MB')\n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type != object and col_type != 'datetime64[ns]':\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                else:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                df[col] = df[col].astype(np.float32)\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose:\n        print(f'Memory usage after optimization is: {end_mem:.2f} MB')\n        print(f'Decreased by {100 * (start_mem - end_mem) / start_mem:.1f}%')\n    return df\n\n# Streamlined feature engineering\ndef add_features(df, feature_cols, max_features=10):\n    new_features = {}\n    new_features['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    new_features['buy_sell_ratio'] = df['buy_qty'] / (df['sell_qty'] + 1e-6)\n    new_features['volume_change'] = df['volume'].astype(np.float32).pct_change().fillna(0)\n    \n    # Limit the number of lag features to reduce memory\n    for col in feature_cols[:max_features]:\n        new_features[f'{col}_lag1'] = df[col].shift(1).fillna(0)\n        new_features[f'{col}_diff'] = df[col].diff().fillna(0)\n    \n    return pd.DataFrame(new_features, index=df.index)\n\n# Data generator with reduced memory footprint\nclass SequenceGenerator(tf.keras.utils.Sequence):\n    def __init__(self, X, y, time_steps, batch_size):\n        self.X = X\n        self.y = y\n        self.time_steps = time_steps\n        self.batch_size = batch_size\n        self.indices = np.arange(len(X) - time_steps)\n    \n    def __len__(self):\n        return int(np.ceil(len(self.indices) / self.batch_size))\n    \n    def __getitem__(self, idx):\n        start_idx = idx * self.batch_size\n        end_idx = min(start_idx + self.batch_size, len(self.indices))\n        batch_indices = self.indices[start_idx:end_idx]\n        X_batch = np.array([self.X[i:i + self.time_steps] for i in batch_indices], dtype=np.float32)\n        y_batch = np.array([self.y[i + self.time_steps] for i in batch_indices], dtype=np.float32)\n        return X_batch, y_batch\n\n# Load data in chunks using pyarrow\ndef load_data_in_chunks(file_path, batch_size=100000):\n    try:\n        # Use pyarrow to read Parquet file in batches\n        parquet_file = pq.ParquetFile(file_path)\n        chunks = []\n        for batch in parquet_file.iter_batches(batch_size=batch_size):\n            chunk = batch.to_pandas()\n            chunk = reduce_mem_usage(chunk, verbose=False)\n            chunks.append(chunk)\n        df = pd.concat(chunks, axis=0)\n        del chunks\n        gc.collect()\n        return df.set_index('timestamp') if 'timestamp' in df.columns else df\n    except FileNotFoundError:\n        print(\"Dataset not found. Creating dummy data.\")\n        n_rows = 10000 if 'train' in file_path else 5000\n        num_features = 10  # Reduced for memory efficiency\n        data = {f'X{i}': np.random.rand(n_rows).astype(np.float32) for i in range(1, num_features + 1)}\n        data.update({\n            'timestamp': pd.to_datetime(pd.date_range(start='2023-03-01', periods=n_rows, freq='min')),\n            'bid_qty': np.random.rand(n_rows).astype(np.float32) * 100,\n            'ask_qty': np.random.rand(n_rows).astype(np.float32) * 100,\n            'buy_qty': np.random.rand(n_rows).astype(np.float32) * 200,\n            'sell_qty': np.random.rand(n_rows).astype(np.float32) * 200,\n            'volume': np.random.rand(n_rows).astype(np.float32) * 500,\n        })\n        if 'train' in file_path:\n            data['label'] = np.random.randn(n_rows).astype(np.float32)\n        df = pd.DataFrame(data).set_index('timestamp')\n        df.iloc[10:20, 5] = np.nan\n        df.iloc[30, 10] = np.inf\n        df.iloc[40, 15] = -np.inf\n        df['all_nan_col'] = np.nan\n        return df\n\n# Load data\nprint(\"Loading data...\")\ntrain_df = load_data_in_chunks('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest_df = load_data_in_chunks('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\n# Memory optimization\nprint(\"Optimizing memory...\")\ntrain_df = reduce_mem_usage(train_df)\ntest_df = reduce_mem_usage(test_df)\ngc.collect()\n\n# Handle duplicate columns\ntrain_df.columns = [f\"{col}_{i}\" if train_df.columns.duplicated()[i] else col for i, col in enumerate(train_df.columns)]\ntest_df.columns = [f\"{col}_{i}\" if test_df.columns.duplicated()[i] else col for i, col in enumerate(test_df.columns)]\n\n# Feature engineering\nfeature_cols = [f'X{i}' for i in range(1, min(11, len(train_df.columns))) if f'X{i}' in train_df.columns]\ntrain_df = pd.concat([train_df, add_features(train_df, feature_cols)], axis=1)\ntest_df = pd.concat([test_df, add_features(test_df, feature_cols)], axis=1)\ngc.collect()\n\n# Preprocessing\nall_numerical_cols = train_df.select_dtypes(include=[np.number]).columns\ntrain_df[all_numerical_cols] = train_df[all_numerical_cols].replace([np.inf, -np.inf], np.nan)\ntest_df[all_numerical_cols] = test_df[all_numerical_cols].replace([np.inf, -np.inf], np.nan)\ninitial_feature_cols = [col for col in all_numerical_cols if col not in ['label']]\nall_nan_in_features = train_df[initial_feature_cols].columns[train_df[initial_feature_cols].isnull().all()].tolist()\nimputer_feature_cols = [col for col in initial_feature_cols if col not in all_nan_in_features]\n\n# Impute missing values\nimputer = SimpleImputer(strategy='mean')\ntrain_df[imputer_feature_cols] = imputer.fit_transform(train_df[imputer_feature_cols])\ntest_df[imputer_feature_cols] = imputer.transform(test_df[imputer_feature_cols])\ngc.collect()\n\n# Feature selection with LightGBM\nprint(\"Selecting top features...\")\nlgbm = LGBMRegressor(n_estimators=50, random_state=42, n_jobs=1)  # CPU for feature selection\nlgbm.fit(train_df[imputer_feature_cols], train_df['label'])\nimportances = pd.Series(lgbm.feature_importances_, index=imputer_feature_cols)\nfinal_train_features = importances.nlargest(20).index.tolist()  # Reduced to 20 features\nprint(f\"Selected {len(final_train_features)} features\")\n\nX = train_df[final_train_features]\ny = train_df['label'].astype(np.float32)\ny = np.clip(y, -10, 10)\n\n# Scale features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X).astype(np.float32)\ntest_X_scaled = scaler.transform(test_df[final_train_features]).astype(np.float32)\ndel train_df, test_df\ngc.collect()\n\n# Parameters\ntime_steps = 10  # Reduced to lower memory usage\nbatch_size = 32  # Increased for faster training\n\n# Training with TimeSeriesSplit\ntscv = TimeSeriesSplit(n_splits=3)  # Reduced splits\nlstm_predictions = []\nlgbm_predictions = []\nval_y_list = []\nfor fold, (train_idx, val_idx) in enumerate(tscv.split(X_scaled)):\n    print(f\"\\n--- Fold {fold+1} ---\")\n    train_generator = SequenceGenerator(X_scaled[train_idx], y[train_idx], time_steps, batch_size)\n    val_generator = SequenceGenerator(X_scaled[val_idx], y[val_idx], time_steps, batch_size)\n    \n    # Simplified LSTM model\n    inputs = Input(shape=(time_steps, len(final_train_features)))\n    x = LSTM(64, return_sequences=True)(inputs)  # Reduced units\n    x = Dropout(0.3)(x)\n    x = LSTM(32)(x)  # Single LSTM layer\n    x = Dense(16, activation='relu')(x)\n    outputs = Dense(1)(x)\n    model = Model(inputs, outputs)\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001), loss='mse')\n    model.fit(train_generator, validation_data=val_generator, epochs=10, verbose=1)\n    \n    val_X, val_y = val_generator[0]\n    lstm_pred = model.predict(val_X, batch_size=batch_size)\n    mse = mean_squared_error(val_y, lstm_pred)\n    print(f\"LSTM Validation MSE: {mse:.4f}\")\n    lstm_predictions.append(lstm_pred.flatten())\n    val_y_list.append(val_y)\n    \n    # LightGBM\n    lgbm = LGBMRegressor(n_estimators=50, learning_rate=0.1, num_leaves=31, random_state=42, n_jobs=1)\n    lgbm.fit(X_scaled[train_idx], y[train_idx])\n    lgbm_pred = lgbm.predict(X_scaled[val_idx][:len(val_y)])\n    mse = mean_squared_error(val_y, lgbm_pred)\n    print(f\"LightGBM Validation MSE: {mse:.4f}\")\n    lgbm_predictions.append(lgbm_pred)\n    \n    del train_generator, val_generator, val_X, val_y, model\n    gc.collect()\n\n# Optimize ensemble weights\nprint(\"Optimizing ensemble weights...\")\nbest_weight = 0.5\nbest_mse = float('inf')\nfor w in np.arange(0.1, 0.9, 0.1):\n    ensemble_pred = w * lstm_predictions[-1] + (1 - w) * lgbm_predictions[-1]\n    mse = mean_squared_error(val_y_list[-1], ensemble_pred)\n    if mse < best_mse:\n        best_mse = mse\n        best_weight = w\nprint(f\"Best ensemble weight: {best_weight}\")\n\n# Train final models\nprint(\"\\nTraining final models...\")\ntrain_generator = SequenceGenerator(X_scaled, y, time_steps, batch_size)\ninputs = Input(shape=(time_steps, len(final_train_features)))\nx = LSTM(64, return_sequences=True)(inputs)\nx = Dropout(0.3)(x)\nx = LSTM(32)(x)\nx = Dense(16, activation='relu')(x)\noutputs = Dense(1)(x)\nmodel = Model(inputs, outputs)\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=0.001), loss='mse')\nmodel.fit(train_generator, epochs=10, verbose=1)\n\nlgbm = LGBMRegressor(n_estimators=50, learning_rate=0.1, num_leaves=31, random_state=42, n_jobs=1)\nlgbm.fit(X_scaled, y)\ndel train_generator, X_scaled, y\ngc.collect()\n\n# Prepare test sequences\ntest_generator = SequenceGenerator(test_X_scaled, np.zeros(len(test_X_scaled)), time_steps, batch_size)\ntest_seq = []\nfor i in range(len(test_generator)):\n    X_batch, _ = test_generator[i]\n    test_seq.append(X_batch)\ntest_seq = np.concatenate(test_seq)[-len(test_X_scaled):]\n\n# Predict and ensemble\nlstm_pred = model.predict(test_seq, batch_size=batch_size)\nlgbm_pred = lgbm.predict(test_X_scaled)\ny_test_pred = best_weight * lstm_pred.flatten() + (1 - best_weight) * lgbm_pred\nsubmission = pd.DataFrame({'id': range(len(y_test_pred)), 'prediction': y_test_pred})\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created: submission.csv\")\n\n# Clean up\ndel test_X_scaled, test_seq, model, lgbm\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T03:40:29.588663Z","iopub.execute_input":"2025-06-03T03:40:29.588956Z","iopub.status.idle":"2025-06-03T04:35:36.707365Z","shell.execute_reply.started":"2025-06-03T03:40:29.588933Z","shell.execute_reply":"2025-06-03T04:35:36.70595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare test sequences to match test_X_scaled length\ndef create_test_sequences(X, time_steps, batch_size):\n    n_samples = len(X)\n    test_seq = np.zeros((n_samples, time_steps, X.shape[1]), dtype=np.float32)\n    for i in range(n_samples):\n        start_idx = max(0, i - time_steps + 1)\n        end_idx = i + 1\n        seq = X[start_idx:end_idx]\n        if len(seq) < time_steps:\n            # Pad with zeros if not enough data\n            pad_width = time_steps - len(seq)\n            seq = np.pad(seq, ((pad_width, 0), (0, 0)), mode='constant', constant_values=0)\n        test_seq[i] = seq\n    return test_seq\n\n# Generate test sequences\ntest_seq = create_test_sequences(test_X_scaled, time_steps, batch_size)\n\n# Predict and ensemble\nlstm_pred = model.predict(test_seq, batch_size=batch_size)\nlgbm_pred = lgbm.predict(test_X_scaled)\ny_test_pred = best_weight * lstm_pred.flatten() + (1 - best_weight) * lgbm_pred\nsubmission = pd.DataFrame({'id': test_index, 'prediction': y_test_pred})\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created: submission.csv\")\n\n# Clean up\ndel test_X_scaled, test_seq, model, lgbm\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T04:44:14.681242Z","iopub.execute_input":"2025-06-03T04:44:14.681528Z","iopub.status.idle":"2025-06-03T04:44:56.534829Z","shell.execute_reply.started":"2025-06-03T04:44:14.681508Z","shell.execute_reply":"2025-06-03T04:44:56.533983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samp=pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsamp.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T04:54:27.917214Z","iopub.execute_input":"2025-06-03T04:54:27.917473Z","iopub.status.idle":"2025-06-03T04:54:28.081755Z","shell.execute_reply.started":"2025-06-03T04:54:27.917454Z","shell.execute_reply":"2025-06-03T04:54:28.08103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_test_sequences(X, time_steps, batch_size):\n    n_samples = len(X)\n    test_seq = np.zeros((n_samples, time_steps, X.shape[1]), dtype=np.float32)\n    for i in range(n_samples):\n        start_idx = max(0, i - time_steps + 1)\n        end_idx = i + 1\n        seq = X[start_idx:end_idx]\n        if len(seq) < time_steps:\n            # Pad with zeros if not enough data\n            pad_width = time_steps - len(seq)\n            seq = np.pad(seq, ((pad_width, 0), (0, 0)), mode='constant', constant_values=0)\n        test_seq[i] = seq\n    return test_seq\n\n# Generate test sequences\ntest_seq = create_test_sequences(test_X_scaled, time_steps, batch_size)\n\n# Predict and ensemble\nlstm_pred = model.predict(test_seq, batch_size=batch_size)\nlgbm_pred = lgbm.predict(test_X_scaled)\nsamp=pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\ny_test_pred = best_weight * lstm_pred.flatten() + (1 - best_weight) * lgbm_pred\nsubmission = pd.DataFrame({'id': samp['ID'], 'prediction': y_test_pred})\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created: submission.csv\")\n\n# Clean up\ndel test_X_scaled, test_seq, model, lgbm\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T04:55:11.852893Z","iopub.execute_input":"2025-06-03T04:55:11.85324Z","iopub.status.idle":"2025-06-03T04:55:11.876601Z","shell.execute_reply.started":"2025-06-03T04:55:11.853219Z","shell.execute_reply":"2025-06-03T04:55:11.875325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samp.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T04:56:11.83083Z","iopub.execute_input":"2025-06-03T04:56:11.831444Z","iopub.status.idle":"2025-06-03T04:56:11.841524Z","shell.execute_reply.started":"2025-06-03T04:56:11.831422Z","shell.execute_reply":"2025-06-03T04:56:11.84093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub=pd.read_csv('/kaggle/working/submission.csv')\nsub.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T04:56:01.770644Z","iopub.execute_input":"2025-06-03T04:56:01.771328Z","iopub.status.idle":"2025-06-03T04:56:01.959566Z","shell.execute_reply.started":"2025-06-03T04:56:01.771297Z","shell.execute_reply":"2025-06-03T04:56:01.958831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load the original file with 'ID' and 'prediction'\ndf_main = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n\n# Load the new file with updated 'prediction'\ndf_new = pd.read_csv('/kaggle/working/submission.csv')\n\n# Replace the 'prediction' column in the main DataFrame\ndf_main['prediction'] = df_new['prediction']\n\n# Optional: Save the updated DataFrame to a new CSV\ndf_main.to_csv('updated_predictions.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T04:58:56.136029Z","iopub.execute_input":"2025-06-03T04:58:56.136663Z","iopub.status.idle":"2025-06-03T04:58:57.445415Z","shell.execute_reply.started":"2025-06-03T04:58:56.136641Z","shell.execute_reply":"2025-06-03T04:58:57.444806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}