{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T08:01:37.028592Z","iopub.execute_input":"2025-06-06T08:01:37.029130Z","iopub.status.idle":"2025-06-06T08:01:37.455712Z","shell.execute_reply.started":"2025-06-06T08:01:37.029097Z","shell.execute_reply":"2025-06-06T08:01:37.454938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom scipy import stats\nfrom scipy.stats import pearsonr\nfrom scipy.optimize import minimize\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import KFold\nfrom sklearn.feature_selection import VarianceThreshold\nfrom sklearn.base import BaseEstimator, TransformerMixin\n\nfrom catboost import CatBoostRegressor\nimport xgboost as xgb\nfrom lightgbm import LGBMRegressor\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.linear_model import LinearRegression, Ridge\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\nimport logging\n\nlogging.basicConfig(level=logging.INFO)\nlogger = logging.getLogger(__name__)\n\nplt.style.use('ggplot')\n%matplotlib inline\npd.options.display.max_columns = 100\npd.options.display.float_format = '{:.4f}'.format\n\nSEED = 42\nN_FOLDS = 5   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T08:01:37.456459Z","iopub.execute_input":"2025-06-06T08:01:37.456925Z","iopub.status.idle":"2025-06-06T08:01:40.097802Z","shell.execute_reply.started":"2025-06-06T08:01:37.456877Z","shell.execute_reply":"2025-06-06T08:01:40.096833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\nimport numpy as np\nimport pandas as pd\n\ndef load_data():\n    train = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\n    test = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n    \n    train = train.drop_duplicates()\n\n    train = optimize_memory_usage(train)\n    test = optimize_memory_usage(test)\n    \n    drop_list = [\n        'X697', 'X698', 'X699', 'X700', 'X701', 'X702', 'X703', 'X704', 'X705', 'X706', \n        'X707', 'X708', 'X709', 'X710', 'X711', 'X712', 'X713', 'X714', 'X715', 'X716',\n        'X717', 'X864', 'X867', 'X869', 'X870', 'X871', 'X872', 'X104', 'X110', 'X116',\n        'X122', 'X128', 'X134', 'X140', 'X146', 'X152', 'X158', 'X164', 'X170', 'X176',\n        'X182', 'X351', 'X357', 'X363', 'X369', 'X375', 'X381', 'X387', 'X393', 'X399',\n        'X405', 'X411', 'X417', 'X423', 'X429'\n    ]\n    \n    train = train.drop(columns=drop_list).reset_index(drop=True)\n    test = test.drop(columns=[\"label\"] + drop_list).reset_index(drop=True)\n    \n    X = train.drop(columns=[\"label\"], axis=1)\n    y = train[\"label\"]\n    \n    X = variance_threshold(X, 0.04)\n    test = test[X.columns]\n    \n    return X, y, test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T08:01:40.098866Z","iopub.execute_input":"2025-06-06T08:01:40.099465Z","iopub.status.idle":"2025-06-06T08:01:40.108315Z","shell.execute_reply.started":"2025-06-06T08:01:40.099439Z","shell.execute_reply":"2025-06-06T08:01:40.107177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def optimize_memory_usage(df, print_size=True):\n    \"\"\"\n    Optimizes memory usage in a DataFrame by downcasting numeric columns.\n\n    Parameters:\n        df (pd.DataFrame): The DataFrame to optimize.\n        print_size (bool): If True, prints memory usage before and after optimization.\n\n    Returns:\n        pd.DataFrame: The optimized DataFrame.\n    \"\"\"\n    # Types for optimization.\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    \n    # Memory usage size before optimize (Mb).\n    before_size = df.memory_usage().sum() / 1024**2\n    \n    for column in df.columns:\n        column_type = df[column].dtype\n        \n        if column_type in numerics:\n            try:\n                if str(column_type).startswith('int'):\n                    df[column] = pd.to_numeric(df[column], downcast='integer')\n                else:\n                    df[column] = pd.to_numeric(df[column], downcast='float')\n                logger.info(f\"Optimized column {column}: {column_type} -> {df[column].dtype}\")\n            except Exception as e:\n                logger.error(f\"Failed to optimize column {column}: {e}\")\n    \n    # Memory usage size after optimize (Mb).\n    after_size = df.memory_usage().sum() / 1024**2\n    \n    if print_size:\n        print(\n            'Memory usage size: before {:5.4f} Mb - after {:5.4f} Mb ({:.1f}%).'.format(\n                before_size, after_size, 100 * (before_size - after_size) / before_size\n            )\n        )\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T08:01:40.109335Z","iopub.execute_input":"2025-06-06T08:01:40.109619Z","iopub.status.idle":"2025-06-06T08:01:40.136375Z","shell.execute_reply.started":"2025-06-06T08:01:40.109589Z","shell.execute_reply":"2025-06-06T08:01:40.135049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def variance_threshold(df, threshold):\n    var_thres = VarianceThreshold(threshold=threshold)\n    var_thres.fit(df)\n    new_cols = var_thres.get_support()\n    return df.iloc[:, new_cols]\n\nclass FeatureGenerator(BaseEstimator, TransformerMixin):\n    def fit(self, X, y=None):\n        return self\n    \n    def transform(self, X):\n        return add_features(X)\n\ndef add_features(df):\n    features = pd.DataFrame(index=df.index)\n    \n    features['bid_ask_spread'] = df['ask_qty'] - df['bid_qty']\n    features['total_liquidity'] = df['bid_qty'] + df['ask_qty']\n    features['trade_imbalance'] = df['buy_qty'] - df['sell_qty']\n    features['total_trades'] = df['buy_qty'] + df['sell_qty']\n    \n    return pd.concat([df, features], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T08:01:40.137477Z","iopub.execute_input":"2025-06-06T08:01:40.138157Z","iopub.status.idle":"2025-06-06T08:01:40.162964Z","shell.execute_reply.started":"2025-06-06T08:01:40.138120Z","shell.execute_reply":"2025-06-06T08:01:40.161758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_model_params():\n    return {\n        'catboost': [\n            {'iterations': 150, 'depth': 6, 'learning_rate': 0.05, 'l2_leaf_reg': 3,\n             'border_count': 32, 'bagging_temperature': 1, 'random_strength': 1},\n             {'iterations': 140, 'depth': 8, 'learning_rate': 0.03, 'l2_leaf_reg': 5,\n              'border_count': 64, 'bagging_temperature': 0.5, 'random_strength': 2},\n        ],\n        'xgb': [\n             {'n_estimators': 140, 'max_depth': 6, 'learning_rate': 0.01,\n              'subsample': 0.8, 'colsample_bytree': 0.8, 'gamma': 0, 'min_child_weight': 1},\n            {'n_estimators': 150, 'max_depth': 8, 'learning_rate': 0.03,\n             'subsample': 0.9, 'colsample_bytree': 0.9, 'gamma': 0.1, 'min_child_weight': 2},\n        ],\n        'lgbm': [\n            {'n_estimators': 150, 'max_depth': 5, 'learning_rate': 0.01,\n             'num_leaves': 31, 'min_data_in_leaf': 20, 'feature_fraction': 0.8, 'bagging_fraction': 0.8},\n             {'n_estimators': 140, 'max_depth': 4, 'learning_rate': 0.05, \n              'num_leaves': 15, 'min_data_in_leaf': 10, 'feature_fraction': 0.7, 'bagging_fraction': 0.7}\n        ],\n        'hgbm': [\n            {'max_iter': 100, 'max_depth': 8, 'learning_rate': 0.03,\n             'min_samples_leaf': 15, 'l2_regularization': 0.2, 'max_bins': 255},\n             {'max_iter': 100, 'max_depth': 6, 'learning_rate': 0.02,\n              'min_samples_leaf': 25, 'l2_regularization': 0.1, 'max_bins': 255},\n        ]\n    }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T08:01:40.167013Z","iopub.execute_input":"2025-06-06T08:01:40.167446Z","iopub.status.idle":"2025-06-06T08:01:40.192699Z","shell.execute_reply.started":"2025-06-06T08:01:40.167410Z","shell.execute_reply":"2025-06-06T08:01:40.191413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_ensemble(X, y, test, model_params, n_folds=N_FOLDS):\n    folds = KFold(n_splits=n_folds, shuffle=True, random_state=SEED)\n    oof_predictions = {}\n    test_predictions = {}\n    \n    models = []\n    for i, params in enumerate(model_params['catboost'], 1):\n        models.append((f'cat_{i}', CatBoostRegressor(**params, verbose=0, thread_count=1)))\n    \n    for i, params in enumerate(model_params['xgb'], 1):\n        models.append((f'xgb_{i}', xgb.XGBRegressor(**params, n_jobs=1)))\n    \n    for i, params in enumerate(model_params['lgbm'], 1):\n        models.append((f'lgb_{i}', LGBMRegressor(**params, verbose=-1, n_jobs=1)))\n\n    for i, params in enumerate(model_params['hgbm'], 1):\n        models.append((f'hgb_{i}', HistGradientBoostingRegressor(**params)))\n    \n    for name, model in models:\n        print(f\"\\nTraining {name}...\")\n        oof = np.zeros(len(X))\n        pred = np.zeros(len(test))\n        \n        for fold, (trn_idx, val_idx) in enumerate(folds.split(X, y)):\n            X_train, y_train = X.iloc[trn_idx], y.iloc[trn_idx]\n            X_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n            \n            model.fit(X_train, y_train)\n            oof[val_idx] = model.predict(X_val)\n            pred += model.predict(test) / folds.n_splits\n            \n            fold_score = pearsonr(y_val, oof[val_idx])[0]\n            print(f'Fold {fold} Pearson: {fold_score:.4f}')\n        \n        full_score = pearsonr(y, oof)[0]\n        print(f'{name} OOF Pearson: {full_score:.4f}')\n        \n        oof_predictions[name] = oof\n        test_predictions[name] = pred\n    \n    return pd.DataFrame(oof_predictions), pd.DataFrame(test_predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T08:01:40.193872Z","iopub.execute_input":"2025-06-06T08:01:40.194356Z","iopub.status.idle":"2025-06-06T08:01:40.220868Z","shell.execute_reply.started":"2025-06-06T08:01:40.194318Z","shell.execute_reply":"2025-06-06T08:01:40.219817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def optimize_weights(oof_df, y_true):\n    model_columns = [col for col in oof_df.columns]\n    \n    def objective(weights):\n        combined = sum(w * oof_df[model] for w, model in zip(weights, model_columns))\n        return -pearsonr(y_true, combined)[0]  \n    \n    constraints = ({'type': 'eq', 'fun': lambda w: np.sum(w) - 1})\n    bounds = [(0, 1)] * len(model_columns)\n    \n    initial_weights = np.ones(len(model_columns)) / len(model_columns)\n    \n    result = minimize(\n        objective,\n        initial_weights,\n        method='SLSQP',\n        bounds=bounds,\n        constraints=constraints\n    )\n    \n    if not result.success:\n        print(\"Optimization warning:\", result.message)\n        return initial_weights\n    \n    return result.x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T08:01:40.221914Z","iopub.execute_input":"2025-06-06T08:01:40.222247Z","iopub.status.idle":"2025-06-06T08:01:40.247934Z","shell.execute_reply.started":"2025-06-06T08:01:40.222223Z","shell.execute_reply":"2025-06-06T08:01:40.246959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_submission(test_predictions, weights, model_names):\n    sample = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n    sample['prediction'] = sum(w * test_predictions[name] for w, name in zip(weights, model_names))\n    sample.to_csv('submission.csv', index=False)\n    return sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T08:01:40.249001Z","iopub.execute_input":"2025-06-06T08:01:40.249444Z","iopub.status.idle":"2025-06-06T08:01:40.273837Z","shell.execute_reply.started":"2025-06-06T08:01:40.249402Z","shell.execute_reply":"2025-06-06T08:01:40.272964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X, y, test = load_data()\n\nfeature_generator = FeatureGenerator()\nX = feature_generator.fit_transform(X)\ntest = feature_generator.transform(test)\n\nmodel_params = get_model_params()\n\noof_results, test_predictions = train_ensemble(X, y, test, model_params)\n\noptimal_weights = optimize_weights(oof_results, y)\n\nprint(\"\\nOptimized weights:\")\nfor name, weight in zip(oof_results.columns, optimal_weights):\n    score = pearsonr(y, oof_results[name])[0]\n    print(f\"{name}: {weight:.4f} (Pearson: {score:.4f})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T08:01:40.275100Z","iopub.execute_input":"2025-06-06T08:01:40.275497Z","iopub.status.idle":"2025-06-06T10:56:43.777944Z","shell.execute_reply.started":"2025-06-06T08:01:40.275455Z","shell.execute_reply":"2025-06-06T10:56:43.775407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = create_submission(test_predictions, optimal_weights, oof_results.columns)\nprint(\"\\nSubmission head:\")\nprint(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T10:56:43.781836Z","iopub.execute_input":"2025-06-06T10:56:43.782391Z","iopub.status.idle":"2025-06-06T10:56:45.529814Z","shell.execute_reply.started":"2025-06-06T10:56:43.782346Z","shell.execute_reply":"2025-06-06T10:56:45.528790Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nweights_df = pd.DataFrame({\n    'Model': oof_results.columns,\n    'Weight': optimal_weights,\n    'Pearson': [pearsonr(y, oof_results[col])[0] for col in oof_results.columns]\n}).sort_values('Weight', ascending=False)\n\nsns.barplot(x='Weight', y='Model', data=weights_df, palette='viridis')\nplt.title('Model Weights in Ensemble')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T10:56:45.531072Z","iopub.execute_input":"2025-06-06T10:56:45.531411Z","iopub.status.idle":"2025-06-06T10:56:46.024417Z","shell.execute_reply.started":"2025-06-06T10:56:45.531382Z","shell.execute_reply":"2025-06-06T10:56:46.023445Z"}},"outputs":[],"execution_count":null}]}