{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:26:39.178282Z","iopub.execute_input":"2025-05-30T06:26:39.178571Z","iopub.status.idle":"2025-05-30T06:26:40.805093Z","shell.execute_reply.started":"2025-05-30T06:26:39.178528Z","shell.execute_reply":"2025-05-30T06:26:40.804423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\nsub = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:26:40.806776Z","iopub.execute_input":"2025-05-30T06:26:40.807073Z","iopub.status.idle":"2025-05-30T06:27:28.100156Z","shell.execute_reply.started":"2025-05-30T06:26:40.807054Z","shell.execute_reply":"2025-05-30T06:27:28.099356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = sub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:28.101081Z","iopub.execute_input":"2025-05-30T06:27:28.101349Z","iopub.status.idle":"2025-05-30T06:27:28.105052Z","shell.execute_reply.started":"2025-05-30T06:27:28.101327Z","shell.execute_reply":"2025-05-30T06:27:28.104400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:28.106335Z","iopub.execute_input":"2025-05-30T06:27:28.106531Z","iopub.status.idle":"2025-05-30T06:27:28.215389Z","shell.execute_reply.started":"2025-05-30T06:27:28.106516Z","shell.execute_reply":"2025-05-30T06:27:28.214813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:28.216057Z","iopub.execute_input":"2025-05-30T06:27:28.216250Z","iopub.status.idle":"2025-05-30T06:27:28.297783Z","shell.execute_reply.started":"2025-05-30T06:27:28.216235Z","shell.execute_reply":"2025-05-30T06:27:28.297018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:28.298521Z","iopub.execute_input":"2025-05-30T06:27:28.298789Z","iopub.status.idle":"2025-05-30T06:27:28.308035Z","shell.execute_reply.started":"2025-05-30T06:27:28.298766Z","shell.execute_reply":"2025-05-30T06:27:28.307325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(dataframe, dataset):\n    print('Reduce Memory Usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n\n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n        \n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n        \n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n\n        else:\n\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage Now: {:.2f} MB'.format(final_mem_usage))\n    print('--- Memory usage Decreased By: {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage)/initial_mem_usage))\n\n    return dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:28.309881Z","iopub.execute_input":"2025-05-30T06:27:28.310069Z","iopub.status.idle":"2025-05-30T06:27:28.326080Z","shell.execute_reply.started":"2025-05-30T06:27:28.310054Z","shell.execute_reply":"2025-05-30T06:27:28.325595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.reset_index(drop=True)\ntest = test.reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:28.326654Z","iopub.execute_input":"2025-05-30T06:27:28.326859Z","iopub.status.idle":"2025-05-30T06:27:35.741487Z","shell.execute_reply.started":"2025-05-30T06:27:28.326833Z","shell.execute_reply":"2025-05-30T06:27:35.740949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_features = ['X863', 'X856', 'X344', 'X598', 'X862', 'X385', 'X852', \n                    'X603', 'X860', 'X674', 'X415', 'X345', 'X137', 'X855', \n                    'X174', 'X302', 'X178', 'X532', 'X168', 'X612', 'bid_qty', \n                    'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n\n\ntrain = train[selected_features + ['label']]\ntest = test[selected_features]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:35.742136Z","iopub.execute_input":"2025-05-30T06:27:35.742308Z","iopub.status.idle":"2025-05-30T06:27:36.290819Z","shell.execute_reply.started":"2025-05-30T06:27:35.742295Z","shell.execute_reply":"2025-05-30T06:27:36.289486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:36.291905Z","iopub.execute_input":"2025-05-30T06:27:36.292314Z","iopub.status.idle":"2025-05-30T06:27:36.314997Z","shell.execute_reply.started":"2025-05-30T06:27:36.292280Z","shell.execute_reply":"2025-05-30T06:27:36.314022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:36.316614Z","iopub.execute_input":"2025-05-30T06:27:36.316888Z","iopub.status.idle":"2025-05-30T06:27:36.338345Z","shell.execute_reply.started":"2025-05-30T06:27:36.316862Z","shell.execute_reply":"2025-05-30T06:27:36.337688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = reduce_mem_usage(train, 'train')\ntest = reduce_mem_usage(test, 'test')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:36.339347Z","iopub.execute_input":"2025-05-30T06:27:36.339694Z","iopub.status.idle":"2025-05-30T06:27:36.604054Z","shell.execute_reply.started":"2025-05-30T06:27:36.339670Z","shell.execute_reply":"2025-05-30T06:27:36.603273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'Train = {train.shape}')\nprint(f'Test = {test.shape}')\nprint(f'Sample = {sample.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:27:36.604874Z","iopub.execute_input":"2025-05-30T06:27:36.605113Z","iopub.status.idle":"2025-05-30T06:27:36.609081Z","shell.execute_reply.started":"2025-05-30T06:27:36.605085Z","shell.execute_reply":"2025-05-30T06:27:36.608396Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Improving the prediction, based on what DeepSeek says**","metadata":{}},{"cell_type":"code","source":"train['liq_imbalance'] = (train['bid_qty'] - train['ask_qty']) / (train['bid_qty'] - train['ask_qty'])\ntest['liq_imbalance'] = (test['bid_qty'] - test['ask_qty']) / (test['bid_qty'] - test['ask_qty'])\n\ntrain['buy_sell_ratio'] = train['buy_qty'] / (train['sell_qty'] + 1e-6)\ntest['buy_sell_ratio'] = test['buy_qty'] / (test['sell_qty'] + 1e-6)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T07:51:03.922784Z","iopub.execute_input":"2025-05-30T07:51:03.923442Z","iopub.status.idle":"2025-05-30T07:51:03.993630Z","shell.execute_reply.started":"2025-05-30T07:51:03.923422Z","shell.execute_reply":"2025-05-30T07:51:03.993070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMV = ['label']\nFEATURES = [col for col in train.columns if not col in RMV]\nprint(f'There are {len(FEATURES)} FEATURES: {FEATURES}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T07:51:20.991976Z","iopub.execute_input":"2025-05-30T07:51:20.992540Z","iopub.status.idle":"2025-05-30T07:51:20.997004Z","shell.execute_reply.started":"2025-05-30T07:51:20.992519Z","shell.execute_reply":"2025-05-30T07:51:20.996248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T07:51:33.782164Z","iopub.execute_input":"2025-05-30T07:51:33.782436Z","iopub.status.idle":"2025-05-30T07:51:33.786009Z","shell.execute_reply.started":"2025-05-30T07:51:33.782416Z","shell.execute_reply":"2025-05-30T07:51:33.785360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Folds = 5\nkf = KFold(n_splits = Folds, shuffle = True, random_state = 42)\n\noof_preds = np.zeros(len(train))\ntest_preds = np.zeros(len(test))\n\nxgb_params = {\n    'tree_method': 'gpu_hist',\n    'colsample_bylevel': 0.4778015829774066,\n    'colsample_bynode': 0.362764358742407,\n    'colsample_bytree': 0.7107423488010493,\n    'gamma': 1.7094857725240398,\n    'learning_rate': 0.02213323588455387,\n    'max_depth': 20,\n    'max_leaves': 12,\n    'min_child_weight': 16,\n    'n_estimators': 1667,\n    'n_jobs': -1,\n    'random_state': 42,\n    'reg_alpha': 39.352415706891264,\n    'reg_lambda': 75.44843704068275,\n    'subsample': 0.06566669853471274,\n    'verbosity': 0\n}\n\nfor i, (train_idx, valid_idx) in enumerate(kf.split(train)):\n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print('#' * 25)\n\n    X_train = train.iloc[train_idx][FEATURES]\n    y_train = train.iloc[train_idx]['label']\n    X_valid = train.iloc[valid_idx][FEATURES]\n    y_valid = train.iloc[valid_idx]['label']\n    X_test = test[FEATURES]\n\n    model = XGBRegressor(**xgb_params)\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=25,\n        verbose=200\n    )\n\n\n    oof_preds[valid_idx] = model.predict(X_valid)\n    test_preds += model.predict(X_test)\n\npearson_score = pearsonr(train['label'], oof_preds)[0]\nprint('Final Pearson Correlation = ', pearson_score)\n\ntest_preds /= Folds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T07:52:46.315070Z","iopub.execute_input":"2025-05-30T07:52:46.315861Z","iopub.status.idle":"2025-05-30T07:53:50.473537Z","shell.execute_reply.started":"2025-05-30T07:52:46.315835Z","shell.execute_reply":"2025-05-30T07:53:50.472792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nprint(torch.cuda.is_available())  # Should return True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T06:28:26.117237Z","iopub.execute_input":"2025-05-30T06:28:26.117858Z","iopub.status.idle":"2025-05-30T06:28:30.408306Z","shell.execute_reply.started":"2025-05-30T06:28:26.117831Z","shell.execute_reply":"2025-05-30T06:28:30.407696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample['prediction'] = test_preds\nsample.to_csv('submission.csv', index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T07:59:35.207098Z","iopub.execute_input":"2025-05-30T07:59:35.207412Z","iopub.status.idle":"2025-05-30T07:59:36.313976Z","shell.execute_reply.started":"2025-05-30T07:59:35.207394Z","shell.execute_reply":"2025-05-30T07:59:36.313171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-30T07:59:43.435691Z","iopub.execute_input":"2025-05-30T07:59:43.436004Z","iopub.status.idle":"2025-05-30T07:59:43.445512Z","shell.execute_reply.started":"2025-05-30T07:59:43.435982Z","shell.execute_reply":"2025-05-30T07:59:43.444911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}