{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# DRW - Crypto Market Prediction","metadata":{}},{"cell_type":"markdown","source":"# 📚 Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore', category=FutureWarning)\nimport gc\nimport catboost as cb\nfrom scipy.stats import pearsonr\nfrom matplotlib import pyplot as plt\nfrom sklearn.model_selection import KFold","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:43.799255Z","start_time":"2025-05-21T21:44:43.795421Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:38:14.140140Z","iopub.execute_input":"2025-05-23T22:38:14.140723Z","iopub.status.idle":"2025-05-23T22:38:14.147984Z","shell.execute_reply.started":"2025-05-23T22:38:14.140691Z","shell.execute_reply":"2025-05-23T22:38:14.146839Z"},"_kg_hide-output":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔍 Reading competition material","metadata":{}},{"cell_type":"code","source":"train = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\n# downgrade to float32 for memory management\ntrain = train.astype('float32')\ntest = test.astype('float32')","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:43.859542Z","start_time":"2025-05-21T21:44:43.834328Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:31:45.411935Z","iopub.execute_input":"2025-05-23T22:31:45.412553Z","iopub.status.idle":"2025-05-23T22:32:57.971686Z","shell.execute_reply.started":"2025-05-23T22:31:45.412524Z","shell.execute_reply":"2025-05-23T22:32:57.970497Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔬 Exploratory Data Analysis","metadata":{}},{"cell_type":"markdown","source":"### train.parquet","metadata":{}},{"cell_type":"code","source":"print(f\"This dataset has {train.shape[0]} rows and {train.shape[1]} columns.\")\nprint(f\"There are {train.isna().sum().sum()} NA's in the dataset.\")\nprint(f\"There are {str(train.duplicated().sum())} duplicates in the dataset.\")\nprint(f\"There are {np.isinf(train).sum().sum()} infinite values in the dataset.\")\n\n# quick look at the data\ntrain.head(3)\n\n# We notice we don't have any NA's in this dataset, but we do have infinite values amounting to ~ 2.3% of the dataset we will need to work with.","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:43.927240Z","start_time":"2025-05-21T21:44:43.907542Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:32:57.973120Z","iopub.execute_input":"2025-05-23T22:32:57.973469Z","iopub.status.idle":"2025-05-23T22:33:40.760524Z","shell.execute_reply.started":"2025-05-23T22:32:57.973441Z","shell.execute_reply":"2025-05-23T22:33:40.759539Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### For now, we'll convert infinite values to NAs, and determine in which columns (proprietary features) the NaNs reside.","metadata":{}},{"cell_type":"code","source":"train = train.replace([np.inf, -np.inf], np.nan)\n\nnan_cols = train.columns[train.isna().any()].tolist() # get list of nan columns\n\nprint('The columns with NaNs and their NaN count:')\nprint(train[nan_cols].isna().sum()) # print NaN count per column\n\n# It appears these 21 columns only contain NaN values, so for now we will remove them from our dataset (both train & test)\ntrain = train.drop(nan_cols, axis=1)\ntest = test.drop(nan_cols, axis=1)\n\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:33:40.763211Z","iopub.execute_input":"2025-05-23T22:33:40.763540Z","iopub.status.idle":"2025-05-23T22:33:49.073566Z","shell.execute_reply.started":"2025-05-23T22:33:40.763513Z","shell.execute_reply":"2025-05-23T22:33:49.072352Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### After reading the idea in this [notebook](https://www.kaggle.com/code/suthcong/drw-cpm-lightgbm), we will also drop non unique columns","metadata":{}},{"cell_type":"code","source":"# Drop columns have exactly 1 value\nNUNIQUE1=[c for c in train.columns if train[c].nunique()==1]\ntrain.drop(NUNIQUE1,axis=1,inplace=True)\ntest.drop(NUNIQUE1,axis=1,inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:34:46.519858Z","iopub.execute_input":"2025-05-23T22:34:46.520163Z","iopub.status.idle":"2025-05-23T22:35:07.818502Z","shell.execute_reply.started":"2025-05-23T22:34:46.520141Z","shell.execute_reply":"2025-05-23T22:35:07.817620Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### To start, we will identify our unique_id & target.  By observing our data, we will initially group our features into categorical & numerical columns as follows:","metadata":{}},{"cell_type":"code","source":"target = 'label'\ncategorical_columns = []\nnumerical_columns = [col for col in train.columns if col != target]","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:43.972440Z","start_time":"2025-05-21T21:44:43.970287Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:35:14.858041Z","iopub.execute_input":"2025-05-23T22:35:14.858471Z","iopub.status.idle":"2025-05-23T22:35:14.863604Z","shell.execute_reply.started":"2025-05-23T22:35:14.858445Z","shell.execute_reply":"2025-05-23T22:35:14.862545Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### We have only numerical features at this point, so we will focus on skewness initially:","metadata":{}},{"cell_type":"code","source":"skewness_threshold = .5 # can tune / experiment with this value\nskewed_cols = [col for col in numerical_columns if train[col].skew() > skewness_threshold]\n\nprint(f'There are {len(skewed_cols)} skewed columns: {str(skewed_cols)}')\n\n# According to our threshold of .5, all of the original features (bid_qty, ask_qty, buy_qty, sell_qty, & volume are skewed, in addition to 262 of the proprietary, derived features.","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:44.094392Z","start_time":"2025-05-21T21:44:44.090993Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:35:18.916279Z","iopub.execute_input":"2025-05-23T22:35:18.916623Z","iopub.status.idle":"2025-05-23T22:35:24.884016Z","shell.execute_reply.started":"2025-05-23T22:35:18.916600Z","shell.execute_reply":"2025-05-23T22:35:24.883041Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Let's make sure all the columns in train exist in test, and vice versa (except for target!)","metadata":{}},{"cell_type":"code","source":"train_unique_cols = [x for x in train.drop([target],axis=1).columns if x not in test.columns]\nprint('All columns in train exist in test.') if not train_unique_cols else print(f'The following train columns are not in test: {train_unique_cols}')\n\ntest_unique_cols = [x for x in test.columns if x not in train.columns]\nprint('All columns in test exist in train.') if not test_unique_cols else print(f'The following test columns are not in train: {test_unique_cols}')\n\n# Perfect, we can use all train columns in our modelling efforts.","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:44.240728Z","start_time":"2025-05-21T21:44:44.236960Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:35:29.597634Z","iopub.execute_input":"2025-05-23T22:35:29.597946Z","iopub.status.idle":"2025-05-23T22:35:30.301317Z","shell.execute_reply.started":"2025-05-23T22:35:29.597923Z","shell.execute_reply":"2025-05-23T22:35:30.300130Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Univariate analysis of non proprietary derived features. We will sample 10% of the data.","metadata":{}},{"cell_type":"code","source":"train_samp = train[['bid_qty','ask_qty', 'buy_qty', 'sell_qty',\t'volume']].sample(frac=0.1, replace=False, random_state=1)\ntrain_samp = train_samp[~(train_samp == 0).any(axis=1)] # we will be taking the logarithmic of these columns for plotting purposes; removing rows w/ zeroes\n\nfor col in ['bid_qty','ask_qty', 'buy_qty', 'sell_qty',\t'volume']:\n    _, axes = plt.subplots(1,2,figsize=(8,4),sharex=False,sharey=False)\n   \n    sns.histplot(data=train_samp.dropna(), x=col,bins=10,ax=axes[0],log_scale=True, color = 'lightblue')\n    axes[0].set_title(f'Log of {col}')\n\n    sns.boxplot(data=train_samp.dropna(),x=col,ax=axes[1],showfliers=False, color = 'honeydew')\n    axes[1].set_title(f'{col} (no outliers)')\n\n    plt.tight_layout()\n    plt.show()\n\ndel train_samp\ngc.collect()","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:45.448824Z","start_time":"2025-05-21T21:44:44.297368Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:35:33.643030Z","iopub.execute_input":"2025-05-23T22:35:33.643728Z","iopub.status.idle":"2025-05-23T22:35:37.306494Z","shell.execute_reply.started":"2025-05-23T22:35:33.643701Z","shell.execute_reply":"2025-05-23T22:35:37.305291Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### test.parquet","metadata":{}},{"cell_type":"code","source":"print(f\"This dataset has {test.shape[0]} rows and {test.shape[1]} columns.\")\nprint(f\"There are {test.isna().sum().sum()} NA's in the dataset.\")\nprint(f\"There are {str(test.duplicated().sum())} duplicates in the dataset.\")\nprint(f\"There are {np.isinf(test).sum().sum()} infinite values in the dataset.\")\n\n# quick look at the data\ntest.head(3)\n\n# Notice there are no infinite values in test because we processed them earlier with train.","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:45.873410Z","start_time":"2025-05-21T21:44:45.862407Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:35:42.211848Z","iopub.execute_input":"2025-05-23T22:35:42.212296Z","iopub.status.idle":"2025-05-23T22:36:21.977093Z","shell.execute_reply.started":"2025-05-23T22:35:42.212254Z","shell.execute_reply":"2025-05-23T22:36:21.976159Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### The target: label","metadata":{}},{"cell_type":"code","source":"train_samp = pd.DataFrame(train[target]).sample(frac=0.1, replace=False, random_state=1).sort_index()\n\n# plot target histplot & boxplot\n_, axes = plt.subplots(1,2,figsize=(8,4),sharex=False,sharey=False)\nsns.histplot(data=train_samp.dropna(), x=target,bins=10,ax=axes[0],log_scale=False, color = 'lightblue')\naxes[0].set_title(f'{target}')\nsns.boxplot(data=train_samp.dropna(),x=target,ax=axes[1],showfliers=False, color = 'honeydew')\naxes[1].set_title(f'{target} (no outliers)')\nplt.tight_layout()\nplt.show()\n\n# plot target for entire timeframe\n_, axes = plt.subplots(1,1,figsize=(8,6),sharex=False,sharey=False)\nsns.lineplot(data=train_samp.dropna(),x=train_samp.index, y=target, color = 'mediumslateblue')\naxes.set_title(f'{target} time series plot (all timeframes)')\nplt.tight_layout()\nplt.show()\n\n# plot target for month of 2023-03\ntrain_samp = train_samp.filter(regex='^2023-03', axis=0)\n_, axes = plt.subplots(1,1,figsize=(8,6),sharex=False,sharey=False)\nsns.lineplot(data=train_samp.dropna(),x=train_samp.index, y=target, color = 'navy')\naxes.set_title(f'{target} time series plot (2023-03 timeframe)')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()\n\ndel train_samp\ngc.collect()\n","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:46.008235Z","start_time":"2025-05-21T21:44:45.935698Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:37:30.583212Z","iopub.execute_input":"2025-05-23T22:37:30.583584Z","iopub.status.idle":"2025-05-23T22:37:32.350446Z","shell.execute_reply.started":"2025-05-23T22:37:30.583559Z","shell.execute_reply":"2025-05-23T22:37:32.348970Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🎯 Model Training","metadata":{}},{"cell_type":"code","source":"X = train.drop([target],axis=1)\ny = train[target]\ncv_method = KFold(n_splits=5, shuffle=True, random_state=1)\nX_test = test.drop([target],axis=1)\ntest_ids = test.index\ndel train, test\ngc.collect()","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:46.074904Z","start_time":"2025-05-21T21:44:46.069896Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:38:28.413869Z","iopub.execute_input":"2025-05-23T22:38:28.417351Z","iopub.status.idle":"2025-05-23T22:38:28.464523Z","shell.execute_reply.started":"2025-05-23T22:38:28.417305Z","shell.execute_reply":"2025-05-23T22:38:28.462982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores = []\noof = np.zeros(len(y), dtype=float)\npreds = np.zeros(X_test.shape[0],dtype=float)\n\nparams = {'iterations': 3077, 'learning_rate': 0.023604883174757015, 'depth': 9, 'l2_leaf_reg': 1.1645944960977261, \n          'subsample': 0.7626370991977192, 'colsample_bylevel': 0.9894180344611699, 'min_data_in_leaf': 75}\n\nfor fold, (idx_tr, idx_va) in enumerate(cv_method.split(X, y), start=1):\n    X_tr = X.iloc[idx_tr]\n    X_va = X.iloc[idx_va]\n    y_tr = y.iloc[idx_tr]\n    y_va = y.iloc[idx_va]\n    \n    model = cb.CatBoostRegressor(**params, boosting_type='Plain', task_type='CPU', random_state=1,cat_features = categorical_columns)\n    model.fit(X_tr, y_tr,verbose=False)\n    \n    y_pred = model.predict(X_va)\n    preds += model.predict(X_test)\n    \n    score = pearsonr(y_va, y_pred)[0]\n    print(f\"# Fold {fold}: {score=:.5f}\")\n    \n    scores.append(score)\n    oof[idx_va] = y_pred\n    score = np.mean(scores)\n    \nprint(f\"# XGBoost_gbtree Overall score: {score}: +/- {np.std(scores)}\")\n\npreds /= 5","metadata":{"ExecuteTime":{"end_time":"2025-05-21T21:44:51.856878Z","start_time":"2025-05-21T21:44:46.181569Z"},"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:38:31.115560Z","iopub.execute_input":"2025-05-23T22:38:31.115852Z","iopub.status.idle":"2025-05-23T22:43:00.305868Z","shell.execute_reply.started":"2025-05-23T22:38:31.115833Z","shell.execute_reply":"2025-05-23T22:43:00.303353Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🤞 Submission","metadata":{}},{"cell_type":"markdown","source":"#### Let's read in the sample submission and review the format.","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsubmission.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:34:09.565433Z","iopub.status.idle":"2025-05-23T22:34:09.565841Z","shell.execute_reply.started":"2025-05-23T22:34:09.565622Z","shell.execute_reply":"2025-05-23T22:34:09.565643Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### We will use the preds list we created earlier and substitute its values in the 'prediction' column of the sample submission df","metadata":{}},{"cell_type":"code","source":"submission['prediction'] = preds\nsubmission.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:34:09.566982Z","iopub.status.idle":"2025-05-23T22:34:09.567433Z","shell.execute_reply.started":"2025-05-23T22:34:09.567230Z","shell.execute_reply":"2025-05-23T22:34:09.567252Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Now we can write our submission.","metadata":{}},{"cell_type":"code","source":"submission.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-23T22:34:09.568795Z","iopub.status.idle":"2025-05-23T22:34:09.569236Z","shell.execute_reply.started":"2025-05-23T22:34:09.568967Z","shell.execute_reply":"2025-05-23T22:34:09.568984Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# If you found this notebook useful please vote and/or leave a suggestion for improvements -- thanks for reviewing!","metadata":{}}]}