{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:04:40.176741Z","iopub.execute_input":"2025-06-02T10:04:40.177054Z","iopub.status.idle":"2025-06-02T10:04:42.429270Z","shell.execute_reply.started":"2025-06-02T10:04:40.177013Z","shell.execute_reply":"2025-06-02T10:04:42.428290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"crypto_submit = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\ncrypto_trn = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ncrypto_txt = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\nclass get_summary:\n    def __init__(self, x):\n        self.x = x if isinstance(x, pd.DataFrame) else pd.DataFrame()\n    def data_set(self):\n        #checks for duplicate\n        duplicate = self.x.duplicated().any()\n        #drop duplicates \n        if duplicate == True:\n            self.x.drop_duplicates(inplace=True)\n            self.x.reset_index(drop=True)\n        #checks for empty values\n        null = self.x.isna().sum().any()\n        #missing values\n        total_missing = self.x.isnull().sum().sum()\n        #data types\n        data_type = self.x.dtypes\n        #shape\n        shapes = self.x.shape\n        return f\"Duplicate: {duplicate}\\nNull: {null}\\nMissing_value: {total_missing}\\nTypes:\\n{data_type}\\nShape: {shapes}\"\n     #missing values\n    def total_missing(self):\n        missing_vals = self.x.isnull().sum()\n        cols_with_missing = missing_vals[missing_vals > 0]\n        if not cols_with_missing.empty:\n            return cols_with_missing.to_dict()\n        else:\n            return f\"{'No missing values detected'}\"\nprint(f\"Training dataset:\\n{get_summary(crypto_trn).data_set()}\\nTest dataset:\\n{get_summary(crypto_txt).data_set()}\")\nprint(f\"columns with missing values train\\n{get_summary(crypto_trn).total_missing()}\\ncolumns with missing values test\\n{get_summary(crypto_txt).total_missing()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:04:59.518917Z","iopub.execute_input":"2025-06-02T10:04:59.519663Z","iopub.status.idle":"2025-06-02T10:07:16.987779Z","shell.execute_reply.started":"2025-06-02T10:04:59.519637Z","shell.execute_reply":"2025-06-02T10:07:16.986986Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"crypto_trn.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:08:39.697404Z","iopub.execute_input":"2025-06-02T10:08:39.697693Z","iopub.status.idle":"2025-06-02T10:08:39.732539Z","shell.execute_reply.started":"2025-06-02T10:08:39.697669Z","shell.execute_reply":"2025-06-02T10:08:39.731790Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"crypto_txt.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:09:08.496192Z","iopub.execute_input":"2025-06-02T10:09:08.496603Z","iopub.status.idle":"2025-06-02T10:09:08.525146Z","shell.execute_reply.started":"2025-06-02T10:09:08.496575Z","shell.execute_reply":"2025-06-02T10:09:08.524229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:09:15.180578Z","iopub.execute_input":"2025-06-02T10:09:15.180981Z","iopub.status.idle":"2025-06-02T10:09:15.192896Z","shell.execute_reply.started":"2025-06-02T10:09:15.180954Z","shell.execute_reply":"2025-06-02T10:09:15.192017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trn_crypto = reduce_mem_usage(crypto_trn, \"crypto_trn\")\ntxt_crypto = reduce_mem_usage(crypto_txt, \"crypto_txt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:09:24.217837Z","iopub.execute_input":"2025-06-02T10:09:24.218995Z","iopub.status.idle":"2025-06-02T10:09:42.169460Z","shell.execute_reply.started":"2025-06-02T10:09:24.218933Z","shell.execute_reply":"2025-06-02T10:09:42.168501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\nimport warnings \nwarnings.filterwarnings(\"ignore\", category=RuntimeWarning)\n\ndef drop_high_corr_columns(crypto_trn: pd.DataFrame, crypto_txt: pd.DataFrame, threshold: float = 0.99):\n  \n    # Convert pandas DataFrames to Polars\n    train_pl = pl.from_pandas(trn_crypto)\n    test_pl = pl.from_pandas(txt_crypto)\n\n    \n    corr_df = train_pl.corr()\n\n    \n    columns = corr_df.columns\n    corr_np = corr_df.to_numpy()\n\n    # Find columns to drop based on upper triangle and threshold\n    upper = np.triu(np.ones(corr_np.shape), k=1)\n    to_drop = [columns[j] for i in range(corr_np.shape[0])\n               for j in range(corr_np.shape[1])\n               if upper[i, j] and corr_np[i, j] > threshold]\n\n    to_drop = list(set(to_drop))\n\n    # Drop columns from train and test Polars DataFrames\n    train_clean = train_pl.drop(to_drop)\n    test_clean = test_pl.drop(to_drop)\n\n    # Convert back to pandas and return\n    return train_clean.to_pandas(), test_clean.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:10:10.247867Z","iopub.execute_input":"2025-06-02T10:10:10.248226Z","iopub.status.idle":"2025-06-02T10:10:10.913132Z","shell.execute_reply.started":"2025-06-02T10:10:10.248203Z","shell.execute_reply":"2025-06-02T10:10:10.912168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trn_crypto, txt_crypto = drop_high_corr_columns(trn_crypto, txt_crypto, threshold=0.99)\ntrn_crypto.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:10:22.597382Z","iopub.execute_input":"2025-06-02T10:10:22.597661Z","iopub.status.idle":"2025-06-02T10:10:58.305224Z","shell.execute_reply.started":"2025-06-02T10:10:22.597640Z","shell.execute_reply":"2025-06-02T10:10:58.304446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"txt_crypto.drop('label', axis=1, inplace=True)\ntxt_crypto.replace([np.inf, -np.inf], np.nan, inplace=True)\ntxt_crypto.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:11:31.021588Z","iopub.execute_input":"2025-06-02T10:11:31.021901Z","iopub.status.idle":"2025-06-02T10:11:34.850693Z","shell.execute_reply.started":"2025-06-02T10:11:31.021880Z","shell.execute_reply":"2025-06-02T10:11:34.850029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = trn_crypto.drop('label', axis=1)\nX.replace([np.inf, -np.inf], np.nan, inplace=True)\n\ny = trn_crypto['label']\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, random_state=12, test_size=0.30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:12:07.478740Z","iopub.execute_input":"2025-06-02T10:12:07.479079Z","iopub.status.idle":"2025-06-02T10:12:15.134560Z","shell.execute_reply.started":"2025-06-02T10:12:07.479048Z","shell.execute_reply":"2025-06-02T10:12:15.133800Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBRegressor\nfrom scipy.stats import pearsonr\n\n\nxgb_params = {\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"n_jobs\": -1,\n    \"random_state\": 12,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.06566669853471274,\n    \"verbosity\": 0\n}\n\n\nmodel = XGBRegressor(**xgb_params, missing=np.nan)\n\nmodel.fit(X_train, y_train)\nmodel_pred = model.predict(X_val)\nscr = pearsonr(y_val, model_pred)\nprint(\"Final Pearson Correlation = \", scr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:12:44.327865Z","iopub.execute_input":"2025-06-02T10:12:44.328400Z","iopub.status.idle":"2025-06-02T10:35:03.222222Z","shell.execute_reply.started":"2025-06-02T10:12:44.328374Z","shell.execute_reply":"2025-06-02T10:35:03.221434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prediction = model.predict(txt_crypto)\nprediction","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:38:35.657405Z","iopub.execute_input":"2025-06-02T10:38:35.657687Z","iopub.status.idle":"2025-06-02T10:38:47.475836Z","shell.execute_reply.started":"2025-06-02T10:38:35.657668Z","shell.execute_reply":"2025-06-02T10:38:47.475191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"crypto_submit.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:40:23.038629Z","iopub.execute_input":"2025-06-02T10:40:23.039567Z","iopub.status.idle":"2025-06-02T10:40:23.058778Z","shell.execute_reply.started":"2025-06-02T10:40:23.039537Z","shell.execute_reply":"2025-06-02T10:40:23.058120Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = crypto_submit\nsubmission['prediction'] = prediction","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:41:39.967616Z","iopub.execute_input":"2025-06-02T10:41:39.968530Z","iopub.status.idle":"2025-06-02T10:41:39.973370Z","shell.execute_reply.started":"2025-06-02T10:41:39.968504Z","shell.execute_reply":"2025-06-02T10:41:39.972552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T10:42:18.488266Z","iopub.execute_input":"2025-06-02T10:42:18.489001Z","iopub.status.idle":"2025-06-02T10:42:19.422793Z","shell.execute_reply.started":"2025-06-02T10:42:18.488973Z","shell.execute_reply":"2025-06-02T10:42:19.422040Z"}},"outputs":[],"execution_count":null}]}