{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":12453798,"sourceType":"datasetVersion","datasetId":7855901}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## See Input Folders","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-21T10:47:09.312541Z","iopub.execute_input":"2025-07-21T10:47:09.313760Z","iopub.status.idle":"2025-07-21T10:47:09.630539Z","shell.execute_reply.started":"2025-07-21T10:47:09.313724Z","shell.execute_reply":"2025-07-21T10:47:09.629721Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Settings","metadata":{}},{"cell_type":"code","source":"#getting data analysis and ML packages\nimport pandas\nfrom pandas.plotting import scatter_matrix\nfrom sklearn import model_selection\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import precision_score\nfrom sklearn.metrics import recall_score\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import median_absolute_error\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.linear_model import SGDRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.linear_model import PassiveAggressiveClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.naive_bayes import BernoulliNB\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.svm import SVC\nfrom sklearn.svm import LinearSVC\nfrom sklearn.svm import SVR\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.ensemble import ExtraTreesRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.kernel_ridge import KernelRidge\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer #transform different types\nfrom sklearn.datasets import fetch_openml\nimport numpy\nfrom numpy import sqrt\nfrom numpy import sum\nfrom numpy import square\nimport seaborn\nimport matplotlib\nimport statsmodels\nimport time\n\nfrom sklearn.decomposition import PCA\nimport xgboost as xgb\nfrom mlxtend.classifier import EnsembleVoteClassifier, StackingClassifier\nfrom mlxtend.regressor import StackingRegressor\nimport matplotlib.pyplot as plt\nfrom matplotlib import cm\nfrom matplotlib.ticker import LinearLocator, FormatStrFormatter\nfrom mpl_toolkits.mplot3d import Axes3D","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T10:47:09.631912Z","iopub.execute_input":"2025-07-21T10:47:09.632415Z","iopub.status.idle":"2025-07-21T10:47:11.586478Z","shell.execute_reply.started":"2025-07-21T10:47:09.632390Z","shell.execute_reply":"2025-07-21T10:47:11.585651Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"import warnings\n\nwarnings.filterwarnings(\"ignore\")\n\n%load_ext cuml.accel\n%load_ext cudf.pandas","metadata":{"execution":{"iopub.status.busy":"2025-07-09T10:12:54.901941Z","iopub.execute_input":"2025-07-09T10:12:54.902421Z","iopub.status.idle":"2025-07-09T10:12:59.600921Z","shell.execute_reply.started":"2025-07-09T10:12:54.902394Z","shell.execute_reply":"2025-07-09T10:12:59.598851Z"}}},{"cell_type":"code","source":"!pip install ipympl\n# Restart the kernel after installing for the changes to take effect.\n# After restarting, you can run the cell again.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T10:47:11.587304Z","iopub.execute_input":"2025-07-21T10:47:11.587686Z","iopub.status.idle":"2025-07-21T10:47:16.322352Z","shell.execute_reply.started":"2025-07-21T10:47:11.587640Z","shell.execute_reply":"2025-07-21T10:47:16.321211Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Processing stages\n### - Convert parquet into dataframe\n### - Descriptive\n#### -- Descriptive statistics (Done)\n#### -- Distribution graphs (hist(Done), KDE (Done), boxplot, violin)\n### - Exploratory\n#### -- Pivot table\n#### -- Scatter (Done), corrplot (Done but shit), bubble plot, time series (need), ribbon (line with confidence interval)\nhttps://r-graph-gallery.com/318-custom-dygraphs-time-series-example.html\nhttps://r-graph-gallery.com/104-plot-lines-with-error-envelopes-ggplot2.html\n#### -- Treemap, stacked bar or heatmap\nhttps://r-graph-gallery.com/237-interactive-treemap.html\nhttps://r-graph-gallery.com/235-treemap-with-subgroups.html\n### - Statistical/Inferential\n#### -- Feature Selection/Dimension Reduction (FA, PCA, Simple Linear Reg)\n#### -- Test effect of insignificant variables (n-way ANOVA)\n#### -- Test independence of variables (Bartlett test of sphericity, X2 test of independence, correlations t test/sample correlation test)\n### - Predictive\n#### -- Feature Engineering\n#### -- Transform\n#### -- HPT with and without transform\n#### -- Post-fit tuning","metadata":{}},{"cell_type":"code","source":"df_train = pandas.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet', engine='pyarrow')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T10:47:16.325132Z","iopub.execute_input":"2025-07-21T10:47:16.325472Z","iopub.status.idle":"2025-07-21T10:47:40.300172Z","shell.execute_reply.started":"2025-07-21T10:47:16.325441Z","shell.execute_reply":"2025-07-21T10:47:40.298978Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T10:47:40.301449Z","iopub.execute_input":"2025-07-21T10:47:40.301966Z","iopub.status.idle":"2025-07-21T10:47:40.428521Z","shell.execute_reply.started":"2025-07-21T10:47:40.301935Z","shell.execute_reply":"2025-07-21T10:47:40.427782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['year']   = df_train.index.year\ndf_train['month']  = df_train.index.month\ndf_train['day']    = df_train.index.day\ndf_train['hour']   = df_train.index.hour\ndf_train['minute'] = df_train.index.minute\ndf_train['weekday'] = df_train.index.weekday\ndf_train['date'] = df_train.index.date","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T10:47:40.429364Z","iopub.execute_input":"2025-07-21T10:47:40.429624Z","iopub.status.idle":"2025-07-21T10:47:40.656452Z","shell.execute_reply.started":"2025-07-21T10:47:40.429605Z","shell.execute_reply":"2025-07-21T10:47:40.655776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['second_cont'] = (df_train.index - df_train.index[0]).total_seconds()\ndf_train['minute_cont'] = (df_train.index - df_train.index[0]).total_seconds() / 60\ndf_train['hour_cont'] = (df_train.index - df_train.index[0]).total_seconds() / 3600\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T10:47:40.657289Z","iopub.execute_input":"2025-07-21T10:47:40.657531Z","iopub.status.idle":"2025-07-21T10:47:40.693575Z","shell.execute_reply.started":"2025-07-21T10:47:40.657513Z","shell.execute_reply":"2025-07-21T10:47:40.692801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## converting index from datetime to numbers. Cannot use pipeline bc not applying to entire df, would need columnTransformer, too lazy\ndf_train.index = range(0,len(df_train.index))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T10:47:40.694328Z","iopub.execute_input":"2025-07-21T10:47:40.694552Z","iopub.status.idle":"2025-07-21T10:47:40.698879Z","shell.execute_reply.started":"2025-07-21T10:47:40.694532Z","shell.execute_reply":"2025-07-21T10:47:40.697994Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Lag k","metadata":{}},{"cell_type":"code","source":"df_train['label'].mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-13T16:07:31.099363Z","iopub.execute_input":"2025-07-13T16:07:31.099625Z","iopub.status.idle":"2025-07-13T16:07:31.120342Z","shell.execute_reply.started":"2025-07-13T16:07:31.099607Z","shell.execute_reply":"2025-07-13T16:07:31.119552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['lag_1'] = df_train['label'].shift(periods=1)\ndf_train['lag_2'] = df_train['label'].shift(periods=2)\ndf_train['lag_3'] = df_train['label'].shift(periods=3)\ndf_train['lag_4'] = df_train['label'].shift(periods=4)\ndf_train['lag_5'] = df_train['label'].shift(periods=5)\ndf_train.loc[:,'lag_1':'lag_5'] = df_train.loc[:,'lag_1':'lag_5'].replace({np.nan: df_train['label'].mean()})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-13T16:07:31.121280Z","iopub.execute_input":"2025-07-13T16:07:31.121516Z","iopub.status.idle":"2025-07-13T16:07:31.229295Z","shell.execute_reply.started":"2025-07-13T16:07:31.121498Z","shell.execute_reply":"2025-07-13T16:07:31.228462Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dropping perfectly one-half of all perfectly correlated pairs of RVs","metadata":{}},{"cell_type":"code","source":"## Dropping perfectly one-half of all perfectly correlated pairs of RVs\ndf_train = df_train.drop(['X104',\n                            'X62',\n                            'X116',\n                            'X74',\n                            'X122',\n                            'X80',\n                            'X128',\n                            'X86',\n                            'X140',\n                            'X98',\n                            'X351',\n                            'X309',\n                            'X357',\n                            'X315',\n                            'X363',\n                            'X321',\n                            'X369',\n                            'X327',\n                            'date'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-13T16:07:31.230139Z","iopub.execute_input":"2025-07-13T16:07:31.230382Z","iopub.status.idle":"2025-07-13T16:07:32.544535Z","shell.execute_reply.started":"2025-07-13T16:07:31.230364Z","shell.execute_reply":"2025-07-13T16:07:32.543655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-13T16:07:32.545366Z","iopub.execute_input":"2025-07-13T16:07:32.545656Z","iopub.status.idle":"2025-07-13T16:07:32.839793Z","shell.execute_reply.started":"2025-07-13T16:07:32.545631Z","shell.execute_reply":"2025-07-13T16:07:32.838954Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dropping infs","metadata":{}},{"cell_type":"code","source":"## Dropping infs\ninfcol = df_train.columns[np.isinf(df_train).any(axis=0)].tolist()\nprint(infcol)\n\ndf_train = df_train.drop(columns=infcol)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-13T16:07:32.840633Z","iopub.execute_input":"2025-07-13T16:07:32.840943Z","iopub.status.idle":"2025-07-13T16:07:35.479194Z","shell.execute_reply.started":"2025-07-13T16:07:32.840918Z","shell.execute_reply":"2025-07-13T16:07:35.478462Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## PLS","metadata":{}},{"cell_type":"code","source":"from sklearn.cross_decomposition import PLSRegression, PLSSVD\n\nX = df_train.loc[:,'X1':'X780']\ny = df_train.loc[:,'label']\n\nX_train, X_test, y_train, y_test = model_selection.train_test_split(X, y, test_size=0.2)\n\npls = PLSRegression(n_components=5)\npls.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T11:06:15.266702Z","iopub.execute_input":"2025-07-21T11:06:15.267380Z","iopub.status.idle":"2025-07-21T11:06:45.125089Z","shell.execute_reply.started":"2025-07-21T11:06:15.267352Z","shell.execute_reply":"2025-07-21T11:06:45.124081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"r2_score(y_test, pls.predict(X_test))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"r2_score(df_train.loc[1250:2250,'label'], pls.predict(df_train.loc[1250:2250,'X1':'X780']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-15T08:39:50.835367Z","iopub.execute_input":"2025-07-15T08:39:50.835665Z","iopub.status.idle":"2025-07-15T08:39:50.862356Z","shell.execute_reply.started":"2025-07-15T08:39:50.835644Z","shell.execute_reply":"2025-07-15T08:39:50.861291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Variance explained in y per component\nexplained_var_y = np.var(pls.y_scores_, axis=0, ddof=0)\ntotal_var_y = np.var(y_train, ddof=0)\nexplained_ratio_y = explained_var_y / total_var_y\ncumulative_var = np.cumsum(explained_ratio_y)\n\n# Plot scree (elbow) plot\nplt.figure(figsize=(30, 30))\nplt.plot(range(1, 276), cumulative_var, marker='v')\n#plt.axhline(0.7, color='r', linestyle='--', label='70% threshold')\nplt.xlabel('Number of PLS Components')\nplt.ylabel('Cumulative Variance Explained in y')\nplt.title('PLS Scree Plot (Elbow Method)')\nplt.grid(True)\nplt.legend()\nplt.tight_layout()\nplt.savefig(\"/kaggle/working/PLS-scree.png\", dpi=300, bbox_inches='tight')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-13T13:51:31.910864Z","iopub.execute_input":"2025-07-13T13:51:31.911132Z","iopub.status.idle":"2025-07-13T13:51:39.661082Z","shell.execute_reply.started":"2025-07-13T13:51:31.911110Z","shell.execute_reply":"2025-07-13T13:51:39.659967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pls = PLSRegression(n_components=762)\npls.fit(X_train, y_train)\nexplained_var_y = np.var(pls.y_scores_, axis=0, ddof=0)\ntotal_var_y = np.var(y_train, ddof=0)\nexplained_ratio_y = explained_var_y / total_var_y\nprint(explained_ratio_y.sum())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate the model performance\nimport shap\n\n# Assuming 'pls_model' is your trained PLS regression model\n# Assuming 'X_train' is your training set features\n# Assuming 'X_test' is your test set features\n\n# Create a SHAP explainer for the PLS regression model using KernelExplainer\nexplainer = shap.KernelExplainer(pls.predict, X_train)\n\n# Calculate SHAP values for the entire test set\nshap_values = explainer.shap_values(X_test)\n\n# Summary plot for all instances\nshap.summary_plot(shap_values, X_test)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Inter-factor correl\ntemp = pandas.DataFrame(pls.transform(X_train)).iloc[:,0:201].corr()\nseaborn.heatmap(temp, annot=False, cmap='coolwarm', vmin=-1, vmax=1)\nplt.title('Inter-latent variable Correlation Heatmap 1m')\nplt.savefig(\"/kaggle/working/inter-factor-pls.png\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T11:06:45.126897Z","iopub.execute_input":"2025-07-21T11:06:45.127218Z","iopub.status.idle":"2025-07-21T11:06:47.888528Z","shell.execute_reply.started":"2025-07-21T11:06:45.127195Z","shell.execute_reply":"2025-07-21T11:06:47.887670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Factor-target correl\nseaborn.heatmap(pandas.DataFrame(temp.corrwith(y_train)), annot=False, cmap='coolwarm', vmin=-1, vmax=1)\nplt.title('Latent variables-target Correlation Heatmap 1m')\nplt.savefig(\"/kaggle/working/target-factor-pls.png\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T11:06:47.889534Z","iopub.execute_input":"2025-07-21T11:06:47.889882Z","iopub.status.idle":"2025-07-21T11:06:48.172336Z","shell.execute_reply.started":"2025-07-21T11:06:47.889852Z","shell.execute_reply":"2025-07-21T11:06:48.171507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pandas.DataFrame(temp.corrwith(y_train)).sort_values(0,ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T11:06:48.173835Z","iopub.execute_input":"2025-07-21T11:06:48.174081Z","iopub.status.idle":"2025-07-21T11:06:48.188078Z","shell.execute_reply.started":"2025-07-21T11:06:48.174062Z","shell.execute_reply":"2025-07-21T11:06:48.187351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\njoblib.dump(pls, \"/kaggle/working/pls.pkl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T11:07:39.313376Z","iopub.execute_input":"2025-07-21T11:07:39.313710Z","iopub.status.idle":"2025-07-21T11:07:39.547636Z","shell.execute_reply.started":"2025-07-21T11:07:39.313686Z","shell.execute_reply":"2025-07-21T11:07:39.546812Z"}},"outputs":[],"execution_count":null}]}