{"metadata":{"kernelspec":{"display_name":"irbackend","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.3"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":777.076344,"end_time":"2025-07-20T14:10:42.981129","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-07-20T13:57:45.904785","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Market States approach\n\n- From the observations that with tabular models (gradient boosting ensemble),\nshuffling the data guarantees no overfitting \nwhile splitting with chronological order overfits / does not capture the test probability distribution, \nit seems like a time-series approach is necessary :\ntimestamps are not independant draws of one underlying random variable.\n- Defining some latent state of the time-series is expected to be an improvement,\n- Also possible to cluster the timesteps into market modes, then to fit a predictor for each market mode.\n- Market modes can be defined from the covariance matrix of the features\n- Covariance matrix filtering techniques to reduce noise ","metadata":{"papermill":{"duration":0.003631,"end_time":"2025-07-20T13:57:50.729214","exception":false,"start_time":"2025-07-20T13:57:50.725583","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nfrom typing import List\nfrom pathlib import Path\nfrom scipy.stats import pearsonr\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.ensemble import RandomForestRegressor\nimport lightgbm as lgb\n\nfor dirname, _, filenames in os.walk(\"/kaggle/\"):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nKAGGLE = True  # define paths accordingly\nSUBMISSION = False  # use smaller datasets during dev\n\nif KAGGLE:\n    crypto_folder = Path(\"/kaggle/input/drw-crypto-market-prediction\")\nelse:\n    crypto_folder = Path(\"../raw_data/crypto\")","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2025-07-20T13:57:50.736615Z","iopub.status.busy":"2025-07-20T13:57:50.736306Z","iopub.status.idle":"2025-07-20T13:58:00.995242Z","shell.execute_reply":"2025-07-20T13:58:00.994253Z"},"papermill":{"duration":10.264046,"end_time":"2025-07-20T13:58:00.996596","exception":false,"start_time":"2025-07-20T13:57:50.73255","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Helper functions","metadata":{"papermill":{"duration":0.002666,"end_time":"2025-07-20T13:58:01.002348","exception":false,"start_time":"2025-07-20T13:58:00.999682","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_clean_crypto_data(train: bool = True) -> pl.LazyFrame:\n    \"\"\"\n    Load and clean crypto data, returning either train or test set.\n\n    Args:\n        train: If True, return training set. If False, return test set.\n\n    Returns:\n        Cleaned lazy frame with columns that have variance and no infinite values.\n    \"\"\"\n\n    filename = \"train.parquet\" if train else \"test.parquet\"\n\n    # load data\n    crypto_lazy = pl.scan_parquet(crypto_folder / filename)\n    n_cols = len(crypto_lazy.collect_schema().names())\n\n    if train and KAGGLE:\n        # rename timestamp column\n        crypto_lazy = crypto_lazy.with_columns(\n            pl.col(\"__index_level_0__\").alias(\"timestamp\")\n        ).drop([\"__index_level_0__\"])\n\n    # Remove columns with zero variance in the training set\n    train_lazy = pl.scan_parquet(crypto_folder / \"train.parquet\")\n    if KAGGLE:\n        train_lazy = train_lazy.with_columns(\n            pl.col(\"__index_level_0__\").alias(\"timestamp\")\n        ).drop([\"__index_level_0__\"])\n\n    # Get column names and calculate variance on training set (for consistency)\n    crypto_var = train_lazy.select(pl.exclude([\"timestamp\"]).var())\n\n    crypto_var_cols = (\n        crypto_var.select(pl.all() == 0.0)\n        .first()\n        .collect()\n        .to_pandas()\n        .T.rename(columns={0: \"is_variance_null\"})\n        .reset_index()\n        .rename(columns={\"index\": \"column_name\"})\n        .groupby(\"is_variance_null\")[\"column_name\"]\n        .unique()\n    )\n\n    crypto_cols_with_var = crypto_var_cols[False]\n\n    try:\n        cols_no_var = crypto_var_cols[True]\n        print(f\"Columns with no variance : {cols_no_var}\")\n    except KeyError:\n        print(\"All columns have variance in the train set\")\n\n    # remove columns that have no variance in the training set\n    train_lazy = train_lazy.select(\n        [\"timestamp\"] + [pl.col(c) for c in crypto_cols_with_var]\n    )\n\n    # Remove columns with infinite values (check on training set)\n    current_columns = train_lazy.collect_schema().names()\n    contains_infinite_cols = (\n        train_lazy.select(pl.exclude(\"timestamp\").abs().max().is_infinite())\n        .collect()\n        .to_pandas()\n        .T.rename(columns={0: \"contains_infinite\"})\n        .reset_index()\n        .rename(columns={\"index\": \"column_name\"})\n        .groupby(\"contains_infinite\")[\"column_name\"]\n        .unique()\n    )\n\n    try:\n        cols_with_inf_vals = contains_infinite_cols[True]\n        print(f\"Columns with infinite values : {cols_with_inf_vals}\")\n    except KeyError:\n        print(\"No columns with infinite values\")\n\n    if not train:\n        # add dummy timestamps\n        crypto_lazy = crypto_lazy.with_columns(\n            ID=range(1, crypto_lazy.select(pl.len()).collect().item() + 1)\n        )\n    # Filter clean columns based on what's available in the current dataset\n    clean_columns = [\n        c for c in current_columns if c in contains_infinite_cols[False]\n    ] + [\"timestamp\", \"ID\"]\n    available_columns = crypto_lazy.collect_schema().names()\n    final_columns = [c for c in clean_columns if c in available_columns]\n    print(f\"Eventually {len(final_columns)}, removed {n_cols - len(final_columns)}\")\n\n    return crypto_lazy.select(final_columns)\n\n\ndef get_diff_features(df: pl.LazyFrame, stats_columns: List[str]):\n    return (\n        df.with_columns(pl.exclude(stats_columns).diff())\n        .with_row_index()\n        .fill_null(strategy=\"backward\")\n        .select(pl.exclude(\"index\"))\n    )\n\n\ndef get_ma_features(df: pl.LazyFrame, cols: List[str]):\n    return df.with_columns(pl.col(cols).rolling_mean(window_size=23, min_samples=1))\n\n\ndef get_rolling_var(df: pl.LazyFrame, cols: List[str]):\n    return df.with_columns(pl.col(cols).rolling_var(window_size=23, min_samples=1))","metadata":{"execution":{"iopub.execute_input":"2025-07-20T13:58:01.010035Z","iopub.status.busy":"2025-07-20T13:58:01.009379Z","iopub.status.idle":"2025-07-20T13:58:01.021835Z","shell.execute_reply":"2025-07-20T13:58:01.021043Z"},"papermill":{"duration":0.018102,"end_time":"2025-07-20T13:58:01.023357","exception":false,"start_time":"2025-07-20T13:58:01.005255","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Metadata","metadata":{"papermill":{"duration":0.002735,"end_time":"2025-07-20T13:58:01.029189","exception":false,"start_time":"2025-07-20T13:58:01.026454","status":"completed"},"tags":[]}},{"cell_type":"code","source":"stats_columns = [\n    \"timestamp\",\n    \"bid_qty\",\n    \"ask_qty\",\n    \"buy_qty\",\n    \"sell_qty\",\n    \"volume\",\n    \"label\",\n]\nstats_columns_test = [\n    \"ID\",\n    \"bid_qty\",\n    \"ask_qty\",\n    \"buy_qty\",\n    \"sell_qty\",\n    \"volume\",\n    \"label\",\n]\nX_exclude = [\"timestamp\", \"label\"]\nX_test_exclude = [\"ID\", \"label\"]","metadata":{"execution":{"iopub.execute_input":"2025-07-20T13:58:01.036667Z","iopub.status.busy":"2025-07-20T13:58:01.03603Z","iopub.status.idle":"2025-07-20T13:58:01.040747Z","shell.execute_reply":"2025-07-20T13:58:01.039986Z"},"papermill":{"duration":0.010025,"end_time":"2025-07-20T13:58:01.042131","exception":false,"start_time":"2025-07-20T13:58:01.032106","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load data","metadata":{"papermill":{"duration":0.002802,"end_time":"2025-07-20T13:58:01.048043","exception":false,"start_time":"2025-07-20T13:58:01.045241","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# create intermediate clean file\ncrypto_lazy_clean = get_clean_crypto_data(train=True)","metadata":{"execution":{"iopub.execute_input":"2025-07-20T13:58:01.054968Z","iopub.status.busy":"2025-07-20T13:58:01.05469Z","iopub.status.idle":"2025-07-20T13:58:29.777066Z","shell.execute_reply":"2025-07-20T13:58:29.775983Z"},"papermill":{"duration":28.727997,"end_time":"2025-07-20T13:58:29.778904","exception":false,"start_time":"2025-07-20T13:58:01.050907","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# transform the dataset to day columns and only X features\n\ncrypto_df = crypto_lazy_clean.select(pl.exclude(stats_columns)).collect().to_pandas().T\ncrypto_df.to_parquet(\"features.parquet\")\ndel crypto_df","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_lazy = pl.scan_parquet(\"features.parquet\").drop(\"__index_level_0__\")\ncols = features_lazy.collect_schema().names()\nfeatures_lazy = features_lazy.select(pl.col(cols[:200]))\nn = len(features_lazy.collect_schema().names())\nlazy_cor = pl.concat(\n    [\n        features_lazy.select(pl.corr(pl.all(), pl.col(f\"{i}\"))).collect()\n        for i in range(n)\n    ]\n)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from matplotlib import pyplot as plt\n\nplt.imshow(lazy_cor.to_numpy())","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"adjacency = 1.0 - lazy_cor.to_numpy()\n\nadjacency","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.save(\"adjancency_matrix.npy\", adjacency)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Filter covariance matrix between features","metadata":{}},{"cell_type":"code","source":"n_samples = 50_000\ncov = crypto[[c for c in cols if c not in stats_columns]].sample(n_samples).cov()\nfrom matplotlib import pyplot as plt\n\nplt.imshow(cov)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cor = crypto[[c for c in cols if c not in stats_columns]].sample(n_samples).corr()\nplt.imshow(cor)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"svd = np.linalg.svd(cov.values, hermitian=True)\nU, S = svd.U, svd.S","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"_, ax = plt.subplots(figsize=(10, 4))\nax.bar(range(len(S)), S)\nax.set_yscale(\"log\")\nax.axhline(0.1, c=\"red\", linestyle=\"--\")\nax.grid()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cov_clean = U @ np.diag(np.where(S > 1.0, S, np.zeros_like(S))) @ U.T\n\nplt.imshow(cov_clean)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Compute day-wise covariance matrix to cluster timestamps into market modes","metadata":{}},{"cell_type":"markdown","source":"### Compute Louvain network communities and plot graph","metadata":{}},{"cell_type":"code","source":"from IPython.display import SVG\nfrom sknetwork.data import karate_club, painters, movie_actor\nfrom sknetwork.clustering import Louvain, get_modularity\nfrom sknetwork.linalg import normalize\nfrom sknetwork.utils import get_membership\nfrom sknetwork.visualization import visualize_graph, visualize_bigraph\nfrom scipy.sparse import csr_matrix\n\nadjacency_np = np.load(\"adjancency_matrix.npy\")\nadjacency = csr_matrix(adjacency_np)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"louvain = Louvain()\nlabels = louvain.fit_predict(adjacency)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_unique, counts = np.unique(labels, return_counts=True)\nprint(labels_unique, counts)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"k = len(labels_unique)\ncx = np.random.randn(2 * k) * 10\ncx = cx.reshape((k, 2))","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# position = np.random.randn(400).reshape((200, 2))\nposition = np.array([cx[l] for l in labels]) + np.random.randn(400).reshape((200, 2))\nimage = visualize_graph(adjacency, position, labels=labels, display_edges=False)\nSVG(image)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import silhouette_score\n\nsilhouette_score(features_lazy.collect().to_numpy().T, labels=labels)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Compute estimated transition matrix","metadata":{}},{"cell_type":"code","source":"transitions = np.zeros((k, k), dtype=np.float64)\nfor l_prev, l in zip(labels[:-1], labels[1:]):\n    transitions[l_prev, l] += 1.0\ntransitions = transitions / np.sum(transitions, axis=1)[:, None]\nplt.imshow(transitions)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- train a classifier to distinguish market modes\n    - PCA to define relevant samples for each modes, then KNN to classify new samples\n    - need train data classification guarantees, otherwise market mode feature will be very lossy","metadata":{}},{"cell_type":"markdown","source":"## Compute adjacency matrix","metadata":{}},{"cell_type":"markdown","source":"## Apply Louvain clustering algorithm of community detection","metadata":{}},{"cell_type":"code","source":"X = crypto_lazy_clean.select(pl.exclude(X_exclude)).collect().to_numpy()\ny = crypto_lazy_clean.select(pl.col(\"label\")).collect().to_numpy().T[0]\n\nif not SUBMISSION:\n    X_train, X_test, y_train, y_test = train_test_split(\n        X,\n        y,\n        test_size=0.2,\n        shuffle=False,  # TODO : question this, whether timestamps are independant draws\n        random_state=42,\n    )\nelse:\n    X_train, y_train = X, y\ndel X\ndel y","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train model","metadata":{"papermill":{"duration":0.003242,"end_time":"2025-07-20T13:58:29.786476","exception":false,"start_time":"2025-07-20T13:58:29.783234","status":"completed"},"tags":[]}},{"cell_type":"code","source":"lr = 1.0\n\nlin = RandomForestRegressor(\n    # fit_intercept=True,\n    n_estimators=80,\n    n_jobs=-1,\n    max_depth=10,\n    min_samples_split=100,\n    min_samples_leaf=50,\n    max_features=\"sqrt\",\n    max_samples=0.5,\n    random_state=41,\n)\n# n_samples = 80_000\nlin.fit(\n    X_train,\n    y_train,\n    # sample_weight=np.flip(1.0 / np.sqrt(np.arange(1, n_samples+1)))\n)\n\ny_train_lin = lin.predict(X_train)\n\nprint(f\"R2 train lin: {r2_score(y_train, y_train_lin)}\")\nprint(f\"Pearson train lin : {pearsonr(y_train, y_train_lin)}\")\n\ny_train_res = y_train - lr * y_train_lin\n\n\nlgb_model = lgb.LGBMRegressor(\n    random_state=42,\n    # weight=np.flip(1.0 / np.sqrt(np.arange(1, len(X_train)+1))),\n    # n_estimators=80,\n    # max_depth=10,\n    n_jobs=-1,\n)\nlgb_model.fit(X_train, y_train_res)\n\ny_train_hat = lgb_model.predict(X_train)\n\nprint(f\"R2 train : {r2_score(y_train, y_train_hat + lr * y_train_lin)}\")\nprint(f\"Pearson train : {pearsonr(y_train, y_train_hat + lr * y_train_lin)}\")\n\nif not SUBMISSION:\n    y_test_lin = lin.predict(X_test)\n\n    print(f\"R2 test lin : {r2_score(y_test, y_test_lin)}\")\n    print(f\"Pearson test lin : {pearsonr(y_test, y_test_lin)}\")\n\n    y_test_hat = lgb_model.predict(X_test)\n\n    print(f\"R2 test : {r2_score(y_test, y_test_hat + lr * y_test_lin)}\")\n    print(f\"Pearson test : {pearsonr(y_test, y_test_hat + lr * y_test_lin)}\")","metadata":{"execution":{"iopub.execute_input":"2025-07-20T13:58:29.794669Z","iopub.status.busy":"2025-07-20T13:58:29.794329Z","iopub.status.idle":"2025-07-20T14:09:33.326111Z","shell.execute_reply":"2025-07-20T14:09:33.32489Z"},"papermill":{"duration":663.54052,"end_time":"2025-07-20T14:09:33.330618","exception":false,"start_time":"2025-07-20T13:58:29.790098","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load test data","metadata":{"papermill":{"duration":0.003269,"end_time":"2025-07-20T14:09:33.3377","exception":false,"start_time":"2025-07-20T14:09:33.334431","status":"completed"},"tags":[]}},{"cell_type":"code","source":"crypto_lazy_test = get_clean_crypto_data(train=False)\n\n# create unique row identifier\nn = crypto_lazy_test.select(pl.len()).collect().item()\ncrypto_lazy_test = crypto_lazy_test.with_columns(ID=range(1, n + 1))\n\nprint(crypto_lazy_test.select(pl.len()).collect().item())\n\ncrypto_lazy_test = crypto_lazy_test.join(\n    get_diff_features(crypto_lazy_test, stats_columns_test),\n    on=stats_columns_test,\n    how=\"inner\",\n    suffix=\"_diff\",\n)\n\n# crypto_lazy_test = get_diff_features(crypto_lazy_test, stats_columns_test)\nassert n == crypto_lazy_test.select(pl.len()).collect().item()","metadata":{"execution":{"iopub.execute_input":"2025-07-20T14:09:33.346111Z","iopub.status.busy":"2025-07-20T14:09:33.345767Z","iopub.status.idle":"2025-07-20T14:09:49.226476Z","shell.execute_reply":"2025-07-20T14:09:49.22552Z"},"papermill":{"duration":15.886968,"end_time":"2025-07-20T14:09:49.228","exception":false,"start_time":"2025-07-20T14:09:33.341032","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Predict target \\& submit","metadata":{"papermill":{"duration":0.003187,"end_time":"2025-07-20T14:09:49.235751","exception":false,"start_time":"2025-07-20T14:09:49.232564","status":"completed"},"tags":[]}},{"cell_type":"code","source":"X_test = crypto_lazy_test.select(pl.exclude(X_test_exclude)).collect().to_numpy()\ny_lin_test = lin.predict(X_test)\ny_hat_lgb_test = lgb_model.predict(X_test)\n\ndel X_test","metadata":{"execution":{"iopub.execute_input":"2025-07-20T14:09:49.243816Z","iopub.status.busy":"2025-07-20T14:09:49.243525Z","iopub.status.idle":"2025-07-20T14:10:31.110731Z","shell.execute_reply":"2025-07-20T14:10:31.107856Z"},"papermill":{"duration":41.873573,"end_time":"2025-07-20T14:10:31.11262","exception":false,"start_time":"2025-07-20T14:09:49.239047","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"crypto_lazy_test = crypto_lazy_test.with_columns(\n    ID=range(1, n + 1), prediction=y_hat_lgb_test + lr * y_lin_test\n)\ncrypto_lazy_test.head(5).collect()\ncrypto_lazy_test.select([pl.col(\"ID\"), pl.col(\"prediction\")]).collect().write_csv(\n    Path(\"submission.csv\")\n)","metadata":{"execution":{"iopub.execute_input":"2025-07-20T14:10:31.125471Z","iopub.status.busy":"2025-07-20T14:10:31.125146Z","iopub.status.idle":"2025-07-20T14:10:38.514189Z","shell.execute_reply":"2025-07-20T14:10:38.509917Z"},"papermill":{"duration":7.39843,"end_time":"2025-07-20T14:10:38.517545","exception":false,"start_time":"2025-07-20T14:10:31.119115","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}