{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#  Install the custom Jane Street package (provided by competition organizers)\nimport os\n\n\n# Kaggle's evaluation module for inference (do not modify)\nfrom kaggle_evaluation import jane_street_inference_server\n\n\n# Import essential libraries for data & ML\nimport numpy as np\nimport pandas as pd\nimport polars as pl       # optional, faster dataframe library\nimport torch              # for deep learning models\nfrom tqdm import tqdm\nimport pickle\n\nimport matplotlib.pyplot as plt\nimport plotly.graph_objects as go\nimport plotly.express as px\n\nimport statsmodels.api as sm\nfrom sklearn.linear_model import ElasticNetCV","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-21T09:01:05.610596Z","iopub.execute_input":"2025-09-21T09:01:05.611009Z","iopub.status.idle":"2025-09-21T09:01:19.032860Z","shell.execute_reply.started":"2025-09-21T09:01:05.610971Z","shell.execute_reply":"2025-09-21T09:01:19.031620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv')\nresponders = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv')\nsample_submission = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/sample_submission.csv')\n\ntest_parquet = pl.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet')\n\nlags_parquet = pl.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet/date_id=0/part-0.parquet')\n\n# 전체 train_parquet, 이 중 partition_id = 6 까지만 사용할 것\ntrain_parquet = []\nfor i in tqdm(range(10), desc='train_parquet') :\n    file = f'/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet'\n    train_parquet.append(pl.read_parquet(file))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T15:40:43.566374Z","iopub.execute_input":"2025-09-20T15:40:43.566978Z","iopub.status.idle":"2025-09-20T15:42:20.985891Z","shell.execute_reply.started":"2025-09-20T15:40:43.566950Z","shell.execute_reply":"2025-09-20T15:42:20.984735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 합치기\nparquet = train_parquet[:7]\nparquets = parquet[0]\nfor i in parquet[1:] :\n    parquets = pl.concat([parquets, i])\nparquets\n\n# symbol_id별로 나누기\nsymbol_ids = parquets['symbol_id'].unique()\n\nparquets_by_symbol_id = []\n\nfor i in tqdm(symbol_ids) :\n    df = parquets.filter(pl.col('symbol_id') == i)\n    parquets_by_symbol_id.append(df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T15:43:06.307411Z","iopub.execute_input":"2025-09-20T15:43:06.307801Z","iopub.status.idle":"2025-09-20T15:43:23.354819Z","shell.execute_reply.started":"2025-09-20T15:43:06.307773Z","shell.execute_reply":"2025-09-20T15:43:23.354030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dump\n\n#parquets_by_symbol_id\nfor i in tqdm(range(len(parquets_by_symbol_id))) :\n    with open(f'parquets_by_symbol_id_{i}', 'wb') as f :\n        pickle.dump(parquets_by_symbol_id[i], f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T15:43:30.287619Z","iopub.execute_input":"2025-09-20T15:43:30.288485Z","iopub.status.idle":"2025-09-20T15:44:35.101107Z","shell.execute_reply.started":"2025-09-20T15:43:30.288456Z","shell.execute_reply":"2025-09-20T15:44:35.099434Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1. ","metadata":{}},{"cell_type":"code","source":"# ElasticNet 기반 coefficient의 절대값 반환 함수\n# X: target_features_non_null.to_numpy()\n# y: target.select(pl.col('responder_6')).to_numpy().ravel()\ndef EN(X, y) :\n    model= ElasticNetCV(alphas=np.logspace(-4, 4, 50), l1_ratio=[.1, .5, .9], cv=5)\n    model.fit(X, y)\n    return pl.DataFrame(dict(zip(features_list, abs(model.coef_))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T17:26:58.830753Z","iopub.execute_input":"2025-09-20T17:26:58.831080Z","iopub.status.idle":"2025-09-20T17:26:58.836141Z","shell.execute_reply.started":"2025-09-20T17:26:58.831056Z","shell.execute_reply":"2025-09-20T17:26:58.835405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def coef(symbol_number) :\n    # coefficient 저장용 데이터프레임 생성\n    features_list = parquets_by_symbol_id[0].columns[4: -9]\n    \n    coefs_by_symbol = []\n    \n    dummy = {col:[0.0] for col in features_list}\n    coefs = pl.DataFrame(dummy).clear()\n    \n    print(f'symbol {symbol_number}의 전체 일별 coefficient 도출')\n    symbol = parquets_by_symbol_id[symbol_number]\n    days = symbol['date_id'].unique().to_list()\n    \n    for day in tqdm(days, desc='일별 도출 중..') :\n        target = symbol.filter(pl.col('date_id')==day)\n        features_list = target.columns[4: -9]\n        \n        target_features = target.select(pl.col(features_list))\n    \n        target_features_non_null = target_features.select([\n            pl.col(col).fill_null(pl.col(col).mean()) if target_features[col].null_count() < target_features.height\n            else pl.col(col).fill_null(0)\n            for col in target_features.columns\n        ])\n    \n        target_features_non_null\n        \n        X = target_features_non_null.to_numpy()\n        y = target.select(pl.col('responder_6')).to_numpy().ravel()\n    \n        coefs = coefs.vstack(EN(X, y))\n        \n    coefs_by_features = {}\n    for col in coefs.columns :\n        coefs_by_features[col] = coefs[col].sum()\n\n    coefs_by_features_index = coefs.columns\n    coefs_by_features_col = 'coef'\n    coefs_by_features_row = []\n    \n    for col in coefs.columns :\n        coefs_by_features_row.append(coefs[col].sum())\n\n    coefs_by_features = pl.DataFrame({'index':coefs_by_features_index, 'coef': coefs_by_features_row})\n    return(coefs_by_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T17:27:25.301095Z","iopub.execute_input":"2025-09-20T17:27:25.301417Z","iopub.status.idle":"2025-09-20T17:27:25.310245Z","shell.execute_reply.started":"2025-09-20T17:27:25.301392Z","shell.execute_reply":"2025-09-20T17:27:25.309363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# symbol별 coefficient 저장용 데이터프레임 생성\nfeatures_list = parquets_by_symbol_id[0].columns[4: -9]\ncoefs_by_symbol = pl.DataFrame({\"index\": features_list})\n\n#symbol별 coefficient 도출\nfor i in list(symbol_ids[0:3]) :\n    print(f'symbol_ids: {i}')\n    coefs_by_symbol = coefs_by_symbol.with_columns(pl.Series(name=f'coef_symbol{i}', values=coef(i)['coef']))\n\ncoefs_by_symbol","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T17:29:20.807361Z","iopub.execute_input":"2025-09-20T17:29:20.807688Z","iopub.status.idle":"2025-09-20T17:47:11.291607Z","shell.execute_reply.started":"2025-09-20T17:29:20.807665Z","shell.execute_reply":"2025-09-20T17:47:11.290766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = px.bar(x=coefs_by_symbol['index'], y=coefs_by_symbol['coef_symbol0'])\nfig.update_layout(title_text=\"coef_symbol_id_0\", title_x=0.5)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T17:52:43.341266Z","iopub.execute_input":"2025-09-20T17:52:43.341898Z","iopub.status.idle":"2025-09-20T17:52:43.390375Z","shell.execute_reply.started":"2025-09-20T17:52:43.341874Z","shell.execute_reply":"2025-09-20T17:52:43.389477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"px.bar(x=coefs_by_symbol['index'], y=coefs_by_symbol['coef_symbol1'])\nfig.update_layout(title_text=\"coef_symbol_id_1\", title_x=0.5)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T17:52:44.362317Z","iopub.execute_input":"2025-09-20T17:52:44.362626Z","iopub.status.idle":"2025-09-20T17:52:44.410012Z","shell.execute_reply.started":"2025-09-20T17:52:44.362604Z","shell.execute_reply":"2025-09-20T17:52:44.409151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"px.bar(x=coefs_by_symbol['index'], y=coefs_by_symbol['coef_symbol2'])\nfig.update_layout(title_text=\"coef_symbol_id_2\", title_x=0.5)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T17:52:44.960092Z","iopub.execute_input":"2025-09-20T17:52:44.960742Z","iopub.status.idle":"2025-09-20T17:52:45.006652Z","shell.execute_reply.started":"2025-09-20T17:52:44.960692Z","shell.execute_reply":"2025-09-20T17:52:45.005923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(coefs_by_symbol.top_k(30, by='coef_symbol1')[\"index\"].to_list())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T17:54:31.449148Z","iopub.execute_input":"2025-09-20T17:54:31.449451Z","iopub.status.idle":"2025-09-20T17:54:31.455229Z","shell.execute_reply.started":"2025-09-20T17:54:31.449429Z","shell.execute_reply":"2025-09-20T17:54:31.454192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def top30fig(symbol) :\n    #symbol = 0\n    coefs_by_symbol_target = coefs_by_symbol['index', f'coef_symbol{symbol}']\n    top30 = coefs_by_symbol_target.top_k(30, by=f'coef_symbol{symbol}')['index']\n    \n    coefs_by_symbol_target\n    coefs_by_symbol_target = coefs_by_symbol_target.with_columns(\n        pl.when(pl.col('index').is_in(top30)).then(pl.lit('#1f77b4'))\n        .otherwise(pl.lit('#7f7f7f'))#1f77b4\n        .alias('color'))\n    \n    fig = go.Figure()\n    colors = coefs_by_symbol_target['color'].to_list()\n    \n    fig.add_trace(go.Bar(x=coefs_by_symbol_target['index'],y=coefs_by_symbol_target[f'coef_symbol{symbol}'], marker_color=colors))\n    fig.update_layout(title_text=f'coef_symbol_id_{symbol}, top30', title_x=0.5)\n    fig.show()\n    \n    print(top30.to_list())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T18:25:20.696822Z","iopub.execute_input":"2025-09-20T18:25:20.697135Z","iopub.status.idle":"2025-09-20T18:25:20.703504Z","shell.execute_reply.started":"2025-09-20T18:25:20.697111Z","shell.execute_reply":"2025-09-20T18:25:20.702502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top30fig(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T18:25:22.393887Z","iopub.execute_input":"2025-09-20T18:25:22.394198Z","iopub.status.idle":"2025-09-20T18:25:22.409094Z","shell.execute_reply.started":"2025-09-20T18:25:22.394174Z","shell.execute_reply":"2025-09-20T18:25:22.408093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top30fig(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T18:25:24.959438Z","iopub.execute_input":"2025-09-20T18:25:24.959810Z","iopub.status.idle":"2025-09-20T18:25:24.975036Z","shell.execute_reply.started":"2025-09-20T18:25:24.959783Z","shell.execute_reply":"2025-09-20T18:25:24.974181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top30fig(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-20T18:25:30.828772Z","iopub.execute_input":"2025-09-20T18:25:30.829330Z","iopub.status.idle":"2025-09-20T18:25:30.844254Z","shell.execute_reply.started":"2025-09-20T18:25:30.829305Z","shell.execute_reply":"2025-09-20T18:25:30.843269Z"}},"outputs":[],"execution_count":null}]}