{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport plotly.express as px\nimport plotly.graph_objects as go\nimport numpy as np\n\ninput_dir = '/kaggle/input/jane-street-real-time-market-data-forecasting/'\ntrain_data = 'train.parquet/partition_id=0/part-0.parquet'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-17T21:41:03.191417Z","iopub.execute_input":"2024-10-17T21:41:03.192352Z","iopub.status.idle":"2024-10-17T21:41:03.197100Z","shell.execute_reply.started":"2024-10-17T21:41:03.192310Z","shell.execute_reply":"2024-10-17T21:41:03.195909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_parquet(f'{input_dir}{train_data}')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:34:29.513412Z","iopub.execute_input":"2024-10-17T21:34:29.513892Z","iopub.status.idle":"2024-10-17T21:34:33.712226Z","shell.execute_reply.started":"2024-10-17T21:34:29.513858Z","shell.execute_reply":"2024-10-17T21:34:33.711117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# New Columns not in features.csv or responders.csv\n\n## date_id\n\nIs date_id sequential across partitions?\n* Yes\n\nHow many days total?\n* 4.65 years worth of data\n\nHow many days per file?\n* 169-170\n\nFrom Docs,\nInteger values that are ordinally sorted, providing a chronological structure to the data, \nalthough the actual time intervals between time_id values may vary.\n\n## time_id\n\n> The actual time intervals between time_id values may vary.\n\nIf stock_id and date_id is in a file, does it always have X time_ids?\n* Yes\n\n\n\n## symbol_id\n\nIdentifies a unique financial instrument. \n\nDo symbol_ids fall in out of the data?\n\n    * Yes\n\n## weight\n\nThe weighting used for calculating the scoring function.\n\n## Features\n\nHow common are NaNs across all the files?\n* Fairly common\n\nHow many columns are always non-null?\n* There are 32 columns that are always non-null in the training dataset.\n* 32 out of 79 columns\n\n","metadata":{}},{"cell_type":"code","source":"id_cols = ['date_id','symbol_id', 'time_id']","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:35:22.401443Z","iopub.execute_input":"2024-10-17T21:35:22.402184Z","iopub.status.idle":"2024-10-17T21:35:22.406714Z","shell.execute_reply.started":"2024-10-17T21:35:22.402142Z","shell.execute_reply":"2024-10-17T21:35:22.405541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols = list(filter(lambda x: 'feature' in x, train_df.columns))\nresponder_cols = list(filter(lambda x: 'responder' in x, train_df.columns))\ndatasets = [input_dir + f'train.parquet/partition_id={i}/part-0.parquet' for i in range(0, 10)]","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:34:41.636258Z","iopub.execute_input":"2024-10-17T21:34:41.636686Z","iopub.status.idle":"2024-10-17T21:34:41.642608Z","shell.execute_reply.started":"2024-10-17T21:34:41.636647Z","shell.execute_reply":"2024-10-17T21:34:41.641347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def row_statistics(file):\n    df = pd.read_parquet(file)\n    return df.shape[0]\n\nrows = list(map(row_statistics, datasets))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:24:05.237133Z","iopub.execute_input":"2024-10-17T22:24:05.237613Z","iopub.status.idle":"2024-10-17T22:25:48.127039Z","shell.execute_reply.started":"2024-10-17T22:24:05.237564Z","shell.execute_reply":"2024-10-17T22:25:48.125746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rows","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:25:48.129955Z","iopub.execute_input":"2024-10-17T22:25:48.130832Z","iopub.status.idle":"2024-10-17T22:25:48.137608Z","shell.execute_reply.started":"2024-10-17T22:25:48.130780Z","shell.execute_reply":"2024-10-17T22:25:48.136535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(rows)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:25:48.138885Z","iopub.execute_input":"2024-10-17T22:25:48.139267Z","iopub.status.idle":"2024-10-17T22:25:48.152303Z","shell.execute_reply.started":"2024-10-17T22:25:48.139232Z","shell.execute_reply":"2024-10-17T22:25:48.151109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# date_id","metadata":{}},{"cell_type":"code","source":"def calculate_date_id_statistics(file):\n    df = pd.read_parquet(file)\n    # All Dates are one \n    assert all(np.diff(np.sort(train_df['date_id'].unique())))\n    return df['date_id'].nunique()\n\ndate_id_stats = list(map(calculate_date_id_statistics, datasets))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T20:16:01.451692Z","iopub.execute_input":"2024-10-17T20:16:01.452747Z","iopub.status.idle":"2024-10-17T20:17:48.250620Z","shell.execute_reply.started":"2024-10-17T20:16:01.452692Z","shell.execute_reply":"2024-10-17T20:17:48.248363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All datasets have 169 - 170 date_id per file\ndate_id_stats","metadata":{"execution":{"iopub.status.busy":"2024-10-17T20:18:33.365867Z","iopub.execute_input":"2024-10-17T20:18:33.366651Z","iopub.status.idle":"2024-10-17T20:18:33.373587Z","shell.execute_reply.started":"2024-10-17T20:18:33.366609Z","shell.execute_reply":"2024-10-17T20:18:33.372339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 4.65 years worth of data\nsum(date_id_stats) / 365","metadata":{"execution":{"iopub.status.busy":"2024-10-17T20:18:53.694224Z","iopub.execute_input":"2024-10-17T20:18:53.694704Z","iopub.status.idle":"2024-10-17T20:18:53.702315Z","shell.execute_reply.started":"2024-10-17T20:18:53.694663Z","shell.execute_reply":"2024-10-17T20:18:53.700895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# time_id","metadata":{}},{"cell_type":"code","source":"def calculate_time_id_statistics(file):\n    df = pd.read_parquet(file)\n    agg_df = train_df.groupby(['date_id','symbol_id'], as_index=False).agg(unique_count = ('time_id','nunique'))\n    is_one = agg_df['unique_count'].nunique()\n    assert 1 == is_one\n    return agg_df['unique_count'].max()\n\ntime_id_stats = list(map(calculate_time_id_statistics, datasets))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T20:24:56.645307Z","iopub.execute_input":"2024-10-17T20:24:56.645813Z","iopub.status.idle":"2024-10-17T20:25:54.579807Z","shell.execute_reply.started":"2024-10-17T20:24:56.645773Z","shell.execute_reply":"2024-10-17T20:25:54.578748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_id_stats","metadata":{"execution":{"iopub.status.busy":"2024-10-17T20:25:54.581664Z","iopub.execute_input":"2024-10-17T20:25:54.582056Z","iopub.status.idle":"2024-10-17T20:25:54.590087Z","shell.execute_reply.started":"2024-10-17T20:25:54.582016Z","shell.execute_reply":"2024-10-17T20:25:54.588803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# symbol_id\n\nThere's not much that I care about when it comes to validating symbol_id","metadata":{}},{"cell_type":"code","source":"def calculate_symbol_id_statistics(file):\n    df = pd.read_parquet(file)\n    agg_df = train_df.groupby(['date_id','symbol_id'], as_index=False).agg(unique_count = ('time_id','nunique'))\n    is_one = agg_df['unique_count'].nunique()\n    assert 1 == is_one\n    return agg_df['unique_count'].max()\n\ntime_id_stats = list(map(calculate_time_id_statistics, datasets))\n\naggregate_train_df = train_df.groupby('date_id', as_index=False).agg(\n    count_of_unique_symbols = ('symbol_id','nunique')\n)\n\npx.bar(\n    aggregate_train_df,\n    'date_id',\n    'count_of_unique_symbols',\n    title='Unique Symbol Counts'\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T20:08:30.311438Z","iopub.execute_input":"2024-10-17T20:08:30.311890Z","iopub.status.idle":"2024-10-17T20:08:30.459335Z","shell.execute_reply.started":"2024-10-17T20:08:30.311847Z","shell.execute_reply":"2024-10-17T20:08:30.458286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weights\n\nNot much that I care about here.","metadata":{}},{"cell_type":"markdown","source":"# Features","metadata":{}},{"cell_type":"code","source":"def null_feature_statistics(file):\n    df = pd.read_parquet(file)\n    data = df[feature_cols].isna().mean()\n    return data\n\n\nnull_feature_stats = list(map(null_feature_statistics, datasets))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:36:51.758760Z","iopub.execute_input":"2024-10-17T21:36:51.759453Z","iopub.status.idle":"2024-10-17T21:38:37.936294Z","shell.execute_reply.started":"2024-10-17T21:36:51.759413Z","shell.execute_reply":"2024-10-17T21:38:37.935154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_feature = list(map(lambda x: pd.DataFrame(x), null_feature_stats))\nfor i, f in enumerate(datasets):\n    null_feature[i]['FILE'] = f\nnull_feature_df = pd.concat(null_feature)\nnull_feature_df = null_feature_df.reset_index().rename(columns={'index':'feature', 0:'null_frequency'})","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:38:37.938056Z","iopub.execute_input":"2024-10-17T21:38:37.938415Z","iopub.status.idle":"2024-10-17T21:38:37.951806Z","shell.execute_reply.started":"2024-10-17T21:38:37.938367Z","shell.execute_reply":"2024-10-17T21:38:37.950832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# null_feature_df.loc[null_feature_df['null_frequency'] == 0].groupby('FILE', as_index=False).agg(\n#     features = ('feature','nunique'),\n#     all_features = ('feature','unique'),\n# )\n\nnon_null_features = null_feature_df.loc[null_feature_df['null_frequency'] == 0].groupby('feature', as_index=False).agg(\n    count = ('FILE','nunique')\n)\n\nnon_null_features = non_null_features.loc[non_null_features['count'] == len(datasets)]['feature'].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:38:37.952934Z","iopub.execute_input":"2024-10-17T21:38:37.953246Z","iopub.status.idle":"2024-10-17T21:38:37.979050Z","shell.execute_reply.started":"2024-10-17T21:38:37.953214Z","shell.execute_reply":"2024-10-17T21:38:37.978089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(feature_cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:38:37.981359Z","iopub.execute_input":"2024-10-17T21:38:37.981717Z","iopub.status.idle":"2024-10-17T21:38:37.987713Z","shell.execute_reply.started":"2024-10-17T21:38:37.981683Z","shell.execute_reply":"2024-10-17T21:38:37.986647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(non_null_features)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:38:37.989249Z","iopub.execute_input":"2024-10-17T21:38:37.989584Z","iopub.status.idle":"2024-10-17T21:38:38.000956Z","shell.execute_reply.started":"2024-10-17T21:38:37.989551Z","shell.execute_reply":"2024-10-17T21:38:38.000037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_null_features","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:38:38.582792Z","iopub.execute_input":"2024-10-17T21:38:38.583124Z","iopub.status.idle":"2024-10-17T21:38:38.589842Z","shell.execute_reply.started":"2024-10-17T21:38:38.583090Z","shell.execute_reply":"2024-10-17T21:38:38.588749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlation","metadata":{}},{"cell_type":"code","source":"non_null_features = ['feature_05',\n 'feature_06',\n 'feature_07',\n 'feature_09',\n 'feature_10',\n 'feature_11',\n 'feature_12',\n 'feature_13',\n 'feature_14',\n 'feature_20',\n 'feature_22',\n 'feature_23',\n 'feature_24',\n 'feature_25',\n 'feature_28',\n 'feature_29',\n 'feature_30',\n 'feature_34',\n 'feature_35',\n 'feature_36',\n 'feature_38',\n 'feature_48',\n 'feature_49',\n 'feature_59',\n 'feature_60',\n#  'feature_61',\n 'feature_67',\n 'feature_68',\n 'feature_69',\n 'feature_70',\n 'feature_71',\n 'feature_72']","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:38:38.002169Z","iopub.execute_input":"2024-10-17T21:38:38.002501Z","iopub.status.idle":"2024-10-17T21:38:38.010594Z","shell.execute_reply.started":"2024-10-17T21:38:38.002468Z","shell.execute_reply":"2024-10-17T21:38:38.009663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation of Non Null Features + Responders\ntrain_df[id_cols + non_null_features + responder_cols]","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:38:38.011727Z","iopub.execute_input":"2024-10-17T21:38:38.012026Z","iopub.status.idle":"2024-10-17T21:38:38.504817Z","shell.execute_reply.started":"2024-10-17T21:38:38.011994Z","shell.execute_reply":"2024-10-17T21:38:38.503833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr_train = train_df.loc[train_df['date_id'] == 0][non_null_features + responder_cols].corr()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:39:56.252332Z","iopub.execute_input":"2024-10-17T21:39:56.253334Z","iopub.status.idle":"2024-10-17T21:39:56.292114Z","shell.execute_reply.started":"2024-10-17T21:39:56.253289Z","shell.execute_reply":"2024-10-17T21:39:56.291135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = corr_train[['responder_6']]\nfig = px.bar(sub,\n                 x=corr_train.index,\n                 y='responder_6',\n                 title='Correlation with Target'\n            )  # Set color scale range\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:52:45.260927Z","iopub.execute_input":"2024-10-17T21:52:45.261740Z","iopub.status.idle":"2024-10-17T21:52:45.320004Z","shell.execute_reply.started":"2024-10-17T21:52:45.261696Z","shell.execute_reply":"2024-10-17T21:52:45.319015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Linear Modeling ","metadata":{}},{"cell_type":"markdown","source":"## Day 0","metadata":{}},{"cell_type":"code","source":"from sklearn import linear_model\n\nday_0 = train_df.loc[train_df['date_id'] == 0]\n\nreg = linear_model.LinearRegression()\n\nX = day_0[non_null_features].values\ny = day_0['responder_6']","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:54:48.550012Z","iopub.execute_input":"2024-10-17T21:54:48.550434Z","iopub.status.idle":"2024-10-17T21:54:48.562708Z","shell.execute_reply.started":"2024-10-17T21:54:48.550394Z","shell.execute_reply":"2024-10-17T21:54:48.561632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reg.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:55:13.156119Z","iopub.execute_input":"2024-10-17T21:55:13.157226Z","iopub.status.idle":"2024-10-17T21:55:13.207110Z","shell.execute_reply.started":"2024-10-17T21:55:13.157177Z","shell.execute_reply":"2024-10-17T21:55:13.206153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = 'responder_6'\nday_0['target_hat'] = reg.predict(day_0[non_null_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T21:58:28.126410Z","iopub.execute_input":"2024-10-17T21:58:28.127274Z","iopub.status.idle":"2024-10-17T21:58:28.137071Z","shell.execute_reply.started":"2024-10-17T21:58:28.127229Z","shell.execute_reply":"2024-10-17T21:58:28.136032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"day_0['RSS'] = day_0['weight'] * ((day_0[target] - day_0['target_hat'])** 2)\nday_0['TSS'] = day_0['weight'] * (day_0[target] ** 2)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:00:31.786538Z","iopub.execute_input":"2024-10-17T22:00:31.787402Z","iopub.status.idle":"2024-10-17T22:00:31.795305Z","shell.execute_reply.started":"2024-10-17T22:00:31.787339Z","shell.execute_reply":"2024-10-17T22:00:31.794229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DAY ONE ERROR is 0.08\n# 8% of Variability of Responder 6 is explained by the feature set...\n\n1 - (day_0['RSS'].sum() / day_0['TSS'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:02:09.275092Z","iopub.execute_input":"2024-10-17T22:02:09.276709Z","iopub.status.idle":"2024-10-17T22:02:09.286890Z","shell.execute_reply.started":"2024-10-17T22:02:09.276656Z","shell.execute_reply":"2024-10-17T22:02:09.285708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# All Days in File 1","metadata":{}},{"cell_type":"code","source":"train_df['target_hat'] = reg.predict(train_df[non_null_features])\n\ntrain_df['w_RSS'] = train_df['weight'] * ((train_df[target] - train_df['target_hat']) ** 2)\ntrain_df['w_TSS'] = train_df['weight'] * (train_df[target] ** 2)\n\ntrain_df['RSS'] = ((train_df[target] - train_df['target_hat']) ** 2)\ntrain_df['TSS'] = (train_df[target] ** 2)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:08:08.869863Z","iopub.execute_input":"2024-10-17T22:08:08.870278Z","iopub.status.idle":"2024-10-17T22:08:09.274705Z","shell.execute_reply.started":"2024-10-17T22:08:08.870232Z","shell.execute_reply":"2024-10-17T22:08:09.273606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df[target].mean())\nprint(train_df['target_hat'].mean())","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:12:55.816602Z","iopub.execute_input":"2024-10-17T22:12:55.817074Z","iopub.status.idle":"2024-10-17T22:12:55.830266Z","shell.execute_reply.started":"2024-10-17T22:12:55.817032Z","shell.execute_reply":"2024-10-17T22:12:55.829035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agg_train_df = train_df.groupby('date_id',as_index=False).agg(\n    RSS_s = ('RSS','sum'),\n    TSS_s = ('TSS','sum'),\n    w_RSS_s = ('w_RSS','sum'),\n    w_TSS_s = ('w_TSS','sum'),\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:12:56.629791Z","iopub.execute_input":"2024-10-17T22:12:56.630635Z","iopub.status.idle":"2024-10-17T22:12:56.739539Z","shell.execute_reply.started":"2024-10-17T22:12:56.630593Z","shell.execute_reply":"2024-10-17T22:12:56.738176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"agg_train_df['R_sq'] = 1 - (agg_train_df['RSS_s'] / agg_train_df['TSS_s'])\nagg_train_df['w_R_sq'] = 1 - (agg_train_df['w_RSS_s'] / agg_train_df['w_TSS_s'])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:12:57.285764Z","iopub.execute_input":"2024-10-17T22:12:57.286169Z","iopub.status.idle":"2024-10-17T22:12:57.293549Z","shell.execute_reply.started":"2024-10-17T22:12:57.286132Z","shell.execute_reply":"2024-10-17T22:12:57.292356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Not only is this a bad model, its performance worsens over time.\nfig = px.line(agg_train_df,\n        'date_id',\n        ['R_sq','w_R_sq'],\n        title='Model_Performance_by_date_id'\n)\n\nfig","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:12:58.063730Z","iopub.execute_input":"2024-10-17T22:12:58.064541Z","iopub.status.idle":"2024-10-17T22:12:58.135957Z","shell.execute_reply.started":"2024-10-17T22:12:58.064499Z","shell.execute_reply":"2024-10-17T22:12:58.134895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:34:41.561509Z","iopub.execute_input":"2024-10-17T22:34:41.561945Z","iopub.status.idle":"2024-10-17T22:34:41.569153Z","shell.execute_reply.started":"2024-10-17T22:34:41.561907Z","shell.execute_reply":"2024-10-17T22:34:41.565892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"1 - (train_df['w_RSS'].sum() / train_df['w_TSS'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:36:26.593231Z","iopub.execute_input":"2024-10-17T22:36:26.594296Z","iopub.status.idle":"2024-10-17T22:36:26.605754Z","shell.execute_reply.started":"2024-10-17T22:36:26.594250Z","shell.execute_reply":"2024-10-17T22:36:26.604262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"1 - (train_df['RSS'].sum() / train_df['TSS'].sum())","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:36:43.729201Z","iopub.execute_input":"2024-10-17T22:36:43.729682Z","iopub.status.idle":"2024-10-17T22:36:43.741632Z","shell.execute_reply.started":"2024-10-17T22:36:43.729640Z","shell.execute_reply":"2024-10-17T22:36:43.740293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_score(train_df[target], train_df['target_hat'], sample_weight=train_df['weight'])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:35:47.256796Z","iopub.execute_input":"2024-10-17T22:35:47.257750Z","iopub.status.idle":"2024-10-17T22:35:47.289978Z","shell.execute_reply.started":"2024-10-17T22:35:47.257691Z","shell.execute_reply":"2024-10-17T22:35:47.288842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_score(train_df[target], train_df['target_hat'])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T22:36:35.875158Z","iopub.execute_input":"2024-10-17T22:36:35.875599Z","iopub.status.idle":"2024-10-17T22:36:35.900985Z","shell.execute_reply.started":"2024-10-17T22:36:35.875558Z","shell.execute_reply":"2024-10-17T22:36:35.899842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}