{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q lofo-importance","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:05:04.109480Z","iopub.execute_input":"2025-01-09T13:05:04.109786Z","iopub.status.idle":"2025-01-09T13:05:08.907011Z","shell.execute_reply.started":"2025-01-09T13:05:04.109756Z","shell.execute_reply":"2025-01-09T13:05:08.905921Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"![alt text](https://raw.githubusercontent.com/aerdem4/lofo-importance/master/docs/lofo_logo.png)\n\nLOFO (Leave One Feature Out) Importance calculates the importances of a set of features based on a metric of choice, for a model of choice, by iteratively removing each feature from the set, and evaluating the performance of the model, with a validation scheme of choice, based on the chosen metric.\n\nLOFO first evaluates the performance of the model with all the input features included, then iteratively removes one feature at a time, retrains the model, and evaluates its performance on a validation set. The mean and standard deviation (across the folds) of the importance of each feature is then reported.\n\nIf a model is not passed as an argument to LOFO Importance, it will run LightGBM as a default model.\n\n## Install\n\nLOFO Importance can be installed using\n\n```\npip install lofo-importance\n```\n\n## Advantages of LOFO Importance\n\nLOFO has several advantages compared to other importance types:\n\n* It does not favor granular features\n* It generalises well to unseen test sets\n* It is model agnostic\n* It gives negative importance to features that hurt performance upon inclusion\n* It can group the features. Especially useful for high dimensional features like TFIDF or OHE features.\n* It can automatically group highly correlated features to avoid underestimating their importance.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os, sys, gc\nfrom tqdm import tqdm\nimport polars as pl\n\n\ndf = pl.scan_parquet(\n    f\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\"\n).filter(\n    pl.col(\"date_id\").gt(1000)\n).collect().sample(fraction=0.05, seed=0).to_pandas()\n\nprint(df.shape)\n\ngc.collect()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:05:08.908058Z","iopub.execute_input":"2025-01-09T13:05:08.908317Z","iopub.status.idle":"2025-01-09T13:05:39.392648Z","shell.execute_reply.started":"2025-01-09T13:05:08.908298Z","shell.execute_reply":"2025-01-09T13:05:39.391268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"anon_features = [f\"feature_{str(i).zfill(2)}\" for i in range(79)]\nfeature_cols = [\"weight\"] + anon_features + [\"time_id\"]\ntarget = \"responder_6\"\n\nlen(feature_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:05:39.393576Z","iopub.execute_input":"2025-01-09T13:05:39.393988Z","iopub.status.idle":"2025-01-09T13:05:39.400383Z","shell.execute_reply.started":"2025-01-09T13:05:39.393953Z","shell.execute_reply":"2025-01-09T13:05:39.399507Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Sliding window validation sets","metadata":{}},{"cell_type":"code","source":"VAL_LEN = 65\n\nmax_date = df[\"date_id\"].max()\n\nval_scheme = []\n\nfor i in range(1, 6):\n    train_ind = np.where(df[\"date_id\"].values < max_date - VAL_LEN*i)[0]\n    val_ind = np.where((df[\"date_id\"].values >= max_date - VAL_LEN*i) & \n                       (df[\"date_id\"].values <= max_date - VAL_LEN*(i-1)))[0]\n    \n    print(len(train_ind), len(val_ind))\n    \n    val_scheme.append((train_ind, val_ind))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:05:39.402302Z","iopub.execute_input":"2025-01-09T13:05:39.402497Z","iopub.status.idle":"2025-01-09T13:05:39.618649Z","shell.execute_reply.started":"2025-01-09T13:05:39.402480Z","shell.execute_reply":"2025-01-09T13:05:39.617705Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Grouping correlated features together","metadata":{}},{"cell_type":"code","source":"import lofo\n\nds = lofo.Dataset(df, target=target, features=feature_cols, auto_group_threshold=0.7)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:05:39.620902Z","iopub.execute_input":"2025-01-09T13:05:39.621187Z","iopub.status.idle":"2025-01-09T13:05:55.350100Z","shell.execute_reply.started":"2025-01-09T13:05:39.621160Z","shell.execute_reply":"2025-01-09T13:05:55.347108Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Defining the model","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\nxgb_param = {\n        'learning_rate': 0.04,\n        'max_depth': 5,\n        'colsample_bynode': 0.6,\n        'reg_alpha': 1,\n        'reg_lambda': 5,\n        'random_state': 0,\n        'device' : 'cuda',\n    \"min_child_weight\": 128,\n    \"n_estimators\":100\n    }\n\n\nmodel = XGBRegressor(\n    **xgb_param\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:05:55.351396Z","iopub.execute_input":"2025-01-09T13:05:55.352323Z","iopub.status.idle":"2025-01-09T13:05:55.578022Z","shell.execute_reply.started":"2025-01-09T13:05:55.352277Z","shell.execute_reply":"2025-01-09T13:05:55.577385Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Running LOFO","metadata":{}},{"cell_type":"code","source":"lofo_imp = lofo.LOFOImportance(ds, cv=val_scheme, scoring=\"neg_mean_squared_error\", \n                               model=model, n_jobs=1, fit_params={'sample_weight': df[\"weight\"]})\nimp_df = lofo_imp.get_importance()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:05:55.578695Z","iopub.execute_input":"2025-01-09T13:05:55.578892Z","iopub.status.idle":"2025-01-09T13:07:50.079395Z","shell.execute_reply.started":"2025-01-09T13:05:55.578874Z","shell.execute_reply":"2025-01-09T13:07:50.078161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lofo.plot_importance(imp_df, figsize=(12, 18))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T13:07:50.079854Z","iopub.status.idle":"2025-01-09T13:07:50.080129Z","shell.execute_reply":"2025-01-09T13:07:50.080020Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imp_df.to_csv(\"importance.csv\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}