{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":119083,"databundleVersionId":15997195}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Importing data:","metadata":{}},{"cell_type":"code","source":"import gc\nimport polars as pl\n\ntrain_path = '/kaggle/input/competitions/cyber-physical-anomaly-detection-for-der-systems/train.csv'\ntest_path = '/kaggle/input/competitions/cyber-physical-anomaly-detection-for-der-systems/test.csv'\n\ntrain_df = pl.read_csv(train_path)\nprint('Training set: ', train_df.shape)\ndisplay(train_df.head(3))\n\ntest_df = pl.read_csv(test_path)\nprint('Test set: ', test_df.shape)\ndisplay(test_df.head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:28:43.089571Z","iopub.execute_input":"2026-03-19T09:28:43.089875Z","iopub.status.idle":"2026-03-19T09:30:45.090522Z","shell.execute_reply.started":"2026-03-19T09:28:43.089852Z","shell.execute_reply":"2026-03-19T09:30:45.089799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Finding number of unique values in each column\nunique_dict = {col:train_df[col].n_unique() for col in train_df.columns}\n\n#removing redundant columns\nunique_count_1_cols = [col for col,count in unique_dict.items() if count==1]\ntrain_df = train_df.drop(unique_count_1_cols)\nprint(f\"Dropped a total of {len(unique_count_1_cols)} columns \\n\")\nprint(\"Dropped columns:\", unique_count_1_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:30:45.091548Z","iopub.execute_input":"2026-03-19T09:30:45.091939Z","iopub.status.idle":"2026-03-19T09:30:57.885622Z","shell.execute_reply.started":"2026-03-19T09:30:45.091903Z","shell.execute_reply":"2026-03-19T09:30:57.884911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#counting columns by dtypes\nfrom collections import Counter\nimport matplotlib.pyplot as plt\n\ndata_types = [str(dtype) for dtype in train_df.dtypes]\ncounts = Counter(data_types)\nprint(counts)\n\nplt.bar(counts.keys(),counts.values())\nplt.xlabel('dtypes')\nplt.ylabel('counts')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:30:57.887343Z","iopub.execute_input":"2026-03-19T09:30:57.887574Z","iopub.status.idle":"2026-03-19T09:30:58.102015Z","shell.execute_reply.started":"2026-03-19T09:30:57.887552Z","shell.execute_reply":"2026-03-19T09:30:58.101341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nulls = (\n    train_df.null_count()\n    .transpose(include_header=True)\n    .rename({\"column\": \"feature\", \"column_0\": \"null_count\"})\n    .filter(pl.col(\"null_count\") > 0)\n)\n\npl.Config.set_tbl_rows(-1) #so that everything gets printed\nprint(nulls)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:30:58.102821Z","iopub.execute_input":"2026-03-19T09:30:58.103034Z","iopub.status.idle":"2026-03-19T09:30:58.190950Z","shell.execute_reply.started":"2026-03-19T09:30:58.103014Z","shell.execute_reply":"2026-03-19T09:30:58.189725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Dropping columns with too many null values\ncols_to_drop = ['DERCtlAC[0].PFWAbs.Ext','DERCtlAC[0].PFWAbsRvrt.Ext','DERMeasureAC[0].ThrotPct','DERMeasureAC[0].ThrotSrc']\ntrain_df = train_df.drop(cols_to_drop)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:30:58.192690Z","iopub.execute_input":"2026-03-19T09:30:58.193222Z","iopub.status.idle":"2026-03-19T09:30:58.204159Z","shell.execute_reply.started":"2026-03-19T09:30:58.193182Z","shell.execute_reply":"2026-03-19T09:30:58.202718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Dropping null values from remaining columns\ntrain_df = train_df.drop_nulls()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:30:58.205927Z","iopub.execute_input":"2026-03-19T09:30:58.206445Z","iopub.status.idle":"2026-03-19T09:31:01.676303Z","shell.execute_reply.started":"2026-03-19T09:30:58.206404Z","shell.execute_reply":"2026-03-19T09:31:01.675719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Seprating label and features\nX = train_df.drop(['Id','Label'])\ny = train_df['Label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:31:01.677214Z","iopub.execute_input":"2026-03-19T09:31:01.677523Z","iopub.status.idle":"2026-03-19T09:31:01.681395Z","shell.execute_reply.started":"2026-03-19T09:31:01.677501Z","shell.execute_reply":"2026-03-19T09:31:01.680826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#clearing up memory since the dataset is huge\ndel train_df\ngc.collect()\nprint(\"Memory cleared!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:31:01.682166Z","iopub.execute_input":"2026-03-19T09:31:01.682382Z","iopub.status.idle":"2026-03-19T09:31:01.740758Z","shell.execute_reply.started":"2026-03-19T09:31:01.682351Z","shell.execute_reply":"2026-03-19T09:31:01.740168Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Basic EDA:\nJust going to check:\n* Target distribution","metadata":{}},{"cell_type":"code","source":"counts = Counter(y)\nprint(counts)\n\nplt.bar(counts.keys(), counts.values())\nplt.xlabel('category')\nplt.ylabel('counts')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:31:01.742878Z","iopub.execute_input":"2026-03-19T09:31:01.743140Z","iopub.status.idle":"2026-03-19T09:31:02.187853Z","shell.execute_reply.started":"2026-03-19T09:31:01.743111Z","shell.execute_reply":"2026-03-19T09:31:02.187030Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Yes, the given training dataset is a balanced one.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train,X_val,y_train,y_val = train_test_split(X,y,\n                                               test_size=0.2,\n                                               stratify=y,\n                                               random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:31:02.188995Z","iopub.execute_input":"2026-03-19T09:31:02.189800Z","iopub.status.idle":"2026-03-19T09:31:40.763595Z","shell.execute_reply.started":"2026-03-19T09:31:02.189775Z","shell.execute_reply":"2026-03-19T09:31:40.762848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = X.select(pl.col(pl.Utf8)).columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:31:40.764930Z","iopub.execute_input":"2026-03-19T09:31:40.765776Z","iopub.status.idle":"2026-03-19T09:31:40.938327Z","shell.execute_reply.started":"2026-03-19T09:31:40.765732Z","shell.execute_reply":"2026-03-19T09:31:40.937712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#clearing up memory since the dataset is huge\ndel X,y\ngc.collect()\nprint(\"Memory cleared!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:31:40.939158Z","iopub.execute_input":"2026-03-19T09:31:40.939365Z","iopub.status.idle":"2026-03-19T09:31:41.226751Z","shell.execute_reply.started":"2026-03-19T09:31:40.939345Z","shell.execute_reply":"2026-03-19T09:31:41.225603Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Preprocessing:","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\n\ncat_pipe = Pipeline([\n('encoder', OrdinalEncoder(handle_unknown='use_encoded_value', unknown_value=-1))\n])\n\npreprocessor = ColumnTransformer(\n    transformers=[('cat', cat_pipe, cat_cols)],\n    remainder='passthrough'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:31:41.228100Z","iopub.execute_input":"2026-03-19T09:31:41.228423Z","iopub.status.idle":"2026-03-19T09:31:41.598034Z","shell.execute_reply.started":"2026-03-19T09:31:41.228394Z","shell.execute_reply":"2026-03-19T09:31:41.596928Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Baseline XGBoost:\nXGBoost is being used since it provides inbuilt categorical handling, GPU support etc...","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\nparams = {\n    \"objective\": \"binary:logistic\",\n    \"eval_metric\": \"logloss\",\n    \"tree_method\": \"hist\",\n    \"device\": \"cuda\",\n    \"learning_rate\": 0.05,\n    \"max_depth\": 6,\n    \"random_state\": 42,\n}\n\nmodel = XGBClassifier(\n    **params\n)\n\nmodel_pipeline = Pipeline([\n    ('prep', preprocessor),\n    ('model',model)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:31:41.599303Z","iopub.execute_input":"2026-03-19T09:31:41.599797Z","iopub.status.idle":"2026-03-19T09:31:42.175919Z","shell.execute_reply.started":"2026-03-19T09:31:41.599756Z","shell.execute_reply":"2026-03-19T09:31:42.175273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_pipeline.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:31:42.177071Z","iopub.execute_input":"2026-03-19T09:31:42.177344Z","iopub.status.idle":"2026-03-19T09:32:32.712057Z","shell.execute_reply.started":"2026-03-19T09:31:42.177309Z","shell.execute_reply":"2026-03-19T09:32:32.711342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_val_perd_probs = model_pipeline.predict_proba(X_val)[:, 1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:32:32.715050Z","iopub.execute_input":"2026-03-19T09:32:32.715530Z","iopub.status.idle":"2026-03-19T09:32:37.939185Z","shell.execute_reply.started":"2026-03-19T09:32:32.715506Z","shell.execute_reply":"2026-03-19T09:32:37.938595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import fbeta_score\n\nbest_score = 0\nbest_thresh = 0\n\nfor t in [i/100 for i in range(10, 90)]:\n    preds = (y_val_perd_probs > t).astype(int)\n    score = fbeta_score(y_val, preds, beta=2) #bets=2 stands for f2_score which is our metric for this competition\n    \n    if score > best_score:\n        best_score = score\n        best_thresh = t\n\nprint(best_thresh, best_score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:32:37.940042Z","iopub.execute_input":"2026-03-19T09:32:37.940299Z","iopub.status.idle":"2026-03-19T09:32:39.981584Z","shell.execute_reply.started":"2026-03-19T09:32:37.940275Z","shell.execute_reply":"2026-03-19T09:32:39.980949Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* threshold = 0.31\n* best_score = 0.8766287986663943","metadata":{}},{"cell_type":"markdown","source":"## Making predictions on test data:","metadata":{}},{"cell_type":"code","source":"cols_to_remove = unique_count_1_cols + cols_to_drop + ['Id']\nX_test = test_df.drop(cols_to_remove)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:32:39.982404Z","iopub.execute_input":"2026-03-19T09:32:39.982710Z","iopub.status.idle":"2026-03-19T09:32:39.999815Z","shell.execute_reply.started":"2026-03-19T09:32:39.982679Z","shell.execute_reply":"2026-03-19T09:32:39.999145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_proba = model_pipeline.predict_proba(X_test)[:,1]\n\ny_preds = (y_test_proba > 0.31).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:32:40.000639Z","iopub.execute_input":"2026-03-19T09:32:40.001283Z","iopub.status.idle":"2026-03-19T09:32:51.633296Z","shell.execute_reply.started":"2026-03-19T09:32:40.001250Z","shell.execute_reply":"2026-03-19T09:32:51.632658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pl.DataFrame({\n    \"Id\": test_df['Id'],\n    \"Label\": y_preds\n})\n\nsubmission.write_csv(\"submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:32:51.634246Z","iopub.execute_input":"2026-03-19T09:32:51.634597Z","iopub.status.idle":"2026-03-19T09:32:51.818323Z","shell.execute_reply.started":"2026-03-19T09:32:51.634562Z","shell.execute_reply":"2026-03-19T09:32:51.817554Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Importance:","metadata":{}},{"cell_type":"code","source":"model = model_pipeline.named_steps['model']\npreprocessor = model_pipeline.named_steps['prep']\n\nfeature_names = preprocessor.get_feature_names_out()\n\nimportance = model.get_booster().get_score(importance_type='gain')\n\n# Map f-index to real feature names\nmapped_features = [\n    feature_names[int(f[1:])] for f in importance.keys()\n]\n\nimportance_df = pl.DataFrame({\n    'feature': mapped_features,\n    'importance': list(importance.values())\n}).sort('importance', descending=True)\n\nprint(importance_df.head(35))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-19T09:39:21.353554Z","iopub.execute_input":"2026-03-19T09:39:21.354383Z","iopub.status.idle":"2026-03-19T09:39:21.363543Z","shell.execute_reply.started":"2026-03-19T09:39:21.354351Z","shell.execute_reply":"2026-03-19T09:39:21.362434Z"}},"outputs":[],"execution_count":null}]}