{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Standard Libraries\nimport os\nimport random\nfrom pathlib import Path\nfrom concurrent.futures import ThreadPoolExecutor\nimport matplotlib.pyplot as plt\n\n# Progress Bar\nfrom tqdm import tqdm\n\n# Data Handling\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\n# Machine Learning Libraries\nfrom sklearn.model_selection import (\n    train_test_split, StratifiedKFold, cross_val_score, cross_val_predict\n)\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.cluster import DBSCAN\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import make_scorer, cohen_kappa_score\nfrom sklearn.ensemble import (\n    VotingClassifier, AdaBoostClassifier, GradientBoostingClassifier\n)\nfrom sklearn.tree import DecisionTreeClassifier\n\n# Gradient Boosting Models\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier, LGBMRegressor \n\n\nfrom scipy.optimize import minimize\nimport optuna\n\nimport lightgbm as lgb\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\nSEED = 42\n\nKAPPA_SCORER = make_scorer(\n    cohen_kappa_score, \n    greater_is_better=True, \n    weights='quadratic',\n)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-05T08:47:53.645436Z","iopub.execute_input":"2024-11-05T08:47:53.646258Z","iopub.status.idle":"2024-11-05T08:47:53.654543Z","shell.execute_reply.started":"2024-11-05T08:47:53.646219Z","shell.execute_reply":"2024-11-05T08:47:53.653566Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Loading\n","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(estimator, X, y_true):\n    y_pred = estimator.predict(X).round()\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:53.658512Z","iopub.execute_input":"2024-11-05T08:47:53.659042Z","iopub.status.idle":"2024-11-05T08:47:53.664818Z","shell.execute_reply.started":"2024-11-05T08:47:53.658996Z","shell.execute_reply":"2024-11-05T08:47:53.663993Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def threshold_rounder(y_pred, thresholds):\n    return np.where(y_pred < thresholds[0], 0,\n                    np.where(y_pred < thresholds[1], 1,\n                             np.where(y_pred < thresholds[2], 2, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:53.668682Z","iopub.execute_input":"2024-11-05T08:47:53.668978Z","iopub.status.idle":"2024-11-05T08:47:53.674155Z","shell.execute_reply.started":"2024-11-05T08:47:53.668918Z","shell.execute_reply":"2024-11-05T08:47:53.673331Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def eval_preds(thresholds, y_true, y_pred):\n    y_pred = threshold_rounder(y_pred, thresholds)\n    score = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    return -score","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:53.683392Z","iopub.execute_input":"2024-11-05T08:47:53.683707Z","iopub.status.idle":"2024-11-05T08:47:53.688198Z","shell.execute_reply.started":"2024-11-05T08:47:53.683676Z","shell.execute_reply":"2024-11-05T08:47:53.687316Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:53.696012Z","iopub.execute_input":"2024-11-05T08:47:53.696532Z","iopub.status.idle":"2024-11-05T08:47:53.758111Z","shell.execute_reply.started":"2024-11-05T08:47:53.696499Z","shell.execute_reply":"2024-11-05T08:47:53.757212Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ts_train = pl.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet') \n# ts_test = pl.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:53.759522Z","iopub.execute_input":"2024-11-05T08:47:53.759809Z","iopub.status.idle":"2024-11-05T08:47:53.763777Z","shell.execute_reply.started":"2024-11-05T08:47:53.759778Z","shell.execute_reply":"2024-11-05T08:47:53.762851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:53.764892Z","iopub.execute_input":"2024-11-05T08:47:53.765171Z","iopub.status.idle":"2024-11-05T08:47:53.803992Z","shell.execute_reply.started":"2024-11-05T08:47:53.765140Z","shell.execute_reply":"2024-11-05T08:47:53.803156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.set_index('id', inplace = True)\ndf_test.set_index('id', inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:53.805977Z","iopub.execute_input":"2024-11-05T08:47:53.806876Z","iopub.status.idle":"2024-11-05T08:47:53.812010Z","shell.execute_reply.started":"2024-11-05T08:47:53.806840Z","shell.execute_reply":"2024-11-05T08:47:53.811016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:53.813212Z","iopub.execute_input":"2024-11-05T08:47:53.813871Z","iopub.status.idle":"2024-11-05T08:47:53.844995Z","shell.execute_reply.started":"2024-11-05T08:47:53.813813Z","shell.execute_reply":"2024-11-05T08:47:53.844127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop columns from PCIAT_01 to PCIAT_20\ncolumns_to_drop = [f\"PCIAT-PCIAT_{i:02}\" for i in range(1, 21)]\ndf_train = df_train.drop(columns=columns_to_drop, axis = 1)\ny = df_train['sii']\ndf_train.drop(['PCIAT-PCIAT_Total', 'sii', 'PCIAT-Season'], axis = 1, inplace = True)\n# df_test.drop(['SDS-SEASON'], axis = 1, inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:53.846535Z","iopub.execute_input":"2024-11-05T08:47:53.846850Z","iopub.status.idle":"2024-11-05T08:47:53.856368Z","shell.execute_reply.started":"2024-11-05T08:47:53.846816Z","shell.execute_reply":"2024-11-05T08:47:53.855559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(df_test.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:53.857756Z","iopub.execute_input":"2024-11-05T08:47:53.858033Z","iopub.status.idle":"2024-11-05T08:47:53.863970Z","shell.execute_reply.started":"2024-11-05T08:47:53.858003Z","shell.execute_reply":"2024-11-05T08:47:53.862955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess(df_train):\n# Step 1: Identify numerical and categorical columns\n    numerical_cols = df_train.select_dtypes(include=['int64', 'float64']).columns.tolist()\n    categorical_cols = df_train.select_dtypes(exclude=['int64', 'float64']).columns.tolist()\n    \n    # Step 2: Impute missing values in features\n    num_imputer = SimpleImputer(strategy='mean')\n    df_train[numerical_cols] = num_imputer.fit_transform(df_train[numerical_cols])\n    \n    cat_imputer = SimpleImputer(strategy='most_frequent')\n    df_train[categorical_cols] = cat_imputer.fit_transform(df_train[categorical_cols])\n    \n    for col in categorical_cols:\n        le = LabelEncoder()\n        df_train[col] = le.fit_transform(df_train[col])\n    \n    return df_train\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:53.866413Z","iopub.execute_input":"2024-11-05T08:47:53.866772Z","iopub.status.idle":"2024-11-05T08:47:53.873591Z","shell.execute_reply.started":"2024-11-05T08:47:53.866721Z","shell.execute_reply":"2024-11-05T08:47:53.872771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming df_train is your DataFrame and 'sii' is the target column\n# y = df_train['sii']\n\ndf_train = preprocess(df_train)\n# Step 5: Apply DBSCAN\ndbscan = DBSCAN(eps=0.5, min_samples=5)  # Adjust parameters as necessary\nclusters = dbscan.fit_predict(df_train)\n\n# Add the cluster labels to the original DataFrame\ndf_train['cluster'] = clusters\ndf_train['sii'] = y  # Ensure the target is still linked","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:53.874996Z","iopub.execute_input":"2024-11-05T08:47:53.875306Z","iopub.status.idle":"2024-11-05T08:47:53.968705Z","shell.execute_reply.started":"2024-11-05T08:47:53.875274Z","shell.execute_reply":"2024-11-05T08:47:53.967920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6: Fill missing values in y based on cluster mode\nfor cluster in df_train['cluster'].unique():\n    if cluster == -1:  # Skip noise points that are not clustered\n        continue\n\n    # Get the most common value in 'sii' for each cluster\n    mode_value = df_train.loc[(df_train['cluster'] == cluster) & (df_train['sii'].notna()), 'sii'].mode()\n    if not mode_value.empty:\n        # Fill missing values in 'sii' for this cluster with the mode value\n        df_train.loc[(df_train['cluster'] == cluster) & (df_train['sii'].isna()), 'sii'] = mode_value[0]\n    else:\n        print(f\"No mode found for cluster {cluster}.\")  # Optional: Log the issue\n\n# Fill any remaining NaNs in 'sii' with the overall mode\noverall_mode = df_train['sii'].mode()\nif not overall_mode.empty:\n    df_train['sii'].fillna(overall_mode[0], inplace=True)\n# Check the result\nprint(df_train['sii'].isna().sum())  # Should show 0 if all missing values are filled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:53.970133Z","iopub.execute_input":"2024-11-05T08:47:53.970495Z","iopub.status.idle":"2024-11-05T08:47:53.996495Z","shell.execute_reply.started":"2024-11-05T08:47:53.970461Z","shell.execute_reply":"2024-11-05T08:47:53.995668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.drop('cluster', axis = 1, inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:53.997390Z","iopub.execute_input":"2024-11-05T08:47:53.997684Z","iopub.status.idle":"2024-11-05T08:47:54.005191Z","shell.execute_reply.started":"2024-11-05T08:47:53.997653Z","shell.execute_reply":"2024-11-05T08:47:54.004170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:54.007447Z","iopub.execute_input":"2024-11-05T08:47:54.007769Z","iopub.status.idle":"2024-11-05T08:47:54.020114Z","shell.execute_reply.started":"2024-11-05T08:47:54.007736Z","shell.execute_reply":"2024-11-05T08:47:54.019211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df_train.drop('sii', axis = 1)\ny = df_train['sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:54.021170Z","iopub.execute_input":"2024-11-05T08:47:54.021455Z","iopub.status.idle":"2024-11-05T08:47:54.029352Z","shell.execute_reply.started":"2024-11-05T08:47:54.021397Z","shell.execute_reply":"2024-11-05T08:47:54.028463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.concat([X, y.rename('output')], axis=1)\n\n# Calculate correlation values between each feature in X and y\ncorrelations = data.corr()['output'].drop('output')  # Correlation with the 'output' column\n\n# Plotting the correlation values\nplt.figure(figsize=(15, 8))\ncorrelations.sort_values().plot(kind='bar', color='skyblue')\nplt.title('Correlation of Features in X with Target Variable y')\nplt.xlabel('Correlation Coefficient')\nplt.ylabel('Features')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.030550Z","iopub.execute_input":"2024-11-05T08:47:54.030908Z","iopub.status.idle":"2024-11-05T08:47:54.749138Z","shell.execute_reply.started":"2024-11-05T08:47:54.030867Z","shell.execute_reply":"2024-11-05T08:47:54.748189Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Dictionary to store models and their performance\n# models = {\n#     \"Decision Tree\": DecisionTreeClassifier(),\n#     \"XGBoost\": XGBClassifier(use_label_encoder=False, eval_metric='mlogloss', enable_categorical = True),\n#     \"CatBoost\": CatBoostClassifier(verbose=0),\n#     \"LightGBM\": LGBMClassifier(),\n#     \"AdaBoost\": AdaBoostClassifier(),\n#     \"GradientBoost\": GradientBoostingClassifier()\n# }","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.750309Z","iopub.execute_input":"2024-11-05T08:47:54.750648Z","iopub.status.idle":"2024-11-05T08:47:54.754874Z","shell.execute_reply.started":"2024-11-05T08:47:54.750614Z","shell.execute_reply":"2024-11-05T08:47:54.753957Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Train and evaluate each model using QWK\n# results = {}\n# for name, model in models.items():\n#     # Train the model\n#     if name == 'CatBoost':\n#         model.fit(X_train, y_train, cat_features=cat_cols)\n\n#     model.fit(X_train, y_train)\n    \n#     # Evaluate using Quadratic Weighted Kappa\n#     qwk_score = quadratic_weighted_kappa(model, X_test, y_test)\n#     results[name] = qwk_score\n#     print(f\"{name} QWK Score: {qwk_score:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.756056Z","iopub.execute_input":"2024-11-05T08:47:54.756373Z","iopub.status.idle":"2024-11-05T08:47:54.763605Z","shell.execute_reply.started":"2024-11-05T08:47:54.756340Z","shell.execute_reply":"2024-11-05T08:47:54.762756Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Display all results sorted by QWK score\n# sorted_results = dict(sorted(results.items(), key=lambda item: item[1], reverse=True))\n# print(\"\\nModel Performance (QWK Scores):\")\n# for model, score in sorted_results.items():\n#     print(f\"{model}: {score:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.764818Z","iopub.execute_input":"2024-11-05T08:47:54.765126Z","iopub.status.idle":"2024-11-05T08:47:54.773654Z","shell.execute_reply.started":"2024-11-05T08:47:54.765095Z","shell.execute_reply":"2024-11-05T08:47:54.772758Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = CatBoostClassifier(verbose=0)","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.776650Z","iopub.execute_input":"2024-11-05T08:47:54.776997Z","iopub.status.idle":"2024-11-05T08:47:54.781513Z","shell.execute_reply.started":"2024-11-05T08:47:54.776966Z","shell.execute_reply":"2024-11-05T08:47:54.780585Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cv = StratifiedKFold(5, shuffle=True, random_state=42)\n\n# val_scores = cross_val_score(\n#     model, X_train, y_train, cv=cv, \n#     scoring=KAPPA_SCORER,\n# )\n\n# print(f'kappa score: {np.mean(val_scores):.4f}')","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.782683Z","iopub.execute_input":"2024-11-05T08:47:54.783192Z","iopub.status.idle":"2024-11-05T08:47:54.788640Z","shell.execute_reply.started":"2024-11-05T08:47:54.783159Z","shell.execute_reply":"2024-11-05T08:47:54.787794Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def objective(trial):\n#     # Define hyperparameters to tune\n#     params = {\n#         'iterations': trial.suggest_int('iterations', 100, 1000),\n#         'depth': trial.suggest_int('depth', 4, 10),\n#         'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n#         'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1e-5, 10.0, log=True),\n#         'random_strength': trial.suggest_float('random_strength', 1, 20),\n#         'bagging_temperature': trial.suggest_float('bagging_temperature', 0.0, 1.0),\n#         'cat_features': cat_cols,  # Assuming you have categorical features defined\n#         'verbose': 0,\n#         'eval_metric': 'Kappa'\n#     }\n\n#     # Create and fit the CatBoost model\n#     model = CatBoostClassifier(**params)\n#     model.fit(X_train, y_train, eval_set=(X_test, y_test), early_stopping_rounds=100, verbose=False)\n\n#     # Predict and compute the validation score\n#     y_pred = model.predict(X_test)\n#     score = cohen_kappa_score(y_test, y_pred)\n\n#     return score","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.789731Z","iopub.execute_input":"2024-11-05T08:47:54.790028Z","iopub.status.idle":"2024-11-05T08:47:54.796035Z","shell.execute_reply.started":"2024-11-05T08:47:54.789997Z","shell.execute_reply":"2024-11-05T08:47:54.795233Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# study = optuna.create_study(direction='maximize')  # We want to maximize the score\n# study.optimize(objective, n_trials=50)  # Run 50 trials for hyperparameter optimization\n\n# # Print the best hyperparameters and score\n# print(\"Best hyperparameters: \", study.best_params)\n# print(\"Best score: \", study.best_value)","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.797094Z","iopub.execute_input":"2024-11-05T08:47:54.797370Z","iopub.status.idle":"2024-11-05T08:47:54.803684Z","shell.execute_reply.started":"2024-11-05T08:47:54.797329Z","shell.execute_reply":"2024-11-05T08:47:54.802801Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.info()","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.804734Z","iopub.execute_input":"2024-11-05T08:47:54.805075Z","iopub.status.idle":"2024-11-05T08:47:54.825405Z","shell.execute_reply.started":"2024-11-05T08:47:54.805032Z","shell.execute_reply":"2024-11-05T08:47:54.824643Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# parameters = {'iterations': 800, \n#               'depth': 5, \n#               'learning_rate': 0.1028121118323938, \n#               'l2_leaf_reg': 0.00022223279616889315, \n#               'random_strength': 13.0866816070664, \n#               'bagging_temperature': 0.09964756950107301,\n# #               'cat_features': cat_cols,\n#               'eval_metric': 'Kappa',\n#               'verbose' : 0\n#              }\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.826364Z","iopub.execute_input":"2024-11-05T08:47:54.826694Z","iopub.status.idle":"2024-11-05T08:47:54.830830Z","shell.execute_reply.started":"2024-11-05T08:47:54.826660Z","shell.execute_reply":"2024-11-05T08:47:54.829831Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Create a list of tuples, each containing a name and a CatBoost model\n# estimators = [\n#     (f'catboost_{i}', CatBoostClassifier(**parameters, random_state=random_state))\n#     for i, random_state in enumerate([12, 22, 32, 42, 52, 62, 72, 82, 92, 102])\n# ]\n\n# # Initialize VotingClassifier with the list of estimators\n# model = VotingClassifier(estimators=estimators, voting='soft') ","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.831998Z","iopub.execute_input":"2024-11-05T08:47:54.832336Z","iopub.status.idle":"2024-11-05T08:47:54.837481Z","shell.execute_reply.started":"2024-11-05T08:47:54.832295Z","shell.execute_reply":"2024-11-05T08:47:54.836527Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model = CatBoostClassifier(**parameters, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.838398Z","iopub.execute_input":"2024-11-05T08:47:54.838738Z","iopub.status.idle":"2024-11-05T08:47:54.845367Z","shell.execute_reply.started":"2024-11-05T08:47:54.838698Z","shell.execute_reply":"2024-11-05T08:47:54.844501Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n    'objective'       : 'l2',\n    'verbosity'       : -1,\n    'n_iter'          : 200,\n    'lambda_l1'       : 0.005116829730239727,\n    'lambda_l2'       : 0.0011520776712645852,\n    'learning_rate'   : 0.02376367323636638,\n    'max_depth'       : 5,\n    'num_leaves'      : 207,\n    'colsample_bytree': 0.7759862336963801,\n    'colsample_bynode': 0.5110355095943208,\n    'bagging_fraction': 0.5485770314992224,\n    'bagging_freq'    : 7,\n    'min_data_in_leaf': 78,\n}\n\nmodel = LGBMRegressor(**params, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:54.846546Z","iopub.execute_input":"2024-11-05T08:47:54.847265Z","iopub.status.idle":"2024-11-05T08:47:54.853103Z","shell.execute_reply.started":"2024-11-05T08:47:54.847222Z","shell.execute_reply":"2024-11-05T08:47:54.852268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2024-11-05T08:47:54.854113Z","iopub.execute_input":"2024-11-05T08:47:54.854568Z","iopub.status.idle":"2024-11-05T08:47:55.071468Z","shell.execute_reply.started":"2024-11-05T08:47:54.854536Z","shell.execute_reply":"2024-11-05T08:47:55.070590Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming df_test is your test DataFrame and model is your trained model\nindex = df_test.index\n\ndf_test = preprocess(df_test)\n\n# Get predictions and round them\nres = model.predict(df_test).round().flatten()  # Ensure it's 1D\n\n# Check the shape of index and res\nprint(f\"Shape of index: {index.shape}, Shape of predictions: {res.shape}\")\n\n# Create the submission DataFrame\nsubm = pd.DataFrame({'id': index, 'sii': res})\n\n# Save the DataFrame to CSV\nsubm.to_csv(\"submission.csv\", index=False)  # Avoid writing index in CSV\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T08:47:55.072573Z","iopub.execute_input":"2024-11-05T08:47:55.072849Z","iopub.status.idle":"2024-11-05T08:47:55.103128Z","shell.execute_reply.started":"2024-11-05T08:47:55.072818Z","shell.execute_reply":"2024-11-05T08:47:55.102178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}