{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-10T10:23:53.821019Z","iopub.execute_input":"2024-10-10T10:23:53.821477Z","iopub.status.idle":"2024-10-10T10:23:57.392527Z","shell.execute_reply.started":"2024-10-10T10:23:53.821436Z","shell.execute_reply":"2024-10-10T10:23:57.391091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install fancyimpute","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:23:57.398575Z","iopub.execute_input":"2024-10-10T10:23:57.399129Z","iopub.status.idle":"2024-10-10T10:23:57.405070Z","shell.execute_reply.started":"2024-10-10T10:23:57.399076Z","shell.execute_reply":"2024-10-10T10:23:57.403664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.impute import KNNImputer\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\n# # from fancyimpute import IterativeSVD, KNN, SoftImpute, BiScaler\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:23:57.406762Z","iopub.execute_input":"2024-10-10T10:23:57.407484Z","iopub.status.idle":"2024-10-10T10:23:58.726473Z","shell.execute_reply.started":"2024-10-10T10:23:57.407441Z","shell.execute_reply":"2024-10-10T10:23:58.725438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    import numpy as np\n    import pandas as pd\n    import os\n    import re\n    from sklearn.base import clone\n    from sklearn.metrics import cohen_kappa_score\n    from sklearn.model_selection import StratifiedKFold\n    from scipy.optimize import minimize\n    from concurrent.futures import ThreadPoolExecutor\n    from tqdm import tqdm\n\n    from colorama import Fore, Style\n    from IPython.display import clear_output\n    import warnings\n    from lightgbm import LGBMRegressor\n    from xgboost import XGBRegressor\n    from catboost import CatBoostRegressor\n    from sklearn.ensemble import VotingRegressor\n    warnings.filterwarnings('ignore')\n    pd.options.display.max_columns = None\n\n    SEED = 42\n    n_splits = 5\n\n    # Load datasets\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n    def process_file(filename, dirname):\n        df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n        df.drop('step', axis=1, inplace=True)\n        return df.describe().values.reshape(-1), filename.split('=')[1]\n\n    def load_time_series(dirname) -> pd.DataFrame:\n        ids = os.listdir(dirname)\n\n        with ThreadPoolExecutor() as executor:\n            results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n        stats, indexes = zip(*results)\n\n        df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n        df['id'] = indexes\n\n        return df\n\n    train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n    test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n    time_series_cols = train_ts.columns.tolist()\n    time_series_cols.remove(\"id\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:23:58.729773Z","iopub.execute_input":"2024-10-10T10:23:58.730276Z","iopub.status.idle":"2024-10-10T10:25:40.194721Z","shell.execute_reply.started":"2024-10-10T10:23:58.730208Z","shell.execute_reply":"2024-10-10T10:25:40.193086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"working on train to get the missing sii values","metadata":{}},{"cell_type":"code","source":"# df_si_nan=train[train[\"sii\"].isna()]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.196452Z","iopub.execute_input":"2024-10-10T10:25:40.196862Z","iopub.status.idle":"2024-10-10T10:25:40.202504Z","shell.execute_reply.started":"2024-10-10T10:25:40.196821Z","shell.execute_reply":"2024-10-10T10:25:40.201131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_si_nan.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.203936Z","iopub.execute_input":"2024-10-10T10:25:40.204364Z","iopub.status.idle":"2024-10-10T10:25:40.216369Z","shell.execute_reply.started":"2024-10-10T10:25:40.204322Z","shell.execute_reply":"2024-10-10T10:25:40.215273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1=train.dropna(subset=[\"sii\"],axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.217738Z","iopub.execute_input":"2024-10-10T10:25:40.218105Z","iopub.status.idle":"2024-10-10T10:25:40.228442Z","shell.execute_reply.started":"2024-10-10T10:25:40.218068Z","shell.execute_reply":"2024-10-10T10:25:40.227109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.229865Z","iopub.execute_input":"2024-10-10T10:25:40.230298Z","iopub.status.idle":"2024-10-10T10:25:40.243856Z","shell.execute_reply.started":"2024-10-10T10:25:40.230223Z","shell.execute_reply":"2024-10-10T10:25:40.242710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cat_col2= df1.select_dtypes(include='object').columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.245135Z","iopub.execute_input":"2024-10-10T10:25:40.245530Z","iopub.status.idle":"2024-10-10T10:25:40.255398Z","shell.execute_reply.started":"2024-10-10T10:25:40.245490Z","shell.execute_reply":"2024-10-10T10:25:40.254114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# id_nan=df_si_nan[\"id\"]\n# id1=df1[\"id\"]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.257166Z","iopub.execute_input":"2024-10-10T10:25:40.257707Z","iopub.status.idle":"2024-10-10T10:25:40.268324Z","shell.execute_reply.started":"2024-10-10T10:25:40.257652Z","shell.execute_reply":"2024-10-10T10:25:40.267062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1=df1.drop(\"id\",axis=1)\n# df_si_nan=df_si_nan.drop(\"id\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.270226Z","iopub.execute_input":"2024-10-10T10:25:40.270770Z","iopub.status.idle":"2024-10-10T10:25:40.283078Z","shell.execute_reply.started":"2024-10-10T10:25:40.270713Z","shell.execute_reply":"2024-10-10T10:25:40.281812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tr=df1.sii","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.284637Z","iopub.execute_input":"2024-10-10T10:25:40.285081Z","iopub.status.idle":"2024-10-10T10:25:40.296081Z","shell.execute_reply.started":"2024-10-10T10:25:40.285037Z","shell.execute_reply":"2024-10-10T10:25:40.293873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tr.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.303550Z","iopub.execute_input":"2024-10-10T10:25:40.304281Z","iopub.status.idle":"2024-10-10T10:25:40.309833Z","shell.execute_reply.started":"2024-10-10T10:25:40.304217Z","shell.execute_reply":"2024-10-10T10:25:40.308546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1=df1.drop(\"sii\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.311447Z","iopub.execute_input":"2024-10-10T10:25:40.311967Z","iopub.status.idle":"2024-10-10T10:25:40.323806Z","shell.execute_reply.started":"2024-10-10T10:25:40.311887Z","shell.execute_reply":"2024-10-10T10:25:40.322677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.325001Z","iopub.execute_input":"2024-10-10T10:25:40.325426Z","iopub.status.idle":"2024-10-10T10:25:40.335464Z","shell.execute_reply.started":"2024-10-10T10:25:40.325386Z","shell.execute_reply":"2024-10-10T10:25:40.334090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for c in cat_col2:\n    \n#     if(c!=\"id\"):df1[c].fillna('missing', inplace=True)\n     ","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.336982Z","iopub.execute_input":"2024-10-10T10:25:40.337487Z","iopub.status.idle":"2024-10-10T10:25:40.345865Z","shell.execute_reply.started":"2024-10-10T10:25:40.337430Z","shell.execute_reply":"2024-10-10T10:25:40.344786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for c in cat_col2:\n    \n#     if(c!=\"id\"):df_si_nan[c].fillna('missing', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.347411Z","iopub.execute_input":"2024-10-10T10:25:40.348481Z","iopub.status.idle":"2024-10-10T10:25:40.358162Z","shell.execute_reply.started":"2024-10-10T10:25:40.348425Z","shell.execute_reply":"2024-10-10T10:25:40.356996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in cat_col2:\n#     if(col!=\"id\"):df1[col] = df1[col].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.360076Z","iopub.execute_input":"2024-10-10T10:25:40.360587Z","iopub.status.idle":"2024-10-10T10:25:40.375357Z","shell.execute_reply.started":"2024-10-10T10:25:40.360544Z","shell.execute_reply":"2024-10-10T10:25:40.374310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in cat_col2:\n#     if(col!=\"id\"):df_si_nan[col] = df_si_nan[col].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.376853Z","iopub.execute_input":"2024-10-10T10:25:40.377284Z","iopub.status.idle":"2024-10-10T10:25:40.386830Z","shell.execute_reply.started":"2024-10-10T10:25:40.377211Z","shell.execute_reply":"2024-10-10T10:25:40.385585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.388263Z","iopub.execute_input":"2024-10-10T10:25:40.388678Z","iopub.status.idle":"2024-10-10T10:25:40.399035Z","shell.execute_reply.started":"2024-10-10T10:25:40.388639Z","shell.execute_reply":"2024-10-10T10:25:40.397681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_si_nan=df_si_nan.drop(\"sii\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.400479Z","iopub.execute_input":"2024-10-10T10:25:40.400844Z","iopub.status.idle":"2024-10-10T10:25:40.410469Z","shell.execute_reply.started":"2024-10-10T10:25:40.400806Z","shell.execute_reply":"2024-10-10T10:25:40.409237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.utils import class_weight\n# from sklearn.model_selection import train_test_split\n\n# # Assuming 'sii' is your target column in the train DataFrame\n# y = tr\n\n# # Calculate class weights using the 'balanced' strategy\n# class_weights = class_weight.compute_class_weight('balanced', classes=np.unique(y), y=y)\n\n# # Convert class weights into a dictionary (use the class index as key)\n# class_weights_dict = {i: weight for i, weight in enumerate(class_weights)}\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.412319Z","iopub.execute_input":"2024-10-10T10:25:40.412699Z","iopub.status.idle":"2024-10-10T10:25:40.421870Z","shell.execute_reply.started":"2024-10-10T10:25:40.412659Z","shell.execute_reply":"2024-10-10T10:25:40.420559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights_dict","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.423557Z","iopub.execute_input":"2024-10-10T10:25:40.424093Z","iopub.status.idle":"2024-10-10T10:25:40.432191Z","shell.execute_reply.started":"2024-10-10T10:25:40.424035Z","shell.execute_reply":"2024-10-10T10:25:40.430964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Modify the Params dictionary to include the calculated class weights\n# Params = {\n#     'learning_rate': 0.05,\n#         'max_depth': 6,\n#         'n_estimators': 200,\n#         'subsample': 0.8,\n#         'colsample_bytree': 0.8,\n#         'reg_alpha': 1,  # Increased from 0.1\n#         'reg_lambda': 5,   # Increased from 2.68e-06\n#     'class_weight': class_weights_dict  # Add the class weights here\n# }\n\n\n# from xgboost import XGBClassifier\n# Light = XGBClassifier(**Params, random_state=SEED, verbose=-1,enable_categorical=True)\n\n\n# # X_resampled","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.433723Z","iopub.execute_input":"2024-10-10T10:25:40.434212Z","iopub.status.idle":"2024-10-10T10:25:40.443535Z","shell.execute_reply.started":"2024-10-10T10:25:40.434155Z","shell.execute_reply":"2024-10-10T10:25:40.442097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train, X_val, y_train, y_val = train_test_split(df1, tr, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.445041Z","iopub.execute_input":"2024-10-10T10:25:40.445542Z","iopub.status.idle":"2024-10-10T10:25:40.455023Z","shell.execute_reply.started":"2024-10-10T10:25:40.445487Z","shell.execute_reply":"2024-10-10T10:25:40.452617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import accuracy_score, classification_report, confusion_matrix\n\n# Light.fit(X_train, y_train)\n\n# y_pred = Light.predict(X_val)\n\n# print(classification_report(y_val, y_pred))\n# print(confusion_matrix(y_val, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.457238Z","iopub.execute_input":"2024-10-10T10:25:40.457740Z","iopub.status.idle":"2024-10-10T10:25:40.463488Z","shell.execute_reply.started":"2024-10-10T10:25:40.457694Z","shell.execute_reply":"2024-10-10T10:25:40.462309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Light.fit(df1, tr)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.465026Z","iopub.execute_input":"2024-10-10T10:25:40.465448Z","iopub.status.idle":"2024-10-10T10:25:40.479228Z","shell.execute_reply.started":"2024-10-10T10:25:40.465408Z","shell.execute_reply":"2024-10-10T10:25:40.477992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pr=Light.predict(df_si_nan)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.480810Z","iopub.execute_input":"2024-10-10T10:25:40.481226Z","iopub.status.idle":"2024-10-10T10:25:40.490686Z","shell.execute_reply.started":"2024-10-10T10:25:40.481186Z","shell.execute_reply":"2024-10-10T10:25:40.489467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pr=pd.DataFrame(pr)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.492202Z","iopub.execute_input":"2024-10-10T10:25:40.492597Z","iopub.status.idle":"2024-10-10T10:25:40.501171Z","shell.execute_reply.started":"2024-10-10T10:25:40.492558Z","shell.execute_reply":"2024-10-10T10:25:40.500053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pr.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.502720Z","iopub.execute_input":"2024-10-10T10:25:40.503174Z","iopub.status.idle":"2024-10-10T10:25:40.513706Z","shell.execute_reply.started":"2024-10-10T10:25:40.503121Z","shell.execute_reply":"2024-10-10T10:25:40.512583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1=pd.merge(train,train_ts,how=\"left\",on=\"id\")\ntest1=pd.merge(test,test_ts,how=\"left\",on=\"id\")","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.515173Z","iopub.execute_input":"2024-10-10T10:25:40.515668Z","iopub.status.idle":"2024-10-10T10:25:40.548944Z","shell.execute_reply.started":"2024-10-10T10:25:40.515612Z","shell.execute_reply":"2024-10-10T10:25:40.547413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.550329Z","iopub.execute_input":"2024-10-10T10:25:40.550751Z","iopub.status.idle":"2024-10-10T10:25:40.559404Z","shell.execute_reply.started":"2024-10-10T10:25:40.550704Z","shell.execute_reply":"2024-10-10T10:25:40.558108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.561131Z","iopub.execute_input":"2024-10-10T10:25:40.561638Z","iopub.status.idle":"2024-10-10T10:25:40.694490Z","shell.execute_reply.started":"2024-10-10T10:25:40.561583Z","shell.execute_reply":"2024-10-10T10:25:40.693343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"sii\"].value_counts().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.695755Z","iopub.execute_input":"2024-10-10T10:25:40.696108Z","iopub.status.idle":"2024-10-10T10:25:40.710422Z","shell.execute_reply.started":"2024-10-10T10:25:40.696069Z","shell.execute_reply":"2024-10-10T10:25:40.709126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.711850Z","iopub.execute_input":"2024-10-10T10:25:40.712317Z","iopub.status.idle":"2024-10-10T10:25:40.846270Z","shell.execute_reply.started":"2024-10-10T10:25:40.712260Z","shell.execute_reply":"2024-10-10T10:25:40.844916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f_col = ['id','Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                    'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                    'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                    'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                    'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                    'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                    'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                    'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                    'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                    'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                    'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                    'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                    'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                    'PreInt_EduHx-computerinternet_hoursday', 'sii']","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.848223Z","iopub.execute_input":"2024-10-10T10:25:40.848752Z","iopub.status.idle":"2024-10-10T10:25:40.858114Z","shell.execute_reply.started":"2024-10-10T10:25:40.848697Z","shell.execute_reply":"2024-10-10T10:25:40.856749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fe_col=f_col+time_series_cols","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.859544Z","iopub.execute_input":"2024-10-10T10:25:40.859966Z","iopub.status.idle":"2024-10-10T10:25:40.868871Z","shell.execute_reply.started":"2024-10-10T10:25:40.859915Z","shell.execute_reply":"2024-10-10T10:25:40.867706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fe_col","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.870284Z","iopub.execute_input":"2024-10-10T10:25:40.871822Z","iopub.status.idle":"2024-10-10T10:25:40.895442Z","shell.execute_reply.started":"2024-10-10T10:25:40.871779Z","shell.execute_reply":"2024-10-10T10:25:40.894070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2=train1[fe_col]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.897469Z","iopub.execute_input":"2024-10-10T10:25:40.897878Z","iopub.status.idle":"2024-10-10T10:25:40.910067Z","shell.execute_reply.started":"2024-10-10T10:25:40.897823Z","shell.execute_reply":"2024-10-10T10:25:40.908497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:40.924485Z","iopub.execute_input":"2024-10-10T10:25:40.924931Z","iopub.status.idle":"2024-10-10T10:25:41.165995Z","shell.execute_reply.started":"2024-10-10T10:25:40.924887Z","shell.execute_reply":"2024-10-10T10:25:41.164799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1=train2.dropna(subset=['sii'],axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.167641Z","iopub.execute_input":"2024-10-10T10:25:41.168111Z","iopub.status.idle":"2024-10-10T10:25:41.180232Z","shell.execute_reply.started":"2024-10-10T10:25:41.168062Z","shell.execute_reply":"2024-10-10T10:25:41.178777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id3=train1[\"id\"]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.181832Z","iopub.execute_input":"2024-10-10T10:25:41.182320Z","iopub.status.idle":"2024-10-10T10:25:41.189971Z","shell.execute_reply.started":"2024-10-10T10:25:41.182267Z","shell.execute_reply":"2024-10-10T10:25:41.188909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1=train1.drop(\"id\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.191808Z","iopub.execute_input":"2024-10-10T10:25:41.192359Z","iopub.status.idle":"2024-10-10T10:25:41.204536Z","shell.execute_reply.started":"2024-10-10T10:25:41.192304Z","shell.execute_reply":"2024-10-10T10:25:41.203194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train1.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.206205Z","iopub.execute_input":"2024-10-10T10:25:41.206679Z","iopub.status.idle":"2024-10-10T10:25:41.214291Z","shell.execute_reply.started":"2024-10-10T10:25:41.206628Z","shell.execute_reply":"2024-10-10T10:25:41.213115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_col1= train1.select_dtypes(include='object').columns.tolist()\n# cat_col2 = train.select_dtypes(include='category').columns.tolist()\n\nf_col.remove(\"id\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.215726Z","iopub.execute_input":"2024-10-10T10:25:41.216240Z","iopub.status.idle":"2024-10-10T10:25:41.226610Z","shell.execute_reply.started":"2024-10-10T10:25:41.216172Z","shell.execute_reply":"2024-10-10T10:25:41.225334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(cat_col1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.228159Z","iopub.execute_input":"2024-10-10T10:25:41.228577Z","iopub.status.idle":"2024-10-10T10:25:41.243787Z","shell.execute_reply.started":"2024-10-10T10:25:41.228537Z","shell.execute_reply":"2024-10-10T10:25:41.242620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_col1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.245277Z","iopub.execute_input":"2024-10-10T10:25:41.245765Z","iopub.status.idle":"2024-10-10T10:25:41.253314Z","shell.execute_reply.started":"2024-10-10T10:25:41.245711Z","shell.execute_reply":"2024-10-10T10:25:41.252283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1[\"sii\"].isna().value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.254633Z","iopub.execute_input":"2024-10-10T10:25:41.255039Z","iopub.status.idle":"2024-10-10T10:25:41.271257Z","shell.execute_reply.started":"2024-10-10T10:25:41.255000Z","shell.execute_reply":"2024-10-10T10:25:41.270026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def mis_cat(df):\n#     for c in cat_col1:\n#         df[c].fillna('missing', inplace=True)\n#     return df    ","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.272673Z","iopub.execute_input":"2024-10-10T10:25:41.273059Z","iopub.status.idle":"2024-10-10T10:25:41.281371Z","shell.execute_reply.started":"2024-10-10T10:25:41.273019Z","shell.execute_reply":"2024-10-10T10:25:41.280118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mis_cat(train1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.282857Z","iopub.execute_input":"2024-10-10T10:25:41.283349Z","iopub.status.idle":"2024-10-10T10:25:41.292543Z","shell.execute_reply.started":"2024-10-10T10:25:41.283299Z","shell.execute_reply":"2024-10-10T10:25:41.291358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_y","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.293983Z","iopub.execute_input":"2024-10-10T10:25:41.294434Z","iopub.status.idle":"2024-10-10T10:25:41.304395Z","shell.execute_reply.started":"2024-10-10T10:25:41.294390Z","shell.execute_reply":"2024-10-10T10:25:41.302932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1.tail(20)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.306187Z","iopub.execute_input":"2024-10-10T10:25:41.306837Z","iopub.status.idle":"2024-10-10T10:25:41.621326Z","shell.execute_reply.started":"2024-10-10T10:25:41.306653Z","shell.execute_reply":"2024-10-10T10:25:41.620109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_data=train1[cat_col1]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.622745Z","iopub.execute_input":"2024-10-10T10:25:41.623128Z","iopub.status.idle":"2024-10-10T10:25:41.629948Z","shell.execute_reply.started":"2024-10-10T10:25:41.623087Z","shell.execute_reply":"2024-10-10T10:25:41.628458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_data = cat_data.reset_index(drop=True)\ncat_data","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.631866Z","iopub.execute_input":"2024-10-10T10:25:41.632441Z","iopub.status.idle":"2024-10-10T10:25:41.655125Z","shell.execute_reply.started":"2024-10-10T10:25:41.632385Z","shell.execute_reply":"2024-10-10T10:25:41.653867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_data","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.656528Z","iopub.execute_input":"2024-10-10T10:25:41.657638Z","iopub.status.idle":"2024-10-10T10:25:41.678823Z","shell.execute_reply.started":"2024-10-10T10:25:41.657591Z","shell.execute_reply":"2024-10-10T10:25:41.677682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def create_mapping(column, dataset):\n#     unique_values = dataset[column].unique()\n#     return {value: idx for idx, value in enumerate(unique_values)}\n\n# for col in cat_col1:\n#     # Create mapping for both train and test datasets\n#     mapping = create_mapping(col, train)\n#     mappingTe = create_mapping(col, test)\n    \n#     # Replace missing or invalid values before converting to integers\n#     cat_data[col] = cat_data[col].replace(mapping).fillna(-1)  # Replace missing values with -1\n#     cat_data[col] = cat_data[col].astype(int)\n\n#     # Uncomment the following line if you want to process the test set similarly\n#     # test[col] = test[col].replace(mappingTe).fillna(-1).astype(int)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.680357Z","iopub.execute_input":"2024-10-10T10:25:41.680786Z","iopub.status.idle":"2024-10-10T10:25:41.688881Z","shell.execute_reply.started":"2024-10-10T10:25:41.680745Z","shell.execute_reply":"2024-10-10T10:25:41.687733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_col1:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test1)\n    \n    cat_data[col] = cat_data[col].replace(mapping).astype(int)\n    test1[col] = test1[col].replace(mappingTe).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.690436Z","iopub.execute_input":"2024-10-10T10:25:41.690927Z","iopub.status.idle":"2024-10-10T10:25:41.749408Z","shell.execute_reply.started":"2024-10-10T10:25:41.690855Z","shell.execute_reply":"2024-10-10T10:25:41.748313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_data.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.750714Z","iopub.execute_input":"2024-10-10T10:25:41.751082Z","iopub.status.idle":"2024-10-10T10:25:41.759660Z","shell.execute_reply.started":"2024-10-10T10:25:41.751042Z","shell.execute_reply":"2024-10-10T10:25:41.758499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_in=train1.drop(cat_col1,axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.761081Z","iopub.execute_input":"2024-10-10T10:25:41.761499Z","iopub.status.idle":"2024-10-10T10:25:41.770088Z","shell.execute_reply.started":"2024-10-10T10:25:41.761458Z","shell.execute_reply":"2024-10-10T10:25:41.768650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t=train_in.drop(\"sii\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.771865Z","iopub.execute_input":"2024-10-10T10:25:41.772260Z","iopub.status.idle":"2024-10-10T10:25:41.782496Z","shell.execute_reply.started":"2024-10-10T10:25:41.772208Z","shell.execute_reply":"2024-10-10T10:25:41.781136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# id2=train1[\"id\"]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.784012Z","iopub.execute_input":"2024-10-10T10:25:41.784484Z","iopub.status.idle":"2024-10-10T10:25:41.792175Z","shell.execute_reply.started":"2024-10-10T10:25:41.784435Z","shell.execute_reply":"2024-10-10T10:25:41.790995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_in1=train_in.drop(\"sii\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.793771Z","iopub.execute_input":"2024-10-10T10:25:41.794277Z","iopub.status.idle":"2024-10-10T10:25:41.807599Z","shell.execute_reply.started":"2024-10-10T10:25:41.794204Z","shell.execute_reply":"2024-10-10T10:25:41.806307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=train_in1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.809314Z","iopub.execute_input":"2024-10-10T10:25:41.809694Z","iopub.status.idle":"2024-10-10T10:25:41.814792Z","shell.execute_reply.started":"2024-10-10T10:25:41.809655Z","shell.execute_reply":"2024-10-10T10:25:41.813628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test3.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.816350Z","iopub.execute_input":"2024-10-10T10:25:41.816806Z","iopub.status.idle":"2024-10-10T10:25:41.824361Z","shell.execute_reply.started":"2024-10-10T10:25:41.816765Z","shell.execute_reply":"2024-10-10T10:25:41.823151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tid=test1[\"id\"]\ntest2=test1.drop(\"id\",axis=1)\ncat_td=test2[cat_col1]\ncat_td=cat_td.reset_index(drop=True)\ntest3=test2.drop(cat_col1,axis=1)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.825998Z","iopub.execute_input":"2024-10-10T10:25:41.826521Z","iopub.status.idle":"2024-10-10T10:25:41.838881Z","shell.execute_reply.started":"2024-10-10T10:25:41.826468Z","shell.execute_reply":"2024-10-10T10:25:41.837568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.experimental import enable_iterative_imputer\n# from sklearn.impute import IterativeImputer\n# from sklearn.decomposition import TruncatedSVD\n# from sklearn.preprocessing import StandardScaler\n# from sklearn.linear_model import BayesianRidge\n# import numpy as np\n# import pandas as pd\n# from sklearn.preprocessing import StandardScaler\n# from tensorflow.keras.models import Model\n# from tensorflow.keras.layers import Input, Dense\n# from sklearn.ensemble import RandomForestRegressor  # Import Random Forest Regressor\n# # This model supports fit and predict\n\n# # Assuming your dataset is loaded into a DataFrame 'df'\n# # df = pd.read_csv('train23.csv')  # Replace with your actual data loading code\n\n# # Standardizing the features (important for SVD)\n# def Fil(df,test):\n#     scaler1 = StandardScaler()\n#     scaler2= StandardScaler()\n#     scaled_data = scaler1.fit_transform(df)\n#     test_scaled=scaler2.fit_transform(test)\n# # Define the SVD model for matrix decomposition\n#     ridge_estimator = BayesianRidge()\n# #     random_forest_estimator1 = RandomForestRegressor(n_estimators=150,max_depth=10 ,random_state=0, n_jobs=-1)\n\n\n# # Define the Iterative Imputer with the Ridge Estimator\n#     imputer = IterativeImputer(estimator=ridge_estimator, max_iter=20, tol=1e-3, random_state=0, verbose=2)\n\n# # Fit the imputer on the dataset and transform the data\n#     imputed_data = imputer.fit_transform(scaled_data)\n#     test1=imputer.transform(test_scaled)\n\n# # Inverse transform the scaled data back to the original scale\n#     data_imputed_original_scale = scaler1.inverse_transform(imputed_data)\n#     testo=scaler2.inverse_transform(test1)\n# # Convert back to DataFrame\n#     train2_new = pd.DataFrame(data_imputed_original_scale, columns=df.columns)\n#     test_n=pd.DataFrame(testo, columns=test.columns)\n#     return train2_new,test_n\n\n\n\n# train_data = train_in1  # Your training data\n# test_data = test3   # Your test data\n\n# # Fill missing values temporarily, since autoencoders need complete data\n# train_data,test_data=Fil(train_data,test_data)\n\n# # Standardize the data\n# scaler = StandardScaler()\n# train_scaled = scaler.fit_transform(train_data)\n# test_scaled = scaler.transform(test_data)\n\n# # Define the dimensions\n# input_dim = train_scaled.shape[1]  # Number of features in the input\n# encoding_dim = 100  # You want to reduce to 100 dimensions\n\n# # Build the autoencoder model\n# input_layer = Input(shape=(input_dim,))\n# encoded = Dense(128, activation='relu')(input_layer)\n# encoded = Dense(encoding_dim, activation='relu')(encoded)\n\n# decoded = Dense(128, activation='relu')(encoded)\n# decoded = Dense(input_dim, activation='sigmoid')(decoded)  # Sigmoid to reconstruct original data\n\n# autoencoder = Model(input_layer, decoded)\n\n# # Compile the model\n# autoencoder.compile(optimizer='adam', loss='mean_squared_error')\n\n# # Train the autoencoder on the training data\n# autoencoder.fit(train_scaled, train_scaled, \n#                 epochs=99, batch_size=32, shuffle=True, \n#                 verbose=1)\n\n# # Create a separate encoder model for dimensionality reduction\n# encoder = Model(input_layer, encoded)\n\n# # Apply the encoder to the training data\n# train2_new = encoder.predict(train_scaled)\n\n# # Apply the encoder to the test data\n# test4 = encoder.predict(test_scaled)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.840646Z","iopub.execute_input":"2024-10-10T10:25:41.841270Z","iopub.status.idle":"2024-10-10T10:25:41.853396Z","shell.execute_reply.started":"2024-10-10T10:25:41.841201Z","shell.execute_reply":"2024-10-10T10:25:41.852171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import necessary libraries\nfrom sklearn.experimental import enable_iterative_imputer  # Enable experimental feature for iterative imputation\nfrom sklearn.impute import IterativeImputer  # For handling missing data\nfrom sklearn.preprocessing import StandardScaler  # For standardizing data\nfrom sklearn.linear_model import BayesianRidge  # Estimator for iterative imputer\nimport numpy as np\nimport pandas as pd\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Input, Dense  # Keras layers for building neural networks\n\n# Define function for filling missing values using IterativeImputer\ndef Fil(df, test):\n    # Initialize scalers for train and test data\n    scaler1 = StandardScaler()\n    scaler2 = StandardScaler()\n    \n    # Standardize the training and test data\n    scaled_data = scaler1.fit_transform(df)\n    test_scaled = scaler2.fit_transform(test)\n    \n    # Define the BayesianRidge estimator for the imputer\n    ridge_estimator = BayesianRidge()\n    \n    # Create the IterativeImputer instance with the BayesianRidge estimator\n    imputer = IterativeImputer(estimator=ridge_estimator, max_iter=13, tol=1e-3, random_state=0, verbose=2)\n    \n    # Apply the imputer to the standardized data to fill missing values\n    imputed_data = imputer.fit_transform(scaled_data)\n    test_imputed = imputer.transform(test_scaled)\n\n    # Inverse transform the data back to the original scale\n    data_imputed_original_scale = scaler1.inverse_transform(imputed_data)\n    test_original_scale = scaler2.inverse_transform(test_imputed)\n\n    # Convert the imputed arrays back to DataFrames for easier handling\n    train2_new = pd.DataFrame(data_imputed_original_scale, columns=df.columns)\n    test_n = pd.DataFrame(test_original_scale, columns=test.columns)\n    \n    return train2_new, test_n\n\n# Load and preprocess the training and test datasets\ntrain_data = train_in1  # Replace this with actual loading code for your training data\ntest_data = test3  # Replace this with actual loading code for your test data\n\n# Fill missing values using the Fil function, which returns imputed DataFrames\ntrain2_new, test4= Fil(train_data, test_data)\n\n# # Standardize the imputed training and test data for use with the autoencoder\n# scaler1 = StandardScaler()\n# scaler2 = StandardScaler()\n# train_scaled = scaler1.fit_transform(train_data)\n# test_scaled = scaler2.fit_transform(test_data)\n\n# # Define the dimensions for the autoencoder\n# input_dim = train_scaled.shape[1]  # Number of features in the input data\n# encoding_dim = 145  # Reduced dimensionality (100 latent features)\n\n# # Build the autoencoder model\n# input_layer = Input(shape=(input_dim,))  # Input layer with the original number of features\n# encoded = Dense(128, activation='relu')(input_layer)  # First hidden layer\n# encoded = Dense(encoding_dim, activation='relu')(encoded)  # Encoding layer (reduced dimension)\n\n# # Decoder part to reconstruct the original input\n# decoded = Dense(128, activation='relu')(encoded)  # Decoder hidden layer\n# decoded = Dense(input_dim, activation='sigmoid')(decoded)  # Output layer with original number of features (use sigmoid)\n\n# # Create the autoencoder model\n# autoencoder = Model(input_layer, decoded)\n\n# # Compile the model\n# autoencoder.compile(optimizer='adam', loss='mean_squared_error')\n\n# # Train the autoencoder model on the training data\n# autoencoder.fit(train_scaled, train_scaled, \n#                 epochs=99, batch_size=32, shuffle=True, \n#                 verbose=1)\n\n# # Create a separate encoder model to obtain the reduced dimensions\n# encoder = Model(input_layer, encoded)\n\n# # Use the encoder to transform (reduce dimensionality of) the training data\n# train2_new1 = encoder.predict(train_scaled)\n# # train2_new=scaler1.inverse_transform(train2_new1)\n\n# # Use the encoder to transform the test data\n# test41 = encoder.predict(test_scaled)\n# # test4=scaler2.inverse_transform(test41)\n\n# # Now, train2_new and test4 have reduced dimensions (100-dimensional latent space) and can be used for further tasks\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:25:41.855018Z","iopub.execute_input":"2024-10-10T10:25:41.855530Z","iopub.status.idle":"2024-10-10T10:31:03.506658Z","shell.execute_reply.started":"2024-10-10T10:25:41.855484Z","shell.execute_reply":"2024-10-10T10:31:03.504729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train2_new=pd.DataFrame(train2_new1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.509537Z","iopub.execute_input":"2024-10-10T10:31:03.511653Z","iopub.status.idle":"2024-10-10T10:31:03.518198Z","shell.execute_reply.started":"2024-10-10T10:31:03.511600Z","shell.execute_reply":"2024-10-10T10:31:03.517017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2_new","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.519631Z","iopub.execute_input":"2024-10-10T10:31:03.520018Z","iopub.status.idle":"2024-10-10T10:31:03.693771Z","shell.execute_reply.started":"2024-10-10T10:31:03.519978Z","shell.execute_reply":"2024-10-10T10:31:03.692482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test4=pd.DataFrame(test41)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.695464Z","iopub.execute_input":"2024-10-10T10:31:03.695934Z","iopub.status.idle":"2024-10-10T10:31:03.701560Z","shell.execute_reply.started":"2024-10-10T10:31:03.695874Z","shell.execute_reply":"2024-10-10T10:31:03.700314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test4.columns = test4.columns.astype(str)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.703230Z","iopub.execute_input":"2024-10-10T10:31:03.703669Z","iopub.status.idle":"2024-10-10T10:31:03.713049Z","shell.execute_reply.started":"2024-10-10T10:31:03.703629Z","shell.execute_reply":"2024-10-10T10:31:03.711797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train2_new.columns = train2_new.columns.astype(str)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.714783Z","iopub.execute_input":"2024-10-10T10:31:03.715321Z","iopub.status.idle":"2024-10-10T10:31:03.723522Z","shell.execute_reply.started":"2024-10-10T10:31:03.715262Z","shell.execute_reply":"2024-10-10T10:31:03.722313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testf=pd.concat([cat_td,test4],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.725094Z","iopub.execute_input":"2024-10-10T10:31:03.725554Z","iopub.status.idle":"2024-10-10T10:31:03.736046Z","shell.execute_reply.started":"2024-10-10T10:31:03.725501Z","shell.execute_reply":"2024-10-10T10:31:03.734839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test4","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.737837Z","iopub.execute_input":"2024-10-10T10:31:03.738465Z","iopub.status.idle":"2024-10-10T10:31:03.953993Z","shell.execute_reply.started":"2024-10-10T10:31:03.738408Z","shell.execute_reply":"2024-10-10T10:31:03.952796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# import pandas as pd\n# from sklearn.preprocessing import StandardScaler\n# from tensorflow.keras.models import Model\n# from tensorflow.keras.layers import Input, Dense\n\n# # Assuming you have train and test data already loaded as train_data and test_data\n# train_data = train_in1  # Your training data\n# test_data = test3   # Your test data\n\n# # Fill missing values temporarily, since autoencoders need complete data\n# train_data,test_data=Fil(train_data,test_data)\n\n# # Standardize the data\n# scaler = StandardScaler()\n# train_scaled = scaler.fit_transform(train_data)\n# test_scaled = scaler.transform(test_data)\n\n# # Define the dimensions\n# input_dim = train_scaled.shape[1]  # Number of features in the input\n# encoding_dim = 100  # You want to reduce to 100 dimensions\n\n# # Build the autoencoder model\n# input_layer = Input(shape=(input_dim,))\n# encoded = Dense(128, activation='relu')(input_layer)\n# encoded = Dense(encoding_dim, activation='relu')(encoded)\n\n# decoded = Dense(128, activation='relu')(encoded)\n# decoded = Dense(input_dim, activation='sigmoid')(decoded)  # Sigmoid to reconstruct original data\n\n# autoencoder = Model(input_layer, decoded)\n\n# # Compile the model\n# autoencoder.compile(optimizer='adam', loss='mean_squared_error')\n\n# # Train the autoencoder on the training data\n# autoencoder.fit(train_scaled, train_scaled, \n#                 epochs=99, batch_size=32, shuffle=True, \n#                 verbose=1)\n\n# # Create a separate encoder model for dimensionality reduction\n# encoder = Model(input_layer, encoded)\n\n# # Apply the encoder to the training data\n# train2_new = encoder.predict(train_scaled)\n\n# # Apply the encoder to the test data\n# test4 = encoder.predict(test_scaled)\n\n# # Now train_encoded and test_encoded have 100 dimensions, you can use them for further prediction\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.955547Z","iopub.execute_input":"2024-10-10T10:31:03.956273Z","iopub.status.idle":"2024-10-10T10:31:03.963066Z","shell.execute_reply.started":"2024-10-10T10:31:03.956207Z","shell.execute_reply":"2024-10-10T10:31:03.961910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# conda install -c rapidsai -c nvidia -c conda-forge cuml=23.10 python=3.10 cudatoolkit=11.8\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.964435Z","iopub.execute_input":"2024-10-10T10:31:03.964806Z","iopub.status.idle":"2024-10-10T10:31:03.978299Z","shell.execute_reply.started":"2024-10-10T10:31:03.964767Z","shell.execute_reply":"2024-10-10T10:31:03.977005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train2_new=train_in1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.979852Z","iopub.execute_input":"2024-10-10T10:31:03.980398Z","iopub.status.idle":"2024-10-10T10:31:03.989846Z","shell.execute_reply.started":"2024-10-10T10:31:03.980340Z","shell.execute_reply":"2024-10-10T10:31:03.988468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train2_ = IterativeSVD().fit_transform(train_in)\n# train2 = pd.DataFrame(train2_, columns=train_in.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:03.991908Z","iopub.execute_input":"2024-10-10T10:31:03.992378Z","iopub.status.idle":"2024-10-10T10:31:04.000802Z","shell.execute_reply.started":"2024-10-10T10:31:03.992324Z","shell.execute_reply":"2024-10-10T10:31:03.999735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install missingpy","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.002399Z","iopub.execute_input":"2024-10-10T10:31:04.002833Z","iopub.status.idle":"2024-10-10T10:31:04.017542Z","shell.execute_reply.started":"2024-10-10T10:31:04.002791Z","shell.execute_reply":"2024-10-10T10:31:04.016328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# threshold = 0.8 # 50%\n# columns_excessive_nan = train_in1.columns[train_in1.isna().mean() > threshold].tolist()\n\n# if columns_excessive_nan:\n#     print(f\"Columns with more than {threshold*100}% NaN values:\", columns_excessive_nan)\n#     # Optionally, drop these columns or handle them differently\n# #     train_in1 = train_in1.drop(columns=columns_excessive_nan)\n# else:\n#     print(f\"No columns with more than {threshold*100}% NaN values.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.019205Z","iopub.execute_input":"2024-10-10T10:31:04.019727Z","iopub.status.idle":"2024-10-10T10:31:04.031436Z","shell.execute_reply.started":"2024-10-10T10:31:04.019673Z","shell.execute_reply":"2024-10-10T10:31:04.030013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.experimental import enable_iterative_imputer\n# from sklearn.impute import IterativeImputer\n\n# imputer = IterativeImputer(max_iter=20, random_state=0)\n# # imputed_data = imputer.fit_transform(X)\n# # imputed_data = imputer.fit_transform(X)\n# train2_new1 = pd.DataFrame(imputer.fit_transform(train_in1), columns=train_in1.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.033019Z","iopub.execute_input":"2024-10-10T10:31:04.033483Z","iopub.status.idle":"2024-10-10T10:31:04.041521Z","shell.execute_reply.started":"2024-10-10T10:31:04.033440Z","shell.execute_reply":"2024-10-10T10:31:04.040202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data=train2_new","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.043058Z","iopub.execute_input":"2024-10-10T10:31:04.043501Z","iopub.status.idle":"2024-10-10T10:31:04.052200Z","shell.execute_reply.started":"2024-10-10T10:31:04.043461Z","shell.execute_reply":"2024-10-10T10:31:04.051016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install cupy-cuda11x  # Make sure to use the correct CUDA version (e.g., cuda11x)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.053720Z","iopub.execute_input":"2024-10-10T10:31:04.054111Z","iopub.status.idle":"2024-10-10T10:31:04.062338Z","shell.execute_reply.started":"2024-10-10T10:31:04.054071Z","shell.execute_reply":"2024-10-10T10:31:04.061315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # from fancyimpute import SoftImpute\n# soft_imputer = SoftImpute(max_iters=100)  # max_iters defines the number of iterations\n# imputed_data = soft_imputer.fit_transform(data)\n\n# # Convert back to DataFrame\n# train2_new = pd.DataFrame(imputed_data, columns=data.columns)\n\n# imputer = KNNImputer(n_neighbors=10)\n# train2_new = pd.DataFrame(imputer.fit_transform(data), columns=data.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.063980Z","iopub.execute_input":"2024-10-10T10:31:04.064454Z","iopub.status.idle":"2024-10-10T10:31:04.072955Z","shell.execute_reply.started":"2024-10-10T10:31:04.064411Z","shell.execute_reply":"2024-10-10T10:31:04.071820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 12344df\n# df","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.074477Z","iopub.execute_input":"2024-10-10T10:31:04.074904Z","iopub.status.idle":"2024-10-10T10:31:04.082391Z","shell.execute_reply.started":"2024-10-10T10:31:04.074863Z","shell.execute_reply":"2024-10-10T10:31:04.081283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2_new","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.083781Z","iopub.execute_input":"2024-10-10T10:31:04.084156Z","iopub.status.idle":"2024-10-10T10:31:04.257805Z","shell.execute_reply.started":"2024-10-10T10:31:04.084115Z","shell.execute_reply":"2024-10-10T10:31:04.256655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train2_new=train_in1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.259081Z","iopub.execute_input":"2024-10-10T10:31:04.259469Z","iopub.status.idle":"2024-10-10T10:31:04.265504Z","shell.execute_reply.started":"2024-10-10T10:31:04.259429Z","shell.execute_reply":"2024-10-10T10:31:04.264336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2_new.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.267690Z","iopub.execute_input":"2024-10-10T10:31:04.268099Z","iopub.status.idle":"2024-10-10T10:31:04.282203Z","shell.execute_reply.started":"2024-10-10T10:31:04.268057Z","shell.execute_reply":"2024-10-10T10:31:04.280804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train2.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.283839Z","iopub.execute_input":"2024-10-10T10:31:04.284265Z","iopub.status.idle":"2024-10-10T10:31:04.291329Z","shell.execute_reply.started":"2024-10-10T10:31:04.284202Z","shell.execute_reply":"2024-10-10T10:31:04.290316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def categorize_values(x):\n#     if x < 0.5:\n#         return 0\n#     elif x < 1.5:\n#         return 1\n#     elif x < 2.5:\n#         return 2\n#     elif x < 3.5:\n#         return 3\n#     else:\n#         return 3\n\n# # Apply the function to the desired column\n# train2_new['sii'] = train2_new['sii'].apply(categorize_values)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.293124Z","iopub.execute_input":"2024-10-10T10:31:04.293592Z","iopub.status.idle":"2024-10-10T10:31:04.303369Z","shell.execute_reply.started":"2024-10-10T10:31:04.293540Z","shell.execute_reply":"2024-10-10T10:31:04.301948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train2_new","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.305103Z","iopub.execute_input":"2024-10-10T10:31:04.305548Z","iopub.status.idle":"2024-10-10T10:31:04.315542Z","shell.execute_reply.started":"2024-10-10T10:31:04.305504Z","shell.execute_reply":"2024-10-10T10:31:04.314293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train2_new[\"sii\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.317222Z","iopub.execute_input":"2024-10-10T10:31:04.317635Z","iopub.status.idle":"2024-10-10T10:31:04.326645Z","shell.execute_reply.started":"2024-10-10T10:31:04.317595Z","shell.execute_reply":"2024-10-10T10:31:04.325348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y=train_in[\"sii\"]\n# train_fina","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.328466Z","iopub.execute_input":"2024-10-10T10:31:04.329011Z","iopub.status.idle":"2024-10-10T10:31:04.338139Z","shell.execute_reply.started":"2024-10-10T10:31:04.328955Z","shell.execute_reply":"2024-10-10T10:31:04.336892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2_n=train2_new","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.339798Z","iopub.execute_input":"2024-10-10T10:31:04.340293Z","iopub.status.idle":"2024-10-10T10:31:04.349403Z","shell.execute_reply.started":"2024-10-10T10:31:04.340217Z","shell.execute_reply":"2024-10-10T10:31:04.347786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2_n=train2_n.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.351207Z","iopub.execute_input":"2024-10-10T10:31:04.351668Z","iopub.status.idle":"2024-10-10T10:31:04.364481Z","shell.execute_reply.started":"2024-10-10T10:31:04.351625Z","shell.execute_reply":"2024-10-10T10:31:04.363115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2_n.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.366264Z","iopub.execute_input":"2024-10-10T10:31:04.367479Z","iopub.status.idle":"2024-10-10T10:31:04.378998Z","shell.execute_reply.started":"2024-10-10T10:31:04.367434Z","shell.execute_reply":"2024-10-10T10:31:04.377648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train22=pd.concat([cat_data,train2_n],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.380526Z","iopub.execute_input":"2024-10-10T10:31:04.381038Z","iopub.status.idle":"2024-10-10T10:31:04.395689Z","shell.execute_reply.started":"2024-10-10T10:31:04.380982Z","shell.execute_reply":"2024-10-10T10:31:04.394363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train22.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.397579Z","iopub.execute_input":"2024-10-10T10:31:04.398895Z","iopub.status.idle":"2024-10-10T10:31:04.409892Z","shell.execute_reply.started":"2024-10-10T10:31:04.398845Z","shell.execute_reply":"2024-10-10T10:31:04.408659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.411413Z","iopub.execute_input":"2024-10-10T10:31:04.411782Z","iopub.status.idle":"2024-10-10T10:31:04.421001Z","shell.execute_reply.started":"2024-10-10T10:31:04.411744Z","shell.execute_reply":"2024-10-10T10:31:04.419628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train22","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.422885Z","iopub.execute_input":"2024-10-10T10:31:04.423419Z","iopub.status.idle":"2024-10-10T10:31:04.594896Z","shell.execute_reply.started":"2024-10-10T10:31:04.423363Z","shell.execute_reply":"2024-10-10T10:31:04.593678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train23","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.596695Z","iopub.execute_input":"2024-10-10T10:31:04.597166Z","iopub.status.idle":"2024-10-10T10:31:04.602611Z","shell.execute_reply.started":"2024-10-10T10:31:04.597117Z","shell.execute_reply":"2024-10-10T10:31:04.601407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y=train_in[\"sii\"]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.604376Z","iopub.execute_input":"2024-10-10T10:31:04.604890Z","iopub.status.idle":"2024-10-10T10:31:04.614636Z","shell.execute_reply.started":"2024-10-10T10:31:04.604820Z","shell.execute_reply":"2024-10-10T10:31:04.613448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.616225Z","iopub.execute_input":"2024-10-10T10:31:04.616791Z","iopub.status.idle":"2024-10-10T10:31:04.628315Z","shell.execute_reply.started":"2024-10-10T10:31:04.616748Z","shell.execute_reply":"2024-10-10T10:31:04.627160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in cat_col1:\n#     if(col!=\"id\"):train22[col] = train22[col].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.630038Z","iopub.execute_input":"2024-10-10T10:31:04.630479Z","iopub.status.idle":"2024-10-10T10:31:04.639828Z","shell.execute_reply.started":"2024-10-10T10:31:04.630437Z","shell.execute_reply":"2024-10-10T10:31:04.638554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_y.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.641605Z","iopub.execute_input":"2024-10-10T10:31:04.642119Z","iopub.status.idle":"2024-10-10T10:31:04.654920Z","shell.execute_reply.started":"2024-10-10T10:31:04.642062Z","shell.execute_reply":"2024-10-10T10:31:04.653689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install imblearn","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.659320Z","iopub.execute_input":"2024-10-10T10:31:04.659752Z","iopub.status.idle":"2024-10-10T10:31:04.664920Z","shell.execute_reply.started":"2024-10-10T10:31:04.659711Z","shell.execute_reply":"2024-10-10T10:31:04.663753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from imblearn.over_sampling import RandomOverSampler\n\n# # Initialize the oversampler\n# oversampler = RandomOverSampler(random_state=42)\n\n# # Fit and resample the data\n# X_resampled, y_resampled = oversampler.fit_resample(train22, train_y)\n\n# # Check the new class distribution\n# new_class_distribution = pd.Series(y_resampled).value_counts()\n# print(\"New class distribution after oversampling:\\n\", new_class_distribution)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.666816Z","iopub.execute_input":"2024-10-10T10:31:04.667601Z","iopub.status.idle":"2024-10-10T10:31:04.675542Z","shell.execute_reply.started":"2024-10-10T10:31:04.667547Z","shell.execute_reply":"2024-10-10T10:31:04.674315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.677059Z","iopub.execute_input":"2024-10-10T10:31:04.677556Z","iopub.status.idle":"2024-10-10T10:31:04.687201Z","shell.execute_reply.started":"2024-10-10T10:31:04.677513Z","shell.execute_reply":"2024-10-10T10:31:04.686020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X.dtypes\nfrom sklearn.ensemble import RandomForestRegressor\nparam_grid = {\n    'n_estimators':  300,\n    'max_depth': 10 ,\n    'min_samples_split':  10,\n    'min_samples_leaf':  4,\n    'max_features': 'auto',\n    'bootstrap': True,\n}\nrandom_forest_estimator = RandomForestRegressor(**param_grid)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.688778Z","iopub.execute_input":"2024-10-10T10:31:04.689288Z","iopub.status.idle":"2024-10-10T10:31:04.698551Z","shell.execute_reply.started":"2024-10-10T10:31:04.689213Z","shell.execute_reply":"2024-10-10T10:31:04.697332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.utils import class_weight\n\n# # Assuming 'sii' is your target column in the train DataFrame\n# y = y_resampled\n\n# # Calculate class weights using the 'balanced' strategy\n# class_weights = class_weight.compute_class_weight('balanced', classes=np.unique(y), y=y)\n\n# # Convert class weights into a dictionary (use the class index as key)\n# class_weights_dict = {i: weight for i, weight in enumerate(class_weights)}\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.699799Z","iopub.execute_input":"2024-10-10T10:31:04.700499Z","iopub.status.idle":"2024-10-10T10:31:04.708370Z","shell.execute_reply.started":"2024-10-10T10:31:04.700445Z","shell.execute_reply":"2024-10-10T10:31:04.707163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights_dict1=class_weights_dict","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.710157Z","iopub.execute_input":"2024-10-10T10:31:04.711197Z","iopub.status.idle":"2024-10-10T10:31:04.718280Z","shell.execute_reply.started":"2024-10-10T10:31:04.711137Z","shell.execute_reply":"2024-10-10T10:31:04.717301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights_dict1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.719932Z","iopub.execute_input":"2024-10-10T10:31:04.720446Z","iopub.status.idle":"2024-10-10T10:31:04.730016Z","shell.execute_reply.started":"2024-10-10T10:31:04.720389Z","shell.execute_reply":"2024-10-10T10:31:04.728756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights_dict1[1]=1.2\n# class_weights_dict1[0]=.6","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.731971Z","iopub.execute_input":"2024-10-10T10:31:04.732406Z","iopub.status.idle":"2024-10-10T10:31:04.740294Z","shell.execute_reply.started":"2024-10-10T10:31:04.732363Z","shell.execute_reply":"2024-10-10T10:31:04.739002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_resampled.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.741809Z","iopub.execute_input":"2024-10-10T10:31:04.742168Z","iopub.status.idle":"2024-10-10T10:31:04.751576Z","shell.execute_reply.started":"2024-10-10T10:31:04.742131Z","shell.execute_reply":"2024-10-10T10:31:04.750225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train, X_val, y_train, y_val = train_test_split(X_resampled, y_resampled, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.753120Z","iopub.execute_input":"2024-10-10T10:31:04.753581Z","iopub.status.idle":"2024-10-10T10:31:04.760853Z","shell.execute_reply.started":"2024-10-10T10:31:04.753538Z","shell.execute_reply":"2024-10-10T10:31:04.759734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.762571Z","iopub.execute_input":"2024-10-10T10:31:04.762997Z","iopub.status.idle":"2024-10-10T10:31:04.771237Z","shell.execute_reply.started":"2024-10-10T10:31:04.762926Z","shell.execute_reply":"2024-10-10T10:31:04.770087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.772836Z","iopub.execute_input":"2024-10-10T10:31:04.773239Z","iopub.status.idle":"2024-10-10T10:31:04.781567Z","shell.execute_reply.started":"2024-10-10T10:31:04.773198Z","shell.execute_reply":"2024-10-10T10:31:04.780312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report, confusion_matrix\n\n# Light.fit(X_train, y_train)\n\n# y_pred = Light.predict(X_val)\n\n# print(classification_report(y_val, y_pred))\n# print(confusion_matrix(y_val, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.783173Z","iopub.execute_input":"2024-10-10T10:31:04.783693Z","iopub.status.idle":"2024-10-10T10:31:04.792626Z","shell.execute_reply.started":"2024-10-10T10:31:04.783636Z","shell.execute_reply":"2024-10-10T10:31:04.791425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#      precision    recall  f1-score   support\n\n#          0.0       0.88      0.72      0.79       346\n#          1.0       0.78      0.87      0.82       297\n#          2.0       0.90      0.97      0.94       319\n#          3.0       0.99      1.00      1.00       314\n\n#     accuracy                           0.89      1276\n#    macro avg       0.89      0.89      0.89      1276\n# weighted avg       0.89      0.89      0.89      1276\n\n# [[250  71  25   0]\n#  [ 28 258   8   3]\n#  [  5   3 311   0]\n#  [  0   0   0 314]]","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-10-10T10:31:04.794339Z","iopub.execute_input":"2024-10-10T10:31:04.794761Z","iopub.status.idle":"2024-10-10T10:31:04.807804Z","shell.execute_reply.started":"2024-10-10T10:31:04.794720Z","shell.execute_reply":"2024-10-10T10:31:04.806364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"    precision    recall  f1-score   support\n\n         0.0       0.91      0.67      0.77       346\n         1.0       0.75      0.89      0.81       297\n         2.0       0.89      0.97      0.93       319\n         3.0       0.98      1.00      0.99       314\n\n    accuracy                           0.88      1276\n   macro avg       0.88      0.88      0.88      1276\nweighted avg       0.88      0.88      0.87      1276\n\n[[232  82  32   0]\n [ 21 264   8   4]\n [  3   5 309   2]\n [  0   0   0 314]]","metadata":{}},{"cell_type":"code","source":"# class_weights_dict","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.809728Z","iopub.execute_input":"2024-10-10T10:31:04.810295Z","iopub.status.idle":"2024-10-10T10:31:04.819351Z","shell.execute_reply.started":"2024-10-10T10:31:04.810202Z","shell.execute_reply":"2024-10-10T10:31:04.818009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights = class_weight.compute_class_weight('balanced', classes=np.unique(y), y=y)\n\n# # Convert class weights into a dictionary (use the class index as key)\n# class_weights_dict2 = {i: weight for i, weight in enumerate(class_weights)}","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.845962Z","iopub.execute_input":"2024-10-10T10:31:04.846451Z","iopub.status.idle":"2024-10-10T10:31:04.852806Z","shell.execute_reply.started":"2024-10-10T10:31:04.846406Z","shell.execute_reply":"2024-10-10T10:31:04.851473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights_dict2\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.854305Z","iopub.execute_input":"2024-10-10T10:31:04.854834Z","iopub.status.idle":"2024-10-10T10:31:04.861971Z","shell.execute_reply.started":"2024-10-10T10:31:04.854782Z","shell.execute_reply":"2024-10-10T10:31:04.860344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights_dict2=class_weights_dict.copy\n# class_weights_dict2[0]=1.3\n# class_weights_dict2[1]=1.3","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.863768Z","iopub.execute_input":"2024-10-10T10:31:04.864173Z","iopub.status.idle":"2024-10-10T10:31:04.873626Z","shell.execute_reply.started":"2024-10-10T10:31:04.864131Z","shell.execute_reply":"2024-10-10T10:31:04.872156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# XGB_Model.fit(X_train, y_train)\n# y_pred = XGB_Model.predict(X_val)\n\n# print(classification_report(y_val, y_pred))\n# print(confusion_matrix(y_val, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.875387Z","iopub.execute_input":"2024-10-10T10:31:04.875853Z","iopub.status.idle":"2024-10-10T10:31:04.884145Z","shell.execute_reply.started":"2024-10-10T10:31:04.875806Z","shell.execute_reply":"2024-10-10T10:31:04.882740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"            precision    recall  f1-score   support\n\n         0.0       0.87      0.79      0.83       346\n         1.0       0.84      0.85      0.85       297\n         2.0       0.91      0.97      0.94       319\n         3.0       0.99      1.00      1.00       314\n\n    accuracy                           0.90      1276\n   macro avg       0.90      0.91      0.90      1276\nweighted avg       0.90      0.90      0.90      1276\n\n[[275  45  26   0]\n [ 35 253   6   3]\n [  5   3 311   0]\n [  0   0   0 314]]","metadata":{}},{"cell_type":"code","source":"len(cat_col1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.885723Z","iopub.execute_input":"2024-10-10T10:31:04.886122Z","iopub.status.idle":"2024-10-10T10:31:04.898918Z","shell.execute_reply.started":"2024-10-10T10:31:04.886081Z","shell.execute_reply":"2024-10-10T10:31:04.897724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cat_c=cat_col1.remove(\"sii\")","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.900411Z","iopub.execute_input":"2024-10-10T10:31:04.900785Z","iopub.status.idle":"2024-10-10T10:31:04.909573Z","shell.execute_reply.started":"2024-10-10T10:31:04.900746Z","shell.execute_reply":"2024-10-10T10:31:04.908323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights = class_weight.compute_class_weight('balanced', classes=np.unique(y), y=y)\n# # \n# # Convert class weights into a dictionary (use the class index as key)\n# class_weights_dict3 = {i: weight for i, weight in enumerate(class_weights)}","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.911136Z","iopub.execute_input":"2024-10-10T10:31:04.911578Z","iopub.status.idle":"2024-10-10T10:31:04.919611Z","shell.execute_reply.started":"2024-10-10T10:31:04.911535Z","shell.execute_reply":"2024-10-10T10:31:04.918355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights_dict3[0]=.6\n# class_weights_dict3[1]=1.1\n# class_weights_dict3[2]=1.95","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.920886Z","iopub.execute_input":"2024-10-10T10:31:04.921306Z","iopub.status.idle":"2024-10-10T10:31:04.929938Z","shell.execute_reply.started":"2024-10-10T10:31:04.921262Z","shell.execute_reply":"2024-10-10T10:31:04.928900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights_list = [class_weights_dict3[i] for i in range(len(class_weights_dict3))]\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.931498Z","iopub.execute_input":"2024-10-10T10:31:04.932018Z","iopub.status.idle":"2024-10-10T10:31:04.941090Z","shell.execute_reply.started":"2024-10-10T10:31:04.931962Z","shell.execute_reply":"2024-10-10T10:31:04.939494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights_list[0]=1.8\n# class_weights_list[1]=1.45\n# class_weights_list[2]=1.3","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.943031Z","iopub.execute_input":"2024-10-10T10:31:04.943816Z","iopub.status.idle":"2024-10-10T10:31:04.953253Z","shell.execute_reply.started":"2024-10-10T10:31:04.943763Z","shell.execute_reply":"2024-10-10T10:31:04.951990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CAT_Model.fit(X_train,y_train)\n# # XGB_Model.fit(X_train, y_train)\n# y_pred = CAT_Model.predict(X_val)\n\n# print(classification_report(y_val, y_pred))\n# print(confusion_matrix(y_val, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.954721Z","iopub.execute_input":"2024-10-10T10:31:04.955148Z","iopub.status.idle":"2024-10-10T10:31:04.965301Z","shell.execute_reply.started":"2024-10-10T10:31:04.955105Z","shell.execute_reply":"2024-10-10T10:31:04.964025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" CatBoost_Params = {\n        'learning_rate': 0.04,\n        'depth': 6,\n        'iterations': 250,\n        \n       'cat_features': cat_col1,\n        'verbose': 0,\n        'l2_leaf_reg': 6 , # Increase this value\n#      'scale_pos_weight': class_weights_dict3\n#      'class_weights':class_weights_list\n#     'auto_class_weights': 'Balanced' ,\n     \n    }\nCAT_Model = CatBoostRegressor(**CatBoost_Params)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.967011Z","iopub.execute_input":"2024-10-10T10:31:04.967440Z","iopub.status.idle":"2024-10-10T10:31:04.984781Z","shell.execute_reply.started":"2024-10-10T10:31:04.967398Z","shell.execute_reply":"2024-10-10T10:31:04.983309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"XGB_Params = {\n        'learning_rate': 0.038,\n        'max_depth': 9,\n        'n_estimators': 300,\n        'subsample': 0.8,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.85,\n        'colsample_bytree': 0.8,\n        'reg_alpha': .9,  # Increased from 0.1\n        'reg_lambda': 3,  # Increased from 1\n       \n#      'class_weight': class_weights_dict2\n    }\n# 'learning_rate': 0.0156,\n#     'max_depth': 12,\n#     'num_leaves': 478,\n#     'min_data_in_leaf': 13,\n#     'feature_fraction': 0.893,\n#     'bagging_fraction': 0.784,\n#     'bagging_freq': 4,\n#     'lambda_l1': .7,  # Increased from 6.59\n#     'lambda_l2': 0.1,  # Increased from 2.68e-06\n#     'class_weight': class_weights_dict1 \nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nXGB_Model = XGBRegressor(**XGB_Params, verbose=-1,enable_categorical=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.986618Z","iopub.execute_input":"2024-10-10T10:31:04.987060Z","iopub.status.idle":"2024-10-10T10:31:04.996451Z","shell.execute_reply.started":"2024-10-10T10:31:04.987003Z","shell.execute_reply":"2024-10-10T10:31:04.995100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Modify the Params dictionary to include the calculated class weights\nParams = {\n    'learning_rate': 0.038,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 4,\n    'lambda_l1': 8,  # Increased from 6.59\n    'lambda_l2': 0.1,  # Increased from 2.68e-06\n#     'class_weight': class_weights_dict1  # Add the class weights here\n}\n\n\nfrom lightgbm import LGBMRegressor\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=350)\n\n\n# X_resampled","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:04.998279Z","iopub.execute_input":"2024-10-10T10:31:04.998732Z","iopub.status.idle":"2024-10-10T10:31:05.006999Z","shell.execute_reply.started":"2024-10-10T10:31:04.998689Z","shell.execute_reply":"2024-10-10T10:31:05.005780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"         0.0       0.65      0.68      0.67       346\n         1.0       0.70      0.59      0.64       297\n         2.0       0.77      0.79      0.78       319\n         3.0       0.93      1.00      0.96       314\n\n    accuracy                           0.77      1276\n   macro avg       0.76      0.77      0.76      1276\nweighted avg       0.76      0.77      0.76      1276\n\n[[236  64  45   1]\n [ 79 174  29  15]\n [ 47  11 253   8]\n [  0   0   0 314]]","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import VotingRegressor\nvoting_model = VotingRegressor(estimators=[\n    ('lgbm', Light),\n    ('xgb', XGB_Model),\n    ('catboost', CAT_Model)\n  \n]) \n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.008557Z","iopub.execute_input":"2024-10-10T10:31:05.009053Z","iopub.status.idle":"2024-10-10T10:31:05.020076Z","shell.execute_reply.started":"2024-10-10T10:31:05.008998Z","shell.execute_reply":"2024-10-10T10:31:05.018819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=train22\ny=train_y","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.021853Z","iopub.execute_input":"2024-10-10T10:31:05.022398Z","iopub.status.idle":"2024-10-10T10:31:05.034219Z","shell.execute_reply.started":"2024-10-10T10:31:05.022315Z","shell.execute_reply":"2024-10-10T10:31:05.032971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# voting_model.fit(X_resampled, y_resampled)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.035984Z","iopub.execute_input":"2024-10-10T10:31:05.036437Z","iopub.status.idle":"2024-10-10T10:31:05.044808Z","shell.execute_reply.started":"2024-10-10T10:31:05.036395Z","shell.execute_reply":"2024-10-10T10:31:05.043555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final_preds = voting_model.predict(X_val)\n# y_pred = voting_model.predict(X_val)\n\n# Evaluate the result\n# predicted_probabilities = voting_model.predict(X_val)\n\n# Convert probabilities to class predictions\n# For multi-class classification, we can use argmax\n# predictions = np.argmax(predicted_probabilities, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.046642Z","iopub.execute_input":"2024-10-10T10:31:05.047193Z","iopub.status.idle":"2024-10-10T10:31:05.056329Z","shell.execute_reply.started":"2024-10-10T10:31:05.047137Z","shell.execute_reply":"2024-10-10T10:31:05.054950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(classification_report(y_val, predicted_probabilities))\n# print(confusion_matrix(y_val, predicted_probabilities))","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.057898Z","iopub.execute_input":"2024-10-10T10:31:05.058347Z","iopub.status.idle":"2024-10-10T10:31:05.066367Z","shell.execute_reply.started":"2024-10-10T10:31:05.058287Z","shell.execute_reply":"2024-10-10T10:31:05.065014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicted_probabilities.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.068011Z","iopub.execute_input":"2024-10-10T10:31:05.068739Z","iopub.status.idle":"2024-10-10T10:31:05.077450Z","shell.execute_reply.started":"2024-10-10T10:31:05.068693Z","shell.execute_reply":"2024-10-10T10:31:05.075982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final_preds1=pd.DataFrame(final_preds)\n# final_preds1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.079179Z","iopub.execute_input":"2024-10-10T10:31:05.079744Z","iopub.status.idle":"2024-10-10T10:31:05.087745Z","shell.execute_reply.started":"2024-10-10T10:31:05.079685Z","shell.execute_reply":"2024-10-10T10:31:05.086357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_val","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.089540Z","iopub.execute_input":"2024-10-10T10:31:05.089965Z","iopub.status.idle":"2024-10-10T10:31:05.097549Z","shell.execute_reply.started":"2024-10-10T10:31:05.089921Z","shell.execute_reply":"2024-10-10T10:31:05.096102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train22.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.099582Z","iopub.execute_input":"2024-10-10T10:31:05.100066Z","iopub.status.idle":"2024-10-10T10:31:05.110195Z","shell.execute_reply.started":"2024-10-10T10:31:05.100019Z","shell.execute_reply":"2024-10-10T10:31:05.108399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.111779Z","iopub.execute_input":"2024-10-10T10:31:05.112225Z","iopub.status.idle":"2024-10-10T10:31:05.414949Z","shell.execute_reply.started":"2024-10-10T10:31:05.112166Z","shell.execute_reply":"2024-10-10T10:31:05.413776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test2.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.416462Z","iopub.execute_input":"2024-10-10T10:31:05.416844Z","iopub.status.idle":"2024-10-10T10:31:05.422219Z","shell.execute_reply.started":"2024-10-10T10:31:05.416804Z","shell.execute_reply":"2024-10-10T10:31:05.420771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_td","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.423855Z","iopub.execute_input":"2024-10-10T10:31:05.424285Z","iopub.status.idle":"2024-10-10T10:31:05.446239Z","shell.execute_reply.started":"2024-10-10T10:31:05.424214Z","shell.execute_reply":"2024-10-10T10:31:05.444872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test3=test2.drop(cat_col1,axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.447861Z","iopub.execute_input":"2024-10-10T10:31:05.448225Z","iopub.status.idle":"2024-10-10T10:31:05.456892Z","shell.execute_reply.started":"2024-10-10T10:31:05.448188Z","shell.execute_reply":"2024-10-10T10:31:05.455829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sc=StandardScaler()\n# data=sc.fit_transform(test3)\n# test4 = imputer.transform(data)\n\n# Inverse transform the scaled data back to the original scale\n# test4 = sc.inverse_transform(imputed_data)\n# test4 = pd.DataFrame(test4, columns=test3.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.458631Z","iopub.execute_input":"2024-10-10T10:31:05.459076Z","iopub.status.idle":"2024-10-10T10:31:05.468590Z","shell.execute_reply.started":"2024-10-10T10:31:05.459034Z","shell.execute_reply":"2024-10-10T10:31:05.467526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test4","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.470025Z","iopub.execute_input":"2024-10-10T10:31:05.470456Z","iopub.status.idle":"2024-10-10T10:31:05.692425Z","shell.execute_reply.started":"2024-10-10T10:31:05.470417Z","shell.execute_reply":"2024-10-10T10:31:05.691168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# knn_imputer = KNNImputer(n_neighbors=5)\n# X_train_imputed = knn_imputer.fit_transform(t)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.693988Z","iopub.execute_input":"2024-10-10T10:31:05.694426Z","iopub.status.idle":"2024-10-10T10:31:05.699769Z","shell.execute_reply.started":"2024-10-10T10:31:05.694384Z","shell.execute_reply":"2024-10-10T10:31:05.698236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_new = pd.DataFrame(imputer.transform(test3), columns=test3.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.701446Z","iopub.execute_input":"2024-10-10T10:31:05.701856Z","iopub.status.idle":"2024-10-10T10:31:05.711728Z","shell.execute_reply.started":"2024-10-10T10:31:05.701808Z","shell.execute_reply":"2024-10-10T10:31:05.710306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_new","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.713449Z","iopub.execute_input":"2024-10-10T10:31:05.714564Z","iopub.status.idle":"2024-10-10T10:31:05.720327Z","shell.execute_reply.started":"2024-10-10T10:31:05.714517Z","shell.execute_reply":"2024-10-10T10:31:05.719291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# testf=pd.concat([cat_td,test4],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.721915Z","iopub.execute_input":"2024-10-10T10:31:05.722687Z","iopub.status.idle":"2024-10-10T10:31:05.731488Z","shell.execute_reply.started":"2024-10-10T10:31:05.722626Z","shell.execute_reply":"2024-10-10T10:31:05.730317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testf","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.732956Z","iopub.execute_input":"2024-10-10T10:31:05.733362Z","iopub.status.idle":"2024-10-10T10:31:05.961806Z","shell.execute_reply.started":"2024-10-10T10:31:05.733321Z","shell.execute_reply":"2024-10-10T10:31:05.960586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample[\"id\"]=tid","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.963162Z","iopub.execute_input":"2024-10-10T10:31:05.963581Z","iopub.status.idle":"2024-10-10T10:31:05.968976Z","shell.execute_reply.started":"2024-10-10T10:31:05.963540Z","shell.execute_reply":"2024-10-10T10:31:05.967706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.970647Z","iopub.execute_input":"2024-10-10T10:31:05.971037Z","iopub.status.idle":"2024-10-10T10:31:05.980855Z","shell.execute_reply.started":"2024-10-10T10:31:05.970997Z","shell.execute_reply":"2024-10-10T10:31:05.979738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testf.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.982735Z","iopub.execute_input":"2024-10-10T10:31:05.983295Z","iopub.status.idle":"2024-10-10T10:31:05.994150Z","shell.execute_reply.started":"2024-10-10T10:31:05.983202Z","shell.execute_reply":"2024-10-10T10:31:05.993022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train.columns==testf.columns","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:05.995741Z","iopub.execute_input":"2024-10-10T10:31:05.996165Z","iopub.status.idle":"2024-10-10T10:31:06.004596Z","shell.execute_reply.started":"2024-10-10T10:31:05.996109Z","shell.execute_reply":"2024-10-10T10:31:06.003516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testf","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:06.006176Z","iopub.execute_input":"2024-10-10T10:31:06.006741Z","iopub.status.idle":"2024-10-10T10:31:06.233418Z","shell.execute_reply.started":"2024-10-10T10:31:06.006698Z","shell.execute_reply":"2024-10-10T10:31:06.232227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for c in cat_col1:\n    \n#     if(col!=\"id\" and col in testf.columns):testf[c].fillna('missing', inplace=True) \n# for col in cat_col1:\n#     if(col!=\"id\" and col in testf.columns):testf[col] = testf[col].astype('category')\n    \n   \n        \n        ","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:06.235146Z","iopub.execute_input":"2024-10-10T10:31:06.235651Z","iopub.status.idle":"2024-10-10T10:31:06.241568Z","shell.execute_reply.started":"2024-10-10T10:31:06.235597Z","shell.execute_reply":"2024-10-10T10:31:06.240206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# testf1=testf.drop(\"id\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:06.243188Z","iopub.execute_input":"2024-10-10T10:31:06.243599Z","iopub.status.idle":"2024-10-10T10:31:06.250905Z","shell.execute_reply.started":"2024-10-10T10:31:06.243552Z","shell.execute_reply":"2024-10-10T10:31:06.249822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:06.252607Z","iopub.execute_input":"2024-10-10T10:31:06.253005Z","iopub.status.idle":"2024-10-10T10:31:06.262829Z","shell.execute_reply.started":"2024-10-10T10:31:06.252966Z","shell.execute_reply":"2024-10-10T10:31:06.261683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testf.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:06.264382Z","iopub.execute_input":"2024-10-10T10:31:06.264805Z","iopub.status.idle":"2024-10-10T10:31:06.277136Z","shell.execute_reply.started":"2024-10-10T10:31:06.264763Z","shell.execute_reply":"2024-10-10T10:31:06.275762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:06.278672Z","iopub.execute_input":"2024-10-10T10:31:06.279094Z","iopub.status.idle":"2024-10-10T10:31:06.286798Z","shell.execute_reply.started":"2024-10-10T10:31:06.279054Z","shell.execute_reply":"2024-10-10T10:31:06.285697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class,test_data):\n    \n#     X = train.drop(['sii'], axis=1)\n#     y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission,model\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:06.288502Z","iopub.execute_input":"2024-10-10T10:31:06.288973Z","iopub.status.idle":"2024-10-10T10:31:06.310392Z","shell.execute_reply.started":"2024-10-10T10:31:06.288932Z","shell.execute_reply":"2024-10-10T10:31:06.308983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission,model = TrainML(voting_model,testf)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:31:06.312058Z","iopub.execute_input":"2024-10-10T10:31:06.312997Z","iopub.status.idle":"2024-10-10T10:34:59.409327Z","shell.execute_reply.started":"2024-10-10T10:31:06.312933Z","shell.execute_reply":"2024-10-10T10:34:59.408061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import GridSearchCV\n\n# # Grid of hyperparameters\n# param_grid = {\n#     'n_estimators': [ 400],\n#     'max_depth': [10 ],\n#     'min_samples_split': [ 10],\n#     'min_samples_leaf': [ 4],\n#     'max_features': ['auto'],\n#     'bootstrap': [True]\n# }\n\n# # Instantiate the Random Forest classifier\n# # rf = RandomForestClassifier()\n\n# # Grid search of parameters\n# grid_search = GridSearchCV(estimator=random_forest_estimator, param_grid=param_grid, cv=3, n_jobs=-1, verbose=2)\n\n# # Fit the model\n# grid_search.fit(X, train_y)\n# print(grid_search.best_params_)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.410937Z","iopub.execute_input":"2024-10-10T10:34:59.411345Z","iopub.status.idle":"2024-10-10T10:34:59.418846Z","shell.execute_reply.started":"2024-10-10T10:34:59.411303Z","shell.execute_reply":"2024-10-10T10:34:59.417680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(grid_search.best_params_)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.420122Z","iopub.execute_input":"2024-10-10T10:34:59.420525Z","iopub.status.idle":"2024-10-10T10:34:59.430939Z","shell.execute_reply.started":"2024-10-10T10:34:59.420485Z","shell.execute_reply":"2024-10-10T10:34:59.429578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Mean Train QWK --> 0.5338\nMean Validation QWK ---> 0.3599\n\n----> || Optimized QWK SCORE ::  0.453\n\nTraining Folds: 100%|██████████| 5/5 [00:43<00:00,  8.64s/it]\nMean Train QWK --> 0.5391\nMean Validation QWK ---> 0.3758\n\n----> || Optimized QWK SCORE ::  0.471","metadata":{}},{"cell_type":"markdown","source":"Mean Train QWK --> 0.8462\nMean Validation QWK ---> 0.3947\n\n----> || Optimized QWK SCORE ::  0.454","metadata":{}},{"cell_type":"code","source":"Submission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.433741Z","iopub.execute_input":"2024-10-10T10:34:59.434153Z","iopub.status.idle":"2024-10-10T10:34:59.452932Z","shell.execute_reply.started":"2024-10-10T10:34:59.434106Z","shell.execute_reply":"2024-10-10T10:34:59.451712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# testf.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.455638Z","iopub.execute_input":"2024-10-10T10:34:59.456049Z","iopub.status.idle":"2024-10-10T10:34:59.461335Z","shell.execute_reply.started":"2024-10-10T10:34:59.456007Z","shell.execute_reply":"2024-10-10T10:34:59.459952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# yp=Light.predict(testf)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.462610Z","iopub.execute_input":"2024-10-10T10:34:59.462968Z","iopub.status.idle":"2024-10-10T10:34:59.474456Z","shell.execute_reply.started":"2024-10-10T10:34:59.462922Z","shell.execute_reply":"2024-10-10T10:34:59.473268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# yp = voting_model.predict(testf)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.476092Z","iopub.execute_input":"2024-10-10T10:34:59.476611Z","iopub.status.idle":"2024-10-10T10:34:59.484686Z","shell.execute_reply.started":"2024-10-10T10:34:59.476556Z","shell.execute_reply":"2024-10-10T10:34:59.483508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# yp1=pd.DataFrame(yp)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.486370Z","iopub.execute_input":"2024-10-10T10:34:59.487194Z","iopub.status.idle":"2024-10-10T10:34:59.494740Z","shell.execute_reply.started":"2024-10-10T10:34:59.487151Z","shell.execute_reply":"2024-10-10T10:34:59.493757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# yp1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.496366Z","iopub.execute_input":"2024-10-10T10:34:59.496894Z","iopub.status.idle":"2024-10-10T10:34:59.505294Z","shell.execute_reply.started":"2024-10-10T10:34:59.496851Z","shell.execute_reply":"2024-10-10T10:34:59.504077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def categorize_values(x):\n#     if x < 0.5:\n#         return 0\n#     elif x < 1.5:\n#         return 1\n#     elif x < 2.5:\n#         return 2\n#     elif x < 3.5:\n#         return 3\n#     else:\n#         return 3\n\n# yp1.iloc[:,0]= yp1.iloc[:,0].apply(categorize_values)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.507074Z","iopub.execute_input":"2024-10-10T10:34:59.508042Z","iopub.status.idle":"2024-10-10T10:34:59.515780Z","shell.execute_reply.started":"2024-10-10T10:34:59.507999Z","shell.execute_reply":"2024-10-10T10:34:59.514651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# yp1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.518296Z","iopub.execute_input":"2024-10-10T10:34:59.518680Z","iopub.status.idle":"2024-10-10T10:34:59.528082Z","shell.execute_reply.started":"2024-10-10T10:34:59.518639Z","shell.execute_reply":"2024-10-10T10:34:59.526846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub=pd.concat([tid,yp1],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.530094Z","iopub.execute_input":"2024-10-10T10:34:59.530935Z","iopub.status.idle":"2024-10-10T10:34:59.538984Z","shell.execute_reply.started":"2024-10-10T10:34:59.530875Z","shell.execute_reply":"2024-10-10T10:34:59.537673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.540744Z","iopub.execute_input":"2024-10-10T10:34:59.541142Z","iopub.status.idle":"2024-10-10T10:34:59.548965Z","shell.execute_reply.started":"2024-10-10T10:34:59.541099Z","shell.execute_reply":"2024-10-10T10:34:59.547793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.550458Z","iopub.execute_input":"2024-10-10T10:34:59.550850Z","iopub.status.idle":"2024-10-10T10:34:59.559421Z","shell.execute_reply.started":"2024-10-10T10:34:59.550811Z","shell.execute_reply":"2024-10-10T10:34:59.558331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import lightgbm as lgb\n# import pandas as pd\n# from sklearn.model_selection import train_test_split, GridSearchCV, cross_val_score\n# from sklearn.metrics import accuracy_score, classification_report, confusion_matrix\n# import matplotlib.pyplot as plt\n\n# # Assuming `train23` is your feature DataFrame and `train_y` is your target Series\n# # Example: train23 = pd.read_csv('your_data.csv'), train_y = pd.read_csv('your_labels.csv')\n\n# # Step 1: Split the dataset into training and validation sets\n# X_train, X_val, y_train, y_val = train_test_split(X_resampled, y_resampled, test_size=0.2, random_state=42)\n\n# # Step 2: Define the LightGBM model\n# lgbm_model = lgb.LGBMClassifier(objective='multiclass', num_class=4, random_state=42,learning_rate=.005, num_leaves=50,n_estimators=75,reg_alpha=.7,reg_lambda=1.2)\n\n# lgbm_model.fit(X_train, y_train,eval_set=[(X_val, y_val)],\n#     eval_metric='multi_logloss',  # You can change this to other metrics if needed\n#          # Number of rounds with no improvement to stop training\n#       )\n\n# # Step 4: Make predictions on the validation set\n# y_pred = lgbm_model.predict(X_val)\n# # # Step 3: Hyperparameter tuning using Grid Search\n# # param_grid = {\n# #     'num_leaves': [31, 50, 100],\n# #     'max_depth': [-1, 10, 20],\n# #     'learning_rate': [0.01, 0.1, 0.5],\n# #     'n_estimators': [100, 200, 500]\n# # }\n\n# # grid_search = GridSearchCV(estimator=lgbm_model, param_grid=param_grid, scoring='accuracy', cv=5)\n# # grid_search.fit(X_train, y_train)\n# # print(\"Best parameters found: \", grid_search.best_params_)\n\n# # # Step 4: Train the model with the best parameters\n# # best_model = grid_search.best_estimator_\n\n# # # Step 5: Train with early stopping\n# # best_model.fit(X_train, y_train, eval_set=[(X_val, y_val)], early_stopping_rounds=100, verbose=10)\n\n# # # Step 6: Predict on the validation set\n# # y_pred = best_model.predict(X_val)\n\n# # # Step 7: Calculate accuracy\n# # accuracy = accuracy_score(y_val, y_pred)\n# # print(f\"LightGBM Accuracy: {accuracy}\")\n\n# # # Step 8: Classification report and confusion matrix\n# print(classification_report(y_val, y_pred))\n# print(confusion_matrix(y_val, y_pred))\n\n# # # Step 9: Cross-validation scores\n# # cv_scores = cross_val_score(best_model, train23, train_y, cv=5, scoring='accuracy')\n# # print(f\"Mean CV Accuracy: {cv_scores.mean()}\")\n\n# # # Step 10: Feature importance analysis\n# # lgb.plot_importance(best_model, max_num_features=10, importance_type='gain')\n# # plt.title('Feature Importance')\n# # plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.561308Z","iopub.execute_input":"2024-10-10T10:34:59.561719Z","iopub.status.idle":"2024-10-10T10:34:59.571270Z","shell.execute_reply.started":"2024-10-10T10:34:59.561678Z","shell.execute_reply":"2024-10-10T10:34:59.570005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(classification_report(y_val, y_pred))\n# print(confusion_matrix(y_val, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.573027Z","iopub.execute_input":"2024-10-10T10:34:59.573460Z","iopub.status.idle":"2024-10-10T10:34:59.583966Z","shell.execute_reply.started":"2024-10-10T10:34:59.573418Z","shell.execute_reply":"2024-10-10T10:34:59.582937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pre=pd.DataFrame(y_pred)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.585341Z","iopub.execute_input":"2024-10-10T10:34:59.585725Z","iopub.status.idle":"2024-10-10T10:34:59.594889Z","shell.execute_reply.started":"2024-10-10T10:34:59.585686Z","shell.execute_reply":"2024-10-10T10:34:59.593672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pre.value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.596320Z","iopub.execute_input":"2024-10-10T10:34:59.596677Z","iopub.status.idle":"2024-10-10T10:34:59.605100Z","shell.execute_reply.started":"2024-10-10T10:34:59.596640Z","shell.execute_reply":"2024-10-10T10:34:59.603837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_val.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.606666Z","iopub.execute_input":"2024-10-10T10:34:59.607121Z","iopub.status.idle":"2024-10-10T10:34:59.615667Z","shell.execute_reply.started":"2024-10-10T10:34:59.607077Z","shell.execute_reply":"2024-10-10T10:34:59.614337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import xgboost as xgb","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.617450Z","iopub.execute_input":"2024-10-10T10:34:59.617884Z","iopub.status.idle":"2024-10-10T10:34:59.625322Z","shell.execute_reply.started":"2024-10-10T10:34:59.617836Z","shell.execute_reply":"2024-10-10T10:34:59.624147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_in_filled = train_in.fillna(0)\n# train_in_filled=train_in_filled.reset_index(drop=True)\n# train_zero=pd.concat([cat_data,train_in_filled],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.626778Z","iopub.execute_input":"2024-10-10T10:34:59.627142Z","iopub.status.idle":"2024-10-10T10:34:59.635835Z","shell.execute_reply.started":"2024-10-10T10:34:59.627093Z","shell.execute_reply":"2024-10-10T10:34:59.634706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_zero","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.637095Z","iopub.execute_input":"2024-10-10T10:34:59.637482Z","iopub.status.idle":"2024-10-10T10:34:59.647101Z","shell.execute_reply.started":"2024-10-10T10:34:59.637431Z","shell.execute_reply":"2024-10-10T10:34:59.645909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_zero=train_zero.drop([\"id\",\"sii\"],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.648569Z","iopub.execute_input":"2024-10-10T10:34:59.649008Z","iopub.status.idle":"2024-10-10T10:34:59.658849Z","shell.execute_reply.started":"2024-10-10T10:34:59.648966Z","shell.execute_reply":"2024-10-10T10:34:59.657614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in cat_col1:\n#     if(col!=\"id\"):train_zero[col] = train_zero[col].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.660513Z","iopub.execute_input":"2024-10-10T10:34:59.661574Z","iopub.status.idle":"2024-10-10T10:34:59.669300Z","shell.execute_reply.started":"2024-10-10T10:34:59.661517Z","shell.execute_reply":"2024-10-10T10:34:59.668110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from imblearn.over_sampling import RandomOverSampler\n\n# # Initialize the oversampler\n# oversampler = RandomOverSampler(random_state=42)\n\n# # Fit and resample the data\n# X_resampled_o, y_resampled_o = oversampler.fit_resample(train_zero, train_y)\n\n# # Check the new class distribution\n# new_class_distribution = pd.Series(y_resampled_o).value_counts()\n# print(\"New class distribution after oversampling:\\n\", new_class_distribution)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.670933Z","iopub.execute_input":"2024-10-10T10:34:59.671449Z","iopub.status.idle":"2024-10-10T10:34:59.680461Z","shell.execute_reply.started":"2024-10-10T10:34:59.671391Z","shell.execute_reply":"2024-10-10T10:34:59.679150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train_o, X_val_o, y_train_o, y_val_o = train_test_split(X_resampled_o, y_resampled_o, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.681955Z","iopub.execute_input":"2024-10-10T10:34:59.682396Z","iopub.status.idle":"2024-10-10T10:34:59.690859Z","shell.execute_reply.started":"2024-10-10T10:34:59.682355Z","shell.execute_reply":"2024-10-10T10:34:59.689679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# xgb_model = xgb.XGBClassifier(\n#     objective='multi:softmax',    # Correct objective for multi-class classification\n#     n_classes=4,                   # This parameter is used for multiclass\n#     learning_rate=0.005,\n#     max_depth=6,\n#     n_estimators=50,\n#     random_state=42,\n#     alpha=0.1,\n#     subsample=.8,\n#     enable_categorical=True,# L1 regularization\n#     reg_lambda=1.5                 # L2 regularization (formerly lambda)\n# )# 4. Train the model\n# xgb_model.fit(X_train_o, y_train_o)\n\n# # 5. Make predictions\n# y_pred_o = xgb_model.predict(X_val_o)\n\n# # 6. Evaluate the model\n# accuracy = accuracy_score(y_val_o, y_pred_o)\n# print(f\"XGBoost Accuracy: {accuracy}\")\n\n# # Print classification report\n# print(classification_report(y_val_o, y_pred_o))","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.692333Z","iopub.execute_input":"2024-10-10T10:34:59.692758Z","iopub.status.idle":"2024-10-10T10:34:59.702816Z","shell.execute_reply.started":"2024-10-10T10:34:59.692717Z","shell.execute_reply":"2024-10-10T10:34:59.701575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n# import matplotlib.pyplot as plt\n\n# # Generate confusion matrix\n# cm = confusion_matrix(y_val, y_pred)\n\n# # Display confusion matrix\n# disp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=[0, 1, 2, 3])\n# disp.plot(cmap=plt.cm.Blues)\n# plt.title('Confusion Matrix')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.704615Z","iopub.execute_input":"2024-10-10T10:34:59.705567Z","iopub.status.idle":"2024-10-10T10:34:59.717145Z","shell.execute_reply.started":"2024-10-10T10:34:59.705510Z","shell.execute_reply":"2024-10-10T10:34:59.715856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pip install missingpy","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.718793Z","iopub.execute_input":"2024-10-10T10:34:59.719302Z","iopub.status.idle":"2024-10-10T10:34:59.729131Z","shell.execute_reply.started":"2024-10-10T10:34:59.719227Z","shell.execute_reply":"2024-10-10T10:34:59.727894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.experimental import enable_iterative_imputer\n# from sklearn.impute import IterativeImputer\n\n# # Define IterativeImputer with BayesianRidge estimator\n# bayesian_imputer = IterativeImputer(estimator=BayesianRidge(), max_iter=10, random_state=0)\n\n# # Fit and transform the training data\n# train_bayesian_imputed = bayesian_imputer.fit_transform(train_in)\n\n# # Convert back to DataFrame\n# train_bayesian = pd.DataFrame(train_bayesian_imputed, columns=train_in.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.730749Z","iopub.execute_input":"2024-10-10T10:34:59.731146Z","iopub.status.idle":"2024-10-10T10:34:59.741788Z","shell.execute_reply.started":"2024-10-10T10:34:59.731104Z","shell.execute_reply":"2024-10-10T10:34:59.740424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.experimental import enable_iterative_imputer  # Required to enable the experimental feature\n# from sklearn.impute import IterativeImputer\n\n# # Instantiate IterativeImputer\n# imputer = IterativeImputer()\n\n# # Fit and transform the data\n# imputed_data = imputer.fit_transform(train_in)\n\n# # Convert it back to a DataFrame (if needed)\n# train_imputed = pd.DataFrame(imputed_data, columns=train_in.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.743492Z","iopub.execute_input":"2024-10-10T10:34:59.743904Z","iopub.status.idle":"2024-10-10T10:34:59.753589Z","shell.execute_reply.started":"2024-10-10T10:34:59.743863Z","shell.execute_reply":"2024-10-10T10:34:59.752310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df=train_in\n# from sklearn.preprocessing import MinMaxScaler\n# from tensorflow.keras.models import Sequential\n# from tensorflow.keras.layers import Dense\n\n# # Fill missing values with mean first\n# df.fillna(df.mean(), inplace=True)\n\n# # Normalize data\n# scaler = MinMaxScaler()\n# data_scaled = scaler.fit_transform(df)\n\n# # Define autoencoder\n# autoencoder = Sequential([\n#     Dense(128, activation='relu', input_shape=(data_scaled.shape[1],)),\n#     Dense(64, activation='relu'),\n#     Dense(128, activation='relu'),\n#     Dense(data_scaled.shape[1], activation='linear')\n# ])\n\n# autoencoder.compile(optimizer='adam', loss='mean_squared_error')\n# autoencoder.fit(data_scaled, data_scaled, epochs=50, batch_size=32, shuffle=True)\n\n# # Predict and fill missing values\n# predicted = autoencoder.predict(data_scaled)\n# df = pd.DataFrame(scaler.inverse_transform(predicted), columns=df.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.755080Z","iopub.execute_input":"2024-10-10T10:34:59.755509Z","iopub.status.idle":"2024-10-10T10:34:59.769318Z","shell.execute_reply.started":"2024-10-10T10:34:59.755467Z","shell.execute_reply":"2024-10-10T10:34:59.767928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.770874Z","iopub.execute_input":"2024-10-10T10:34:59.771356Z","iopub.status.idle":"2024-10-10T10:34:59.780567Z","shell.execute_reply.started":"2024-10-10T10:34:59.771287Z","shell.execute_reply":"2024-10-10T10:34:59.779323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1=pd.concat([cat_data,df],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.782126Z","iopub.execute_input":"2024-10-10T10:34:59.782560Z","iopub.status.idle":"2024-10-10T10:34:59.791548Z","shell.execute_reply.started":"2024-10-10T10:34:59.782517Z","shell.execute_reply":"2024-10-10T10:34:59.790342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df1","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.793143Z","iopub.execute_input":"2024-10-10T10:34:59.793557Z","iopub.status.idle":"2024-10-10T10:34:59.803210Z","shell.execute_reply.started":"2024-10-10T10:34:59.793516Z","shell.execute_reply":"2024-10-10T10:34:59.801863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2=df1.drop(\"id\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.805178Z","iopub.execute_input":"2024-10-10T10:34:59.805777Z","iopub.status.idle":"2024-10-10T10:34:59.815405Z","shell.execute_reply.started":"2024-10-10T10:34:59.805717Z","shell.execute_reply":"2024-10-10T10:34:59.814025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df2","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.816839Z","iopub.execute_input":"2024-10-10T10:34:59.817330Z","iopub.status.idle":"2024-10-10T10:34:59.829303Z","shell.execute_reply.started":"2024-10-10T10:34:59.817277Z","shell.execute_reply":"2024-10-10T10:34:59.827997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in cat_col1:\n#     if(col!=\"id\"):df2[col] = df2[col].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.830972Z","iopub.execute_input":"2024-10-10T10:34:59.831421Z","iopub.status.idle":"2024-10-10T10:34:59.841005Z","shell.execute_reply.started":"2024-10-10T10:34:59.831380Z","shell.execute_reply":"2024-10-10T10:34:59.839810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from imblearn.over_sampling import RandomOverSampler\n\n# # Initialize the oversampler\n# oversampler = RandomOverSampler(random_state=42)\n\n# # Fit and resample the data\n# X_resampled_c, y_resampled_c = oversampler.fit_resample(df2, train_y)\n\n# # Check the new class distribution\n# new_class_distribution = pd.Series(y_resampled_c).value_counts()\n# print(\"New class distribution after oversampling:\\n\", new_class_distribution)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.842563Z","iopub.execute_input":"2024-10-10T10:34:59.842979Z","iopub.status.idle":"2024-10-10T10:34:59.855407Z","shell.execute_reply.started":"2024-10-10T10:34:59.842938Z","shell.execute_reply":"2024-10-10T10:34:59.854152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# from catboost import CatBoostClassifier, Pool\n# from sklearn.model_selection import train_test_split\n# from sklearn.metrics import accuracy_score, classification_report","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.857122Z","iopub.execute_input":"2024-10-10T10:34:59.857672Z","iopub.status.idle":"2024-10-10T10:34:59.868329Z","shell.execute_reply.started":"2024-10-10T10:34:59.857615Z","shell.execute_reply":"2024-10-10T10:34:59.867037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train_c, X_val_c, y_train_c, y_val_c = train_test_split(X_resampled_c, y_resampled_c, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.870191Z","iopub.execute_input":"2024-10-10T10:34:59.870707Z","iopub.status.idle":"2024-10-10T10:34:59.883656Z","shell.execute_reply.started":"2024-10-10T10:34:59.870660Z","shell.execute_reply":"2024-10-10T10:34:59.882188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# catc=[]\n# for c in cat_col1:\n#     if(c!=\"id\" and c!=\"sii\"):catc.append(c)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.885520Z","iopub.execute_input":"2024-10-10T10:34:59.886163Z","iopub.status.idle":"2024-10-10T10:34:59.893989Z","shell.execute_reply.started":"2024-10-10T10:34:59.886101Z","shell.execute_reply":"2024-10-10T10:34:59.892580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# catc","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.895665Z","iopub.execute_input":"2024-10-10T10:34:59.896119Z","iopub.status.idle":"2024-10-10T10:34:59.906187Z","shell.execute_reply.started":"2024-10-10T10:34:59.896076Z","shell.execute_reply":"2024-10-10T10:34:59.904948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_pool = Pool(X_train_c, y_train_c, cat_features=catc)\n# val_pool = Pool(X_val_c, y_val_c, cat_features=catc)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.907810Z","iopub.execute_input":"2024-10-10T10:34:59.908211Z","iopub.status.idle":"2024-10-10T10:34:59.915260Z","shell.execute_reply.started":"2024-10-10T10:34:59.908171Z","shell.execute_reply":"2024-10-10T10:34:59.914049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# catboost_model = CatBoostClassifier(iterations=100, learning_rate=0.1, depth=6, random_seed=42, verbose=10)\n# catboost_model.fit(train_pool, eval_set=val_pool, early_stopping_rounds=10)\n\n# # Make predictions\n# y_pred_c = catboost_model.predict(X_val_c)\n\n# # Evaluate the model\n# accuracy = accuracy_score(y_val_c, y_pred_c)\n# print(f\"CatBoost Accuracy: {accuracy}\")\n# print(classification_report(y_val_c, y_pred_c))","metadata":{"execution":{"iopub.status.busy":"2024-10-10T10:34:59.916815Z","iopub.execute_input":"2024-10-10T10:34:59.917612Z","iopub.status.idle":"2024-10-10T10:34:59.926892Z","shell.execute_reply.started":"2024-10-10T10:34:59.917560Z","shell.execute_reply":"2024-10-10T10:34:59.925747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}