{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"libraries","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier, VotingClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import GridSearchCV, cross_val_predict\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np \n\nimport statsmodels.api as sm\n\nimport scipy.stats as stats\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor as vif\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.model_selection import train_test_split\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.719650Z","iopub.execute_input":"2024-12-11T13:40:00.720160Z","iopub.status.idle":"2024-12-11T13:40:00.728956Z","shell.execute_reply.started":"2024-12-11T13:40:00.720122Z","shell.execute_reply":"2024-12-11T13:40:00.727634Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reading csv files","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndf_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n#Checking shapes\ntrain.shape, df_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.730816Z","iopub.execute_input":"2024-12-11T13:40:00.731169Z","iopub.status.idle":"2024-12-11T13:40:00.801276Z","shell.execute_reply.started":"2024-12-11T13:40:00.731136Z","shell.execute_reply":"2024-12-11T13:40:00.800012Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Checking shapes","metadata":{}},{"cell_type":"code","source":"# Get the intersection of columns in train and df_test\ncommon_columns = train.columns.intersection(df_test.columns)\n\n# Create the new DataFrame 'df_train' with only the common columns\ndf_train = train[common_columns].copy()\n\n# Add the 'sii' column \nif 'sii' in train.columns:\n    df_train.loc[:, 'sii'] = train['sii']\n\ndf_train.shape, df_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.803310Z","iopub.execute_input":"2024-12-11T13:40:00.803778Z","iopub.status.idle":"2024-12-11T13:40:00.821917Z","shell.execute_reply.started":"2024-12-11T13:40:00.803731Z","shell.execute_reply":"2024-12-11T13:40:00.820626Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Checking missing data","metadata":{}},{"cell_type":"code","source":"def missing_data_summary(df):\n    \"\"\"\n    This function summarizes missing data, \n    showing count and percentage of missing values for each column.\n    \"\"\"\n    return (pd.DataFrame(df.isna().sum())\n            .reset_index()\n            .rename(columns={'index': 'Column', 0: 'mis_count'})\n            .query('mis_count > 0')\n            .assign(Missing_Percentage=lambda x: x['mis_count'] / df.shape[0] * 100)\n            .sort_values('mis_count', ascending=False)\n            .reset_index(drop=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.824679Z","iopub.execute_input":"2024-12-11T13:40:00.825012Z","iopub.status.idle":"2024-12-11T13:40:00.837294Z","shell.execute_reply.started":"2024-12-11T13:40:00.824977Z","shell.execute_reply":"2024-12-11T13:40:00.836191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_data_summary(df_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.838749Z","iopub.execute_input":"2024-12-11T13:40:00.839188Z","iopub.status.idle":"2024-12-11T13:40:00.874390Z","shell.execute_reply.started":"2024-12-11T13:40:00.839146Z","shell.execute_reply":"2024-12-11T13:40:00.873100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_data_summary(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.875994Z","iopub.execute_input":"2024-12-11T13:40:00.876414Z","iopub.status.idle":"2024-12-11T13:40:00.902720Z","shell.execute_reply.started":"2024-12-11T13:40:00.876379Z","shell.execute_reply":"2024-12-11T13:40:00.901311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify columns in 'train' that are missing in 'df_test'\nmissing_columns = [col for col in train.columns if col not in df_test.columns]\n\n# Create a DataFrame 'missing_data' for all data in 'train' columns missing from 'df_test'\nmissing_data = train[missing_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.904141Z","iopub.execute_input":"2024-12-11T13:40:00.905039Z","iopub.status.idle":"2024-12-11T13:40:00.913383Z","shell.execute_reply.started":"2024-12-11T13:40:00.904968Z","shell.execute_reply":"2024-12-11T13:40:00.911827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_data_summary(missing_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.915009Z","iopub.execute_input":"2024-12-11T13:40:00.915375Z","iopub.status.idle":"2024-12-11T13:40:00.944756Z","shell.execute_reply.started":"2024-12-11T13:40:00.915332Z","shell.execute_reply":"2024-12-11T13:40:00.943668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_data.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.946751Z","iopub.execute_input":"2024-12-11T13:40:00.947234Z","iopub.status.idle":"2024-12-11T13:40:00.955752Z","shell.execute_reply.started":"2024-12-11T13:40:00.947186Z","shell.execute_reply":"2024-12-11T13:40:00.954629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Checking if PCIAT-PCIAT_Total correctness ","metadata":{}},{"cell_type":"code","source":"# Identify all columns that contribute to 'PCIAT-PCIAT_Total', excluding 'PCIAT-PCIAT_Total' itself\npciat_columns = [col for col in train.columns if col.startswith('PCIAT-PCIAT_') and col != 'PCIAT-PCIAT_Total']\n\n# Create a new boolean column to check the correctness of 'PCIAT-PCIAT_Total'\nmissing_data.loc[:,'PCIAT_Total_Correct'] = train[pciat_columns].fillna(0).sum(axis=1) == train['PCIAT-PCIAT_Total']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.959013Z","iopub.execute_input":"2024-12-11T13:40:00.959435Z","iopub.status.idle":"2024-12-11T13:40:00.974598Z","shell.execute_reply.started":"2024-12-11T13:40:00.959377Z","shell.execute_reply":"2024-12-11T13:40:00.973222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_data['PCIAT_Total_Correct'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.976244Z","iopub.execute_input":"2024-12-11T13:40:00.976649Z","iopub.status.idle":"2024-12-11T13:40:00.989104Z","shell.execute_reply.started":"2024-12-11T13:40:00.976615Z","shell.execute_reply":"2024-12-11T13:40:00.987589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Imputing missing data in PCIAT ","metadata":{}},{"cell_type":"code","source":"# Identify the PCIAT-PCIAT_number columns (excluding 'PCIAT-PCIAT_Total')\npciat_columns = [col for col in train.columns if col.startswith('PCIAT-PCIAT_') and col != 'PCIAT-PCIAT_Total']\n\n# Initialize the KNNImputer with a reasonable number of neighbors (e.g., 5)\nimputer = KNNImputer(n_neighbors=5)\n\n# Apply the imputer to the PCIAT-PCIAT_number columns\nimputed_data = imputer.fit_transform(train[pciat_columns])\n\n# Round the imputed values to the nearest integer and cast them to int type\ntrain[pciat_columns] = imputed_data.round().astype(int)\n\n# Confirm that missing values have been imputed\nprint(train[pciat_columns].isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:00.990625Z","iopub.execute_input":"2024-12-11T13:40:00.991052Z","iopub.status.idle":"2024-12-11T13:40:02.281716Z","shell.execute_reply.started":"2024-12-11T13:40:00.991016Z","shell.execute_reply":"2024-12-11T13:40:02.280446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Update the 'PCIAT-PCIAT_Total' column with the new sum\ntrain['PCIAT-PCIAT_Total'] = train[pciat_columns].sum(axis=1)\n\n# Confirm the updated values\nprint(train['PCIAT-PCIAT_Total'].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.283257Z","iopub.execute_input":"2024-12-11T13:40:02.284030Z","iopub.status.idle":"2024-12-11T13:40:02.296352Z","shell.execute_reply.started":"2024-12-11T13:40:02.283991Z","shell.execute_reply":"2024-12-11T13:40:02.294828Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Updating sii","metadata":{}},{"cell_type":"code","source":"# Update the 'sii' column based on the rules defined in the data dictionary\ntrain['sii'] = 0  # Default to 0 (None)\ntrain.loc[(train['PCIAT-PCIAT_Total'] > 30) & (train['PCIAT-PCIAT_Total'] <= 49), 'sii'] = 1  # Mild\ntrain.loc[(train['PCIAT-PCIAT_Total'] > 49) & (train['PCIAT-PCIAT_Total'] <= 79), 'sii'] = 2  # Moderate\ntrain.loc[train['PCIAT-PCIAT_Total'] >= 80, 'sii'] = 3  # Severe\n\n# Confirm the updated values\nprint(train['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.297922Z","iopub.execute_input":"2024-12-11T13:40:02.298387Z","iopub.status.idle":"2024-12-11T13:40:02.313799Z","shell.execute_reply.started":"2024-12-11T13:40:02.298334Z","shell.execute_reply":"2024-12-11T13:40:02.312248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['sii']=train['sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.315314Z","iopub.execute_input":"2024-12-11T13:40:02.315747Z","iopub.status.idle":"2024-12-11T13:40:02.328033Z","shell.execute_reply.started":"2024-12-11T13:40:02.315714Z","shell.execute_reply":"2024-12-11T13:40:02.326723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.T.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.329462Z","iopub.execute_input":"2024-12-11T13:40:02.329862Z","iopub.status.idle":"2024-12-11T13:40:02.369619Z","shell.execute_reply.started":"2024-12-11T13:40:02.329828Z","shell.execute_reply":"2024-12-11T13:40:02.368384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_columns = df_train.select_dtypes(include=['object', 'category']).columns\ndf_train[cat_columns].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.371339Z","iopub.execute_input":"2024-12-11T13:40:02.371829Z","iopub.status.idle":"2024-12-11T13:40:02.389908Z","shell.execute_reply.started":"2024-12-11T13:40:02.371770Z","shell.execute_reply":"2024-12-11T13:40:02.388419Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Starting pipeline","metadata":{}},{"cell_type":"code","source":"df_train.drop('id', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.391922Z","iopub.execute_input":"2024-12-11T13:40:02.392814Z","iopub.status.idle":"2024-12-11T13:40:02.401766Z","shell.execute_reply.started":"2024-12-11T13:40:02.392750Z","shell.execute_reply":"2024-12-11T13:40:02.400313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dividing data on categorical and numerical (without target column)\n\ncat_columns = df_train.select_dtypes(include=['object', 'category']).columns\nnum_columns = df_train.select_dtypes(include=['int64', 'float64']).columns.drop('sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.403180Z","iopub.execute_input":"2024-12-11T13:40:02.403537Z","iopub.status.idle":"2024-12-11T13:40:02.416616Z","shell.execute_reply.started":"2024-12-11T13:40:02.403480Z","shell.execute_reply":"2024-12-11T13:40:02.414849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_columns = (\n            [#'Basic_Demos-Enroll_Season',\n             'CGAS-Season',\n             'Physical-Season',\n             'Fitness_Endurance-Season',\n             'FGC-Season', 'BIA-Season',\n             'PAQ_A-Season',\n             'PAQ_C-Season',\n             'SDS-Season',\n             'PreInt_EduHx-Season'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.418156Z","iopub.execute_input":"2024-12-11T13:40:02.418628Z","iopub.status.idle":"2024-12-11T13:40:02.429200Z","shell.execute_reply.started":"2024-12-11T13:40:02.418584Z","shell.execute_reply":"2024-12-11T13:40:02.427942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define transformers for numerical and categorical columns\nnum_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', StandardScaler())\n])\n\ncat_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant', fill_value='missing')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore', sparse_output = False))\n])\n\n# Combine transformers using ColumnTransformer\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', num_transformer, num_columns),\n        ('cat', cat_transformer, cat_columns)\n    ],remainder = 'drop')\n\n# Create a pipeline with the preprocessor\npipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor)])\n\n# Apply the pipeline \nX = df_train.drop('sii', axis=1)\ny = df_train['sii'] #normalize dependent variable np.log(\nX_preprocessed = pipeline.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.430278Z","iopub.execute_input":"2024-12-11T13:40:02.430590Z","iopub.status.idle":"2024-12-11T13:40:02.487979Z","shell.execute_reply.started":"2024-12-11T13:40:02.430562Z","shell.execute_reply":"2024-12-11T13:40:02.486798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#checking\nX_preprocessed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.489442Z","iopub.execute_input":"2024-12-11T13:40:02.489818Z","iopub.status.idle":"2024-12-11T13:40:02.497862Z","shell.execute_reply.started":"2024-12-11T13:40:02.489786Z","shell.execute_reply":"2024-12-11T13:40:02.496749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Apply the pipline to test data\n\ndf_test_preprocessed = pipeline.transform(df_test) #using transform ( not fit_transform!!! - produces the same amount of features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.499620Z","iopub.execute_input":"2024-12-11T13:40:02.500107Z","iopub.status.idle":"2024-12-11T13:40:02.519826Z","shell.execute_reply.started":"2024-12-11T13:40:02.500060Z","shell.execute_reply":"2024-12-11T13:40:02.518280Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_preprocessed.shape, X_preprocessed.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.521260Z","iopub.execute_input":"2024-12-11T13:40:02.521615Z","iopub.status.idle":"2024-12-11T13:40:02.530249Z","shell.execute_reply.started":"2024-12-11T13:40:02.521583Z","shell.execute_reply":"2024-12-11T13:40:02.528969Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Solving imbalanced data by SMOTE","metadata":{}},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\nfrom sklearn.model_selection import train_test_split\n\n# Initialize SMOTE\nsmote = SMOTE(random_state=42)\n\n# Apply SMOTE to the data\nX_train_smote, y_train_smote = smote.fit_resample(X_preprocessed, y)\n\n# Check the distribution of the resampled data\nfrom collections import Counter\nprint(\"Original training distribution:\", Counter(y))\nprint(\"Resampled training distribution:\", Counter(y_train_smote))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.532066Z","iopub.execute_input":"2024-12-11T13:40:02.532486Z","iopub.status.idle":"2024-12-11T13:40:02.688768Z","shell.execute_reply.started":"2024-12-11T13:40:02.532451Z","shell.execute_reply":"2024-12-11T13:40:02.687573Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Creating train and validation set","metadata":{}},{"cell_type":"code","source":"# Split the data into training and testing sets\nX_train, X_valid, y_train, y_valid = train_test_split(X_train_smote, y_train_smote, test_size=0.25, random_state=123)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:40:02.690014Z","iopub.execute_input":"2024-12-11T13:40:02.690363Z","iopub.status.idle":"2024-12-11T13:40:02.702813Z","shell.execute_reply.started":"2024-12-11T13:40:02.690328Z","shell.execute_reply":"2024-12-11T13:40:02.701625Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training Random Forest\n### on test and validation dataset","metadata":{"execution":{"iopub.status.busy":"2024-12-10T14:29:23.646384Z","iopub.execute_input":"2024-12-10T14:29:23.646885Z","iopub.status.idle":"2024-12-10T14:29:23.652413Z","shell.execute_reply.started":"2024-12-10T14:29:23.646844Z","shell.execute_reply":"2024-12-10T14:29:23.651347Z"}}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier, VotingClassifier\n\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Random Forest\nrf = RandomForestClassifier(random_state=123)\nrf_params = {\n    'n_estimators': [300],\n    'max_depth': [None, 10, 20],\n    'min_samples_split': [2, 3]\n}\n\nrf_grid = GridSearchCV(rf, rf_params, cv=5, scoring='accuracy', n_jobs=-1)\nrf_grid.fit(X_train, y_train)\nprint(\"Best Parameters for Random Forest:\", rf_grid.best_params_)\nprint(\"Random Forest Performance:\\n\", classification_report(y_valid, rf_grid.best_estimator_.predict(X_valid)))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:44:41.009912Z","iopub.execute_input":"2024-12-11T13:44:41.010342Z","iopub.status.idle":"2024-12-11T13:46:23.635534Z","shell.execute_reply.started":"2024-12-11T13:44:41.010308Z","shell.execute_reply":"2024-12-11T13:46:23.634284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion Matrix\ny_pred = rf_grid.best_estimator_.predict(X_valid)\ncm = confusion_matrix(y_valid, y_pred)\n\n# Visualize the Confusion Matrix\nplt.figure(figsize=(4, 4))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=rf_grid.best_estimator_.classes_, yticklabels=rf_grid.best_estimator_.classes_)\nplt.xlabel('Predicted Labels')\nplt.ylabel('True Labels')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:47:22.248964Z","iopub.execute_input":"2024-12-11T13:47:22.249385Z","iopub.status.idle":"2024-12-11T13:47:22.774139Z","shell.execute_reply.started":"2024-12-11T13:47:22.249350Z","shell.execute_reply":"2024-12-11T13:47:22.772973Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training Random Forest\n### on whole train dataset","metadata":{}},{"cell_type":"code","source":"# Ensure X_train, X_valid, y_train, y_valid are converted to pandas objects if they're NumPy arrays\nX_train = pd.DataFrame(X_train) if isinstance(X_train, (np.ndarray)) else X_train\nX_valid = pd.DataFrame(X_valid) if isinstance(X_valid, (np.ndarray)) else X_valid\ny_train = pd.Series(y_train) if isinstance(y_train, (np.ndarray)) else y_train\ny_valid = pd.Series(y_valid) if isinstance(y_valid, (np.ndarray)) else y_valid\n\n\n# Concatenate Train and Validation datasets\nX_final = pd.concat([X_train, X_valid])\ny_final = pd.concat([y_train, y_valid])\n\n# Random Forest\nrf = RandomForestClassifier(random_state=123)\nrf_params = {\n    'n_estimators': [200, 300],\n    'max_depth': [None, 10, 20],\n    'min_samples_split': [2, 3]\n}\n\nrf_grid = GridSearchCV(rf, rf_params, cv=5, scoring='accuracy', n_jobs=-1)\nrf_grid.fit(X_final, y_final)\n\nprint(\"Best Parameters for Random Forest:\", rf_grid.best_params_)\n\n# Evaluate on the final dataset using cross-validation\ny_pred_final = cross_val_predict(rf_grid.best_estimator_, X_final, y_final, cv=5)\n\n# Performance Metrics\nprint(\"Final Random Forest Performance:\\n\", classification_report(y_final, y_pred_final))\n\n# Confusion Matrix\ncm_final = confusion_matrix(y_final, y_pred_final)\n\n# Visualize the Confusion Matrix\nplt.figure(figsize=(4, 4))\nsns.heatmap(\n    cm_final, \n    annot=True, \n    fmt='d', \n    cmap='Blues', \n    xticklabels=rf_grid.best_estimator_.classes_, \n    yticklabels=rf_grid.best_estimator_.classes_\n)\nplt.xlabel('Predicted Labels')\nplt.ylabel('True Labels')\nplt.title('Final Confusion Matrix')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:47:31.629689Z","iopub.execute_input":"2024-12-11T13:47:31.630084Z","iopub.status.idle":"2024-12-11T13:52:14.318372Z","shell.execute_reply.started":"2024-12-11T13:47:31.630053Z","shell.execute_reply":"2024-12-11T13:52:14.316844Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict target for test data","metadata":{}},{"cell_type":"code","source":"# Predict target for test data\ndf_test_preprocessed = pd.DataFrame(df_test_preprocessed) if isinstance(df_test_preprocessed, (np.ndarray)) else df_test_preprocessed\ny_test_pred = rf_grid.best_estimator_.predict(df_test_preprocessed)\n\n# Output predictions\nprint(\"Test Data Predictions:\", y_test_pred)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:52:28.473819Z","iopub.execute_input":"2024-12-11T13:52:28.475250Z","iopub.status.idle":"2024-12-11T13:52:28.509770Z","shell.execute_reply.started":"2024-12-11T13:52:28.475200Z","shell.execute_reply":"2024-12-11T13:52:28.508536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission\n\ndf_submit = df_test[['id']].copy()\ndf_submit['sii'] = y_test_pred\n\ndf_submit.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T13:52:32.863167Z","iopub.execute_input":"2024-12-11T13:52:32.863682Z","iopub.status.idle":"2024-12-11T13:52:32.874572Z","shell.execute_reply.started":"2024-12-11T13:52:32.863636Z","shell.execute_reply":"2024-12-11T13:52:32.873178Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n**Author:** Beata Faron  \n[LinkedIn](https://www.linkedin.com/in/beata-faron-24764832/) • [Kaggle](https://www.kaggle.com/beatafaron)\n\n*Data Scientist with a background in business, design, and machine learning. Focused on time series forecasting and real-world applications.*\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}