{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:08.751612Z","iopub.execute_input":"2024-12-04T10:03:08.752042Z","iopub.status.idle":"2024-12-04T10:03:10.285538Z","shell.execute_reply.started":"2024-12-04T10:03:08.751994Z","shell.execute_reply":"2024-12-04T10:03:10.284280Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.metrics import classification_report, accuracy_score\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import GridSearchCV, train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, accuracy_score\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom xgboost import XGBClassifier ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:10.287113Z","iopub.execute_input":"2024-12-04T10:03:10.287716Z","iopub.status.idle":"2024-12-04T10:03:11.094995Z","shell.execute_reply.started":"2024-12-04T10:03:10.287668Z","shell.execute_reply":"2024-12-04T10:03:11.094128Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Exploration","metadata":{}},{"cell_type":"code","source":"# Saving the files\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\ndf_train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndf_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:11.096330Z","iopub.execute_input":"2024-12-04T10:03:11.096787Z","iopub.status.idle":"2024-12-04T10:03:11.158509Z","shell.execute_reply.started":"2024-12-04T10:03:11.096731Z","shell.execute_reply":"2024-12-04T10:03:11.157589Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Comment this cells after visualization to remove clutter and save time in new runs","metadata":{}},{"cell_type":"code","source":"# Reading the dictionary\n# data_dict.style","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:11.159716Z","iopub.execute_input":"2024-12-04T10:03:11.160125Z","iopub.status.idle":"2024-12-04T10:03:11.164895Z","shell.execute_reply.started":"2024-12-04T10:03:11.160082Z","shell.execute_reply":"2024-12-04T10:03:11.163561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualizng the df\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:11.166498Z","iopub.execute_input":"2024-12-04T10:03:11.166926Z","iopub.status.idle":"2024-12-04T10:03:11.203268Z","shell.execute_reply.started":"2024-12-04T10:03:11.166866Z","shell.execute_reply":"2024-12-04T10:03:11.202206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:11.204981Z","iopub.execute_input":"2024-12-04T10:03:11.205318Z","iopub.status.idle":"2024-12-04T10:03:11.229829Z","shell.execute_reply.started":"2024-12-04T10:03:11.205286Z","shell.execute_reply":"2024-12-04T10:03:11.228548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature Selection\n# Choose only relevant columns for the analysis from the training dataset.\n# This step ensures the model doesn't get bogged down by unrelated features.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:11.231133Z","iopub.execute_input":"2024-12-04T10:03:11.231486Z","iopub.status.idle":"2024-12-04T10:03:11.240540Z","shell.execute_reply.started":"2024-12-04T10:03:11.231446Z","shell.execute_reply":"2024-12-04T10:03:11.239521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Getting the useable columns in test\ndf_train = df_train[['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']]\ndf_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:11.245538Z","iopub.execute_input":"2024-12-04T10:03:11.245896Z","iopub.status.idle":"2024-12-04T10:03:11.258782Z","shell.execute_reply.started":"2024-12-04T10:03:11.245856Z","shell.execute_reply":"2024-12-04T10:03:11.257742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Checking values\nfor column in df_train.columns:\n    print(f\"NaN {column}: {df_train[column].isna().sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:11.260129Z","iopub.execute_input":"2024-12-04T10:03:11.260795Z","iopub.status.idle":"2024-12-04T10:03:11.290168Z","shell.execute_reply.started":"2024-12-04T10:03:11.260762Z","shell.execute_reply":"2024-12-04T10:03:11.289003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:11.291683Z","iopub.execute_input":"2024-12-04T10:03:11.292534Z","iopub.status.idle":"2024-12-04T10:03:11.306968Z","shell.execute_reply.started":"2024-12-04T10:03:11.292466Z","shell.execute_reply":"2024-12-04T10:03:11.305869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handle Missing Values\ntrain_supervised = df_train[df_train['sii'].notnull()]\n\n# Cleaning and Imputation\n# Remove rows with excessive missing data and impute missing values based on column types.\n\n\nrow_missing_threshold = 0.5  \ntrain_supervised = train_supervised[train_supervised.isnull().mean(axis=1) <= row_missing_threshold]\n\nmissing_percentages = train_supervised.isnull().mean() * 100\nmissing_percentages = missing_percentages[missing_percentages > 0].sort_values()\n\nplt.figure(figsize=(12, 6))\nsns.barplot(x=missing_percentages.index, y=missing_percentages.values)\nplt.xticks(rotation=90)\nplt.xlabel(\"Features\")\nplt.ylabel(\"Percentage of Missing Values\")\nplt.title(\"Missing Values Percentage After Row Filtering\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:11.308146Z","iopub.execute_input":"2024-12-04T10:03:11.308470Z","iopub.status.idle":"2024-12-04T10:03:12.018534Z","shell.execute_reply.started":"2024-12-04T10:03:11.308438Z","shell.execute_reply":"2024-12-04T10:03:12.017392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Feature Preprocessing\n### Separate numerical and categorical features for tailored preprocessing.\n### Use scaling and one-hot encoding to normalize and prepare the data for modeling.\n\nfrom sklearn.impute import SimpleImputer\n\ny_train = train_supervised['sii']\ntrain_supervised.drop(columns=['sii'], inplace=True)\n\nnumerical_features = train_supervised.select_dtypes(include=['int64', 'float64']).columns\ncategorical_features = train_supervised.select_dtypes(include=['object']).columns\n\nnumerical_imputer = SimpleImputer(strategy='mean')\ntrain_supervised[numerical_features] = numerical_imputer.fit_transform(train_supervised[numerical_features])\n\n# categorical_imputer = SimpleImputer(strategy='most_frequent')\n# train_supervised[categorical_features] = categorical_imputer.fit_transform(train_supervised[categorical_features])\ntrain_supervised[categorical_features] = df_train[categorical_features].fillna('Missing')  # Replace NaN with 'Missing'\nprint(train_supervised.isnull().sum().sum()) \n\n# train_supervised = train_supervised[df_test.columns]\n# df_train[numerical_columns] = df_train[numerical_columns].fillna(0)  # Replace NaN with 0\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:12.019882Z","iopub.execute_input":"2024-12-04T10:03:12.020207Z","iopub.status.idle":"2024-12-04T10:03:12.058369Z","shell.execute_reply.started":"2024-12-04T10:03:12.020175Z","shell.execute_reply":"2024-12-04T10:03:12.057205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\n\n# One-hot encode kolom kategorikal dan scale kolom numerikal\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', StandardScaler(), numerical_features),\n        ('cat', OneHotEncoder(), categorical_features)\n    ])\n\nX_processed = preprocessor.fit_transform(train_supervised)\n# Memisahkan fitur (X) dan target (y)\nX = train_supervised\ny = y_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:12.059589Z","iopub.execute_input":"2024-12-04T10:03:12.059946Z","iopub.status.idle":"2024-12-04T10:03:12.090268Z","shell.execute_reply.started":"2024-12-04T10:03:12.059914Z","shell.execute_reply":"2024-12-04T10:03:12.089189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Shape of X_processed:\", X_processed.shape)\nprint(\"Shape of y_train:\", y_train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:12.091597Z","iopub.execute_input":"2024-12-04T10:03:12.091977Z","iopub.status.idle":"2024-12-04T10:03:12.097701Z","shell.execute_reply.started":"2024-12-04T10:03:12.091943Z","shell.execute_reply":"2024-12-04T10:03:12.096632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X.head())  # Fitur\nprint(y.head())  # Kolom target","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:12.099186Z","iopub.execute_input":"2024-12-04T10:03:12.099536Z","iopub.status.idle":"2024-12-04T10:03:12.123150Z","shell.execute_reply.started":"2024-12-04T10:03:12.099505Z","shell.execute_reply":"2024-12-04T10:03:12.122054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_supervised.head())  # Tampilkan beberapa baris awal untuk memverifikasi","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:12.124567Z","iopub.execute_input":"2024-12-04T10:03:12.124999Z","iopub.status.idle":"2024-12-04T10:03:12.147769Z","shell.execute_reply.started":"2024-12-04T10:03:12.124939Z","shell.execute_reply":"2024-12-04T10:03:12.146588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_processed = preprocessor.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:12.149095Z","iopub.execute_input":"2024-12-04T10:03:12.149433Z","iopub.status.idle":"2024-12-04T10:03:12.179730Z","shell.execute_reply.started":"2024-12-04T10:03:12.149391Z","shell.execute_reply":"2024-12-04T10:03:12.178582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dropping NaN sii since this is the target variable that we are trying to predict and we need its true values\n# df_train.dropna(subset=['sii'], axis = 'index', inplace = True)\n# Rechecking nan values\nfor column in train_supervised.columns:\n    print(f\"NaN {column}: {train_supervised[column].isna().sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:12.181179Z","iopub.execute_input":"2024-12-04T10:03:12.181619Z","iopub.status.idle":"2024-12-04T10:03:12.198649Z","shell.execute_reply.started":"2024-12-04T10:03:12.181572Z","shell.execute_reply":"2024-12-04T10:03:12.197458Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Applying PCA","metadata":{}},{"cell_type":"code","source":"# New Test to fix bugs\nnumerical_features = train_supervised.select_dtypes(include=['int64', 'float64']).columns\ncategorical_features = train_supervised.select_dtypes(include=['object']).columns\n# Scale numerical features\nscaler = StandardScaler()\nX_scaled_numerical = scaler.fit_transform(train_supervised[numerical_features])\n\n# One-hot encode categorical features\nencoder = OneHotEncoder(drop='first', sparse=False)\nX_encoded_categorical = encoder.fit_transform(train_supervised[categorical_features])\n\n# Combine numerical and categorical features\nX_scaled = np.hstack((X_scaled_numerical, X_encoded_categorical))\n\n# PCA\npca = PCA(n_components=0.8)  # Retain x% of variance\nX_pca = pca.fit_transform(X_scaled)\n\n# Outputs\nprint(\"Number of components selected:\", pca.n_components_)\nprint(\"Explained variance ratio:\", pca.explained_variance_ratio_)\nprint(\"Cumulative variance explained:\", pca.explained_variance_ratio_.cumsum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:12.200121Z","iopub.execute_input":"2024-12-04T10:03:12.200566Z","iopub.status.idle":"2024-12-04T10:03:12.276096Z","shell.execute_reply.started":"2024-12-04T10:03:12.200518Z","shell.execute_reply":"2024-12-04T10:03:12.275293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split into train and validation\nX_train_split, X_val_split, y_train_split, y_val_split = train_test_split(X_pca, y_train, test_size=0.2, random_state=42, stratify=y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:12.277127Z","iopub.execute_input":"2024-12-04T10:03:12.277480Z","iopub.status.idle":"2024-12-04T10:03:12.290398Z","shell.execute_reply.started":"2024-12-04T10:03:12.277441Z","shell.execute_reply":"2024-12-04T10:03:12.289061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model Sheneningans\nfrom sklearn.model_selection import StratifiedKFold\n# Define classifiers\nmodels = {\n    'SVM': SVC(probability=True, decision_function_shape='ovo'),\n    'Random Forest': RandomForestClassifier(),\n    'XGBoost': XGBClassifier()\n}\n\n# Define hyperparameter grids\nparam_grids = {\n    'SVM': {\n        'C': [0.1, 1, 10],\n        'kernel': ['linear', 'rbf']\n    },\n    'Random Forest': {\n        'n_estimators': [50, 100, 200],\n        'max_depth': [None, 10, 20],\n        'class_weight': ['balanced']\n    },\n    'XGBoost': {\n        'n_estimators': [50, 100, 200],\n        'max_depth': [3, 5, 7],\n        'learning_rate': [0.01, 0.1, 0.2]\n    }\n}\nskf = StratifiedKFold(n_splits=5)\n# Create a function to perform GridSearchCV on each model\nbest_models = {}\nfor model_name in models:\n    model = models[model_name]\n    param_grid = param_grids[model_name]\n    \n    grid_search = GridSearchCV(model, param_grid, cv=skf, n_jobs=-1, verbose=1)\n    grid_search.fit(X_train_split, y_train_split)\n    \n    best_models[model_name] = grid_search.best_estimator_\n\n# Evaluate the best models\nfor model_name, model in best_models.items():\n    print(f\"Best Model: {model_name}\")\n    y_pred = model.predict(X_val_split)\n    print(f\"Accuracy: {accuracy_score(y_val_split, y_pred)}\")\n    print(f\"Classification Report:\\n{classification_report(y_val_split, y_pred, zero_division=0)}\")\n    print(\"-\" * 60)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:03:12.292798Z","iopub.execute_input":"2024-12-04T10:03:12.295656Z","iopub.status.idle":"2024-12-04T10:05:14.757189Z","shell.execute_reply.started":"2024-12-04T10:03:12.295608Z","shell.execute_reply":"2024-12-04T10:05:14.755979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluation\n# Use classification reports and accuracy scores to evaluate model performance on validation data.\n# Also, plot confusion matrices to understand the classification outcomes.\n\n## confusion matrix\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\ny_pred_svm = best_models['SVM'].predict(X_val_split)\n\ncm = confusion_matrix(y_val_split, y_pred_svm)\n\nplt.figure(figsize=(10, 8))  \nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=[0, 1, 2, 3], yticklabels=[0, 1, 2, 3], cbar=False)\nplt.xlabel('Predicted', fontsize=14)\nplt.ylabel('True', fontsize=14)\nplt.title('Confusion Matrix - SVM Model', fontsize=16)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:14.761325Z","iopub.execute_input":"2024-12-04T10:05:14.761716Z","iopub.status.idle":"2024-12-04T10:05:15.120483Z","shell.execute_reply.started":"2024-12-04T10:05:14.761671Z","shell.execute_reply":"2024-12-04T10:05:15.119293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Clases en los datos de validación:\")\nprint(np.unique(y_val_split))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:15.122084Z","iopub.execute_input":"2024-12-04T10:05:15.122581Z","iopub.status.idle":"2024-12-04T10:05:15.129303Z","shell.execute_reply.started":"2024-12-04T10:05:15.122529Z","shell.execute_reply":"2024-12-04T10:05:15.128136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate and visualize ROC curves and AUC for each class in the multiclass classification task.\n## ROC Curves and AUC\n\nfrom sklearn.metrics import roc_curve, auc\nfrom sklearn.preprocessing import label_binarize\nfrom sklearn.metrics import roc_auc_score\n\ny_val_bin = label_binarize(y_val_split, classes=[0, 1, 2, 3])\n\ny_pred_proba = best_models['SVM'].predict_proba(X_val_split)\n\nplt.figure(figsize=(10, 8))\n\nfor i in range(4):\n    fpr, tpr, _ = roc_curve(y_val_bin[:, i], y_pred_proba[:, i])\n    roc_auc = auc(fpr, tpr)\n    plt.plot(fpr, tpr, lw=2, label=f'Class {i} (AUC = {roc_auc:.2f})')\n\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\n\nplt.xlabel('False Positive Rate', fontsize=14)\nplt.ylabel('True Positive Rate', fontsize=14)\nplt.title('ROC Curve - SVM Model', fontsize=16)\nplt.legend(loc='lower right', fontsize=12)\nplt.tight_layout()\nplt.show()\n\nauc_scores = [roc_auc_score(y_val_bin[:, i], y_pred_proba[:, i]) for i in range(4)]\nprint(\"AUC Scores for each class:\")\nfor i, auc_score in enumerate(auc_scores):\n    print(f\"Class {i}: {auc_score:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:15.130648Z","iopub.execute_input":"2024-12-04T10:05:15.130974Z","iopub.status.idle":"2024-12-04T10:05:15.600301Z","shell.execute_reply.started":"2024-12-04T10:05:15.130928Z","shell.execute_reply":"2024-12-04T10:05:15.599136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Predictions for test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:05:15.601757Z","iopub.execute_input":"2024-12-04T10:05:15.602637Z","iopub.status.idle":"2024-12-04T10:05:15.607204Z","shell.execute_reply.started":"2024-12-04T10:05:15.602586Z","shell.execute_reply":"2024-12-04T10:05:15.606105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nif 'id' not in df_test.columns:\n    raise ValueError(\"Error\")\n\n# Preprocess test data\n# 1. Impute missing values using the fitted imputer\ndf_test[numerical_features] = numerical_imputer.transform(df_test[numerical_features])\n\n# 2. Scale numerical features using the fitted scaler\ndf_test_numerical = scaler.transform(df_test[numerical_features])\n\n# 3. Encode categorical features using the fitted encoder\ndf_test_categorical = encoder.transform(df_test[categorical_features].fillna('Missing'))\n\n# 4. Combine transformed features\ndf_test_transformed = np.hstack((df_test_numerical, df_test_categorical))\n\n# 5. Use PCA for dimensionality reduction (reuse fitted PCA)\ndf_test_pca = pca.transform(df_test_transformed)\n\n# Predict with the best model\ntest_predictions = best_models['SVM'].predict(df_test_pca)\n\n# Create the submission DataFrame\nsubmission = pd.DataFrame({\n    'id': df_test['id'],  # Identificador único de cada fila\n    'sii': test_predictions  # Predicciones del modelo\n})\n\n# Submission\n# Create a submission file to be uploaded as the final result.\noutput_path = 'submission.csv'\nsubmission.to_csv(output_path, index=False)\n\nprint(f\"Sudah disimpan dengan nama '{output_path}'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-04T10:06:04.895668Z","iopub.execute_input":"2024-12-04T10:06:04.896101Z","iopub.status.idle":"2024-12-04T10:06:04.926155Z","shell.execute_reply.started":"2024-12-04T10:06:04.896065Z","shell.execute_reply":"2024-12-04T10:06:04.924911Z"}},"outputs":[],"execution_count":null}]}