{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:06:18.211214Z","iopub.execute_input":"2024-12-05T10:06:18.211636Z","iopub.status.idle":"2024-12-05T10:06:19.351429Z","shell.execute_reply.started":"2024-12-05T10:06:18.211598Z","shell.execute_reply":"2024-12-05T10:06:19.350147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import library yang diperlukan\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV, cross_val_score\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import GradientBoostingRegressor, RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom scipy.stats import uniform, randint\nfrom sklearn.ensemble import VotingRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:06:19.353528Z","iopub.execute_input":"2024-12-05T10:06:19.353907Z","iopub.status.idle":"2024-12-05T10:06:19.360429Z","shell.execute_reply.started":"2024-12-05T10:06:19.353872Z","shell.execute_reply":"2024-12-05T10:06:19.359253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load dataset\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:06:19.361767Z","iopub.execute_input":"2024-12-05T10:06:19.362158Z","iopub.status.idle":"2024-12-05T10:06:19.425325Z","shell.execute_reply.started":"2024-12-05T10:06:19.362124Z","shell.execute_reply":"2024-12-05T10:06:19.424232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hapus baris dengan nilai NaN di target\ntrain_df = train_df.dropna(subset=['sii'])\n\n# Pisahkan fitur dan target\nX = train_df.drop(columns=['id', 'sii'])\ny = train_df['sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:06:19.428089Z","iopub.execute_input":"2024-12-05T10:06:19.428888Z","iopub.status.idle":"2024-12-05T10:06:19.439232Z","shell.execute_reply.started":"2024-12-05T10:06:19.428836Z","shell.execute_reply":"2024-12-05T10:06:19.438245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cari kolom yang ada di train tetapi tidak ada di test\nmissing_cols_in_test = set(X.columns) - set(test_df.columns)\nprint(f\"Kolom di train tetapi tidak ada di test: {missing_cols_in_test}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:06:19.440531Z","iopub.execute_input":"2024-12-05T10:06:19.440834Z","iopub.status.idle":"2024-12-05T10:06:19.448764Z","shell.execute_reply.started":"2024-12-05T10:06:19.440804Z","shell.execute_reply":"2024-12-05T10:06:19.447817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hapus kolom yang tidak ada di test dari data train\nX = X.drop(columns=missing_cols_in_test)\n\n# Identifikasi kolom numerik dan kategorikal\nnumeric_cols = X.select_dtypes(include=['int64', 'float64']).columns\ncategorical_cols = X.select_dtypes(include=['object']).columns\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:06:19.449886Z","iopub.execute_input":"2024-12-05T10:06:19.450218Z","iopub.status.idle":"2024-12-05T10:06:19.468197Z","shell.execute_reply.started":"2024-12-05T10:06:19.450156Z","shell.execute_reply":"2024-12-05T10:06:19.467189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pipeline untuk kolom numerik\nnumeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    ('scaler', StandardScaler())\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:06:19.469405Z","iopub.execute_input":"2024-12-05T10:06:19.469777Z","iopub.status.idle":"2024-12-05T10:06:19.480131Z","shell.execute_reply.started":"2024-12-05T10:06:19.469741Z","shell.execute_reply":"2024-12-05T10:06:19.478912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pipeline untuk kolom kategorikal\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:06:19.481506Z","iopub.execute_input":"2024-12-05T10:06:19.481861Z","iopub.status.idle":"2024-12-05T10:06:19.492886Z","shell.execute_reply.started":"2024-12-05T10:06:19.481825Z","shell.execute_reply":"2024-12-05T10:06:19.491623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Gabungkan pipeline dengan ColumnTransformer\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, numeric_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:07:12.431798Z","iopub.execute_input":"2024-12-05T10:07:12.432300Z","iopub.status.idle":"2024-12-05T10:07:12.444078Z","shell.execute_reply.started":"2024-12-05T10:07:12.432252Z","shell.execute_reply":"2024-12-05T10:07:12.442934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model utama\ngb_model = GradientBoostingRegressor(random_state=42)\nrf_model = RandomForestRegressor(random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:07:20.574277Z","iopub.execute_input":"2024-12-05T10:07:20.574697Z","iopub.status.idle":"2024-12-05T10:07:20.579967Z","shell.execute_reply.started":"2024-12-05T10:07:20.574659Z","shell.execute_reply":"2024-12-05T10:07:20.578773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pipeline utama\ngb_pipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('regressor', gb_model)\n])\n\nrf_pipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('regressor', rf_model)\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:07:34.458616Z","iopub.execute_input":"2024-12-05T10:07:34.459034Z","iopub.status.idle":"2024-12-05T10:07:34.464633Z","shell.execute_reply.started":"2024-12-05T10:07:34.458996Z","shell.execute_reply":"2024-12-05T10:07:34.463422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split data menjadi training dan validation\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:07:46.558324Z","iopub.execute_input":"2024-12-05T10:07:46.559315Z","iopub.status.idle":"2024-12-05T10:07:46.567405Z","shell.execute_reply.started":"2024-12-05T10:07:46.559263Z","shell.execute_reply":"2024-12-05T10:07:46.566385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hyperparameter untuk RandomizedSearchCV\nparam_dist = {\n    'regressor__n_estimators': randint(50, 300),\n    'regressor__learning_rate': uniform(0.01, 0.2),\n    'regressor__max_depth': randint(3, 10),\n    'regressor__subsample': uniform(0.6, 0.4)\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:07:58.663141Z","iopub.execute_input":"2024-12-05T10:07:58.663968Z","iopub.status.idle":"2024-12-05T10:07:58.672100Z","shell.execute_reply.started":"2024-12-05T10:07:58.663928Z","shell.execute_reply":"2024-12-05T10:07:58.670837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# RandomizedSearchCV untuk Gradient Boosting\ngb_search = RandomizedSearchCV(\n    gb_pipeline,\n    param_distributions=param_dist,\n    n_iter=30,\n    cv=5,\n    scoring='neg_mean_squared_error',\n    n_jobs=-1,\n    verbose=2,\n    random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:08:06.902367Z","iopub.execute_input":"2024-12-05T10:08:06.902748Z","iopub.status.idle":"2024-12-05T10:08:06.908125Z","shell.execute_reply.started":"2024-12-05T10:08:06.902715Z","shell.execute_reply":"2024-12-05T10:08:06.907049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Melatih model Gradient Boosting\ngb_search.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:10:57.714326Z","iopub.execute_input":"2024-12-05T10:10:57.714738Z","iopub.status.idle":"2024-12-05T10:14:46.333015Z","shell.execute_reply.started":"2024-12-05T10:10:57.714699Z","shell.execute_reply":"2024-12-05T10:14:46.331866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hasil terbaik dari RandomizedSearch\nprint(f\"Best parameters (Gradient Boosting): {gb_search.best_params_}\")\nprint(f\"Best score (Gradient Boosting): {np.sqrt(-gb_search.best_score_)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:16:13.815345Z","iopub.execute_input":"2024-12-05T10:16:13.815750Z","iopub.status.idle":"2024-12-05T10:16:13.821981Z","shell.execute_reply.started":"2024-12-05T10:16:13.815715Z","shell.execute_reply":"2024-12-05T10:16:13.820774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluasi model terbaik pada data validasi\nbest_gb_model = gb_search.best_estimator_\ny_pred_gb = best_gb_model.predict(X_val)\ngb_mse = mean_squared_error(y_val, y_pred_gb)\nprint(f'Gradient Boosting Validation MSE: {gb_mse}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:16:25.774477Z","iopub.execute_input":"2024-12-05T10:16:25.774864Z","iopub.status.idle":"2024-12-05T10:16:25.797672Z","shell.execute_reply.started":"2024-12-05T10:16:25.774830Z","shell.execute_reply":"2024-12-05T10:16:25.795993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Gabungkan model (Ensemble)\nvoting_model = VotingRegressor(\n    estimators=[('gb', best_gb_model), ('rf', rf_pipeline.fit(X_train, y_train))]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:16:33.536150Z","iopub.execute_input":"2024-12-05T10:16:33.537314Z","iopub.status.idle":"2024-12-05T10:16:37.858158Z","shell.execute_reply.started":"2024-12-05T10:16:33.537262Z","shell.execute_reply":"2024-12-05T10:16:37.857146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluasi Ensemble Model\nvoting_model.fit(X_train, y_train)\ny_pred_ensemble = voting_model.predict(X_val)\nensemble_mse = mean_squared_error(y_val, y_pred_ensemble)\nprint(f'Ensemble Validation MSE: {ensemble_mse}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:16:41.378377Z","iopub.execute_input":"2024-12-05T10:16:41.378767Z","iopub.status.idle":"2024-12-05T10:16:47.888065Z","shell.execute_reply.started":"2024-12-05T10:16:41.378731Z","shell.execute_reply":"2024-12-05T10:16:47.887020Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prediksi pada data test\nX_test = test_df.drop(columns=['id'])\nX_test = X_test.reindex(columns=X.columns, fill_value=np.nan)\ntest_predictions = voting_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:16:51.910061Z","iopub.execute_input":"2024-12-05T10:16:51.910918Z","iopub.status.idle":"2024-12-05T10:16:51.939287Z","shell.execute_reply.started":"2024-12-05T10:16:51.910873Z","shell.execute_reply":"2024-12-05T10:16:51.937894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Simpan file submission\ntest_df['sii'] = test_predictions.round().astype(int)\nsubmission = test_df[['id', 'sii']]\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file saved as submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T10:16:59.100221Z","iopub.execute_input":"2024-12-05T10:16:59.100658Z","iopub.status.idle":"2024-12-05T10:16:59.110412Z","shell.execute_reply.started":"2024-12-05T10:16:59.100602Z","shell.execute_reply":"2024-12-05T10:16:59.109231Z"}},"outputs":[],"execution_count":null}]}