{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:28:49.594072Z","iopub.execute_input":"2024-12-05T08:28:49.595068Z","iopub.status.idle":"2024-12-05T08:28:54.229551Z","shell.execute_reply.started":"2024-12-05T08:28:49.595019Z","shell.execute_reply":"2024-12-05T08:28:54.228501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Import libraries\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier, VotingClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import f1_score\n\n# Step 2: Load the training dataset\ntrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\n# Step 3: Prepare the dataset\ntrain_data['t'] = (train_data['sii'] == 1).astype(int)  # Convert 'sii' to binary target variable\n\n# Selecting feature columns\nX = train_data[['Physical-BMI', 'Fitness_Endurance-Time_Mins', 'PreInt_EduHx-computerinternet_hoursday']]\ny = train_data['t']\n\n# Impute missing values\nimputer = SimpleImputer(strategy='mean')\nX_imputed = imputer.fit_transform(X)\n\n# Scale the features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X_imputed)\n\n# Split data for training and validation\nX_train, X_val, y_train, y_val = train_test_split(X_scaled, y, test_size=0.2, random_state=42)\n\n# Step 4: Initialize classifiers\nrf_model = RandomForestClassifier(random_state=42, n_estimators=100, max_features='sqrt')\nxgb_model = XGBClassifier(use_label_encoder=False, eval_metric='logloss')\nlgb_model = LGBMClassifier()\n\n# Step 5: Create Voting Classifier with multiple models\nvoting_model = VotingClassifier(\n    estimators=[\n        ('rf', rf_model), \n        ('xgb', xgb_model), \n        ('lgb', lgb_model)\n    ],\n    voting='soft'  # Use soft voting for better probability handling\n)\n\n# Step 6: Fit the Voting Classifier model\nvoting_model.fit(X_train, y_train)\n\n# Step 7: Validate the model\ny_pred = voting_model.predict(X_val)\n\n# Calculate the F1 Score\nf1 = f1_score(y_val, y_pred)\nprint(f\"Validation F1 Score: {f1:.4f}\")\n\n# Step 8: Load the test dataset\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n# Prepare test features using the same feature names\ntest_features = test_data[['Physical-BMI', 'Fitness_Endurance-Time_Mins', 'PreInt_EduHx-computerinternet_hoursday']]\ntest_features_imputed = imputer.transform(test_features)\ntest_features_scaled = scaler.transform(test_features_imputed)\n\n# Step 9: Make predictions on the test dataset\npredictions_classified = voting_model.predict(test_features_scaled)\n\n# Step 10: Save predictions to CSV\nsubmission = pd.DataFrame({'id': test_data['id'], 'sii': predictions_classified})\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\n\nprint(\"Submission file created successfully as 'submission.csv'.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:26:33.710405Z","iopub.execute_input":"2024-12-05T09:26:33.710833Z","iopub.status.idle":"2024-12-05T09:26:36.481311Z","shell.execute_reply.started":"2024-12-05T09:26:33.710796Z","shell.execute_reply":"2024-12-05T09:26:36.479919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}