{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n #   for filename in filenames:\n  #      print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nimport pandas as pd \nimport optuna\nimport numpy as np \nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport seaborn as sns \nimport re\nimport math\nfrom io import StringIO\nfrom colorama import Fore, Style, init;\nfrom IPython.display import display, HTML\nfrom scipy.stats import skew  \nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import VotingClassifier, VotingRegressor\nfrom sklearn.model_selection import KFold, RepeatedStratifiedKFold, cross_val_score\nfrom sklearn.model_selection import RandomizedSearchCV\nfrom sklearn.metrics import cohen_kappa_score\nfrom tqdm import tqdm\nfrom functools import partial\nfrom lightgbm import LGBMRegressor\nfrom sklearn.preprocessing import LabelEncoder, MinMaxScaler , StandardScaler , QuantileTransformer, PowerTransformer\nimport xgboost as xgb\nimport lightgbm as lgb\nfrom sklearn.model_selection import cross_val_score, StratifiedKFold\nfrom catboost import CatBoostClassifier\nfrom sklearn.metrics import *\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier, ExtraTreesClassifier\nfrom xgboost import XGBClassifier\npd.set_option('display.max_columns', None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-23T17:40:36.985843Z","iopub.execute_input":"2024-09-23T17:40:36.986308Z","iopub.status.idle":"2024-09-23T17:40:37.003019Z","shell.execute_reply.started":"2024-09-23T17:40:36.986260Z","shell.execute_reply":"2024-09-23T17:40:37.001190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load training and testing datasets\ntrain_ds = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_ds = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsubmission = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:37.006213Z","iopub.execute_input":"2024-09-23T17:40:37.008252Z","iopub.status.idle":"2024-09-23T17:40:37.076621Z","shell.execute_reply.started":"2024-09-23T17:40:37.008179Z","shell.execute_reply":"2024-09-23T17:40:37.075347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:37.078009Z","iopub.execute_input":"2024-09-23T17:40:37.078394Z","iopub.status.idle":"2024-09-23T17:40:37.183711Z","shell.execute_reply.started":"2024-09-23T17:40:37.078354Z","shell.execute_reply":"2024-09-23T17:40:37.182477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:37.185109Z","iopub.execute_input":"2024-09-23T17:40:37.185505Z","iopub.status.idle":"2024-09-23T17:40:37.204308Z","shell.execute_reply.started":"2024-09-23T17:40:37.185452Z","shell.execute_reply":"2024-09-23T17:40:37.202740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:37.208384Z","iopub.execute_input":"2024-09-23T17:40:37.208832Z","iopub.status.idle":"2024-09-23T17:40:37.224266Z","shell.execute_reply.started":"2024-09-23T17:40:37.208787Z","shell.execute_reply":"2024-09-23T17:40:37.223117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds.describe()\ntest_ds.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:37.225459Z","iopub.execute_input":"2024-09-23T17:40:37.225848Z","iopub.status.idle":"2024-09-23T17:40:37.353815Z","shell.execute_reply.started":"2024-09-23T17:40:37.225807Z","shell.execute_reply":"2024-09-23T17:40:37.352328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handle missing values\nprint(train_ds.dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:37.355700Z","iopub.execute_input":"2024-09-23T17:40:37.357786Z","iopub.status.idle":"2024-09-23T17:40:37.366339Z","shell.execute_reply.started":"2024-09-23T17:40:37.357701Z","shell.execute_reply":"2024-09-23T17:40:37.365073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for column in test_ds.select_dtypes(include=['number']):\n # plt.figure()\n  #sns.histplot(test_ds[column])\n  #plt.title(f'Distribution of {column}')\n  #plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:37.368125Z","iopub.execute_input":"2024-09-23T17:40:37.368661Z","iopub.status.idle":"2024-09-23T17:40:37.379396Z","shell.execute_reply.started":"2024-09-23T17:40:37.368582Z","shell.execute_reply":"2024-09-23T17:40:37.378057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate features and target variable\nX = train_ds.drop('sii', axis=1)\ny = train_ds['sii']\n\n# Identify categorical and numerical columns\ncategorical_cols = X.select_dtypes(include=['object']).columns\nnumerical_cols = X.select_dtypes(exclude=['object']).columns\n\n# Handle 'PCIAT-Season' (if present)\nif 'PCIAT-Season' in categorical_cols:\n  categorical_cols = categorical_cols.drop('PCIAT-Season')\n  X = X.drop('PCIAT-Season', axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:37.381168Z","iopub.execute_input":"2024-09-23T17:40:37.381754Z","iopub.status.idle":"2024-09-23T17:40:37.400266Z","shell.execute_reply.started":"2024-09-23T17:40:37.381694Z","shell.execute_reply":"2024-09-23T17:40:37.399224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\n# Apply one-hot encoding to categorical features\nencoder = OneHotEncoder(handle_unknown='ignore', sparse_output=False)\nencoded_features = encoder.fit_transform(X[categorical_cols])\nencoded_df = pd.DataFrame(encoded_features)\nencoded_df.columns = encoder.get_feature_names_out(categorical_cols)\n\n# Concatenate numerical and encoded features\nX = pd.concat([X[numerical_cols], encoded_df], axis=1)\n\n# --- Impute missing values in features ---\nimputer = SimpleImputer(strategy='mean')\nX = imputer.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:37.401820Z","iopub.execute_input":"2024-09-23T17:40:37.402441Z","iopub.status.idle":"2024-09-23T17:40:38.373137Z","shell.execute_reply.started":"2024-09-23T17:40:37.402382Z","shell.execute_reply":"2024-09-23T17:40:38.371929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# --- Impute missing values in target variable (if any) ---\n# Use a SimpleImputer with strategy='most_frequent' for categorical target\nimputer_y = SimpleImputer(strategy='most_frequent')  \ny = imputer_y.fit_transform(y.values.reshape(-1, 1))\ny = y.ravel() # Convert back to 1D array","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:38.374845Z","iopub.execute_input":"2024-09-23T17:40:38.375251Z","iopub.status.idle":"2024-09-23T17:40:38.384729Z","shell.execute_reply.started":"2024-09-23T17:40:38.375211Z","shell.execute_reply":"2024-09-23T17:40:38.383305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# --- Scale the features ---\nscaler = StandardScaler()\nX = scaler.fit_transform(X)\n\n# Now, X is a feature matrix ready for machine learning model training\n# and y is the target variable.\n\n# --- Example: Split into train and validation sets ---\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:38.386434Z","iopub.execute_input":"2024-09-23T17:40:38.386961Z","iopub.status.idle":"2024-09-23T17:40:38.857824Z","shell.execute_reply.started":"2024-09-23T17:40:38.386903Z","shell.execute_reply":"2024-09-23T17:40:38.856400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n\n# Initialize models\nmodels = {\n    \"Logistic Regression\": LogisticRegression(),\n    \"Decision Tree\": DecisionTreeClassifier(),\n    \"Random Forest\": RandomForestClassifier(),\n    \"Gradient Boosting\": GradientBoostingClassifier(),\n    \"Support Vector Machine\": SVC(),\n    \"K-Nearest Neighbors\": KNeighborsClassifier()\n}\n\n# Train and evaluate models\nfor model_name, model in models.items():\n  model.fit(X_train, y_train)\n  y_pred = model.predict(X_val)\n\n  accuracy = accuracy_score(y_val, y_pred)\n  precision = precision_score(y_val, y_pred, average='weighted')\n  recall = recall_score(y_val, y_pred, average='weighted')\n  f1 = f1_score(y_val, y_pred, average='weighted')\n\n  print(f\"{model_name}:\")\n  print(f\"  Accuracy: {accuracy:.4f}\")\n  print(f\"  Precision: {precision:.4f}\")\n  print(f\"  Recall: {recall:.4f}\")\n  print(f\"  F1-Score: {f1:.4f}\")\n  print(\"-\" * 20)\n\n# You can choose the best model based on the evaluation metrics\n# and then train it on the full training data (X, y) and make predictions on the test data.","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:40:38.859384Z","iopub.execute_input":"2024-09-23T17:40:38.859757Z","iopub.status.idle":"2024-09-23T17:43:34.852891Z","shell.execute_reply.started":"2024-09-23T17:40:38.859718Z","shell.execute_reply":"2024-09-23T17:43:34.851401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Load the test data\n\nX_test = test_ds.copy()\nids = test_ds['id']  # Assuming 'id' is the column name for IDs\n\n# Identify categorical and numerical columns in test data\ncategorical_cols_test = X_test.select_dtypes(include=['object']).columns\nnumerical_cols_test = X_test.select_dtypes(exclude=['object']).columns\n\n# Handle 'PCIAT-Season' (if present)\nif 'PCIAT-Season' in categorical_cols_test:\n    categorical_cols_test = categorical_cols_test.drop('PCIAT-Season')\n    X_test = X_test.drop('PCIAT-Season', axis=1)\n\n# Apply one-hot encoding to categorical features in test data using the same encoder as training\nencoded_features_test = encoder.transform(X_test[categorical_cols_test])\nencoded_df_test = pd.DataFrame(encoded_features_test)\nencoded_df_test.columns = encoder.get_feature_names_out(categorical_cols_test)\n\n# Concatenate numerical and encoded features in test data\nX_test = pd.concat([X_test[numerical_cols_test], encoded_df_test], axis=1)\n\n# Align columns between training and test sets to ensure they have the same features\n# Convert X back to DataFrame for reindexing\nX = pd.DataFrame(X) # Convert X back to a DataFrame\nX_test = X_test.reindex(columns=X.columns, fill_value=0)\n\n\n# --- Impute missing values in test data ---\nX_test = imputer.transform(X_test)\n\n# --- Scale the test data ---\nX_test = scaler.transform(X_test)\n\n# Now, X_test is ready to use for making predictions\n\n# Make predictions using the best model \nbest_model = models['Random Forest']  \ny_pred_test = best_model.predict(X_test)\n\n# Create a DataFrame with the predictions and IDs\nsubmission_df = pd.DataFrame({'id': ids, 'sii': y_pred_test})\n\n# Save the predictions to a CSV file\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-23T17:59:11.434262Z","iopub.execute_input":"2024-09-23T17:59:11.434772Z","iopub.status.idle":"2024-09-23T17:59:11.539526Z","shell.execute_reply.started":"2024-09-23T17:59:11.434724Z","shell.execute_reply":"2024-09-23T17:59:11.538196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}