{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T11:50:06.544011Z","iopub.execute_input":"2024-12-05T11:50:06.544334Z","iopub.status.idle":"2024-12-05T11:50:08.252165Z","shell.execute_reply.started":"2024-12-05T11:50:06.544302Z","shell.execute_reply":"2024-12-05T11:50:08.250792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, confusion_matrix\nfrom sklearn.ensemble import RandomForestClassifier\nimport lightgbm as lgb\nfrom catboost import CatBoostClassifier\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Input, Dense, Dropout\nfrom imblearn.over_sampling import SMOTE\n\n# Load data\ntrain_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntest_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\n\ntrain_data = pd.read_csv(train_path)\ntest_data = pd.read_csv(test_path)\n\n# Define target and selected features\ntarget_column = 'sii'\nselected_features = [\n    'Basic_Demos-Age',\n    'Basic_Demos-Sex',\n    'Physical-BMI',\n    'Physical-Height',\n    'Physical-Weight',\n    'PreInt_EduHx-computerinternet_hoursday'\n]\n\n# Check if all features are available in the dataset\nmissing_features = [col for col in selected_features if col not in train_data.columns]\nif missing_features:\n    raise ValueError(f\"Columns not found in dataset: {missing_features}\")\n\n# Drop rows where target is missing\ntrain_data = train_data.dropna(subset=[target_column])\n\n# Handle missing values in selected features\ntrain_data[selected_features] = train_data[selected_features].fillna(train_data[selected_features].median())\ntest_data[selected_features] = test_data[selected_features].fillna(train_data[selected_features].median())\n\n# Split features and target\nX = train_data[selected_features]\ny = train_data[target_column]\nX_test = test_data[selected_features]\n\n# Standardize features\nscaler = StandardScaler()\nX = scaler.fit_transform(X)\nX_test = scaler.transform(X_test)\n\n# Handle class imbalance using SMOTE\nsmote = SMOTE(random_state=42)\nX, y = smote.fit_resample(X, y)\n\n# Split data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Hyperparameters for models\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.01,\n    'force_col_wise': True\n}\n\nRF_Params = {\n    'n_estimators': 200,\n    'max_depth': 6,\n    'max_features': 0.8,\n    'min_samples_split': 2,\n    'min_samples_leaf': 1,\n    'bootstrap': True,\n    'random_state': 42\n}\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': 42,\n    'verbose': 0,\n    'l2_leaf_reg': 10,\n    'task_type': 'CPU'\n}\n\n# Model Training with LightGBM\nlgb_model = lgb.LGBMClassifier(**Params)\nlgb_model.fit(X_train, y_train)\nlgb_val_pred = lgb_model.predict(X_val)\nlgb_val_qwk = cohen_kappa_score(y_val, lgb_val_pred, weights=\"quadratic\")\n\n# Model Training with Random Forest\nrf_model = RandomForestClassifier(**RF_Params)\nrf_model.fit(X_train, y_train)\nrf_val_pred = rf_model.predict(X_val)\nrf_val_qwk = cohen_kappa_score(y_val, rf_val_pred, weights=\"quadratic\")\n\n# Model Training with CatBoost\ncatboost_model = CatBoostClassifier(**CatBoost_Params)\ncatboost_model.fit(X_train, y_train)\ncatboost_val_pred = catboost_model.predict(X_val)\ncatboost_val_qwk = cohen_kappa_score(y_val, catboost_val_pred, weights=\"quadratic\")\n\n# Model Training with Neural Network\nnn_model = Sequential([\n    Input(shape=(X_train.shape[1],)),\n    Dense(128, activation='relu'),\n    Dropout(0.3),\n    Dense(64, activation='relu'),\n    Dropout(0.3),\n    Dense(len(np.unique(y_train)), activation='softmax')  # Multi-class output\n])\n\nnn_model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\nnn_model.fit(X_train, y_train, epochs=50, batch_size=32, validation_data=(X_val, y_val))\n\n# Evaluate the Neural Network\nnn_val_pred = np.argmax(nn_model.predict(X_val), axis=1)\nnn_val_qwk = cohen_kappa_score(y_val, nn_val_pred, weights=\"quadratic\")\n\n# Print QWK scores for each model\nprint(f\"LightGBM Validation QWK Score: {lgb_val_qwk:.4f}\")\nprint(f\"Random Forest Validation QWK Score: {rf_val_qwk:.4f}\")\nprint(f\"CatBoost Validation QWK Score: {catboost_val_qwk:.4f}\")\nprint(f\"Neural Network Validation QWK Score: {nn_val_qwk:.4f}\")\n\n# Choose the best model based on QWK\nbest_model = max([(lgb_model, lgb_val_qwk),\n                  (rf_model, rf_val_qwk),\n                  (catboost_model, catboost_val_qwk),\n                  (nn_model, nn_val_qwk)], key=lambda x: x[1])\n\nprint(f\"\\nBest Model: {best_model[0]}\")\nprint(f\"Best Model Validation QWK Score: {best_model[1]:.4f}\")\n\n# Use the best model to predict on test data\nif best_model[0] == lgb_model:\n    predictions = lgb_model.predict(X_test)\nelif best_model[0] == rf_model:\n    predictions = rf_model.predict(X_test)\nelif best_model[0] == catboost_model:\n    predictions = catboost_model.predict(X_test)\nelse:\n    predictions = np.argmax(nn_model.predict(X_test), axis=1)\n\n# Save predictions to CSV\nsubmission = pd.DataFrame({'id': test_data['id'], 'sii': predictions.flatten()})\nsubmission.to_csv('submission.csv', index=False)\n\n# Analysis of Predictions on Test Data\nseverity_counts = pd.Series(predictions.flatten()).value_counts()\nseverity_levels = ['None', 'Mild', 'Moderate', 'Severe']\npredicted_severity = dict(zip(severity_levels[:len(severity_counts)], severity_counts))\n\nprint(\"\\nPredicted Severity Levels Breakdown:\")\nfor severity, count in predicted_severity.items():\n    print(f\"{severity}: {count}\")\n\nprint(\"\\nFinal Results and Performance Metrics:\")\nprint(f\"LightGBM Validation QWK Score: {lgb_val_qwk:.4f}\")\nprint(f\"Random Forest Validation QWK Score: {rf_val_qwk:.4f}\")\nprint(f\"CatBoost Validation QWK Score: {catboost_val_qwk:.4f}\")\nprint(f\"Neural Network Validation QWK Score: {nn_val_qwk:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T11:50:08.253741Z","iopub.execute_input":"2024-12-05T11:50:08.254244Z","iopub.status.idle":"2024-12-05T11:50:42.085039Z","shell.execute_reply.started":"2024-12-05T11:50:08.254208Z","shell.execute_reply":"2024-12-05T11:50:42.083649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}