{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:06.425983Z","iopub.execute_input":"2024-12-05T01:37:06.426393Z","iopub.status.idle":"2024-12-05T01:37:07.526399Z","shell.execute_reply.started":"2024-12-05T01:37:06.426358Z","shell.execute_reply":"2024-12-05T01:37:07.525341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\nfrom sklearn.metrics import classification_report, confusion_matrix","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.528034Z","iopub.execute_input":"2024-12-05T01:37:07.528997Z","iopub.status.idle":"2024-12-05T01:37:07.534814Z","shell.execute_reply.started":"2024-12-05T01:37:07.528959Z","shell.execute_reply":"2024-12-05T01:37:07.533458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the CSV data\ntrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.536049Z","iopub.execute_input":"2024-12-05T01:37:07.536396Z","iopub.status.idle":"2024-12-05T01:37:07.595993Z","shell.execute_reply.started":"2024-12-05T01:37:07.536365Z","shell.execute_reply":"2024-12-05T01:37:07.595012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.598092Z","iopub.execute_input":"2024-12-05T01:37:07.598415Z","iopub.status.idle":"2024-12-05T01:37:07.622720Z","shell.execute_reply.started":"2024-12-05T01:37:07.598385Z","shell.execute_reply":"2024-12-05T01:37:07.621715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.623943Z","iopub.execute_input":"2024-12-05T01:37:07.624244Z","iopub.status.idle":"2024-12-05T01:37:07.650047Z","shell.execute_reply.started":"2024-12-05T01:37:07.624215Z","shell.execute_reply":"2024-12-05T01:37:07.648976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handle missing values and categorical columns (same as before)\nthreshold = 0.5 * len(train_data)\ncolumns_with_data = train_data.columns[train_data.isnull().sum() < threshold]\ntrain_data = train_data[columns_with_data]\ntrain_data = train_data.fillna(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.651348Z","iopub.execute_input":"2024-12-05T01:37:07.651756Z","iopub.status.idle":"2024-12-05T01:37:07.669108Z","shell.execute_reply.started":"2024-12-05T01:37:07.651713Z","shell.execute_reply":"2024-12-05T01:37:07.667883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove rows with NaN in the target column\ntarget_column = 'sii'\ntrain_data_cleaned = train_data.dropna(subset=[target_column])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.670709Z","iopub.execute_input":"2024-12-05T01:37:07.671681Z","iopub.status.idle":"2024-12-05T01:37:07.678795Z","shell.execute_reply.started":"2024-12-05T01:37:07.671607Z","shell.execute_reply":"2024-12-05T01:37:07.677802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handle categorical columns\nseason_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'FGC-Season', 'BIA-Season', \n               'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\nseason_mapping = {'Spring': 0, 'Summer': 1, 'Fall': 2, 'Winter': 3}\nfor col in season_cols:\n    if col in train_data_cleaned.columns:\n        train_data_cleaned[col] = train_data_cleaned[col].replace(season_mapping)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.680069Z","iopub.execute_input":"2024-12-05T01:37:07.680411Z","iopub.status.idle":"2024-12-05T01:37:07.712637Z","shell.execute_reply.started":"2024-12-05T01:37:07.680380Z","shell.execute_reply":"2024-12-05T01:37:07.711551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare the feature matrix X and target vector y\ncommon_columns = train_data_cleaned.columns.intersection(test_data.columns)\nX = train_data_cleaned[common_columns].drop(columns=['id'])  # Drop 'id' and 'sii' for features\ny = train_data_cleaned['sii']  # Target is 'sii'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.713927Z","iopub.execute_input":"2024-12-05T01:37:07.714212Z","iopub.status.idle":"2024-12-05T01:37:07.722715Z","shell.execute_reply.started":"2024-12-05T01:37:07.714184Z","shell.execute_reply":"2024-12-05T01:37:07.721493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.726119Z","iopub.execute_input":"2024-12-05T01:37:07.726428Z","iopub.status.idle":"2024-12-05T01:37:07.734879Z","shell.execute_reply.started":"2024-12-05T01:37:07.726398Z","shell.execute_reply":"2024-12-05T01:37:07.733824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify numerical columns in the dataset\nnumerical_columns = X.select_dtypes(include=['float64', 'int64']).columns\n\n# Select only the numeric columns for scaling\nX_train_numerical = X_train[numerical_columns]\nX_test_numerical = X_test[numerical_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.736154Z","iopub.execute_input":"2024-12-05T01:37:07.736979Z","iopub.status.idle":"2024-12-05T01:37:07.747767Z","shell.execute_reply.started":"2024-12-05T01:37:07.736922Z","shell.execute_reply":"2024-12-05T01:37:07.746569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Select only the numeric columns for scaling\nX_train_numerical = X_train[numerical_columns]\nX_test_numerical = X_test[numerical_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.749263Z","iopub.execute_input":"2024-12-05T01:37:07.749733Z","iopub.status.idle":"2024-12-05T01:37:07.764170Z","shell.execute_reply.started":"2024-12-05T01:37:07.749687Z","shell.execute_reply":"2024-12-05T01:37:07.763098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature Scaling\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train_numerical)\nX_test_scaled = scaler.transform(X_test_numerical)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.765232Z","iopub.execute_input":"2024-12-05T01:37:07.765561Z","iopub.status.idle":"2024-12-05T01:37:07.784756Z","shell.execute_reply.started":"2024-12-05T01:37:07.765518Z","shell.execute_reply":"2024-12-05T01:37:07.783427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply PCA\npca = PCA(n_components=0.95)  # Keep 95% of the variance\nX_train_pca = pca.fit_transform(X_train)\nX_test_pca = pca.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.786142Z","iopub.execute_input":"2024-12-05T01:37:07.786832Z","iopub.status.idle":"2024-12-05T01:37:07.812977Z","shell.execute_reply.started":"2024-12-05T01:37:07.786798Z","shell.execute_reply":"2024-12-05T01:37:07.811029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Sequential()\n\n# Add input layer\nmodel.add(Dense(128, activation='relu', input_dim=X_train_pca.shape[1]))  # Larger input layer\n\n# Add hidden layers with more neurons and Batch Normalization\nmodel.add(Dense(64, activation='relu'))\nmodel.add(BatchNormalization())  # Adding BatchNormalization layer\nmodel.add(Dropout(0.3))  # Dropout to prevent overfitting\n\nmodel.add(Dense(32, activation='relu'))\nmodel.add(BatchNormalization())  # Adding BatchNormalization layer\nmodel.add(Dropout(0.3))  # Dropout to prevent overfitting\n\n# Output layer (binary classification: use sigmoid for binary output)\nmodel.add(Dense(1, activation='sigmoid'))  # For binary classificati","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.814122Z","iopub.execute_input":"2024-12-05T01:37:07.814535Z","iopub.status.idle":"2024-12-05T01:37:07.985910Z","shell.execute_reply.started":"2024-12-05T01:37:07.814490Z","shell.execute_reply":"2024-12-05T01:37:07.984932Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compile the model\nmodel.compile(optimizer=Adam(learning_rate=0.001), \n              loss='binary_crossentropy',  # Binary classification loss\n              metrics=['accuracy'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.987284Z","iopub.execute_input":"2024-12-05T01:37:07.987754Z","iopub.status.idle":"2024-12-05T01:37:07.998575Z","shell.execute_reply.started":"2024-12-05T01:37:07.987708Z","shell.execute_reply":"2024-12-05T01:37:07.997514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use ReduceLROnPlateau to reduce learning rate when the validation loss plateaus\nlr_scheduler = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=5, min_lr=1e-6)\n\n# Train the model\nhistory = model.fit(X_train_pca, y_train, epochs=50, batch_size=32, validation_data=(X_test_pca, y_test), callbacks=[lr_scheduler])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:07.999995Z","iopub.execute_input":"2024-12-05T01:37:08.000441Z","iopub.status.idle":"2024-12-05T01:37:22.972506Z","shell.execute_reply.started":"2024-12-05T01:37:08.000393Z","shell.execute_reply":"2024-12-05T01:37:22.971579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate the model\ny_pred = (model.predict(X_test_pca) > 0.5).astype(\"int32\")  # Convert probabilities to 0 or 1 for classification\n\n# Classification Report","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:22.974479Z","iopub.execute_input":"2024-12-05T01:37:22.974965Z","iopub.status.idle":"2024-12-05T01:37:23.227574Z","shell.execute_reply.started":"2024-12-05T01:37:22.974915Z","shell.execute_reply":"2024-12-05T01:37:23.226695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Classification Report\nprint(\"Classification Report:\")\nprint(classification_report(y_test, y_pred))\n\n# Confusion Matrix\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_test, y_pred))\n\n# Accuracy\naccuracy = model.evaluate(X_test_pca, y_test)\nprint(f\"Model Accuracy: {accuracy[1]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:23.229001Z","iopub.execute_input":"2024-12-05T01:37:23.229359Z","iopub.status.idle":"2024-12-05T01:37:23.352000Z","shell.execute_reply.started":"2024-12-05T01:37:23.229325Z","shell.execute_reply":"2024-12-05T01:37:23.350925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot training and validation loss\nplt.plot(history.history['loss'], label='train loss')\nplt.plot(history.history['val_loss'], label='val loss')\nplt.title('Model Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:23.353297Z","iopub.execute_input":"2024-12-05T01:37:23.353586Z","iopub.status.idle":"2024-12-05T01:37:23.627363Z","shell.execute_reply.started":"2024-12-05T01:37:23.353558Z","shell.execute_reply":"2024-12-05T01:37:23.626322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot training and validation accuracy\nplt.plot(history.history['accuracy'], label='train accuracy')\nplt.plot(history.history['val_accuracy'], label='val accuracy')\nplt.title('Model Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:23.628591Z","iopub.execute_input":"2024-12-05T01:37:23.628970Z","iopub.status.idle":"2024-12-05T01:37:23.902696Z","shell.execute_reply.started":"2024-12-05T01:37:23.628935Z","shell.execute_reply":"2024-12-05T01:37:23.901723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess the test data (same steps as the training data)\nfor col in season_cols:\n    if col in test_data.columns:\n        test_data[col] = test_data[col].replace(season_mapping)\n\ntest_data.fillna(0, inplace=True)\nX_test_data = test_data[common_columns].drop(columns=['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:23.904183Z","iopub.execute_input":"2024-12-05T01:37:23.904618Z","iopub.status.idle":"2024-12-05T01:37:23.921769Z","shell.execute_reply.started":"2024-12-05T01:37:23.904571Z","shell.execute_reply":"2024-12-05T01:37:23.920580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Scale and apply PCA to test data\nX_test_scaled = scaler.transform(X_test_data)\nX_test_pca = pca.transform(X_test_scaled)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:23.923338Z","iopub.execute_input":"2024-12-05T01:37:23.923812Z","iopub.status.idle":"2024-12-05T01:37:23.936569Z","shell.execute_reply.started":"2024-12-05T01:37:23.923764Z","shell.execute_reply":"2024-12-05T01:37:23.935463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on the test set\npredictions = (model.predict(X_test_pca) > 0.5).astype(\"int32\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:23.938063Z","iopub.execute_input":"2024-12-05T01:37:23.938513Z","iopub.status.idle":"2024-12-05T01:37:24.019845Z","shell.execute_reply.started":"2024-12-05T01:37:23.938480Z","shell.execute_reply":"2024-12-05T01:37:24.018768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a submission DataFrame\nsubmission = pd.DataFrame({\n    'id': test_data['id'],\n    'sii': predictions.flatten()  # Flatten to convert array to 1D\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:24.021771Z","iopub.execute_input":"2024-12-05T01:37:24.022156Z","iopub.status.idle":"2024-12-05T01:37:24.027845Z","shell.execute_reply.started":"2024-12-05T01:37:24.022122Z","shell.execute_reply":"2024-12-05T01:37:24.026567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the submission DataFrame to a CSV file\nsubmission.to_csv('submission.csv', index=False)\n\n# Check the submission DataFrame\nprint(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T01:37:24.029301Z","iopub.execute_input":"2024-12-05T01:37:24.030104Z","iopub.status.idle":"2024-12-05T01:37:24.047493Z","shell.execute_reply.started":"2024-12-05T01:37:24.030067Z","shell.execute_reply":"2024-12-05T01:37:24.046239Z"}},"outputs":[],"execution_count":null}]}