{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:48:20.524837Z","iopub.execute_input":"2024-12-07T04:48:20.525267Z","iopub.status.idle":"2024-12-07T04:48:23.593365Z","shell.execute_reply.started":"2024-12-07T04:48:20.525226Z","shell.execute_reply":"2024-12-07T04:48:23.592092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import libraries\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import classification_report, accuracy_score\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization, Activation\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:53:00.504050Z","iopub.execute_input":"2024-12-07T04:53:00.504474Z","iopub.status.idle":"2024-12-07T04:53:00.510921Z","shell.execute_reply.started":"2024-12-07T04:53:00.504439Z","shell.execute_reply":"2024-12-07T04:53:00.509786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load Dataset\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_ds = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:53:54.492657Z","iopub.execute_input":"2024-12-07T04:53:54.493109Z","iopub.status.idle":"2024-12-07T04:53:54.564435Z","shell.execute_reply.started":"2024-12-07T04:53:54.493058Z","shell.execute_reply":"2024-12-07T04:53:54.563291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:54:06.769255Z","iopub.execute_input":"2024-12-07T04:54:06.769688Z","iopub.status.idle":"2024-12-07T04:54:06.819854Z","shell.execute_reply.started":"2024-12-07T04:54:06.769654Z","shell.execute_reply":"2024-12-07T04:54:06.818579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ds.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:54:18.797610Z","iopub.execute_input":"2024-12-07T04:54:18.798026Z","iopub.status.idle":"2024-12-07T04:54:18.825796Z","shell.execute_reply.started":"2024-12-07T04:54:18.797989Z","shell.execute_reply":"2024-12-07T04:54:18.824583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clean the training dataset\nthreshold = 0.5 * len(train_df)\nvalid_columns = train_df.columns[train_df.isnull().sum() < threshold]\ntrain_df = train_df[valid_columns].fillna(0)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:54:31.261437Z","iopub.execute_input":"2024-12-07T04:54:31.261839Z","iopub.status.idle":"2024-12-07T04:54:31.276596Z","shell.execute_reply.started":"2024-12-07T04:54:31.261805Z","shell.execute_reply":"2024-12-07T04:54:31.275431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove rows with missing target values\ntarget_column = 'sii'\ntrain_df_cleaned = train_df.dropna(subset=[target_column])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:54:38.986837Z","iopub.execute_input":"2024-12-07T04:54:38.987243Z","iopub.status.idle":"2024-12-07T04:54:38.995927Z","shell.execute_reply.started":"2024-12-07T04:54:38.987208Z","shell.execute_reply":"2024-12-07T04:54:38.994697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encode seasonal columns\nseason_columns = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'FGC-Season', 'BIA-Season', \n                  'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\nseason_values = {'Spring': 0, 'Summer': 1, 'Fall': 2, 'Winter': 3}\nfor column in season_columns:\n    if column in train_df_cleaned.columns:\n        train_df_cleaned[column] = train_df_cleaned[column].replace(season_values)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:54:47.746900Z","iopub.execute_input":"2024-12-07T04:54:47.747332Z","iopub.status.idle":"2024-12-07T04:54:47.780966Z","shell.execute_reply.started":"2024-12-07T04:54:47.747267Z","shell.execute_reply":"2024-12-07T04:54:47.779792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract features and target\ncommon_features = train_df_cleaned.columns.intersection(test_ds.columns)\nX = train_df_cleaned[common_features]\nif 'id' in X.columns:\n    X = X.drop(columns=['id'])\ny = train_df_cleaned['sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:55:03.009680Z","iopub.execute_input":"2024-12-07T04:55:03.010101Z","iopub.status.idle":"2024-12-07T04:55:03.020231Z","shell.execute_reply.started":"2024-12-07T04:55:03.010058Z","shell.execute_reply":"2024-12-07T04:55:03.018916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:55:10.194154Z","iopub.execute_input":"2024-12-07T04:55:10.194564Z","iopub.status.idle":"2024-12-07T04:55:10.207221Z","shell.execute_reply.started":"2024-12-07T04:55:10.194530Z","shell.execute_reply":"2024-12-07T04:55:10.206127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Scale numeric features\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:55:19.062175Z","iopub.execute_input":"2024-12-07T04:55:19.062575Z","iopub.status.idle":"2024-12-07T04:55:19.079795Z","shell.execute_reply.started":"2024-12-07T04:55:19.062544Z","shell.execute_reply":"2024-12-07T04:55:19.078300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply PCA for dimensionality reduction\npca = PCA(n_components=0.95)\nX_train_pca = pca.fit_transform(X_train_scaled)\nX_test_pca = pca.transform(X_test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:55:30.656240Z","iopub.execute_input":"2024-12-07T04:55:30.657490Z","iopub.status.idle":"2024-12-07T04:55:30.710636Z","shell.execute_reply.started":"2024-12-07T04:55:30.657423Z","shell.execute_reply":"2024-12-07T04:55:30.706508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build the neural network network_model\nnetwork_model = Sequential([\n    Dense(128, activation='relu', input_dim=X_train_pca.shape[1]),\n    Dense(64, activation='relu'),\n    BatchNormalization(),\n    Dropout(0.3),\n    Dense(32, activation='relu'),\n    BatchNormalization(),\n    Dropout(0.3),\n    Dense(1, activation='sigmoid')\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:55:39.166945Z","iopub.execute_input":"2024-12-07T04:55:39.167349Z","iopub.status.idle":"2024-12-07T04:55:39.308323Z","shell.execute_reply.started":"2024-12-07T04:55:39.167315Z","shell.execute_reply":"2024-12-07T04:55:39.306946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compile the network_model\nnetwork_model.compile(optimizer=Adam(learning_rate=0.001), loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:55:53.957383Z","iopub.execute_input":"2024-12-07T04:55:53.957789Z","iopub.status.idle":"2024-12-07T04:55:53.976887Z","shell.execute_reply.started":"2024-12-07T04:55:53.957741Z","shell.execute_reply":"2024-12-07T04:55:53.975327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up callbacks\nlr_scheduler = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=5, min_lr=1e-6)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:56:03.138818Z","iopub.execute_input":"2024-12-07T04:56:03.139202Z","iopub.status.idle":"2024-12-07T04:56:03.146259Z","shell.execute_reply.started":"2024-12-07T04:56:03.139169Z","shell.execute_reply":"2024-12-07T04:56:03.143893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model\nhistory = network_model.fit(X_train_pca, y_train, epochs=50, batch_size=32, validation_data=(X_test_pca, y_test), callbacks=[lr_scheduler])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:56:11.226632Z","iopub.execute_input":"2024-12-07T04:56:11.227022Z","iopub.status.idle":"2024-12-07T04:56:30.036884Z","shell.execute_reply.started":"2024-12-07T04:56:11.226988Z","shell.execute_reply":"2024-12-07T04:56:30.035715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate the model\ny_pred = network_model.predict(X_test_pca)\ny_pred_class = (y_pred > 0.5).astype(int)\nprint(\"Classification Report:\\n\", classification_report(y_test, y_pred_class, zero_division=1))\nprint(f\"Accuracy: {accuracy_score(y_test, y_pred_class) * 100:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:56:39.695550Z","iopub.execute_input":"2024-12-07T04:56:39.695966Z","iopub.status.idle":"2024-12-07T04:56:39.994886Z","shell.execute_reply.started":"2024-12-07T04:56:39.695929Z","shell.execute_reply":"2024-12-07T04:56:39.993574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot training and validation loss\nplt.figure(figsize=(8, 6)) \nplt.plot(history.history['loss'], label='Train Loss', color='blue')\nplt.plot(history.history['val_loss'], label='Validation Loss', color='red')\nplt.title('Model Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()  \nplt.grid(True) \nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:56:47.161528Z","iopub.execute_input":"2024-12-07T04:56:47.161992Z","iopub.status.idle":"2024-12-07T04:56:47.451732Z","shell.execute_reply.started":"2024-12-07T04:56:47.161954Z","shell.execute_reply":"2024-12-07T04:56:47.450487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot training and validation accuracy\nplt.figure(figsize=(8, 6)) \nplt.plot(history.history['accuracy'], label='Train Accuracy', color='blue')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy', color='red')\nplt.title('Model Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()  \nplt.grid(True) \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:56:57.680993Z","iopub.execute_input":"2024-12-07T04:56:57.681435Z","iopub.status.idle":"2024-12-07T04:56:57.941354Z","shell.execute_reply.started":"2024-12-07T04:56:57.681396Z","shell.execute_reply":"2024-12-07T04:56:57.940200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 15: Preprocess the test datasetset\ntest_ds_cleaned = test_ds.fillna(0)\n\n# Handle 'Season' columns in test datasetset\nfor column in season_columns:\n    if column in test_ds_cleaned.columns:\n        test_ds_cleaned[column] = test_ds_cleaned[column].replace(season_values)\n\n# Ensure the test datasetset has the same feature set as training\nX_test_final = test_ds_cleaned[common_features]\nif 'id' in X_test_final.columns:\n    ids = X_test_final['id']  \n    X_test_final = X_test_final.drop(columns=['id'])\n\n# Scale and apply PCA on test dataset\nX_test_final_scaled = scaler.transform(X_test_final)\nX_test_final_pca = pca.transform(X_test_final_scaled)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:57:07.190897Z","iopub.execute_input":"2024-12-07T04:57:07.191345Z","iopub.status.idle":"2024-12-07T04:57:07.216932Z","shell.execute_reply.started":"2024-12-07T04:57:07.191306Z","shell.execute_reply":"2024-12-07T04:57:07.215554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 16: Predict on test dataset\ntest_predictions = network_model.predict(X_test_final_pca)\ntest_predictions_class = (test_predictions > 0.5).astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:57:15.629581Z","iopub.execute_input":"2024-12-07T04:57:15.630003Z","iopub.status.idle":"2024-12-07T04:57:15.710250Z","shell.execute_reply.started":"2024-12-07T04:57:15.629960Z","shell.execute_reply":"2024-12-07T04:57:15.709198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 17: Prepare submission\nsubmission = pd.DataFrame({'id': ids, 'sii': test_predictions_class.flatten()})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file 'submission.csv' has been created.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:57:25.470097Z","iopub.execute_input":"2024-12-07T04:57:25.470705Z","iopub.status.idle":"2024-12-07T04:57:25.482228Z","shell.execute_reply.started":"2024-12-07T04:57:25.470662Z","shell.execute_reply":"2024-12-07T04:57:25.480848Z"}},"outputs":[],"execution_count":null}]}