{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:23.355939Z","iopub.execute_input":"2024-12-07T04:27:23.356323Z","iopub.status.idle":"2024-12-07T04:27:27.647580Z","shell.execute_reply.started":"2024-12-07T04:27:23.356279Z","shell.execute_reply":"2024-12-07T04:27:27.646476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import libraries\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import classification_report, accuracy_score\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, BatchNormalization, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:27.649659Z","iopub.execute_input":"2024-12-07T04:27:27.650060Z","iopub.status.idle":"2024-12-07T04:27:40.444539Z","shell.execute_reply.started":"2024-12-07T04:27:27.650029Z","shell.execute_reply":"2024-12-07T04:27:40.443424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load Dataset\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_ds = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.445845Z","iopub.execute_input":"2024-12-07T04:27:40.446403Z","iopub.status.idle":"2024-12-07T04:27:40.517970Z","shell.execute_reply.started":"2024-12-07T04:27:40.446370Z","shell.execute_reply":"2024-12-07T04:27:40.516928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.519556Z","iopub.execute_input":"2024-12-07T04:27:40.519860Z","iopub.status.idle":"2024-12-07T04:27:40.558020Z","shell.execute_reply.started":"2024-12-07T04:27:40.519830Z","shell.execute_reply":"2024-12-07T04:27:40.557031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ds.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.559420Z","iopub.execute_input":"2024-12-07T04:27:40.559830Z","iopub.status.idle":"2024-12-07T04:27:40.582528Z","shell.execute_reply.started":"2024-12-07T04:27:40.559785Z","shell.execute_reply":"2024-12-07T04:27:40.581513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Clean the training dataset\nthreshold = 0.5 * len(train_df)\nvalid_columns = train_df.columns[train_df.isnull().sum() < threshold]\ntrain_df = train_df[valid_columns].fillna(0)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.585249Z","iopub.execute_input":"2024-12-07T04:27:40.585561Z","iopub.status.idle":"2024-12-07T04:27:40.605502Z","shell.execute_reply.started":"2024-12-07T04:27:40.585532Z","shell.execute_reply":"2024-12-07T04:27:40.604752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove rows with missing target values\ntarget_column = 'sii'\ntrain_df_cleaned = train_df.dropna(subset=[target_column])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.606702Z","iopub.execute_input":"2024-12-07T04:27:40.607085Z","iopub.status.idle":"2024-12-07T04:27:40.614283Z","shell.execute_reply.started":"2024-12-07T04:27:40.607039Z","shell.execute_reply":"2024-12-07T04:27:40.613165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encode seasonal columns\nseason_columns = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'FGC-Season', 'BIA-Season', \n                  'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\nseason_values = {'Spring': 0, 'Summer': 1, 'Fall': 2, 'Winter': 3}\nfor column in season_columns:\n    if column in train_df_cleaned.columns:\n        train_df_cleaned[column] = train_df_cleaned[column].replace(season_values)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.615394Z","iopub.execute_input":"2024-12-07T04:27:40.615706Z","iopub.status.idle":"2024-12-07T04:27:40.646434Z","shell.execute_reply.started":"2024-12-07T04:27:40.615663Z","shell.execute_reply":"2024-12-07T04:27:40.645441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract features and target\ncommon_features = train_df_cleaned.columns.intersection(test_ds.columns)\nX = train_df_cleaned[common_features]\nif 'id' in X.columns:\n    X = X.drop(columns=['id'])\ny = train_df_cleaned['sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.647715Z","iopub.execute_input":"2024-12-07T04:27:40.648019Z","iopub.status.idle":"2024-12-07T04:27:40.658305Z","shell.execute_reply.started":"2024-12-07T04:27:40.647989Z","shell.execute_reply":"2024-12-07T04:27:40.657340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.659623Z","iopub.execute_input":"2024-12-07T04:27:40.659943Z","iopub.status.idle":"2024-12-07T04:27:40.673455Z","shell.execute_reply.started":"2024-12-07T04:27:40.659898Z","shell.execute_reply":"2024-12-07T04:27:40.672367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Scale numeric features\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.675325Z","iopub.execute_input":"2024-12-07T04:27:40.675673Z","iopub.status.idle":"2024-12-07T04:27:40.691719Z","shell.execute_reply.started":"2024-12-07T04:27:40.675629Z","shell.execute_reply":"2024-12-07T04:27:40.690711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply PCA for dimensionality reduction\npca = PCA(n_components=0.95)\nX_train_pca = pca.fit_transform(X_train_scaled)\nX_test_pca = pca.transform(X_test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.693209Z","iopub.execute_input":"2024-12-07T04:27:40.693739Z","iopub.status.idle":"2024-12-07T04:27:40.737995Z","shell.execute_reply.started":"2024-12-07T04:27:40.693695Z","shell.execute_reply":"2024-12-07T04:27:40.734376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build the neural network network_model\nnetwork_model = Sequential([\n    Dense(128, activation='relu', input_dim=X_train_pca.shape[1]),\n    Dense(64, activation='relu'),\n    BatchNormalization(),\n    Dropout(0.3),\n    Dense(32, activation='relu'),\n    BatchNormalization(),\n    Dropout(0.3),\n    Dense(1, activation='sigmoid')\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.739020Z","iopub.execute_input":"2024-12-07T04:27:40.743580Z","iopub.status.idle":"2024-12-07T04:27:40.928674Z","shell.execute_reply.started":"2024-12-07T04:27:40.743534Z","shell.execute_reply":"2024-12-07T04:27:40.927867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compile the network_model\nnetwork_model.compile(optimizer=Adam(learning_rate=0.001), loss='binary_crossentropy', metrics=['accuracy'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.929801Z","iopub.execute_input":"2024-12-07T04:27:40.930097Z","iopub.status.idle":"2024-12-07T04:27:40.943403Z","shell.execute_reply.started":"2024-12-07T04:27:40.930068Z","shell.execute_reply":"2024-12-07T04:27:40.942331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up callbacks\nlr_scheduler = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=5, min_lr=1e-6)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.944953Z","iopub.execute_input":"2024-12-07T04:27:40.945671Z","iopub.status.idle":"2024-12-07T04:27:40.950279Z","shell.execute_reply.started":"2024-12-07T04:27:40.945625Z","shell.execute_reply":"2024-12-07T04:27:40.949216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model\nhistory = network_model.fit(X_train_pca, y_train, epochs=50, batch_size=32, validation_data=(X_test_pca, y_test), callbacks=[lr_scheduler])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:40.951644Z","iopub.execute_input":"2024-12-07T04:27:40.952345Z","iopub.status.idle":"2024-12-07T04:27:55.708330Z","shell.execute_reply.started":"2024-12-07T04:27:40.952313Z","shell.execute_reply":"2024-12-07T04:27:55.707528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate the model\ny_pred = network_model.predict(X_test_pca)\ny_pred_class = (y_pred > 0.5).astype(int)\nprint(\"Classification Report:\\n\", classification_report(y_test, y_pred_class, zero_division=1))\nprint(f\"Accuracy: {accuracy_score(y_test, y_pred_class) * 100:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:37:24.100523Z","iopub.execute_input":"2024-12-07T04:37:24.100904Z","iopub.status.idle":"2024-12-07T04:37:24.231543Z","shell.execute_reply.started":"2024-12-07T04:37:24.100870Z","shell.execute_reply":"2024-12-07T04:37:24.230269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot training and validation loss\nplt.figure(figsize=(8, 6)) \nplt.plot(history.history['loss'], label='Train Loss', color='blue')\nplt.plot(history.history['val_loss'], label='Validation Loss', color='red')\nplt.title('Model Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()  \nplt.grid(True) \nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:56.520663Z","iopub.status.idle":"2024-12-07T04:27:56.520998Z","shell.execute_reply.started":"2024-12-07T04:27:56.520834Z","shell.execute_reply":"2024-12-07T04:27:56.520851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot training and validation accuracy\nplt.figure(figsize=(8, 6)) \nplt.plot(history.history['accuracy'], label='Train Accuracy', color='blue')\nplt.plot(history.history['val_accuracy'], label='Validation Accuracy', color='red')\nplt.title('Model Accuracy')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()  \nplt.grid(True) \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:56.522055Z","iopub.status.idle":"2024-12-07T04:27:56.522442Z","shell.execute_reply.started":"2024-12-07T04:27:56.522273Z","shell.execute_reply":"2024-12-07T04:27:56.522291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 15: Preprocess the test datasetset\ntest_ds_cleaned = test_ds.fillna(0)\n\n# Handle 'Season' columns in test datasetset\nfor column in season_columns:\n    if column in test_ds_cleaned.columns:\n        test_ds_cleaned[column] = test_ds_cleaned[column].replace(season_values)\n\n# Ensure the test datasetset has the same feature set as training\nX_test_final = test_ds_cleaned[common_features]\nif 'id' in X_test_final.columns:\n    ids = X_test_final['id']  \n    X_test_final = X_test_final.drop(columns=['id'])\n\n# Scale and apply PCA on test dataset\nX_test_final_scaled = scaler.transform(X_test_final)\nX_test_final_pca = pca.transform(X_test_final_scaled)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:56.523616Z","iopub.status.idle":"2024-12-07T04:27:56.523978Z","shell.execute_reply.started":"2024-12-07T04:27:56.523805Z","shell.execute_reply":"2024-12-07T04:27:56.523823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 16: Predict on test dataset\ntest_predictions = network_model.predict(X_test_final_pca)\ntest_predictions_class = (test_predictions > 0.5).astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:56.525408Z","iopub.status.idle":"2024-12-07T04:27:56.525770Z","shell.execute_reply.started":"2024-12-07T04:27:56.525588Z","shell.execute_reply":"2024-12-07T04:27:56.525605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 17: Prepare submission\nsubmission = pd.DataFrame({'id': ids, 'sii': test_predictions_class.flatten()})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file 'submission.csv' has been created.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T04:27:56.527062Z","iopub.status.idle":"2024-12-07T04:27:56.527446Z","shell.execute_reply.started":"2024-12-07T04:27:56.527270Z","shell.execute_reply":"2024-12-07T04:27:56.527288Z"}},"outputs":[],"execution_count":null}]}