{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-31T16:47:45.736309Z","iopub.execute_input":"2024-10-31T16:47:45.737653Z","iopub.status.idle":"2024-10-31T16:47:51.005733Z","shell.execute_reply.started":"2024-10-31T16:47:45.737587Z","shell.execute_reply":"2024-10-31T16:47:51.004606Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.getcwd())\nprint(os.listdir(os.getcwd()))\nprint(os.listdir(\"../../\"))\nprint(os.listdir(\"../\"))\nprint(os.listdir(\"../input/\"))\nprint(os.listdir(\"../../kaggle\"))","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.007978Z","iopub.execute_input":"2024-10-31T16:47:51.008995Z","iopub.status.idle":"2024-10-31T16:47:51.017135Z","shell.execute_reply.started":"2024-10-31T16:47:51.008950Z","shell.execute_reply":"2024-10-31T16:47:51.015847Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# Convert Dictionary into Dataframe","metadata":{}},{"cell_type":"code","source":"# obtain the documents containing the sentiment provided\nimport pandas as pd\n\n# Load the dataset\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.019092Z","iopub.execute_input":"2024-10-31T16:47:51.019922Z","iopub.status.idle":"2024-10-31T16:47:51.159087Z","shell.execute_reply.started":"2024-10-31T16:47:51.019866Z","shell.execute_reply":"2024-10-31T16:47:51.157857Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## Check the data label","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.160510Z","iopub.execute_input":"2024-10-31T16:47:51.160924Z","iopub.status.idle":"2024-10-31T16:47:51.196574Z","shell.execute_reply.started":"2024-10-31T16:47:51.160880Z","shell.execute_reply":"2024-10-31T16:47:51.195446Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.200241Z","iopub.execute_input":"2024-10-31T16:47:51.200668Z","iopub.status.idle":"2024-10-31T16:47:51.243098Z","shell.execute_reply.started":"2024-10-31T16:47:51.200590Z","shell.execute_reply":"2024-10-31T16:47:51.241696Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.244825Z","iopub.execute_input":"2024-10-31T16:47:51.245280Z","iopub.status.idle":"2024-10-31T16:47:51.263185Z","shell.execute_reply.started":"2024-10-31T16:47:51.245228Z","shell.execute_reply":"2024-10-31T16:47:51.261939Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## If there is null value in each column","metadata":{}},{"cell_type":"code","source":"print(\"Null occurrences in Columns:\")\nprint(train_df.isnull().sum(axis=0))","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.264603Z","iopub.execute_input":"2024-10-31T16:47:51.265007Z","iopub.status.idle":"2024-10-31T16:47:51.279369Z","shell.execute_reply.started":"2024-10-31T16:47:51.264968Z","shell.execute_reply":"2024-10-31T16:47:51.278088Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## If there is null value in each row","metadata":{}},{"cell_type":"code","source":"print(\"Null occurrences in Rows:\")\nprint(train_df.isnull().sum(axis=1))","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.281153Z","iopub.execute_input":"2024-10-31T16:47:51.281531Z","iopub.status.idle":"2024-10-31T16:47:51.297180Z","shell.execute_reply.started":"2024-10-31T16:47:51.281491Z","shell.execute_reply":"2024-10-31T16:47:51.295878Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Check for the duplicated value","metadata":{}},{"cell_type":"code","source":"train_df.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.298731Z","iopub.execute_input":"2024-10-31T16:47:51.299203Z","iopub.status.idle":"2024-10-31T16:47:51.331372Z","shell.execute_reply.started":"2024-10-31T16:47:51.299150Z","shell.execute_reply":"2024-10-31T16:47:51.330051Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### There is no duplicate data in dataset.","metadata":{}},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.334859Z","iopub.execute_input":"2024-10-31T16:47:51.335262Z","iopub.status.idle":"2024-10-31T16:47:51.341700Z","shell.execute_reply.started":"2024-10-31T16:47:51.335220Z","shell.execute_reply":"2024-10-31T16:47:51.340500Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.sii.value_counts())\n\nupperbound = max(train_df.sii.value_counts() + 100)\n\nplt.style.use('seaborn-v0_8-whitegrid')\n\n# plot barchart for train.csv\ntrain_df.sii.value_counts().plot(kind = 'bar',\n                                    title = 'Sii distribution',\n                                    ylim = [0, upperbound],        \n                                    rot = 0, fontsize = 11, figsize = (8,3))","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.343272Z","iopub.execute_input":"2024-10-31T16:47:51.343745Z","iopub.status.idle":"2024-10-31T16:47:51.719950Z","shell.execute_reply.started":"2024-10-31T16:47:51.343700Z","shell.execute_reply":"2024-10-31T16:47:51.718729Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## Convert the season datatype into numerical representation","metadata":{}},{"cell_type":"markdown","source":"### Load the data to check the content of 'seasom'","metadata":{}},{"cell_type":"code","source":"print(train_df['Basic_Demos-Enroll_Season'].unique())","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.721531Z","iopub.execute_input":"2024-10-31T16:47:51.721943Z","iopub.status.idle":"2024-10-31T16:47:51.728898Z","shell.execute_reply.started":"2024-10-31T16:47:51.721901Z","shell.execute_reply":"2024-10-31T16:47:51.727698Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Check every label with content 'season'","metadata":{}},{"cell_type":"code","source":"season_columns = [col for col in train_df.columns if 'season' in col.lower()]\nprint(season_columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.730763Z","iopub.execute_input":"2024-10-31T16:47:51.731214Z","iopub.status.idle":"2024-10-31T16:47:51.746766Z","shell.execute_reply.started":"2024-10-31T16:47:51.731173Z","shell.execute_reply":"2024-10-31T16:47:51.745453Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Ordinal Encoding (When Seasons Have a Natural Order)","metadata":{}},{"cell_type":"code","source":"# Define the mapping for ordinal encoding\nseason_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\n\n# Apply the mapping to season columns\ntrain_df['Basic_Demos-Enroll_Season'] = train_df['Basic_Demos-Enroll_Season'].map(season_mapping)\ntrain_df['CGAS-Season'] = train_df['CGAS-Season'].map(season_mapping)\ntrain_df['Physical-Season'] = train_df['Physical-Season'].map(season_mapping)\ntrain_df['Fitness_Endurance-Season'] = train_df['Fitness_Endurance-Season'].map(season_mapping)\ntrain_df['FGC-Season'] = train_df['FGC-Season'].map(season_mapping)\ntrain_df['BIA-Season'] = train_df['BIA-Season'].map(season_mapping)\ntrain_df['PAQ_A-Season'] = train_df['PAQ_A-Season'].map(season_mapping)\ntrain_df['PAQ_C-Season'] = train_df['PAQ_C-Season'].map(season_mapping)\ntrain_df['PCIAT-Season'] = train_df['PCIAT-Season'].map(season_mapping)\ntrain_df['SDS-Season'] = train_df['SDS-Season'].map(season_mapping)\ntrain_df['PreInt_EduHx-Season'] = train_df['PreInt_EduHx-Season'].map(season_mapping)\n\n# Display the updated DataFrame\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.752077Z","iopub.execute_input":"2024-10-31T16:47:51.753127Z","iopub.status.idle":"2024-10-31T16:47:51.819719Z","shell.execute_reply.started":"2024-10-31T16:47:51.753082Z","shell.execute_reply":"2024-10-31T16:47:51.818552Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## Another Issue arises：There are too much null value in each label. In this case, we use Regression-based imputation to predict these missing values.","metadata":{}},{"cell_type":"markdown","source":"### Convert 'id' data from object into numerical data.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Convert 'id' column to numerical if it’s not already numerical\nif train_df['id'].dtype == 'object':\n    label_encoder = LabelEncoder()\n    train_df['id'] = label_encoder.fit_transform(train_df['id'])","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:51.821229Z","iopub.execute_input":"2024-10-31T16:47:51.821612Z","iopub.status.idle":"2024-10-31T16:47:52.497326Z","shell.execute_reply.started":"2024-10-31T16:47:51.821572Z","shell.execute_reply":"2024-10-31T16:47:52.496102Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:52.499340Z","iopub.execute_input":"2024-10-31T16:47:52.500466Z","iopub.status.idle":"2024-10-31T16:47:52.521836Z","shell.execute_reply.started":"2024-10-31T16:47:52.500408Z","shell.execute_reply":"2024-10-31T16:47:52.520110Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Okay, now all the data are numerical, we can implement Regression-based imputation.","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\n# Define columns to impute based on a threshold for high null counts\nnull_threshold = 10  # Define threshold for 'many' nulls\ncolumns_to_impute = [col for col in train_df.columns if train_df[col].isnull().sum() > null_threshold]\n\n# Loop through each column with many nulls\nfor target_column in columns_to_impute:\n    print(f\"Imputing column: {target_column}\")\n\n    # Split the data into rows with known and unknown target values\n    known_data = train_df[train_df[target_column].notnull()]\n    unknown_data = train_df[train_df[target_column].isnull()]\n\n    # Select features for regression, excluding the current target column and columns with excessive nulls\n    feature_columns = [col for col in train_df.columns if col != target_column and train_df[col].isnull().sum() < null_threshold]\n\n    # Ensure there are enough features for regression\n    if len(feature_columns) == 0:\n        print(f\"Skipping {target_column} due to insufficient features.\")\n        continue\n\n    # Prepare training data\n    X_train = known_data[feature_columns]\n    y_train = known_data[target_column]\n\n    # Train the regression model\n    model = RandomForestRegressor(random_state=0)\n    model.fit(X_train, y_train)\n\n    # Prepare test data (rows with nulls in the target column)\n    X_test = unknown_data[feature_columns]\n\n    # Predict missing values\n    predicted_values = model.predict(X_test)\n\n    # Impute the predicted values back into the original DataFrame\n    train_df.loc[train_df[target_column].isnull(), target_column] = predicted_values\n\n    print(f\"Completed imputation for {target_column}\")\n\n# Final check to see if there are any remaining nulls\nprint(train_df.isnull().sum())\n","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:47:52.523556Z","iopub.execute_input":"2024-10-31T16:47:52.524070Z","iopub.status.idle":"2024-10-31T16:53:47.610518Z","shell.execute_reply.started":"2024-10-31T16:47:52.524013Z","shell.execute_reply":"2024-10-31T16:53:47.609293Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"## Sampling","metadata":{}},{"cell_type":"code","source":"train_sample = train_df.sample(n = 1000)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:53:47.611912Z","iopub.execute_input":"2024-10-31T16:53:47.612282Z","iopub.status.idle":"2024-10-31T16:53:47.623841Z","shell.execute_reply.started":"2024-10-31T16:53:47.612242Z","shell.execute_reply":"2024-10-31T16:53:47.622565Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_sample)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:53:47.625141Z","iopub.execute_input":"2024-10-31T16:53:47.625563Z","iopub.status.idle":"2024-10-31T16:53:47.639627Z","shell.execute_reply.started":"2024-10-31T16:53:47.625520Z","shell.execute_reply":"2024-10-31T16:53:47.638504Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sample[:10]","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:53:47.641017Z","iopub.execute_input":"2024-10-31T16:53:47.641386Z","iopub.status.idle":"2024-10-31T16:53:47.691903Z","shell.execute_reply.started":"2024-10-31T16:53:47.641345Z","shell.execute_reply":"2024-10-31T16:53:47.690660Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count occurrences of each sii in train_df and train_sample\nsii_counts_train_df = train_df['sii'].value_counts()\nsii_counts_train_sample = train_sample['sii'].value_counts()\n\n# Create a common index with all sii from both dataframes\nall_sii = sii_counts_train_df.index.union(sii_counts_train_sample.index)\n\n# Align the counts for plotting (fill missing sii with 0)\nsii_counts_train_df = sii_counts_train_df.reindex(all_sii, fill_value=0)\nsii_counts_train_ample = sii_counts_train_sample.reindex(all_sii, fill_value=0)\n\n# Plot\nwidth = 0.15  # Width of the bars\nfig, ax = plt.subplots(figsize=(10, 6))\n\n# Create an index for positioning the bars\nindices = range(len(all_sii))\n\n# Bar plot for TRain (blue)\nax.bar(indices, sii_counts_train_df, width, label='Train DF', color='blue')\n\n# Bar plot for Train Sample (orange), with some offset for side-by-side comparison\nax.bar([i + width for i in indices], sii_counts_train_sample, width, label='Train Sample', color='orange')\n\n# Customize the plot\nax.set_xlabel('Sii Number')\nax.set_ylabel('Count')\nax.set_title('Sii Counts in Train (blue) vs Train Sample (orange)')\nax.set_xticks([i + width / 2 for i in indices])  # Center the x-axis labels\nax.set_xticklabels(all_sii)\nax.legend()\n\n# Display the plot\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:53:47.693394Z","iopub.execute_input":"2024-10-31T16:53:47.693781Z","iopub.status.idle":"2024-10-31T16:53:48.083736Z","shell.execute_reply.started":"2024-10-31T16:53:47.693737Z","shell.execute_reply":"2024-10-31T16:53:48.082516Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"## PCA to implement dimensionality reduction","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# Define feature columns (exclude the target column)\nfeature_columns = [col for col in train_df.columns if col != 'sii']\nX = train_df[feature_columns]\n\n# Standardize the features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:53:48.085352Z","iopub.execute_input":"2024-10-31T16:53:48.085773Z","iopub.status.idle":"2024-10-31T16:53:48.113152Z","shell.execute_reply.started":"2024-10-31T16:53:48.085730Z","shell.execute_reply":"2024-10-31T16:53:48.112041Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\n# Initialize PCA, and set n_components to the number of components you'd like to keep\n# Here, we’ll start by explaining 95% of the variance\npca = PCA(n_components=0.95)  # Automatically selects the number of components to explain 95% variance\nX_pca = pca.fit_transform(X_scaled)\n\n# Print explained variance to understand how much variance each component covers\nprint(\"Explained variance ratio:\", pca.explained_variance_ratio_)\nprint(\"Number of components selected:\", pca.n_components_)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:53:48.114816Z","iopub.execute_input":"2024-10-31T16:53:48.115316Z","iopub.status.idle":"2024-10-31T16:53:48.186333Z","shell.execute_reply.started":"2024-10-31T16:53:48.115260Z","shell.execute_reply":"2024-10-31T16:53:48.185287Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert PCA output to a DataFrame\npca_columns = [f'PC{i+1}' for i in range(pca.n_components_)]\nX_pca_df = pd.DataFrame(X_pca, columns=pca_columns)\n\n# Concatenate the principal components with the target variable\ndata_pca = pd.concat([X_pca_df, train_df['sii'].reset_index(drop=True)], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:53:48.188036Z","iopub.execute_input":"2024-10-31T16:53:48.188905Z","iopub.status.idle":"2024-10-31T16:53:48.199517Z","shell.execute_reply.started":"2024-10-31T16:53:48.188851Z","shell.execute_reply":"2024-10-31T16:53:48.198469Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\n\n# Load test data\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:53:48.201242Z","iopub.execute_input":"2024-10-31T16:53:48.207177Z","iopub.status.idle":"2024-10-31T16:53:48.234418Z","shell.execute_reply.started":"2024-10-31T16:53:48.207114Z","shell.execute_reply":"2024-10-31T16:53:48.233365Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the mapping for ordinal encoding\nseason_mapping = {'Spring': 1, 'Summer': 2, 'Fall': 3, 'Winter': 4}\n\n# Apply the mapping to season columns\ntest_df['Basic_Demos-Enroll_Season'] = test_df['Basic_Demos-Enroll_Season'].map(season_mapping)\ntest_df['CGAS-Season'] = test_df['CGAS-Season'].map(season_mapping)\ntest_df['Physical-Season'] = test_df['Physical-Season'].map(season_mapping)\ntest_df['Fitness_Endurance-Season'] = test_df['Fitness_Endurance-Season'].map(season_mapping)\ntest_df['FGC-Season'] = test_df['FGC-Season'].map(season_mapping)\ntest_df['BIA-Season'] = test_df['BIA-Season'].map(season_mapping)\ntest_df['PAQ_A-Season'] = test_df['PAQ_A-Season'].map(season_mapping)\ntest_df['PAQ_C-Season'] = test_df['PAQ_C-Season'].map(season_mapping)\n##test_df['PCIAT-Season'] = test_df['PCIAT-Season'].map(season_mapping)\ntest_df['SDS-Season'] = test_df['SDS-Season'].map(season_mapping)\ntest_df['PreInt_EduHx-Season'] = test_df['PreInt_EduHx-Season'].map(season_mapping)\n\n# Display the updated DataFrame\ntest_df","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:53:48.239831Z","iopub.execute_input":"2024-10-31T16:53:48.242793Z","iopub.status.idle":"2024-10-31T16:53:48.328143Z","shell.execute_reply.started":"2024-10-31T16:53:48.242717Z","shell.execute_reply":"2024-10-31T16:53:48.326898Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## As we can see object type 'id' column, we need to convert it into integer but we need to convert it back to object type in submission file. Furthermore, we need to fill the null value using regression-based imputation just like we did in train_df. ","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Convert 'id' column to numerical if it’s not already numerical\nif test_df['id'].dtype == 'object':\n    label_encoder = LabelEncoder()\n    test_df['id'] = label_encoder.fit_transform(test_df['id'])\ntest_df","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:55:09.589122Z","iopub.execute_input":"2024-10-31T16:55:09.589592Z","iopub.status.idle":"2024-10-31T16:55:09.660323Z","shell.execute_reply.started":"2024-10-31T16:55:09.589548Z","shell.execute_reply":"2024-10-31T16:55:09.656130Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### In this case, we do not use random forest directly since there is NaN value in the dataset. Simple imputation step (like filling with the mean or median of the feature columns) is required in advance.","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nfrom sklearn.impute import SimpleImputer\nimport pandas as pd\n\n# Define columns to impute based on a threshold for high null counts\nnull_threshold = 1  # Define threshold for 'many' nulls\ncolumns_to_impute = [col for col in test_df.columns if test_df[col].isnull().sum() > null_threshold]\n\n# Loop through each column with many nulls\nfor target_column in columns_to_impute:\n    print(f\"Imputing column: {target_column}\")\n\n    # Split the data into rows with known and unknown target values\n    known_data = test_df[test_df[target_column].notnull()]\n    unknown_data = test_df[test_df[target_column].isnull()]\n\n    # Select features for regression, excluding the current target column and columns with excessive nulls\n    feature_columns = [col for col in test_df.columns if col != target_column and test_df[col].isnull().sum() < null_threshold]\n\n    # Ensure there are enough features for regression\n    if len(feature_columns) == 0:\n        print(f\"Skipping {target_column} due to insufficient features.\")\n        continue\n\n    # Prepare training data\n    X_train = known_data[feature_columns]\n    y_train = known_data[target_column]\n\n    # Impute missing values in X_train using median imputation\n    imputer = SimpleImputer(strategy='median')\n    X_train_imputed = imputer.fit_transform(X_train)\n\n    # Train the regression model\n    model = RandomForestRegressor(random_state=0)\n    model.fit(X_train_imputed, y_train)\n\n    # Prepare test data (rows with nulls in the target column) and impute missing values\n    X_test = unknown_data[feature_columns]\n    X_test_imputed = imputer.transform(X_test)\n\n    # Predict missing values\n    predicted_values = model.predict(X_test_imputed)\n\n    # Impute the predicted values back into the original DataFrame\n    test_df.loc[test_df[target_column].isnull(), target_column] = predicted_values\n\n    print(f\"Completed imputation for {target_column}\")\n\n# Final check to see if there are any remaining nulls\nprint(test_df.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2024-10-31T16:59:09.334647Z","iopub.execute_input":"2024-10-31T16:59:09.335508Z","iopub.status.idle":"2024-10-31T16:59:09.591573Z","shell.execute_reply.started":"2024-10-31T16:59:09.335453Z","shell.execute_reply":"2024-10-31T16:59:09.590372Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure consistent features across train and test by filtering test columns\ncommon_features = [col for col in train_df.columns if col != 'sii' and col in test_df.columns]\nX = train_df[common_features]\ny = train_df['sii']\nX_test = test_df[common_features]\n\n# 70/30 train-validation split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.3, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:00:58.187962Z","iopub.execute_input":"2024-10-31T17:00:58.188438Z","iopub.status.idle":"2024-10-31T17:00:58.203739Z","shell.execute_reply.started":"2024-10-31T17:00:58.188395Z","shell.execute_reply":"2024-10-31T17:00:58.202602Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_squared_error  # Import for MSE\n\n# Reload test.csv to ensure we have the original 'id' column\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')  # Replace with the actual path if needed\n\n# Step 1: Standardize the data across training, validation, and test\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)\nX_test_scaled = scaler.transform(X_test)\n\n# Step 2: Apply PCA with the same number of components across all datasets\npca = PCA(n_components=0.95)  # or specify an integer number for fixed components\nX_train_pca = pca.fit_transform(X_train_scaled)\nX_val_pca = pca.transform(X_val_scaled)\nX_test_pca = pca.transform(X_test_scaled)\n\n# Step 3: Train the model using the PCA-transformed training data\nmodel = RandomForestRegressor(random_state=0)\nmodel.fit(X_train_pca, y_train)\n\n# Step 4: Validate the model\ny_val_pred = model.predict(X_val_pca)\nmse = mean_squared_error(y_val, y_val_pred)\nprint(\"Validation Mean Squared Error:\", mse)\n\n# Step 5: Predict on test data and prepare for submission\ntest_predictions = model.predict(X_test_pca)\n\n# Convert predictions to integers by rounding\ntest_predictions = test_predictions.round().astype(int)\n\n# Create the submission DataFrame with 'id' from reloaded test_df and 'sii' as integer type\nsubmission = pd.DataFrame({\n    'id': test_df['id'],  # Use the original 'id' column from the reloaded test data\n    'sii': test_predictions  # 'sii' is now integer as required\n})\n\n# Save the submission file\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\nprint(\"Submission file created: submission1030.csv\")\nsubmission","metadata":{"execution":{"iopub.status.busy":"2024-10-31T17:07:23.340508Z","iopub.execute_input":"2024-10-31T17:07:23.340997Z","iopub.status.idle":"2024-10-31T17:07:30.323569Z","shell.execute_reply.started":"2024-10-31T17:07:23.340954Z","shell.execute_reply":"2024-10-31T17:07:30.322137Z"},"trusted":true},"outputs":[],"execution_count":null}]}