{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:01.786258Z","iopub.execute_input":"2024-12-03T12:16:01.787061Z","iopub.status.idle":"2024-12-03T12:16:03.713773Z","shell.execute_reply.started":"2024-12-03T12:16:01.787010Z","shell.execute_reply":"2024-12-03T12:16:03.712433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the CSV data\ntrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ntrain_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:03.716136Z","iopub.execute_input":"2024-12-03T12:16:03.716653Z","iopub.status.idle":"2024-12-03T12:16:03.847839Z","shell.execute_reply.started":"2024-12-03T12:16:03.716594Z","shell.execute_reply":"2024-12-03T12:16:03.846756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get Statistical details\ntrain_data.describe().transpose()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:03.848952Z","iopub.execute_input":"2024-12-03T12:16:03.849257Z","iopub.status.idle":"2024-12-03T12:16:04.008115Z","shell.execute_reply.started":"2024-12-03T12:16:03.849227Z","shell.execute_reply":"2024-12-03T12:16:04.006842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:04.010977Z","iopub.execute_input":"2024-12-03T12:16:04.011467Z","iopub.status.idle":"2024-12-03T12:16:04.043110Z","shell.execute_reply.started":"2024-12-03T12:16:04.011419Z","shell.execute_reply":"2024-12-03T12:16:04.041891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of Label Data\ntrain_data['sii'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:04.044326Z","iopub.execute_input":"2024-12-03T12:16:04.044686Z","iopub.status.idle":"2024-12-03T12:16:04.055473Z","shell.execute_reply.started":"2024-12-03T12:16:04.044653Z","shell.execute_reply":"2024-12-03T12:16:04.054132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# We'll start by selecting columns with more than 50% non-null values and filling missing values\nthreshold = 0.5 * len(train_data)\ncolumns_with_data = train_data.columns[train_data.isnull().sum() < threshold]\ntrain_data = train_data[columns_with_data]\n# Replace all missing values with 0\ntrain_data = train_data.fillna(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:04.056806Z","iopub.execute_input":"2024-12-03T12:16:04.057144Z","iopub.status.idle":"2024-12-03T12:16:04.082692Z","shell.execute_reply.started":"2024-12-03T12:16:04.057111Z","shell.execute_reply":"2024-12-03T12:16:04.081474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the target column\ntarget_column = 'sii'\n# Remove rows where the target column 'sii' is NaN (if there are any)\ntrain_data_cleaned = train_data.dropna(subset=[target_column])\n# Check the results\ntrain_data_cleaned.head()\ntrain_data_cleaned.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:04.084747Z","iopub.execute_input":"2024-12-03T12:16:04.085274Z","iopub.status.idle":"2024-12-03T12:16:04.105555Z","shell.execute_reply.started":"2024-12-03T12:16:04.085228Z","shell.execute_reply":"2024-12-03T12:16:04.104415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Categorical columns in the dataset\ncategorical_columns = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n                       'FGC-Season', 'BIA-Season', 'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n# Plotting boxplots for 'sii' against each categorical column\nplt.figure(figsize=(16, 24))\nfor i, col in enumerate(categorical_columns, 1):\n    plt.subplot(4, 2, i)  # 4 rows, 2 columns, plot i\n    sns.boxplot(x=col, y='sii', data=train_data_cleaned)\n    plt.xticks(rotation=45)\n    plt.title(f\"'sii' vs {col}\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:04.107020Z","iopub.execute_input":"2024-12-03T12:16:04.107476Z","iopub.status.idle":"2024-12-03T12:16:06.206958Z","shell.execute_reply.started":"2024-12-03T12:16:04.107432Z","shell.execute_reply":"2024-12-03T12:16:06.205844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot target column 'sii' with numerical columns\nnumerical_cols = train_data_cleaned.select_dtypes(include=['float64', 'int64']).columns\n# Set the number of plots per row\nplots_per_row = 5\nn_rows = (len(numerical_cols) + plots_per_row - 1) // plots_per_row\nplt.figure(figsize=(20, 4 * n_rows))\nfor i, col in enumerate(numerical_cols):\n    plt.subplot(n_rows, plots_per_row, i + 1)\n    sns.boxplot(x='sii', y=col, data=train_data_cleaned)\n    plt.title(col)\n    plt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:06.208409Z","iopub.execute_input":"2024-12-03T12:16:06.208777Z","iopub.status.idle":"2024-12-03T12:16:59.104646Z","shell.execute_reply.started":"2024-12-03T12:16:06.208743Z","shell.execute_reply":"2024-12-03T12:16:59.103533Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify categorical columns for seasons\nseason_cols = [\n    'Basic_Demos-Enroll_Season', \n    'CGAS-Season', \n    'Physical-Season', \n    'FGC-Season', \n    'BIA-Season', \n    'PCIAT-Season', \n    'SDS-Season', \n    'PreInt_EduHx-Season'\n]\n# Create a mapping dictionary for seasons\nseason_mapping = {\n    'Spring': 0,\n    'Summer': 1,\n    'Fall': 2,\n    'Winter': 3\n}\n# Apply manual encoding to the categorical columns\nfor col in season_cols:\n    if col in train_data_cleaned.columns:\n        train_data_cleaned[col] = train_data_cleaned[col].replace(season_mapping)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:59.107966Z","iopub.execute_input":"2024-12-03T12:16:59.108335Z","iopub.status.idle":"2024-12-03T12:16:59.140837Z","shell.execute_reply.started":"2024-12-03T12:16:59.108300Z","shell.execute_reply":"2024-12-03T12:16:59.139731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop the 'id' column if present\ntrain_data_no_id = train_data_cleaned.drop(columns=['id'], errors='ignore')\n# Calculate the correlation matrix\ncorrelation_matrix = train_data_no_id.corr()\n# Plot the heatmap\nplt.figure(figsize=(30, 30))\nsns.heatmap(correlation_matrix, annot=True, fmt='.1f', cmap='coolwarm', square=True)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:16:59.142238Z","iopub.execute_input":"2024-12-03T12:16:59.142583Z","iopub.status.idle":"2024-12-03T12:17:08.767532Z","shell.execute_reply.started":"2024-12-03T12:16:59.142551Z","shell.execute_reply":"2024-12-03T12:17:08.766065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the intersection of columns in train_data_cleaned and test_data\ncommon_columns = train_data_cleaned.columns.intersection(test_data.columns)\n# Prepare the feature matrix X and target vector y\nX = train_data_cleaned[common_columns].drop(columns=['id'])\ny = train_data_cleaned['sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:17:08.769110Z","iopub.execute_input":"2024-12-03T12:17:08.769577Z","iopub.status.idle":"2024-12-03T12:17:08.782273Z","shell.execute_reply.started":"2024-12-03T12:17:08.769532Z","shell.execute_reply":"2024-12-03T12:17:08.780934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:17:08.783902Z","iopub.execute_input":"2024-12-03T12:17:08.784350Z","iopub.status.idle":"2024-12-03T12:17:08.797821Z","shell.execute_reply.started":"2024-12-03T12:17:08.784304Z","shell.execute_reply":"2024-12-03T12:17:08.796610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature Scaling\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)# Instantiate the Model\nrf_model = RandomForestClassifier(n_estimators=100, random_state=42, \n                                  max_depth=10, \n                                  min_samples_split=10, \n                                  min_samples_leaf=4)\nX_test_scaled = scaler.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:17:08.799502Z","iopub.execute_input":"2024-12-03T12:17:08.800098Z","iopub.status.idle":"2024-12-03T12:17:08.823949Z","shell.execute_reply.started":"2024-12-03T12:17:08.800049Z","shell.execute_reply":"2024-12-03T12:17:08.822396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply PCA\npca = PCA(n_components=0.95)  # Keep 95% of the variance\nX_train_pca = pca.fit_transform(X_train_scaled)\nX_test_pca = pca.transform(X_test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:17:08.825530Z","iopub.execute_input":"2024-12-03T12:17:08.826015Z","iopub.status.idle":"2024-12-03T12:17:09.223614Z","shell.execute_reply.started":"2024-12-03T12:17:08.825964Z","shell.execute_reply":"2024-12-03T12:17:09.219433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Instantiate the Model\nrf_model = RandomForestClassifier(n_estimators=100, random_state=42, \n                                  max_depth=10, \n                                  min_samples_split=10, \n                                  min_samples_leaf=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:17:09.226565Z","iopub.execute_input":"2024-12-03T12:17:09.227063Z","iopub.status.idle":"2024-12-03T12:17:09.237209Z","shell.execute_reply.started":"2024-12-03T12:17:09.227016Z","shell.execute_reply":"2024-12-03T12:17:09.236279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fit the Model on PCA-transformed data\nrf_model.fit(X_train_pca, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:17:09.238435Z","iopub.execute_input":"2024-12-03T12:17:09.241729Z","iopub.status.idle":"2024-12-03T12:17:10.406720Z","shell.execute_reply.started":"2024-12-03T12:17:09.241655Z","shell.execute_reply":"2024-12-03T12:17:10.405566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on the test set\ny_pred = rf_model.predict(X_test_pca)\n# Generate a classification report\nprint(\"Classification Report:\")\nprint(classification_report(y_test, y_pred))\n# Generate a confusion matrix\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_test, y_pred))\n# Calculate the accuracy on the test set\naccuracy = rf_model.score(X_test_pca, y_test)\nprint(f\"Model Accuracy: {accuracy}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:17:10.408198Z","iopub.execute_input":"2024-12-03T12:17:10.408567Z","iopub.status.idle":"2024-12-03T12:17:10.464145Z","shell.execute_reply.started":"2024-12-03T12:17:10.408535Z","shell.execute_reply":"2024-12-03T12:17:10.463063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the season columns\nseason_cols = [\n    'Basic_Demos-Enroll_Season', \n    'CGAS-Season', \n    'Physical-Season', \n    'FGC-Season', \n    'BIA-Season', \n    'PCIAT-Season', \n    'SDS-Season', \n    'PreInt_EduHx-Season'\n]\n# Create a mapping dictionary for seasons\nseason_mapping = {\n    'Spring': 0,\n    'Summer': 1,\n    'Fall': 2,\n    'Winter': 3\n}\n# Replace season values in test data using the mapping\nfor col in season_cols:\n    if col in test_data.columns:  # Check if the column exists in test_data\n        test_data[col] = test_data[col].map(season_mapping)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:17:10.465797Z","iopub.execute_input":"2024-12-03T12:17:10.466294Z","iopub.status.idle":"2024-12-03T12:17:10.481540Z","shell.execute_reply.started":"2024-12-03T12:17:10.466226Z","shell.execute_reply":"2024-12-03T12:17:10.479983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Replace null values with 0 in test data\ntest_data.fillna(0, inplace=True)\n# Get the intersection of columns in test_data and training data\ncommon_columns = train_data_cleaned.columns.intersection(test_data.columns)\n# Prepare the test data using the common columns\nX_test_data = test_data[common_columns].drop(columns=['id'])  # Drop 'id' column\n# Scale the test data using the same scaler fitted on training data\nX_test_scaled = scaler.transform(X_test_data)\n# Apply PCA to the test data\nX_test_pca = pca.transform(X_test_scaled)\n# Make predictions on the test set\npredictions = rf_model.predict(X_test_pca)\n# Create a submission DataFrame\nsubmission = pd.DataFrame({\n    'id': test_data['id'],  # Include 'id' from test_data\n    'sii': predictions  # Predictions from the model\n})\n# Save the submission DataFrame to a CSV file\nsubmission.to_csv('submission.csv', index=False)\n# Check the submission DataFrame\nprint(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-03T12:17:10.482949Z","iopub.execute_input":"2024-12-03T12:17:10.483341Z","iopub.status.idle":"2024-12-03T12:17:10.515049Z","shell.execute_reply.started":"2024-12-03T12:17:10.483297Z","shell.execute_reply":"2024-12-03T12:17:10.513883Z"}},"outputs":[],"execution_count":null}]}