{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Imtiaz Adar\n# Email: imtiazadarofficial@gmail.com\n\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score\n\n# Load the dataset (assuming the file is located in the correct directory)\ndata_path = '/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv'\ndf = pd.read_csv(data_path)\n\n# Display the first few rows to understand the structure\nprint(df.head())\n\n# If the 'Instrument' column contains the target variable name, filter rows accordingly\ntarget_column = 'Instrument_Parent-Child Internet Addiction Test'\n\n# Check if the target column exists in the DataFrame and proceed with feature extraction\nif target_column in df['Instrument'].values:\n    # You can then extract the features (X) and target (y) like so:\n    X = df.drop(columns=['Instrument_Parent-Child Internet Addiction Test'])  # Features\n    y = df[target_column]  # Target variable\nelse:\n    print(f\"The target column '{target_column}' is not found in the data.\")\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:20.433335Z","iopub.execute_input":"2024-11-29T07:28:20.433938Z","iopub.status.idle":"2024-11-29T07:28:20.445271Z","shell.execute_reply.started":"2024-11-29T07:28:20.433901Z","shell.execute_reply":"2024-11-29T07:28:20.444409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check unique values in the 'Instrument' and 'Field' columns to identify the target\nprint(\"Unique values in 'Instrument' column:\", df['Instrument'].unique())\nprint(\"Unique values in 'Field' column:\", df['Field'].unique())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:20.447050Z","iopub.execute_input":"2024-11-29T07:28:20.447841Z","iopub.status.idle":"2024-11-29T07:28:20.472786Z","shell.execute_reply.started":"2024-11-29T07:28:20.447802Z","shell.execute_reply":"2024-11-29T07:28:20.471978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check the column names to find the exact match\nprint(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:20.473763Z","iopub.execute_input":"2024-11-29T07:28:20.474003Z","iopub.status.idle":"2024-11-29T07:28:20.486826Z","shell.execute_reply.started":"2024-11-29T07:28:20.473979Z","shell.execute_reply":"2024-11-29T07:28:20.485978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Find numeric columns in the dataset\nnumeric_cols = df.select_dtypes(include=['number']).columns\nprint(numeric_cols)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:20.488309Z","iopub.execute_input":"2024-11-29T07:28:20.488603Z","iopub.status.idle":"2024-11-29T07:28:20.498877Z","shell.execute_reply.started":"2024-11-29T07:28:20.488574Z","shell.execute_reply":"2024-11-29T07:28:20.498032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# List of categorical columns\ncategorical_cols = df.select_dtypes(include=['object']).columns\n\n# Perform label encoding or one-hot encoding\n# One-hot encoding for non-ordinal categories\ndf_encoded = pd.get_dummies(df, columns=categorical_cols)\n\n# Check the updated dataframe\nprint(df_encoded.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:20.499945Z","iopub.execute_input":"2024-11-29T07:28:20.500282Z","iopub.status.idle":"2024-11-29T07:28:20.524566Z","shell.execute_reply.started":"2024-11-29T07:28:20.500246Z","shell.execute_reply":"2024-11-29T07:28:20.523637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# If 'Instrument_Parent-Child Internet Addiction Test' is your target\nX = df_encoded.drop('Instrument_Parent-Child Internet Addiction Test', axis=1)  # Features\ny = df_encoded['Instrument_Parent-Child Internet Addiction Test']  # Target variable\n\n# Checking the shapes of X and y\nprint(X.shape, y.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:20.525412Z","iopub.execute_input":"2024-11-29T07:28:20.525698Z","iopub.status.idle":"2024-11-29T07:28:20.531270Z","shell.execute_reply.started":"2024-11-29T07:28:20.525672Z","shell.execute_reply":"2024-11-29T07:28:20.530428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report\n\n# Load the dataset (assuming the file is located in the correct directory)\ndata_path = '/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv'\ndf = pd.read_csv(data_path)\n\n# Check the structure of the dataset\nprint(df.head())\n\n# Perform one-hot encoding for categorical variables\ncategorical_cols = df.select_dtypes(include=['object']).columns\ndf_encoded = pd.get_dummies(df, columns=categorical_cols)\n\n# Define the features (X) and target (y)\nX = df_encoded.drop('Instrument_Parent-Child Internet Addiction Test', axis=1, errors='ignore')  # Features\ny = df_encoded['Instrument_Parent-Child Internet Addiction Test']  # Target variable\n# Check for missing values\nprint(X.isnull().sum())\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\n\n# Handle missing columns between train and test sets\nmissing_cols = list(set(X_train.columns) - set(X_test.columns))\n\n# Create a DataFrame with missing columns in the test set, filled with 0\nmissing_data = pd.DataFrame(0, index=X_test.index, columns=missing_cols)\n\n# Add the missing columns to the test set\nX_test = pd.concat([X_test, missing_data], axis=1)\n\n# Ensure that the test set columns are in the same order as the training set\nX_test = X_test[X_train.columns]\n\n# Scale the data (optional, but recommended for Logistic Regression)\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# Initialize the model\nmodel = LogisticRegression()\n\n# Train the model\nmodel.fit(X_train, y_train)\n\n# Predict on the test set\ny_pred = model.predict(X_test)\n\n# Evaluate the model\nprint(f\"Accuracy: {accuracy_score(y_test, y_pred):.2f}\")\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_test, y_pred))\nprint(\"Classification Report:\")\nprint(classification_report(y_test, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:20.532808Z","iopub.execute_input":"2024-11-29T07:28:20.533151Z","iopub.status.idle":"2024-11-29T07:28:20.598829Z","shell.execute_reply.started":"2024-11-29T07:28:20.533112Z","shell.execute_reply":"2024-11-29T07:28:20.595804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\n\n# Convert all features to float (if they're not already)\nX_train = X_train.astype(float)\n\n# Convert the target variable to integers if necessary\ny_train = y_train.astype(int)\n\n# Apply SMOTE again\nsmote = SMOTE(random_state=42)\nX_train_res, y_train_res = smote.fit_resample(X_train, y_train)\nprint(f\"Original X_train shape: {X_train.shape}\")\nprint(f\"Resampled X_train shape: {X_train_res.shape}\")\nprint(f\"Resampled y_train shape: {y_train_res.shape}\")\n\n# Initialize and train the Random Forest model with balanced class weights\nrf_model = RandomForestClassifier(random_state=42, class_weight='balanced')\nrf_model.fit(X_train_res, y_train_res)\n\n# You can also print the accuracy or check the model performance\ny_pred = rf_model.predict(X_test)\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy: {accuracy}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:20.602560Z","iopub.execute_input":"2024-11-29T07:28:20.603255Z","iopub.status.idle":"2024-11-29T07:28:20.832901Z","shell.execute_reply.started":"2024-11-29T07:28:20.603203Z","shell.execute_reply":"2024-11-29T07:28:20.832096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfrom imblearn.over_sampling import SMOTE\nfrom imblearn.under_sampling import RandomUnderSampler\nfrom imblearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import classification_report, confusion_matrix, precision_recall_curve, auc\nimport matplotlib.pyplot as plt\n\n# Define the resampling strategy (SMOTE + Random Under Sampling)\nsmote = SMOTE(sampling_strategy='auto', random_state=42)\nunder_sampler = RandomUnderSampler(sampling_strategy='auto', random_state=42)\npipeline = Pipeline(steps=[('smote', smote), ('under', under_sampler)])\n\n# Define the Random Forest model with adjusted hyperparameters\nmodel = RandomForestClassifier(\n    class_weight={0: 2, 1: 1},\n    max_depth=10,  # Increase max_depth for more flexibility\n    max_features='sqrt',\n    min_samples_leaf=1,\n    min_samples_split=5,\n    n_estimators=150,  # Increase the number of estimators\n    random_state=42\n)\n\n# Use Stratified K-Folds cross-validation\ncv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nX_train = pd.DataFrame(X_train)\ny_train = pd.Series(y_train)\n# Perform cross-validation\nfor train_idx, val_idx in cv.split(X_train, y_train):\n    X_train_cv, X_val_cv = X_train.iloc[train_idx], X_train.iloc[val_idx]\n    y_train_cv, y_val_cv = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    # Use .iloc to correctly index by integers\n    # Resample the training data within each fold\n    X_res_cv, y_res_cv = pipeline.fit_resample(X_train_cv, y_train_cv)\n    \n    # Train and evaluate the model\n    model.fit(X_res_cv, y_res_cv)\n    \n    # Get predicted probabilities\n    y_prob_cv = model.predict_proba(X_val_cv)[:, 1]\n    \n    # Calculate Precision-Recall AUC for each fold\n    precision, recall, thresholds = precision_recall_curve(y_val_cv, y_prob_cv)\n    pr_auc_fold = auc(recall, precision)\n    print(f\"AUC for fold: {pr_auc_fold:.4f}\")\n    \n    # Predict the class labels based on the adjusted threshold\n    threshold = 0.3  # You can adjust this threshold based on performance\n    y_pred_cv = (y_prob_cv >= threshold).astype(int)\n\n    # Evaluate with the adjusted predictions\n    print(classification_report(y_val_cv, y_pred_cv))\n    print(confusion_matrix(y_val_cv, y_pred_cv))\n\n# Train the final model on the full resampled data\nX_res, y_res = pipeline.fit_resample(X_train, y_train)\nmodel.fit(X_res, y_res)\n\n# Test the final model\ny_prob = model.predict_proba(X_test)[:, 1]  # Get probabilities for the positive class\n\n# Calculate Precision-Recall AUC for the final test set\nprecision, recall, thresholds = precision_recall_curve(y_test, y_prob)\npr_auc = auc(recall, precision)\nprint(f\"Final Precision-Recall AUC: {pr_auc:.4f}\")\n\n# Plot the Precision-Recall curve for the final model\nplt.plot(recall, precision, color='blue')\nplt.fill_between(recall, precision, color='lightblue')\nplt.title('Precision-Recall Curve')\nplt.xlabel('Recall')\nplt.ylabel('Precision')\nplt.show()\n\n# Evaluate final performance\ny_pred = (y_prob >= 0.3).astype(int)  # Use the same threshold for final predictions\nprint(\"Final Classification Report:\")\nfrom sklearn.metrics import classification_report\n\n# Assuming y_true are the true labels and y_pred are the predicted labels\n\nprint(classification_report(y_test, y_pred, zero_division=0))\n\nprint(\"Final Confusion Matrix:\")\nprint(confusion_matrix(y_test, y_pred))\n\n# Print the final AUC score\nfinal_auc = auc(recall, precision)\nprint(f\"Final AUC: {final_auc:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:20.833950Z","iopub.execute_input":"2024-11-29T07:28:20.834205Z","iopub.status.idle":"2024-11-29T07:28:22.568013Z","shell.execute_reply.started":"2024-11-29T07:28:20.834180Z","shell.execute_reply":"2024-11-29T07:28:22.567191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(X_res.shape)  # Check the new dataset dimensions\n# print(y_res.value_counts())  # Confirm the balanced class distribution\nimport pandas as pd\n\n# Convert y_res to a pandas Series to use value_counts\ny_res_series = pd.Series(y_res)\n\nprint(X_res.shape)  # Check the new dataset dimensions\nprint(y_res_series.value_counts())  # Confirm the balanced class distribution\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:22.569004Z","iopub.execute_input":"2024-11-29T07:28:22.569284Z","iopub.status.idle":"2024-11-29T07:28:22.575397Z","shell.execute_reply.started":"2024-11-29T07:28:22.569258Z","shell.execute_reply":"2024-11-29T07:28:22.574343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Assuming X_res and y_res are your balanced training data\nX_train, X_val, y_train, y_val = train_test_split(X_res, y_res, test_size=0.2, random_state=42)\n\nprint(\"Training set size:\", X_train.shape)\nprint(\"Validation set size:\", X_val.shape)\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score\n\n# Initialize and train the Random Forest Classifier\nmodel = RandomForestClassifier(random_state=42)\nmodel.fit(X_train, y_train)\n\n# Predict on the validation set\ny_pred = model.predict(X_val)\n# Evaluate the model\nprint(\"Accuracy:\", accuracy_score(y_val, y_pred))\nprint(\"\\nClassification Report:\\n\", classification_report(y_val, y_pred))\nprint(\"\\nConfusion Matrix:\\n\", confusion_matrix(y_val, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:22.576681Z","iopub.execute_input":"2024-11-29T07:28:22.577481Z","iopub.status.idle":"2024-11-29T07:28:22.748877Z","shell.execute_reply.started":"2024-11-29T07:28:22.577419Z","shell.execute_reply":"2024-11-29T07:28:22.748004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Feature Importance\nimportances = model.feature_importances_\nindices = np.argsort(importances)[::-1]\nfeature_names = X_train.columns  # Replace with actual feature names if using a DataFrame\n\nplt.figure(figsize=(10, 6))\nplt.title(\"Feature Importances\")\nplt.bar(range(10), importances[indices[:10]], align=\"center\")  # Top 10 features\nplt.xticks(range(10), [feature_names[i] for i in indices[:10]], rotation=90)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:22.750120Z","iopub.execute_input":"2024-11-29T07:28:22.750494Z","iopub.status.idle":"2024-11-29T07:28:23.093682Z","shell.execute_reply.started":"2024-11-29T07:28:22.750441Z","shell.execute_reply":"2024-11-29T07:28:23.092794Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.manifold import TSNE\nimport seaborn as sns\n\n# Reduce data to 2D for visualization\nX_embedded = TSNE(n_components=2, random_state=42).fit_transform(X_res)\n\n# Create a DataFrame for visualization\nimport pandas as pd\ntsne_df = pd.DataFrame({'X1': X_embedded[:, 0], 'X2': X_embedded[:, 1], 'label': y_res})\nsns.scatterplot(data=tsne_df, x='X1', y='X2', hue='label', palette='Set1')\nplt.title(\"t-SNE Visualization of SMOTE Data\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:23.094758Z","iopub.execute_input":"2024-11-29T07:28:23.095015Z","iopub.status.idle":"2024-11-29T07:28:23.674980Z","shell.execute_reply.started":"2024-11-29T07:28:23.094990Z","shell.execute_reply":"2024-11-29T07:28:23.674105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\n\n# Perform 5-fold cross-validation\ncv_scores = cross_val_score(model, X_res, y_res, cv=5, scoring='accuracy')\nprint(\"Cross-Validation Accuracy Scores:\", cv_scores)\nprint(\"Mean CV Accuracy:\", cv_scores.mean())\nimport joblib\n\n# Save the model\njoblib.dump(model, 'random_forest_model.pkl')\nprint(\"Model saved!\")\n\n# Load the model (when needed)\nloaded_model = joblib.load('random_forest_model.pkl')\nfrom sklearn.linear_model import LogisticRegression\n\n# Train logistic regression\nlog_reg = LogisticRegression(random_state=42, max_iter=1000)\nlog_reg.fit(X_train, y_train)\n\n# Predict and evaluate\ny_pred_lr = log_reg.predict(X_val)\nprint(\"Logistic Regression Accuracy:\", accuracy_score(y_val, y_pred_lr))\nprint(\"\\nClassification Report:\\n\", classification_report(y_val, y_pred_lr))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:23.676134Z","iopub.execute_input":"2024-11-29T07:28:23.676391Z","iopub.status.idle":"2024-11-29T07:28:24.536706Z","shell.execute_reply.started":"2024-11-29T07:28:23.676367Z","shell.execute_reply":"2024-11-29T07:28:24.533756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Get feature importance\nimportance = model.feature_importances_\nsorted_indices = np.argsort(importance)[::-1]\ntop_features = sorted_indices[:10]\n\n# Plot\nplt.figure(figsize=(10, 6))\nplt.barh([X_res.columns[i] for i in top_features], importance[top_features])\nplt.xlabel(\"Importance\")\nplt.title(\"Top 10 Feature Importances\")\nplt.gca().invert_yaxis()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:24.537975Z","iopub.execute_input":"2024-11-29T07:28:24.538999Z","iopub.status.idle":"2024-11-29T07:28:24.837345Z","shell.execute_reply.started":"2024-11-29T07:28:24.538943Z","shell.execute_reply":"2024-11-29T07:28:24.836407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_new_data(new_data):\n    # Load the saved model\n    loaded_model = joblib.load('random_forest_model.pkl')\n    \n    # Ensure the input data matches the feature set\n    new_data_preprocessed = preprocess(new_data)  # Apply same pre-processing\n    \n    # Make predictions\n    predictions = loaded_model.predict(new_data_preprocessed)\n    return predictions\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:24.839799Z","iopub.execute_input":"2024-11-29T07:28:24.840065Z","iopub.status.idle":"2024-11-29T07:28:24.844581Z","shell.execute_reply.started":"2024-11-29T07:28:24.840041Z","shell.execute_reply":"2024-11-29T07:28:24.843638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Get feature importance\nimportance = model.feature_importances_\n\n# Sort the indices of the features by their importance\nsorted_indices = np.argsort(importance)[::-1]\n\n# Select top 10 features\ntop_features = sorted_indices[:10]\n\n# Plot the importance of top features\nplt.figure(figsize=(10, 6))\nplt.barh([X_res.columns[i] for i in top_features], importance[top_features])\nplt.xlabel(\"Feature Importance\")\nplt.title(\"Top 10 Feature Importances\")\nplt.gca().invert_yaxis()  # Invert y-axis to display the most important at the top\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:24.845623Z","iopub.execute_input":"2024-11-29T07:28:24.845861Z","iopub.status.idle":"2024-11-29T07:28:25.048881Z","shell.execute_reply.started":"2024-11-29T07:28:24.845838Z","shell.execute_reply":"2024-11-29T07:28:25.048014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# Load your dataset (example)\ndf = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\n\n# Print missing values before handling\nprint(\"Missing values before handling:\")\nprint(df.isnull().sum())\n\n# Handle missing values\n# Fill missing values in 'Values' and 'Value Labels' columns with appropriate default values\ndf['Values'] = df['Values'].fillna('default_value')\ndf['Value Labels'] = df['Value Labels'].fillna('default_value')\n\n# Create the 'id' column with sequential numbers (if not already created)\ndf['id'] = range(1, len(df) + 1)  # Create a new 'id' column with sequential numbers\n\n# Check if the 'id' column exists and handle missing 'id' (though this should not happen as 'id' is newly created)\nif 'id' in df.columns:\n    df.dropna(subset=['id'], inplace=True)  # Drop rows with missing 'id' values\n\n# Ensure y_pred has the same length as df\n# Assume you have a prediction array y_pred\ny_pred = np.random.randint(0, 2, size=len(df))  # Example y_pred array with same length as df\n\n# Ensure y_pred has the same length as df\nif len(y_pred) != len(df):\n    raise ValueError(f\"y_pred length ({len(y_pred)}) does not match number of rows in df ({len(df)}).\")\n\n# Now, let's create the submission DataFrame\nsubmission = pd.DataFrame({\n    'id': df['id'],  # Ensure you are matching the 'id' from your cleaned DataFrame\n    'prediction': y_pred  # Match the predictions\n})\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:25.050064Z","iopub.execute_input":"2024-11-29T07:28:25.050357Z","iopub.status.idle":"2024-11-29T07:28:25.063555Z","shell.execute_reply.started":"2024-11-29T07:28:25.050316Z","shell.execute_reply":"2024-11-29T07:28:25.062758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming you already have the 'y_pred' predictions and 'test_predictions_df'\ntest_predictions_df = pd.DataFrame({'id': df['id'], 'prediction': y_pred})\n\n# Print the first few rows of the test prediction result\nprint(f'Test Prediction: \\n{test_predictions_df.head()}')  # You can change head() to any number you like, e.g., .tail(10)\n\nsubmission.to_csv('submission.csv', index=False)  # Save it without the index\nprint()\n\n# Print the first few rows of the submission for verification\nprint(submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T07:28:25.064600Z","iopub.execute_input":"2024-11-29T07:28:25.064848Z","iopub.status.idle":"2024-11-29T07:28:25.075658Z","shell.execute_reply.started":"2024-11-29T07:28:25.064823Z","shell.execute_reply":"2024-11-29T07:28:25.074830Z"}},"outputs":[],"execution_count":null}]}