{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\n\n\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.model_selection import RandomizedSearchCV\nfrom skopt import BayesSearchCV\nfrom skopt.space import Real, Integer, Categorical\n\n\n\nfrom tqdm.notebook import tqdm\nimport plotly.express as px\nplt.style.use(\"seaborn-colorblind\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-14T16:23:26.455750Z","iopub.execute_input":"2023-05-14T16:23:26.456561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/asl-signs/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = df[df['sign'].isin(['cat', 'bug'])]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EDA : Exploratory Data Analysis** ","metadata":{}},{"cell_type":"code","source":"sub_df.to_csv('sub_df.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of unique signs\nsub_df[\"sign\"].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[\"sign\"].value_counts().head(30).sort_values().plot(\n    kind=\"barh\", figsize=(8, 6), title=\"Top 30 signs of train data\"\n)\nplt.xlabel(\"NO. of training samples\")\nplt.ylabel(\"Signs\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code snippet generates a horizontal bar plot that displays the frequency count of the top 30 sign classes in the dataset. The plot is created using the plot() method from Pandas, with the argument kind=\"barh\" to specify the plot type as a horizontal bar plot. The plot size is specified using figsize=(8, 6). The plot title is set using title=\"Top 30 signs of train data\". The x and y-axis labels are set using plt.xlabel(\"NO. of training samples\") and plt.ylabel(\"Signs\"), respectively. The value_counts() method is used to count the frequency of each unique sign class, head(30) is used to select the top 30 sign classes by frequency, and sort_values() is used to sort the sign classes in ascending order by frequency.","metadata":{}},{"cell_type":"code","source":"sub_df[\"sign\"].value_counts().tail(30).sort_values().plot(\n    kind=\"barh\", figsize=(8, 6), title=\"bottom 30 signs of train data\"\n)\nplt.xlabel(\"NO. of training samples\")\nplt.ylabel(\"Signs\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The second code snippet generates a similar plot, but this time for the bottom 30 sign classes in the dataset. The tail(30) method is used to select the bottom 30 sign classes by frequency.","metadata":{}},{"cell_type":"markdown","source":"Since the bug have less training samples compared to cat, we can do **Data Augmentation** to generate more data !","metadata":{}},{"cell_type":"markdown","source":"# Analysing Single Parquet file\n  **for sign = \"cat\"**","metadata":{}},{"cell_type":"code","source":"p1 = sub_df.query(\"sign == 'cat'\")[\"path\"].iloc[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_dir = \"/kaggle/input/asl-signs/\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p1_file = pd.read_parquet(root_dir + p1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frames = p1_file[\"frame\"]\ntypes = p1_file[\"type\"]\n\nprint(\"frame:\\n\", frames.value_counts())\nprint(f\"this file has {frames.nunique()} unique frames \\n\")\nprint(\"type:\\n\", types.value_counts())\nprint(f\"this file has {types.nunique()} unique types\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code reads the \"frame\" and \"type\" columns from the \"p1_file\" dataframe and prints some information about the unique values and their counts in each column.\n\nThe first line assigns the \"frame\" column to the \"frames\" variable and the \"type\" column to the \"types\" variable.\n\nThe second line prints the counts of each unique value in the \"frames\" column using the \"value_counts()\" method. This provides an overview of how frequently each unique value appears in the \"frames\" column.\n\nThe third line prints the number of unique values in the \"frames\" column using the \"nunique()\" method.\n\nThe fourth line prints the counts of each unique value in the \"types\" column using the \"value_counts()\" method.\n\nThe fifth line prints the number of unique values in the \"types\" column using the \"nunique()\" method.","metadata":{}},{"cell_type":"markdown","source":"The output is showing the number of occurrences of each frame and type in the subset of data that contains only the \"cat\" sign.\n\nFor the \"frame\" column, there are 11 unique frames (frame numbers 21 to 31), and each frame has 543 samples.\n\nFor the \"type\" column, there are 4 unique types: \"face\", \"pose\", \"left_hand\", and \"right_hand\". The \"face\" type is the most frequent with 5148 samples, followed by \"pose\" with 363 samples, \"left_hand\" with 231 samples, and \"right_hand\" with 231 samples.","metadata":{}},{"cell_type":"markdown","source":"The \"frame\" column represents the frame number in the raw video where the landmark data was extracted, and it can be used to identify the time when the sign was performed by the candidate. In this case, the sub dataset only includes landmark data for signs of cats, so the output is showing the number of samples in each of the 11 frames where the cat sign was performed.","metadata":{}},{"cell_type":"markdown","source":"# Metadata for training","metadata":{}},{"cell_type":"markdown","source":"Counting the number of occurrences of each type of hand pose in the selected ASL sign language gesture file (specified by the p1 variable) {for cat and bug}","metadata":{}},{"cell_type":"code","source":"p1_file[\"type\"].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Counting the number of occurrences of each type of hand pose in the selected ASL sign language gesture file that has non-null values for the x, y, and z coordinates of each hand landmark.","metadata":{}},{"cell_type":"code","source":"p1_file.dropna(subset=[\"x\", \"y\", \"z\"])[\"type\"].value_counts()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The following loop  is then used to iterate through each row of the selected subset of the ASL sign language dataset (sub_df) and for each row, the following actions are performed:\n\n1. Reading in the corresponding ASL sign language gesture file specified by the path column of the current row.\n\n2. Counting the number of non-null x, y, and z coordinates for each hand pose type in the current file using the dropna method and creating a dictionary (meta) that stores this information along with the number of unique frames in the file.\n\n3. Calculating summary statistics (minimum, maximum, and mean) for the x, y, and z coordinates of each hand landmark in the current file and storing these statistics in the meta dictionary.\n\n4. Storing the meta dictionary in another dictionary (metadata) with the current file's path as the key.","metadata":{}},{"cell_type":"code","source":"new_data = []\nfor i, d in tqdm(sub_df.iterrows(), total=len(sub_df)):\n    file_path = d[\"path\"]\n    parquet_file = pd.read_parquet(root_dir + file_path)\n    new_data.append(parquet_file)\n\ndf = pd.concat(new_data) # Combine all Parquet files into one DataFrame\ngrouped_data = df.groupby(['type', 'landmark_index']).mean().reset_index() # Groupby 'type' and 'landmark' and calculate mean\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grouped_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = grouped_data[['x', 'y', 'z']]\ny = grouped_data['type']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the number of labeled samples\nn_labeled = int(0.01 * len(grouped_data)) # 1% of the total samples\n\n# Calculate the number of unlabeled samples\nn_unlabeled = len(grouped_data) - n_labeled\n\nindices = np.arange(len(X))\nrng = np.random.RandomState(42)\nrng.shuffle(indices)\n\nX_labeled = X.iloc[indices[:n_labeled]]\ny_labeled = y.iloc[indices[:n_labeled]]\nX_unlabeled = X.iloc[indices[n_labeled:]]\ny_unlabeled = y.iloc[indices[n_labeled:]]\nX_unlabeled = X_unlabeled.reset_index(drop=True)\ny_unlabeled = y_unlabeled.reset_index(drop=True)\n\nn_iterations = 10\nn_samples_per_iter = 10\n\ntrain_accuracy_list = []\ntest_accuracy_list = []\n\n# Split the unlabeled dataset into a test set and a smaller unlabeled set\nX_test, X_unlabeled, y_test, y_unlabeled = train_test_split(X_unlabeled, y_unlabeled, test_size=0.3)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Active Learning usign SVM ","metadata":{}},{"cell_type":"markdown","source":"These code blocks are used to create and train a Support Vector Machine (SVM) classifier. \n\nThe first line creates an instance of an SVM classifier with the \"probability\" parameter set to \"True\". This means that the classifier will be able to output probability estimates for each class in addition to the predicted class.\n\nThe second line trains the classifier using the labeled dataset X_labeled and y_labeled. This is done using the \"fit\" method, which fits the SVM to the data and determines the decision boundary that separates the different classes. Once the SVM is trained, it can be used to make predictions on new, unlabeled data.","metadata":{}},{"cell_type":"code","source":"\n# Create an SVM classifier\nclf = SVC(probability=True)\n\n\n# Train the classifier on the initial labeled set\nclf.fit(X_labeled, y_labeled)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These code blocks are used to create a pipeline that includes a StandardScaler for feature scaling, an SVM classifier with RBF kernel, and L2 regularization. The pipeline is then used for hyperparameter tuning using GridSearchCV.\n\nThe first line creates the pipeline using the \"make_pipeline\" function from Scikit-Learn. The pipeline includes a StandardScaler, which scales the features to have zero mean and unit variance, and an SVM classifier with an RBF kernel. The \"probability\" parameter is set to \"True\" to allow for probability estimates, and the \"class_weight\" parameter is set to \"balanced\" to account for class imbalance. The \"random_state\" parameter is set to 42 for reproducibility.\n\nThe second line sets up the parameter grid for the hyperparameter tuning using GridSearchCV. The \"C\" parameter and \"gamma\" parameter are both varied over a range of values to find the optimal combination of hyperparameters.\n\nThe third line creates a GridSearchCV object, which takes the pipeline and parameter grid as input. The \"cv\" parameter is set to 5 for 5-fold cross-validation. The GridSearchCV object then searches over the specified hyperparameter space using cross-validation to determine the optimal hyperparameters for the SVM classifier in the pipeline.","metadata":{}},{"cell_type":"code","source":"# Create a pipeline with L2 regularization and SVM classifier\npipeline = make_pipeline(StandardScaler(), SVC(kernel='rbf', probability=True, class_weight='balanced', random_state=42))\n\n# Set up the parameter grid for grid search\nparam_grid = {'svc__C': [0.001, 0.01, 0.1, 1, 10, 100],\n              'svc__gamma': [0.001, 0.01, 0.1, 1, 10, 100]}\n\n# Create the grid search object\ngrid_search = GridSearchCV(pipeline, param_grid=param_grid, cv=2)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These code blocks define two functions for measuring uncertainty in a set of predicted probabilities.\n\nThe first function, \"least_confident\", takes a matrix of predicted probabilities as input and returns an array of values that represent the least confident prediction for each instance in the input. This is done by taking the maximum probability for each instance and subtracting it from 1, which gives the probability of the least confident prediction.\n\nThe second function, \"entropy\", also takes a matrix of predicted probabilities as input and returns an array of values that represent the entropy of the predicted probabilities for each instance in the input. This is done by taking the sum of the product of each probability and its logarithm (with a small epsilon added to avoid numerical instability), and negating the result. The intuition behind this measure is that higher entropy indicates greater uncertainty in the predicted probabilities, since the probabilities are more evenly spread across the classes.","metadata":{}},{"cell_type":"code","source":"def least_confident(proba):\n    return 1 - np.max(proba, axis=1)\n\ndef entropy(proba):\n    return -np.sum(proba * np.log2(proba + 1e-10), axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code block defines a function that implements an active learning loop using a specified query strategy. The function takes as input the following:\n\n- The query strategy to use (which should be a function that takes a matrix of predicted probabilities and returns an array of uncertainty scores).\n- The initial labeled set (as X_labeled and y_labeled).\n- The initial unlabeled set (as X_unlabeled and y_unlabeled).\n- The test set (as X_test and y_test).\n- The number of iterations to run (as n_iterations).\n- The number of samples to select per iteration (as n_samples_per_iter).\n\nThe function then iteratively selects the n_samples_per_iter samples from the unlabeled set with the highest uncertainty scores using the specified query strategy. It then adds these samples to the labeled set and retrains the classifier on the updated labeled set. After each iteration, the function calculates the accuracy of the classifier on both the training set and the test set and stores these values in lists for plotting. If there are no more unlabeled samples to select, the function stops early and prints a message.\n\nFinally, the function plots the accuracy values for each iteration of the active learning loop.\n\nIt should be noted that this code block assumes that the grid_search and clf objects have already been defined and that the train_accuracy_list and test_accuracy_list arrays have already been initialized outside of the function.","metadata":{}},{"cell_type":"code","source":"def active_learning_loop(query_strategy, X_labeled, y_labeled, X_unlabeled, y_unlabeled, X_test, y_test, n_iterations, n_samples_per_iter, test_accuracy_list):\n    \n    # Calculate the initial percentage of labeled data\n    initial_percentage_labeled = len(X_labeled) / (len(X_labeled) + len(X_unlabeled))\n\n    for i in range(n_iterations):\n        if len(X_unlabeled) == 0:\n            print(\"No more unlabeled samples to select.\")\n            break\n\n        # Train the classifier on the initial labeled set\n        grid_search.fit(X_labeled, y_labeled)\n\n        # Get the best estimator from the grid search\n        clf = grid_search.best_estimator_\n\n        # Predict the labels and probabilities for the test and unlabeled sets\n        y_pred_test = clf.predict(X_test)\n        y_proba_unlabeled = clf.predict_proba(X_unlabeled)\n\n        # Calculate uncertainty scores using the least confident method\n        uncertainty_scores = 1 - np.max(y_proba_unlabeled, axis=1)\n\n        # Select samples with the highest uncertainty scores\n        selected_indices = np.argsort(-uncertainty_scores)[:n_samples_per_iter]\n\n        # Add selected samples to the labeled set\n        X_labeled = pd.concat([X_labeled, X_unlabeled.iloc[selected_indices]])\n        y_labeled = pd.concat([y_labeled, y_unlabeled.iloc[selected_indices]])\n\n        # Remove selected samples from the unlabeled set\n        X_unlabeled = X_unlabeled.drop(X_unlabeled.index[selected_indices])\n        y_unlabeled = y_unlabeled.drop(y_unlabeled.index[selected_indices])\n        X_unlabeled = X_unlabeled.reset_index(drop=True)\n        y_unlabeled = y_unlabeled.reset_index(drop=True)\n\n        # Retrain the classifier on the updated labeled set\n        clf.fit(X_labeled, y_labeled)\n\n        # Check the performance of the classifier on the training set\n        y_pred_train = clf.predict(X_labeled)\n        train_accuracy = accuracy_score(y_labeled, y_pred_train)\n        train_accuracy_list.append(train_accuracy)\n\n        # Check the performance of the classifier on the test set\n        y_pred_test = clf.predict(X_test)\n        test_accuracy = accuracy_score(y_test, y_pred_test)\n        test_accuracy_list.append(test_accuracy)\n\n        # Calculate the percentage of labeled data\n        percentage_labeled = len(X_labeled) / (len(X_labeled) + len(X_unlabeled))\n\n        print(f\"Iteration {i + 1}: Percentage of Labeled Data = {percentage_labeled:.2%}, Train Accuracy = {train_accuracy:.2f}, Test Accuracy = {test_accuracy:.2f}\")\n\n    if len(X_unlabeled) > 0:\n        print(\"Stopped before all iterations completed: No more unlabeled samples to select.\")\n        \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reset the train_accuracy_list and test_accuracy_list before each call\ntrain_accuracy_list_svm = []\ntest_accuracy_list_svm = []\n\n# Call the active_learning_loop function for each query strategy\nactive_learning_loop('entropy', X_labeled, y_labeled, X_unlabeled, y_unlabeled, X_test, y_test, n_iterations, n_samples_per_iter, test_accuracy_list_svm)\n\ntest_accuracy_list_1_svm = test_accuracy_list_svm.copy()\n\n# Reset the train_accuracy_list and test_accuracy_list before each call\ntrain_accuracy_list_svm = []\ntest_accuracy_list_svm = []\n\nactive_learning_loop('least_confident', X_labeled, y_labeled, X_unlabeled, y_unlabeled, X_test, y_test, n_iterations, n_samples_per_iter, test_accuracy_list_svm)\n\ntest_accuracy_list_2_svm = test_accuracy_list_svm.copy()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate mean and standard deviation for each strategy\nmean_test_accuracy_1 = np.mean(test_accuracy_list_1_svm)\nstd_test_accuracy_1 = np.std(test_accuracy_list_1_svm)\n\nmean_test_accuracy_2 = np.mean(test_accuracy_list_2_svm)\nstd_test_accuracy_2 = np.std(test_accuracy_list_2_svm)\n\n# Create x-axis values for plotting\nx_values = np.linspace(0, 100, num=len(test_accuracy_list_1_svm))\n\n# Create upper and lower bounds for shading\nupper_entropy = mean_test_accuracy_1 + std_test_accuracy_1\nlower_entropy = mean_test_accuracy_1 - std_test_accuracy_1\nupper_least_confident = mean_test_accuracy_2 + std_test_accuracy_2\nlower_least_confident = mean_test_accuracy_2 - std_test_accuracy_2\n\n# Plot mean test accuracies and shade area between upper and lower bounds for each strategy\nplt.plot(x_values, test_accuracy_list_1_svm, label='Entropy')\nplt.fill_between(x_values, lower_entropy, upper_entropy, alpha=0.2)\nplt.plot(x_values, test_accuracy_list_2_svm, label='Least Confident')\nplt.fill_between(x_values, lower_least_confident, upper_least_confident, alpha=0.2)\n\n# Set plot title, axis labels, and legend\nplt.title('Active Learning Test Accuracy Comparison')\nplt.xlabel('Percentage of Labeled Data')\nplt.ylabel('Test Accuracy')\nplt.legend()\n\n# Show plot\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Active Learning with RandomForestClassifier","metadata":{}},{"cell_type":"markdown","source":"This code block creates a `RandomForestClassifier` object and trains it on the initial labeled data `X_labeled` and `y_labeled` using the `fit` method. The `RandomForestClassifier` is an ensemble learning method that combines multiple decision trees to produce a more accurate and robust model. The `random_state` parameter is set to 42 to ensure reproducibility of the results.","metadata":{}},{"cell_type":"code","source":"# Create a Random Forest classifier\nclf = RandomForestClassifier(random_state=42)\n\n\n# Train the classifier on the initial labeled set\nclf.fit(X_labeled, y_labeled)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code block creates a pipeline object that chains two transformers: `StandardScaler` and `RandomForestClassifier` using the `make_pipeline` function. \n\nThe `StandardScaler` is used to standardize the input data by subtracting the mean and dividing by the standard deviation. The `RandomForestClassifier` is an ensemble learning method that combines multiple decision trees to produce a more accurate and robust model. The `random_state` parameter is set to 42 to ensure reproducibility of the results.\n\nThe `param_grid` dictionary specifies the hyperparameters to be tuned using grid search cross-validation. The `n_estimators` hyperparameter controls the number of decision trees in the forest, and the `max_depth` hyperparameter controls the maximum depth of each tree.\n\nThe `GridSearchCV` object is created to perform the grid search with 5-fold cross-validation, and it will search over all combinations of hyperparameters specified in `param_grid`.","metadata":{}},{"cell_type":"code","source":"# Create a pipeline with StandardScaler and Random Forest classifier\npipeline = make_pipeline(StandardScaler(), RandomForestClassifier(random_state=42))\n\n# Set up the parameter grid for grid search\nparam_grid = {'randomforestclassifier__n_estimators': [10, 50, 100, 200, 500],\n              'randomforestclassifier__max_depth': [None, 5, 10, 20, 50]}\n\n# Create the grid search object\ngrid_search = GridSearchCV(pipeline, param_grid=param_grid, cv=2)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we are again with the active learning loop, but this time we replace the SVM classifier with the RandomForestClassifier ","metadata":{}},{"cell_type":"code","source":"def active_learning_loop(query_strategy, X_labeled, y_labeled, X_unlabeled, y_unlabeled, X_test, y_test, n_iterations, n_samples_per_iter, test_accuracy_list):\n    \n    # Calculate the initial percentage of labeled data\n    initial_percentage_labeled = len(X_labeled) / (len(X_labeled) + len(X_unlabeled))\n\n    for i in range(n_iterations):\n        if len(X_unlabeled) == 0:\n            print(\"No more unlabeled samples to select.\")\n            break\n\n        # Train the classifier on the initial labeled set\n        grid_search.fit(X_labeled, y_labeled)\n\n        # Get the best estimator from the grid search\n        clf = grid_search.best_estimator_\n\n        # Predict the labels and probabilities for the test and unlabeled sets\n        y_pred_test = clf.predict(X_test)\n        y_proba_unlabeled = clf.predict_proba(X_unlabeled)\n\n        # Calculate uncertainty scores using the least confident method\n        uncertainty_scores = 1 - np.max(y_proba_unlabeled, axis=1)\n\n        # Select samples with the highest uncertainty scores\n        selected_indices = np.argsort(-uncertainty_scores)[:n_samples_per_iter]\n\n        # Add selected samples to the labeled set\n        X_labeled = pd.concat([X_labeled, X_unlabeled.iloc[selected_indices]])\n        y_labeled = pd.concat([y_labeled, y_unlabeled.iloc[selected_indices]])\n\n        # Remove selected samples from the unlabeled set\n        X_unlabeled = X_unlabeled.drop(X_unlabeled.index[selected_indices])\n        y_unlabeled = y_unlabeled.drop(y_unlabeled.index[selected_indices])\n        X_unlabeled = X_unlabeled.reset_index(drop=True)\n        y_unlabeled = y_unlabeled.reset_index(drop=True)\n\n        # Retrain the classifier on the updated labeled set\n        clf.fit(X_labeled, y_labeled)\n\n        # Check the performance of the classifier on the training set\n        y_pred_train = clf.predict(X_labeled)\n        train_accuracy = accuracy_score(y_labeled, y_pred_train)\n        train_accuracy_list.append(train_accuracy)\n\n        # Check the performance of the classifier on the test set\n        y_pred_test = clf.predict(X_test)\n        test_accuracy = accuracy_score(y_test, y_pred_test)\n        test_accuracy_list.append(test_accuracy)\n\n        # Calculate the percentage of labeled data\n        percentage_labeled = len(X_labeled) / (len(X_labeled) + len(X_unlabeled))\n\n        print(f\"Iteration {i + 1}: Percentage of Labeled Data = {percentage_labeled:.2%}, Train Accuracy = {train_accuracy:.2f}, Test Accuracy = {test_accuracy:.2f}\")\n\n    if len(X_unlabeled) > 0:\n        print(\"Stopped before all iterations completed: No more unlabeled samples to select.\")\n        \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here's the call for the active learning loop for both queries : 'entropy' and 'least_confident'","metadata":{}},{"cell_type":"code","source":"# Reset the train_accuracy_list and test_accuracy_list before each call\ntrain_accuracy_list = []\ntest_accuracy_list = []\n\n# Call the active_learning_loop function for each query strategy\nactive_learning_loop('entropy', X_labeled, y_labeled, X_unlabeled, y_unlabeled, X_test, y_test, n_iterations, n_samples_per_iter, test_accuracy_list)\n\ntest_accuracy_list_1 = test_accuracy_list.copy()\n\n# Reset the train_accuracy_list and test_accuracy_list before each call\ntrain_accuracy_list = []\ntest_accuracy_list = []\n\nactive_learning_loop('least_confident', X_labeled, y_labeled, X_unlabeled, y_unlabeled, X_test, y_test, n_iterations, n_samples_per_iter, test_accuracy_list)\n\ntest_accuracy_list_2 = test_accuracy_list.copy()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Calculate mean and standard deviation for each strategy\nmean_test_accuracy_1 = np.mean(test_accuracy_list_1)\nstd_test_accuracy_1 = np.std(test_accuracy_list_1)\n\nmean_test_accuracy_2 = np.mean(test_accuracy_list_2)\nstd_test_accuracy_2 = np.std(test_accuracy_list_2)\n\n# Create x-axis values for plotting\nx_values = np.linspace(0, 100, num=len(test_accuracy_list_1))\n\n# Create upper and lower bounds for shading\nupper_entropy = mean_test_accuracy_1 + std_test_accuracy_1\nlower_entropy = mean_test_accuracy_1 - std_test_accuracy_1\nupper_least_confident = mean_test_accuracy_2 + std_test_accuracy_2\nlower_least_confident = mean_test_accuracy_2 - std_test_accuracy_2\n\n# Plot mean test accuracies and shade area between upper and lower bounds for each strategy\nplt.plot(x_values, test_accuracy_list_1, label='Entropy')\nplt.fill_between(x_values, lower_entropy, upper_entropy, alpha=0.2)\nplt.plot(x_values, test_accuracy_list_2, label='Least Confident')\nplt.fill_between(x_values, lower_least_confident, upper_least_confident, alpha=0.2)\n\n# Set plot title, axis labels, and legend\nplt.title('Active Learning Test Accuracy Comparison')\nplt.xlabel('Percentage of Labeled Data')\nplt.ylabel('Test Accuracy')\nplt.legend()\n\n# Show plot\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Active Learning with Neural Networks (MLPClassifier)","metadata":{}},{"cell_type":"markdown","source":"**MLPClassifier** is a class in the scikit-learn library used for implementing a Multi-layer Perceptron (MLP) neural network. An MLP is a feedforward artificial neural network that is commonly used for classification tasks. The MLPClassifier allows the user to specify various hyperparameters such as the number of hidden layers, the number of neurons in each hidden layer, the activation function, the solver for weight optimization, and regularization parameters. It uses backpropagation to train the network by adjusting the weights to minimize the error between the predicted and actual outputs. Once trained, the MLP can be used to predict the class labels of new data.","metadata":{}},{"cell_type":"markdown","source":"This is a function definition for an active learning loop that performs iterative model training and query selection to improve the accuracy of a machine learning model. The function takes in several arguments:\n\n- `query_strategy`: a string indicating the method for selecting the most informative samples to label at each iteration. The options are 'least_confident' or 'entropy'.\n- `X_labeled`: a pandas dataframe containing the features of the labeled data.\n- `y_labeled`: a pandas series containing the labels of the labeled data.\n- `X_unlabeled`: a pandas dataframe containing the features of the unlabeled data.\n- `y_unlabeled`: a pandas series containing the labels of the unlabeled data.\n- `X_test`: a pandas dataframe containing the features of the test data.\n- `y_test`: a pandas series containing the labels of the test data.\n- `n_iterations`: an integer indicating the number of iterations to run the active learning loop.\n- `n_samples_per_iter`: an integer indicating the number of samples to select and label at each iteration.\n\nThe function uses a multi-layer perceptron (MLP) classifier with a grid search over hyperparameters to fit the model to the labeled data at each iteration. It then uses the chosen query strategy to select the most informative samples from the unlabeled data and adds them to the labeled data for the next iteration. The accuracy scores of the model on the training and test data are calculated and stored for each iteration, and the function returns the lists of these accuracy scores, as well as the best test accuracy score and hyperparameters found during the iterations.","metadata":{}},{"cell_type":"code","source":"def active_learning_loop(query_strategy, X_labeled, y_labeled, X_unlabeled, y_unlabeled, X_test, y_test, n_iterations, n_samples_per_iter, test_accuracy_list):\n    #train_accuracy_list = []\n    #test_accuracy_list = []\n    best_score = 0\n    best_params = None\n    \n    # Calculate the initial percentage of labeled data\n    initial_percentage_labeled = len(X_labeled) / (len(X_labeled) + len(X_unlabeled))\n    \n    for i in range(n_iterations):\n        clf = MLPClassifier(random_state=1, max_iter=1000)\n        \n        # Define the parameter grid to search over\n        param_grid = {\n            'hidden_layer_sizes': [(32,), (64,), (128,), (32, 16), (64, 32), (128, 64)],\n            'activation': ['relu', 'tanh'],\n            'solver': ['adam', 'sgd'],\n            'alpha': [0.0001, 0.001, 0.01, 0.1],\n            'learning_rate': ['constant', 'adaptive'],\n        }\n\n        # perform a grid search over the hyperparameter grid\n        from sklearn.model_selection import GridSearchCV\n        grid_search = GridSearchCV(clf, param_grid=param_grid, cv=2, n_jobs=-1, verbose=1)\n        grid_search.fit(X_labeled, y_labeled)\n        \n        # get the best model and its test accuracy score\n        clf = grid_search.best_estimator_\n        test_score = clf.score(X_test, y_test)\n        \n        # update the best model if it has the highest test accuracy score so far\n        if test_score > best_score:\n            best_score = test_score\n            best_params = grid_search.best_params_\n\n        # fit the model to the labeled data\n        clf.fit(X_labeled, y_labeled)\n\n        # use the query strategy to select the most informative samples\n        y_proba_unlabeled = clf.predict_proba(X_unlabeled)\n        if query_strategy == 'least_confident':\n            uncertainty_scores = least_confident(y_proba_unlabeled)\n        elif query_strategy == 'entropy':\n            uncertainty_scores = entropy(y_proba_unlabeled)\n        else:\n            raise ValueError(\"Invalid query strategy\")\n\n        selected_indices = np.argsort(-uncertainty_scores)[:n_samples_per_iter]\n\n        # add the selected samples to the labeled data\n        X_labeled = pd.concat([X_labeled, X_unlabeled.iloc[selected_indices]])\n        y_labeled = pd.concat([y_labeled, y_unlabeled.iloc[selected_indices]])\n\n        # remove the selected samples from the unlabeled data\n        X_unlabeled = X_unlabeled.drop(X_unlabeled.index[selected_indices])\n        y_unlabeled = y_unlabeled.drop(y_unlabeled.index[selected_indices])\n\n        # calculate and store the accuracy scores for the training and test data\n        y_pred_train = clf.predict(X_labeled)\n        train_accuracy = accuracy_score(y_labeled, y_pred_train)\n        train_accuracy_list.append(train_accuracy)\n\n        y_pred_test = clf.predict(X_test)\n        test_accuracy = accuracy_score(y_test, y_pred_test)\n        test_accuracy_list.append(test_accuracy)\n        \n        # Calculate the percentage of labeled data\n        percentage_labeled = len(X_labeled) / (len(X_labeled) + len(X_unlabeled))\n\n        # print the iteration number and accuracy scores\n        print(f\"Iteration {i + 1}: Train Accuracy = {train_accuracy:.2f}, Test Accuracy = {test_accuracy:.2f}\")\n\n\n\n    # Return the accuracy lists after all iterations are completed\n    #return train_accuracy_list, test_accuracy_list, best_score, best_params\n\n    if len(X_unlabeled) > 0:\n            print(\"Stopped before all iterations completed: No more unlabeled samples to select.\")\n   ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reset the train_accuracy_list and test_accuracy_list before each call\ntrain_accuracy_list = []\ntest_accuracy_list = []\n\n# Call the active_learning_loop function for each query strategy\nactive_learning_loop('entropy', X_labeled, y_labeled, X_unlabeled, y_unlabeled, X_test, y_test, n_iterations, n_samples_per_iter, test_accuracy_list)\n\ntest_accuracy_list_1 = test_accuracy_list.copy()\n\n# Reset the train_accuracy_list and test_accuracy_list before each call\ntrain_accuracy_list = []\ntest_accuracy_list = []\n\nactive_learning_loop('least_confident', X_labeled, y_labeled, X_unlabeled, y_unlabeled, X_test, y_test, n_iterations, n_samples_per_iter, test_accuracy_list)\n\ntest_accuracy_list_2 = test_accuracy_list.copy()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Calculate mean and standard deviation for each strategy\nmean_test_accuracy_1 = np.mean(test_accuracy_list_1)\nstd_test_accuracy_1 = np.std(test_accuracy_list_1)\n\nmean_test_accuracy_2 = np.mean(test_accuracy_list_2)\nstd_test_accuracy_2 = np.std(test_accuracy_list_2)\n\n# Create x-axis values for plotting\nx_values = np.linspace(0, 100, num=len(test_accuracy_list_1))\n\n# Create upper and lower bounds for shading\nupper_entropy = mean_test_accuracy_1 + std_test_accuracy_1\nlower_entropy = mean_test_accuracy_1 - std_test_accuracy_1\nupper_least_confident = mean_test_accuracy_2 + std_test_accuracy_2\nlower_least_confident = mean_test_accuracy_2 - std_test_accuracy_2\n\n# Plot mean test accuracies and shade area between upper and lower bounds for each strategy\nplt.plot(x_values, test_accuracy_list_1, label='Entropy')\nplt.fill_between(x_values, lower_entropy, upper_entropy, alpha=0.2)\nplt.plot(x_values, test_accuracy_list_2, label='Least Confident')\nplt.fill_between(x_values, lower_least_confident, upper_least_confident, alpha=0.2)\n\n# Set plot title, axis labels, and legend\nplt.title('Active Learning Test Accuracy Comparison')\nplt.xlabel('Percentage of Labeled Data')\nplt.ylabel('Test Accuracy')\nplt.legend()\n\n# Show plot\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}