{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19991,"databundleVersionId":1117522,"sourceType":"competition"},{"sourceId":2378330,"sourceType":"datasetVersion","datasetId":492658}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Support Vector Machine","metadata":{}},{"cell_type":"markdown","source":"**Importing RAPIDS CuML library**\n\nThe RAPIDS library will be used to enable Kaggle notebook acceleration through GPU usage. In particular, the cudf and cuml packages will be used to accomplish this.","metadata":{}},{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n# import matplotlib.pyplot as plt\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:14:05.504968Z","iopub.execute_input":"2023-11-26T23:14:05.505299Z","iopub.status.idle":"2023-11-26T23:14:05.510503Z","shell.execute_reply.started":"2023-11-26T23:14:05.505273Z","shell.execute_reply":"2023-11-26T23:14:05.509445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Setup\n\n# import sys\n# !cp ../input/rapids/rapids.0.18.0 /opt/conda/envs/rapids.tar.gz\n# !cd /opt/conda/envs/ && tar -xzvf rapids.tar.gz > /dev/null\n# sys.path = [\"/opt/conda/envs/rapids/lib/python3.7/site-packages\"] + sys.path\n# sys.path = [\"/opt/conda/envs/rapids/lib/python3.7\"] + sys.path\n# sys.path = [\"/opt/conda/envs/rapids/lib\"] + sys.path \n# !cp /opt/conda/envs/rapids/lib/libxgboost.so /opt/conda/lib/","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:14:05.512387Z","iopub.execute_input":"2023-11-26T23:14:05.512651Z","iopub.status.idle":"2023-11-26T23:14:05.528565Z","shell.execute_reply.started":"2023-11-26T23:14:05.512630Z","shell.execute_reply":"2023-11-26T23:14:05.526688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !conda install -c conda-forge cupy -y","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:14:05.531660Z","iopub.execute_input":"2023-11-26T23:14:05.532562Z","iopub.status.idle":"2023-11-26T23:14:05.538965Z","shell.execute_reply.started":"2023-11-26T23:14:05.532522Z","shell.execute_reply":"2023-11-26T23:14:05.537357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !conda install -c rapidsai -c conda-forge -c nvidia rmm cuda-version=11.8 -y","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:14:05.541713Z","iopub.execute_input":"2023-11-26T23:14:05.542087Z","iopub.status.idle":"2023-11-26T23:14:05.549602Z","shell.execute_reply.started":"2023-11-26T23:14:05.542059Z","shell.execute_reply":"2023-11-26T23:14:05.548690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import cudf\n# import cuml","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:14:05.551410Z","iopub.execute_input":"2023-11-26T23:14:05.552116Z","iopub.status.idle":"2023-11-26T23:14:05.560121Z","shell.execute_reply.started":"2023-11-26T23:14:05.552089Z","shell.execute_reply":"2023-11-26T23:14:05.558910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Function to track memory usage***","metadata":{}},{"cell_type":"code","source":"def mem_usage():\n    pid = os.getpid()\n    py = psutil.Process(pid)\n    memory_use = py.memory_info()[0] / 2. ** 30\n    return 'memory usage: ' + str(np.round(memory_use, 2)) + \" GB\\n\"","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:14:05.561801Z","iopub.execute_input":"2023-11-26T23:14:05.562416Z","iopub.status.idle":"2023-11-26T23:14:05.571876Z","shell.execute_reply.started":"2023-11-26T23:14:05.562390Z","shell.execute_reply":"2023-11-26T23:14:05.569874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Importing SVM and other libarries****","metadata":{}},{"cell_type":"code","source":"# !pip install scikit-image --","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:14:05.573468Z","iopub.execute_input":"2023-11-26T23:14:05.574844Z","iopub.status.idle":"2023-11-26T23:14:05.581811Z","shell.execute_reply.started":"2023-11-26T23:14:05.574788Z","shell.execute_reply":"2023-11-26T23:14:05.580678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import datasets, metrics, svm\nfrom sklearn.metrics import ConfusionMatrixDisplay\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.preprocessing import StandardScaler\n\nimport matplotlib.pyplot as plt\n# import seaborn as sns\nfrom skimage import io, transform\nimport numpy as np\n# import pandas as pd\n\nimport time\nimport psutil\nimport os\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-26T23:14:05.583414Z","iopub.execute_input":"2023-11-26T23:14:05.584090Z","iopub.status.idle":"2023-11-26T23:14:05.595106Z","shell.execute_reply.started":"2023-11-26T23:14:05.584050Z","shell.execute_reply":"2023-11-26T23:14:05.593202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define functions to load data, convert to one-dimensional array (feature vector), then split into training and testing datasets","metadata":{}},{"cell_type":"code","source":"gc.collect();\n\ndataset_dir = \"/kaggle/input/alaska2-image-steganalysis\"\nclasses = [\"all three stego algorithms\", \"JMiPOD\", \"JUNIWARD\", \"UERD\"]\n\n# print(\"some cover images' paths: \", str(image_paths_cover[:10]))\n# print(\"\\nsome JMiPOD images' paths: \", str(image_paths_JMiPOD[:10]))\n# print(\"\\nsome JUNIWARD images' paths: \", str(image_paths_JUNIWRD[:10]))\n# print(\"\\nsome UERD images' paths: \", str(image_paths_UERD[:10]))\n\ndef collectImages(image_paths, images, labels):\n    for image_path in image_paths:\n        image = io.imread(image_path)\n#         image = transform.resize(image, (512, 512), mode='constant')\n        images.append(image)\n        labels.append(\"cover\" if \"Cover\" in image_path else \"stego\")\n        \ndef defineData(stego_class):\n    print(\"Binary classification between Cover and\", stego_class)\n    \n    # # size of each class\n    class_size = 1000\n    \n    # # A list to store images\n    images = []\n\n    # # A list to store labels\n    labels = []\n    \n    # # define paths of each image to be sampled\n    image_paths_cover = [dataset_dir + \"/Cover/\" + image_path for image_path in os.listdir(dataset_dir + \"/Cover/\") if image_path.endswith(\".jpg\")][: class_size]\n    collectImages(image_paths_cover, images, labels)\n\n    if stego_class == \"all three stego algorithms\":\n        image_paths_JMiPOD = [dataset_dir + \"/JMiPOD/\" + image_path for image_path in os.listdir(dataset_dir + \"/JMiPOD/\") if image_path.endswith(\".jpg\")][ : (class_size // 3)]\n        image_paths_JUNIWRD = [dataset_dir + \"/JUNIWARD/\" + image_path for image_path in os.listdir(dataset_dir + \"/JUNIWARD/\") if image_path.endswith(\".jpg\")][(class_size // 3) : (2 * class_size // 3)]\n        image_paths_UERD = [dataset_dir + \"/UERD/\" + image_path for image_path in os.listdir(dataset_dir + \"/UERD/\") if image_path.endswith(\".jpg\")][(2 * class_size // 3) : class_size]\n        \n        collectImages(image_paths_JMiPOD, images, labels)\n        collectImages(image_paths_JUNIWRD, images, labels)\n        collectImages(image_paths_UERD, images, labels)\n    else:\n        image_paths_stego = [dataset_dir + \"/\"+ stego_class + \"/\" + image_path for image_path in os.listdir(dataset_dir + \"/\" + stego_class +\"/\") if image_path.endswith(\".jpg\")][: class_size]\n\n        collectImages(image_paths_stego, images, labels)\n\n    images = np.array(images)\n    labels = np.array(labels)\n\n    n_samples = len(images)\n\n    # Convert to vector\n    data = images.reshape((n_samples, -1))\n    print(\"\\nvector shape:\")\n    print(data.shape)\n\n    # flatten data\n    for element in data:\n        element = element / 255\n    \n    print(mem_usage())\n    return data, labels\n    \n# print(\"\\ntraining sets:\")\n# print(\"x training: \", str(X_train))\n# print(\"y training: \", str(y_train))","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:14:05.599154Z","iopub.execute_input":"2023-11-26T23:14:05.599914Z","iopub.status.idle":"2023-11-26T23:14:05.780482Z","shell.execute_reply.started":"2023-11-26T23:14:05.599871Z","shell.execute_reply":"2023-11-26T23:14:05.779177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's create a SVM model, train it, and then make predictions on the test dataset.**\n\nThe support vector model will run with the creation of multiple support vector classifiers utilizing different kernals to create decision boundries of varying accuracy and results.\n\n* Linear kernel\n* Polynomial kernel\n* RBF kernel","metadata":{}},{"cell_type":"code","source":"# starting time\nstart_time = time.time()\n\n# best accuracy so far: \n#      * linear: ~53% w/ ~50/50 train/test split, random_state = 0\n#      * poly: \ndef create_svc(kernel_in, data, labels, algorithm_class):\n    \n    X_train, X_test, y_train, y_test = train_test_split(data, labels, random_state = 0, test_size=0.3, shuffle=True)\n    X_train = X_train ** 2\n\n    print(\"Finished splitting the data\")\n    print(mem_usage())\n    \n    print(\"Creating SVC from svm\")\n    if kernel_in == 'linear':\n        classifier = svm.SVC(kernel = kernel_in, C=10, random_state = 0)\n    elif kernel_in == 'poly':\n        classifier = svm.SVC(kernel = kernel_in, C=10, random_state = 0, degree = 2, gamma = 'auto')\n    else:\n        sc = StandardScaler();\n        X_train = sc.fit_transform(X_train)\n        X_test = sc.fit_transform(X_test)\n        classifier = svm.SVC(kernel = kernel_in, C=1000, random_state = 1234, gamma = 'auto')\n\n    print(mem_usage())\n\n    print(\"fitting SVC\")\n    model = classifier.fit(X = X_train, y = y_train)\n    print(mem_usage())\n\n    print(\"finished fitting SVC, testing SVC\")\n    prediction = classifier.predict(X_test)\n    print(mem_usage())\n    \n#     print(\"Predicted classes: \", prediction)\n#     print(\"Test classes: \", y_test)\n    print(\"Classification report for linear classifier %s:\\n%s\\n\"\n      % (classifier, metrics.classification_report(y_test, prediction)))\n    \n    np.set_printoptions(precision=2)\n    disp = ConfusionMatrixDisplay.from_estimator(\n        classifier,\n        X_test,\n        y_test,\n        display_labels = [\"cover\", \"stego\"],\n        cmap = plt.cm.Blues,\n        normalize = None,\n    )\n    disp.ax_.set_title(\"Binary classification of images: cover vs \" + algorithm_class)\n\n    print(algorithm_class)\n    print(disp.confusion_matrix)\n    plt.show()\n\n    return model, classifier, prediction\n\nfor algorithm_class in classes:\n    data, labels = defineData(algorithm_class)\n    create_svc('linear', data, labels, algorithm_class)\n\n# testing out hyperparameter tuning\n# param_grid_svm = {\n#     'C': [1, 10, 100],                   \n#     'gamma': [1, 0.1],                \n#     'kernel': ['linear'],\n#     'class_weight': ['balanced']                    \n# }\n# clf = GridSearchCV(estimator = SVC(random_state = 1234, probability = True) , param_grid = param_grid_svm, verbose = 1, cv = 10, n_jobs = -1)\n# clf.fit(X_train, y_train)\n# predicted = clf.predict(X_test)\n\nprint(f\"Total time taken: {(time.time() - start_time):.2f} seconds\")\n ","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:14:05.782301Z","iopub.execute_input":"2023-11-26T23:14:05.782787Z","iopub.status.idle":"2023-11-26T23:16:22.390010Z","shell.execute_reply.started":"2023-11-26T23:14:05.782749Z","shell.execute_reply":"2023-11-26T23:16:22.389175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"SVM classifer visualization","metadata":{}},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# import seaborn as sns\n\n# plt.figure(figsize=(10, 8))\n# # Plotting our two-features-space\n# sns.scatterplot(x=X_train[:, 0], \n#                 y=X_train[:, 1], \n#                 hue=y_train, \n#                 s=8);\n# # Constructing a hyperplane using a formula.\n# w = mdl.coef_[0]           # w consists of 2 elements\n# b = mdl.intercept_[0]      # b consists of 1 element\n# x_points = np.linspace(-1, 1)    # generating x-points from -1 to 1\n# y_points = -(w[0] / w[1]) * x_points - b / w[1]  # getting corresponding y-points\n# # Plotting a red hyperplane\n# plt.plot(x_points, y_points, c='r');","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:16:22.391480Z","iopub.execute_input":"2023-11-26T23:16:22.392743Z","iopub.status.idle":"2023-11-26T23:16:22.396443Z","shell.execute_reply.started":"2023-11-26T23:16:22.392713Z","shell.execute_reply":"2023-11-26T23:16:22.395808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n\n# # create a mesh to plot in\n# x_min, x_max = X_train[:, 0].min() - 1, X_train[:, 0].max() + 1\n# y_min, y_max = X_train[:, 1].min() - 1, X_train[:, 1].max() + 1\n# xx, yy = np.meshgrid(np.arange(x_min, x_max, 0.2),\n#                      np.arange(y_min, y_max, 0.2))\n\n# # title for the plots\n\n\n# # Plot the decision boundary. For that, we will assign a color to each\n# # point in the mesh [x_min, x_max]x[y_min, y_max].\n# plt.subplot(2, 2, 1)\n# plt.subplots_adjust(wspace=0.4, hspace=0.4)\n\n# # Put the result into a color plot\n# predicted = predicted.reshape(xx.shape)\n# plt.contourf(xx, yy, predicted, cmap=plt.cm.coolwarm, alpha=0.8)\n\n# # Plot also the training points\n# plt.scatter(X[:, 0], X[:, 1], c=y, cmap=plt.cm.coolwarm)\n# plt.xlabel('Sepal length')\n# plt.ylabel('Sepal width')\n# plt.xlim(xx.min(), xx.max())\n# plt.ylim(yy.min(), yy.max())\n# plt.xticks(())\n# plt.yticks(())\n# plt.title(titles[i])","metadata":{"execution":{"iopub.status.busy":"2023-11-26T23:16:22.397585Z","iopub.execute_input":"2023-11-26T23:16:22.398429Z","iopub.status.idle":"2023-11-26T23:16:22.419146Z","shell.execute_reply.started":"2023-11-26T23:16:22.398404Z","shell.execute_reply":"2023-11-26T23:16:22.417216Z"},"trusted":true},"execution_count":null,"outputs":[]}]}