{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pickle\nimport os\nimport copy\nfrom collections import defaultdict\nimport torch\nimport numpy as np\nimport random\nfrom torchvision import datasets, transforms\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2022-12-12T00:48:22.912964Z","iopub.execute_input":"2022-12-12T00:48:22.913441Z","iopub.status.idle":"2022-12-12T00:48:23.895886Z","shell.execute_reply.started":"2022-12-12T00:48:22.913346Z","shell.execute_reply":"2022-12-12T00:48:23.894844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net(torch.nn.Module):\n    def __init__(self, input_size, hidden_size, output_size):\n        super(Net, self).__init__()\n        \n        self.input_size = input_size\n        self.hidden_size = hidden_size\n        self.output_size = output_size\n        \n        self.i2h = torch.nn.Linear(input_size, hidden_size, bias=False)\n        #self.h2h = torch.nn.Linear(hidden_size, hidden_size)\n        self.h2o = torch.nn.Linear(hidden_size, output_size, bias=False)\n        \n        self.logsoftmax = torch.nn.LogSoftmax()\n        \n    def forward(self, x):\n        x = F.relu(self.i2h(x))\n        #x = F.relu(self.h2h(x))\n        x = self.logsoftmax(self.h2o(x))\n        return x","metadata":{"execution":{"iopub.status.busy":"2022-12-12T00:48:23.89928Z","iopub.execute_input":"2022-12-12T00:48:23.900406Z","iopub.status.idle":"2022-12-12T00:48:23.908348Z","shell.execute_reply.started":"2022-12-12T00:48:23.900364Z","shell.execute_reply":"2022-12-12T00:48:23.90719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(model, criterion, optimizer, x, y):\n    x = Variable(x.flatten(2,3), requires_grad=False)\n    y = Variable(y, requires_grad=False)\n    \n    x = x.to(device)\n    y = y.to(device)\n    # Reset gradient\n    optimizer.zero_grad()\n\n    # Forward\n    fx = model.forward(x)\n    \n    y_pred = torch.zeros((16,10))\n    y_pred[:, :] = fx[:, 0, :]\n    y_pred = y_pred.to(device)\n    loss = criterion(y_pred, y)\n    \n    # Backward\n    loss.backward()\n\n    # Update parameters\n    optimizer.step()\n\n    return loss.item()\n\ndef predict(model, x, y):\n    x = Variable(x.flatten(2,3), requires_grad=False)  \n    x = x.to(device)\n    outputs = model(x)\n    outputs = outputs.to(device)\n    _, predicted = torch.max(outputs.data, 2) #for each output, get the predicted value (torch.max returns (index, value) tuple)\n    predicted = predicted.view(16)\n    predicted = predicted.to(device)\n    y = y.to(device)\n    correct = (predicted == y) #how many predicted values equal the labels\n    return correct.sum().item() \n\ndef accuracy(pruned_model, test_dl, s):\n    correct = 0\n    pruned_model = pruned_model.to(device)\n    for teX, teY in test_dl:\n        teX = teX.to(device)\n        teY= teY.to(device)\n        correct += predict(pruned_model, teX, teY)\n        \n    return 100*correct/(16*len(test_dl))","metadata":{"execution":{"iopub.status.busy":"2022-12-12T00:48:23.909923Z","iopub.execute_input":"2022-12-12T00:48:23.910333Z","iopub.status.idle":"2022-12-12T00:48:23.923315Z","shell.execute_reply.started":"2022-12-12T00:48:23.910298Z","shell.execute_reply":"2022-12-12T00:48:23.922138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.autograd import Variable\nimport torch.nn.functional as F\ndef get_mnist(\n        batch_size=16, binarize=False, download=True, downsample_params=None, net=Net(784, 10000, 10), epochs = 50, learning_rate = 3e-4, momentum = 0.9):\n    trn_kwargs = {\n        'batch_size': batch_size,\n        'num_workers': 0,\n        'shuffle': True,\n    }\n    test_kwargs = {\n        'batch_size': batch_size,\n        'num_workers': 0,\n        'shuffle': False,\n    }\n    transform = transforms.Compose([\n            transforms.ToTensor(),\n            #transforms.Normalize((0.1307,), (0.3081,)),\n            ])\n    trn_ds = datasets.MNIST(\n            root='./data', train=True, download=download, transform=transform)\n    test_ds = datasets.MNIST(\n            root='./data', train=False, download=download, transform=transform)\n    \n    trn_dl = torch.utils.data.DataLoader(trn_ds , **trn_kwargs)\n\n    # test dataset is never modified during downsampling\n    test_dl = torch.utils.data.DataLoader(test_ds, **test_kwargs)\n    \n    #return trn_ds, trn_ds.targets, test_ds, test_ds.targets\n    \n    \n#     epochs = 50\n#     learning_rate = 3e-4\n#     momentum = 0.9\n    \n#     net = Net(784, 10000, 10)\n    net = net.to(device)\n    criterion = torch.nn.NLLLoss()\n    optimizer = torch.optim.SGD(net.parameters(), lr=learning_rate, momentum=momentum)\n    \n#     plot_loss = []\n#     plot_correct = []\n    acc = []\n    for e in range(1, epochs+1):\n        loss = 0.\n        correct = 0.\n        for trX, trY in trn_dl:\n            loss += train(net, criterion, optimizer, trX, trY)\n        for teX, teY in test_dl:\n            correct += predict(net, teX, teY)\n#         plot_loss.append(loss/len(trn_dl))\n#         plot_correct.append(100*correct/(len(test_dl)*batch_size))\n        print(\"Epoch %02d, loss = %f, accuracy = %.2f%%\" % (e, loss / (len(trn_dl)), 100*correct/(len(test_dl)*batch_size)))\n        acc.append(100*correct/(16*len(test_dl)))\n    return net, trn_ds.data.flatten(1,2), trn_ds.targets, test_dl, acc\n","metadata":{"execution":{"iopub.status.busy":"2022-12-12T00:48:23.92615Z","iopub.execute_input":"2022-12-12T00:48:23.92666Z","iopub.status.idle":"2022-12-12T00:48:24.016412Z","shell.execute_reply.started":"2022-12-12T00:48:23.926608Z","shell.execute_reply":"2022-12-12T00:48:24.015444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Loss(torch.nn.Module):\n\n    def __init__(self):\n        super().__init__()\n        \n    def forward(self, x, target):\n        return 0.5*torch.sum((target - x)**2)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T00:48:24.018454Z","iopub.execute_input":"2022-12-12T00:48:24.019044Z","iopub.status.idle":"2022-12-12T00:48:24.024533Z","shell.execute_reply.started":"2022-12-12T00:48:24.019007Z","shell.execute_reply":"2022-12-12T00:48:24.023736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gradDescent(model, full_pruned_weights_model, threek_pruned_model, indexer, data):\n    \n    loss = Loss()\n    optimizer = torch.optim.Adam(full_pruned_weights_model.parameters(), lr=1e-8)\n    \n    for _ in range(50):\n        optimizer.zero_grad()\n\n        pruned_model_output = threek_pruned_model(data)\n        og_model_output = model(data)\n        \n        train_loss = loss(pruned_model_output, og_model_output)     \n        #print(train_loss)\n        \n        train_loss.backward()\n        optimizer.step()\n    \n    full_pruned_weights_model.i2h.weight.data[indexer, :] = threek_pruned_model.i2h.weight.data[:indexer.shape[0],:] \n    full_pruned_weights_model.h2o.weight.data[:, indexer] = threek_pruned_model.h2o.weight.data[:, :indexer.shape[0]]\n    \n    return threek_pruned_model","metadata":{"execution":{"iopub.status.busy":"2022-12-12T00:48:24.026341Z","iopub.execute_input":"2022-12-12T00:48:24.026929Z","iopub.status.idle":"2022-12-12T00:48:24.037532Z","shell.execute_reply.started":"2022-12-12T00:48:24.026723Z","shell.execute_reply":"2022-12-12T00:48:24.036479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cs_prune_layer(model, X_train, results, num_neurons=100, total_iter=10):\n    \"\"\"prunes a two-layer neural network to a fixed number of neurons using\n    CoSAMP algorithm -- any number of output neurons can be present\"\"\"\n\n    #device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    assert model.hidden_size > num_neurons, \"Error: pruned size is larger than dense size.\"\n    X_train = X_train.to(device)\n\n    model = model.to(device)\n    \n    k_pruned_model = Net(input_size=model.input_size, hidden_size=num_neurons, output_size=model.output_size)\n    k_pruned_model = k_pruned_model.to(device)\n    \n    k_pruned_model.i2h.weight.data.fill_(0.00)\n    k_pruned_model.h2o.weight.data.fill_(0.00)\n    \n    threek_pruned_model = Net(input_size=model.input_size, hidden_size=3*num_neurons, output_size=model.output_size)\n    threek_pruned_model = threek_pruned_model.to(device)\n    \n    threek_pruned_model.i2h.weight.data.fill_(0.00)\n    threek_pruned_model.h2o.weight.data.fill_(0.00)\n    \n    full_pruned_weights_model = Net(input_size=model.input_size, hidden_size=model.hidden_size, output_size=model.output_size)\n    full_pruned_weights_model = full_pruned_weights_model.to(device)\n    \n    full_pruned_weights_model.i2h.weight.data = copy.deepcopy(model.i2h.weight.data.cpu())\n    full_pruned_weights_model.h2o.weight.data = copy.deepcopy(model.h2o.weight.data.cpu())\n  \n    k_sparse_indices = set([])\n    residual = None\n    og_model_hidden = None\n    og_model_output = None\n    #eps = 1e-5 # stopping criterion for cosamp iterations\n    iteration = 0\n\n    while iteration < total_iter:\n        # initial residual is just the model output\n        if residual is None:\n            with torch.no_grad():\n                #what does hidden represent in single layer model?\n                og_model_hidden = torch.sum(model.i2h.weight.data, dim=0).squeeze()\n                print(model)\n                print(X_train)\n                og_model_output = model(X_train)\n                residual = copy.deepcopy(og_model_output.cpu())\n                residual = residual.to(device)\n        \n        # perform a cosamp iteration\n        with torch.no_grad():\n            # take max on residual after multiplication so negative weight/act combos are not negated\n            hid_weight = model.h2o.weight.data\n            importance = torch.mm(residual, hid_weight) # B x m (i.e., hid neuron importance over entire batch)\n            importance = torch.sum(importance, dim=0).squeeze() # NOTE: may not be the best choice...\n            imp_idxs = torch.argsort(importance, descending=True)[:2*num_neurons]\n            tmp_imp_neurons = set(imp_idxs.cpu().tolist())\n            bigger_neuron_set = tmp_imp_neurons.union(k_sparse_indices)\n            indexer = torch.LongTensor(sorted(list(bigger_neuron_set))).to(device)\n            \n            #first projection to 3k model\n            threek_pruned_model.i2h.weight.data[:indexer.shape[0],:] = copy.deepcopy(model.i2h.weight.data[indexer, :])\n            threek_pruned_model.h2o.weight.data[:, :indexer.shape[0]] = copy.deepcopy(model.h2o.weight.data[:, indexer])\n        \n        \n        with torch.no_grad():\n            #old_k = copy.deepcopy(k_sparse_indices)\n            hidden_sizes = torch.sum(threek_pruned_model.h2o.weight.data, dim=0).squeeze()#og_model_hidden[indexer]\n            k_sparse_indices = torch.argsort(hidden_sizes, descending=True)[:num_neurons]\n            k_sparse_indices = set(k_sparse_indices.cpu().tolist())\n            k_indexer = torch.LongTensor(list(k_sparse_indices)).to(device)\n            k_pruned_model.i2h.weight.data = threek_pruned_model.i2h.weight.data[k_indexer, :]\n            k_pruned_model.h2o.weight.data = threek_pruned_model.h2o.weight.data[:, k_indexer]\n\n        # compute the new residual\n        with torch.no_grad():\n            pruned_model_output = k_pruned_model(X_train)\n            residual = og_model_output - pruned_model_output\n        iteration += 1\n        frob = float(torch.sum(residual**2)*(10**-7))\n        exp_str = f'{num_neurons}_{\"baseline\"}'\n        results[exp_str].append(frob)\n    return k_pruned_model, k_sparse_indices\n","metadata":{"execution":{"iopub.status.busy":"2022-12-12T00:48:24.039075Z","iopub.execute_input":"2022-12-12T00:48:24.039438Z","iopub.status.idle":"2022-12-12T00:48:24.059504Z","shell.execute_reply.started":"2022-12-12T00:48:24.039401Z","shell.execute_reply":"2022-12-12T00:48:24.057252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cs_prune_layer_grad(model, X_train, results, num_neurons=100, total_iter=10):\n    \"\"\"prunes a two-layer neural network to a fixed number of neurons using\n    CoSAMP algorithm -- any number of output neurons can be present\"\"\"\n\n    #device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    assert model.hidden_size > num_neurons, \"Error: pruned size is larger than dense size.\"\n    X_train = X_train.to(device)\n\n    model = model.to(device)\n    \n    k_pruned_model = Net(input_size=model.input_size, hidden_size=num_neurons, output_size=model.output_size)\n    k_pruned_model = k_pruned_model.to(device)\n    \n    k_pruned_model.i2h.weight.data.fill_(0.00)\n    k_pruned_model.h2o.weight.data.fill_(0.00)\n    \n    threek_pruned_model = Net(input_size=model.input_size, hidden_size=3*num_neurons, output_size=model.output_size)\n    threek_pruned_model = threek_pruned_model.to(device)\n    \n    threek_pruned_model.i2h.weight.data.fill_(0.00)\n    threek_pruned_model.h2o.weight.data.fill_(0.00)\n    \n    full_pruned_weights_model = Net(input_size=model.input_size, hidden_size=model.hidden_size, output_size=model.output_size)\n    full_pruned_weights_model = full_pruned_weights_model.to(device)\n    \n    full_pruned_weights_model.i2h.weight.data = copy.deepcopy(model.i2h.weight.data)\n    full_pruned_weights_model.h2o.weight.data = copy.deepcopy(model.h2o.weight.data)\n  \n    k_sparse_indices = set([])\n    residual = None\n    og_model_hidden = None\n    og_model_output = None\n    #eps = 1e-5 # stopping criterion for cosamp iterations\n    iteration = 0\n\n    while iteration < total_iter:\n        # initial residual is just the model output\n        if residual is None:\n            with torch.no_grad():\n                #what does hidden represent in single layer model?\n                og_model_hidden = torch.sum(model.i2h.weight.data, dim=0).squeeze()\n                og_model_output = model(X_train)\n                residual = copy.deepcopy(og_model_output.cpu())\n                residual = residual.to(device)\n        \n        # perform a cosamp iteration\n        with torch.no_grad():\n            # take max on residual after multiplication so negative weight/act combos are not negated\n            hid_weight = full_pruned_weights_model.h2o.weight.data # n x m (i.e., dims are switched)\n            importance = torch.mm(residual, hid_weight) # B x m (i.e., hid neuron importance over entire batch)\n            importance = torch.sum(importance, dim=0).squeeze() # NOTE: may not be the best choice...\n            imp_idxs = torch.argsort(importance, descending=True)[:2*num_neurons]\n            tmp_imp_neurons = set(imp_idxs.cpu().tolist())\n            bigger_neuron_set = tmp_imp_neurons.union(k_sparse_indices)\n            indexer = torch.LongTensor(sorted(list(bigger_neuron_set))).to(device)\n            \n            #first projection to 3k model\n            \n            threek_pruned_model.i2h.weight.data[:indexer.shape[0],:] = full_pruned_weights_model.i2h.weight.data[indexer, :]\n            threek_pruned_model.h2o.weight.data[:, :indexer.shape[0]] = full_pruned_weights_model.h2o.weight.data[:, indexer]\n          \n        #print(\"iteration: \", iteration)\n        ## update weights here ## Step 3 of algorithm\n        threek_pruned_model = gradDescent(model, full_pruned_weights_model, threek_pruned_model, indexer, X_train)\n        \n        with torch.no_grad():\n            #old_k = copy.deepcopy(k_sparse_indices)\n            hidden_sizes = torch.sum(threek_pruned_model.h2o.weight.data, dim=0).squeeze()#og_model_hidden[indexer]\n            k_sparse_indices = torch.argsort(hidden_sizes, descending=True)[:num_neurons]\n            k_sparse_indices = set(k_sparse_indices.cpu().tolist())\n            k_indexer = torch.LongTensor(list(k_sparse_indices)).to(device)\n            k_pruned_model.i2h.weight.data = threek_pruned_model.i2h.weight.data[k_indexer, :]\n            k_pruned_model.h2o.weight.data = threek_pruned_model.h2o.weight.data[:, k_indexer]\n\n        # compute the new residual\n        with torch.no_grad():\n            pruned_model_output = k_pruned_model(X_train)\n            residual = og_model_output - pruned_model_output\n        iteration += 1\n        frob = float(torch.sum(residual**2)*(10**-7))\n        exp_str = f'{num_neurons}_{\"gradDescent\"}'\n        results[exp_str].append(frob)\n    return k_pruned_model, k_sparse_indices\n","metadata":{"execution":{"iopub.status.busy":"2022-12-12T00:48:24.062216Z","iopub.execute_input":"2022-12-12T00:48:24.06373Z","iopub.status.idle":"2022-12-12T00:48:24.082913Z","shell.execute_reply.started":"2022-12-12T00:48:24.063694Z","shell.execute_reply":"2022-12-12T00:48:24.081873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dense_model, X_train, y_train, test_dl, d_acc = get_mnist(learning_rate=1e-8, epochs=30)\n# X_train = X_train/255\n# torch.save(dense_model, \"dense_model\")\nmodel_path = \"/kaggle/input/mnist-saved-model/dense_model\"\ndense_model, X_train, y_train, test_dl, d_acc = get_mnist(epochs=0)\nX_train = X_train/255\n","metadata":{"execution":{"iopub.status.busy":"2022-12-12T05:54:08.280419Z","iopub.execute_input":"2022-12-12T05:54:08.280843Z","iopub.status.idle":"2022-12-12T05:54:08.356787Z","shell.execute_reply.started":"2022-12-12T05:54:08.280758Z","shell.execute_reply":"2022-12-12T05:54:08.355267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = defaultdict(list)\ngd_results = defaultdict(list)","metadata":{"execution":{"iopub.status.busy":"2022-12-11T20:24:43.275291Z","iopub.execute_input":"2022-12-11T20:24:43.275618Z","iopub.status.idle":"2022-12-11T20:24:43.281052Z","shell.execute_reply.started":"2022-12-11T20:24:43.27559Z","shell.execute_reply":"2022-12-11T20:24:43.279135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dense_model","metadata":{"execution":{"iopub.status.busy":"2022-12-11T20:24:43.282463Z","iopub.execute_input":"2022-12-11T20:24:43.282999Z","iopub.status.idle":"2022-12-11T20:24:43.294189Z","shell.execute_reply.started":"2022-12-11T20:24:43.282965Z","shell.execute_reply":"2022-12-11T20:24:43.293121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iterations = 20\n# s_list=[100, 200, 300]\ns_list = [100, 250, 500, 750, 1000, 1250]\nacc = []\ndense_model = torch.load(model_path)\nfor s in s_list:\n    pruned_model, indexer = cs_prune_layer(dense_model, X_train, results, s, iterations)\n    acc.append(accuracy(pruned_model, test_dl, s))\n    torch.save(pruned_model, f\"pruned_model_{s}\")\n#     new_model, X_train, y_train, test_dl = get_mnist(net=pruned_model)\n#     acc.append(accuracy(pruned_model, test_dl, s))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-12-11T20:24:43.295601Z","iopub.execute_input":"2022-12-11T20:24:43.296075Z","iopub.status.idle":"2022-12-11T20:24:48.7819Z","shell.execute_reply.started":"2022-12-11T20:24:43.296019Z","shell.execute_reply":"2022-12-11T20:24:48.780957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iterations = 20\ns_list = [100, 250, 500, 750, 1000, 1250]\n\nacc = []\ndense_model = torch.load(model_path)\nfor s in s_list:\n    pruned_model, indexer = cs_prune_layer_grad(dense_model, X_train, gd_results, s, iterations)\n    acc.append(accuracy(pruned_model, test_dl, s))\n    torch.save(pruned_model, f\"pruned_model_gd_{s}\")\n#     new_model, X_train, y_train, test_dl = get_mnist(net=pruned_model)\n#     acc.append(accuracy(pruned_model, test_dl, s))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-12-11T20:24:48.783491Z","iopub.execute_input":"2022-12-11T20:24:48.78392Z","iopub.status.idle":"2022-12-11T20:39:40.772935Z","shell.execute_reply.started":"2022-12-11T20:24:48.783878Z","shell.execute_reply":"2022-12-11T20:39:40.77193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_acc = []\nlabels = []\nfor s in s_list:\n    pruned_model = torch.load(f\"pruned_model_{s}\")\n    new_model, X_train, y_train, test_dl, acc = get_mnist(net=pruned_model)\n    labels.append(f\"pruned_model_{s}\")\n    new_acc.append(acc)","metadata":{"execution":{"iopub.status.busy":"2022-12-12T02:01:18.424014Z","iopub.execute_input":"2022-12-12T02:01:18.424405Z","iopub.status.idle":"2022-12-12T02:01:18.530709Z","shell.execute_reply.started":"2022-12-12T02:01:18.424373Z","shell.execute_reply":"2022-12-12T02:01:18.529172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# new_acc = []\nfor s in s_list:\n    pruned_model = torch.load(f\"pruned_model_gd_{s}\")\n    new_model, X_train, y_train, test_dl, acc = get_mnist(net=pruned_model)\n    labels.append(f\"pruned_model_gd_{s}\")\n    new_acc.append(acc)","metadata":{"execution":{"iopub.status.busy":"2022-12-11T21:08:05.353454Z","iopub.execute_input":"2022-12-11T21:08:05.354047Z","iopub.status.idle":"2022-12-11T21:36:31.757245Z","shell.execute_reply.started":"2022-12-11T21:08:05.354009Z","shell.execute_reply":"2022-12-11T21:36:31.756197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results","metadata":{"execution":{"iopub.status.busy":"2022-12-11T21:36:31.758574Z","iopub.execute_input":"2022-12-11T21:36:31.76034Z","iopub.status.idle":"2022-12-11T21:36:31.768679Z","shell.execute_reply.started":"2022-12-11T21:36:31.760302Z","shell.execute_reply":"2022-12-11T21:36:31.767736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfor x, y in zip(new_acc, labels):\n    plt.plot(x, label=y)\n\n    \n# plt.axhline(y=d_acc, color='r', linestyle='-')\nplt.xlabel('Iterations')\nplt.ylabel('accuracy')\nplt.title('Tuning parameters')\n# plt.ylim(70,100)\n# plt.yscale('log')\nplt.legend()\nplt.rcParams[\"figure.figsize\"] = (10,10)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-11T21:36:31.770084Z","iopub.execute_input":"2022-12-11T21:36:31.770608Z","iopub.status.idle":"2022-12-11T21:36:32.020001Z","shell.execute_reply.started":"2022-12-11T21:36:31.770572Z","shell.execute_reply":"2022-12-11T21:36:32.019004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels","metadata":{"execution":{"iopub.status.busy":"2022-12-11T21:52:16.959582Z","iopub.execute_input":"2022-12-11T21:52:16.960049Z","iopub.status.idle":"2022-12-11T21:52:16.9714Z","shell.execute_reply.started":"2022-12-11T21:52:16.960006Z","shell.execute_reply":"2022-12-11T21:52:16.97027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"[a[-1] for a in new_acc]","metadata":{"execution":{"iopub.status.busy":"2022-12-11T21:55:25.761354Z","iopub.execute_input":"2022-12-11T21:55:25.76173Z","iopub.status.idle":"2022-12-11T21:55:25.770686Z","shell.execute_reply.started":"2022-12-11T21:55:25.761698Z","shell.execute_reply":"2022-12-11T21:55:25.769471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(s_list, [a[-1] for a in new_acc[:len(s_list)]], label=\"Pruned_i-spasp\", marker=\"o\")\nplt.plot(s_list, [a[-1] for a in new_acc[len(s_list):]], label=\"Pruned_GD\", marker=\"o\")\n\n# plt.axhline(y=d_acc[-1], color='r', linestyle='-')\nplt.xlabel('Pruned Size')\nplt.ylabel('accuracy')\nplt.title('Tuning parameters')\n# plt.ylim(70,100)\n# plt.yscale('log')\nplt.legend()\nplt.rcParams[\"figure.figsize\"] = (10,10)\nplt.show() ","metadata":{"execution":{"iopub.status.busy":"2022-12-11T21:58:18.372502Z","iopub.execute_input":"2022-12-11T21:58:18.372854Z","iopub.status.idle":"2022-12-11T21:58:18.70123Z","shell.execute_reply.started":"2022-12-11T21:58:18.372825Z","shell.execute_reply":"2022-12-11T21:58:18.699886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfor key in results.keys():\n    plt.plot(results[key], label = key)\nfor key in gd_results.keys():\n    plt.plot(gd_results[key], label = key)\n\nplt.xlabel('Iterations')\nplt.ylabel('Residual')\nplt.title('Tuning parameters')\nplt.yscale('log')\nplt.legend()\nplt.rcParams[\"figure.figsize\"] = (15,10)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-11T21:36:32.021403Z","iopub.execute_input":"2022-12-11T21:36:32.021733Z","iopub.status.idle":"2022-12-11T21:36:32.530252Z","shell.execute_reply.started":"2022-12-11T21:36:32.021699Z","shell.execute_reply":"2022-12-11T21:36:32.527661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}