{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom keras.callbacks import EarlyStopping, TensorBoard\nfrom keras.layers import Dense\nfrom keras.models import Sequential\nfrom sklearn.metrics import *\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.preprocessing import StandardScaler\n\nfrom tqdm import *\nimport numpy as np\nimport random\nimport pickle as pkl\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-21T05:36:44.444242Z","iopub.execute_input":"2022-01-21T05:36:44.444685Z","iopub.status.idle":"2022-01-21T05:36:46.351649Z","shell.execute_reply.started":"2022-01-21T05:36:44.444594Z","shell.execute_reply":"2022-01-21T05:36:46.350865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset['diagnosis']","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.352998Z","iopub.execute_input":"2022-01-21T05:36:46.353385Z","iopub.status.idle":"2022-01-21T05:36:46.357781Z","shell.execute_reply.started":"2022-01-21T05:36:46.353348Z","shell.execute_reply":"2022-01-21T05:36:46.357154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#np.unique(dataset['diagnosis'])","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.358998Z","iopub.execute_input":"2022-01-21T05:36:46.359439Z","iopub.status.idle":"2022-01-21T05:36:46.362404Z","shell.execute_reply.started":"2022-01-21T05:36:46.359398Z","shell.execute_reply":"2022-01-21T05:36:46.361773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ny_train = dataset['diagnosis']\nx_train = dataset.drop(labels =['diagnosis'],axis=1)\n\nohe = OneHotEncoder()\nle = LabelEncoder()\n\ncols = x_train.columns.values\nfor col in cols:\n    x_train[col] = le.fit_transform(x_train[col])\n\ny_train = le.fit_transform(y_train)\n\nohe = OneHotEncoder()\nx_train = ohe.fit_transform(x_train).toarray()\nsc = StandardScaler()\nx_train = sc.fit_transform(x_train)\n\n\nx_train, x_test, y_train, y_test = train_test_split(x_train,y_train, test_size = 0.99, random_state = 42)\nx_valid, x_test, y_valid, y_test = train_test_split(x_test,y_test, test_size = 0.99, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.365043Z","iopub.execute_input":"2022-01-21T05:36:46.365427Z","iopub.status.idle":"2022-01-21T05:36:46.730023Z","shell.execute_reply.started":"2022-01-21T05:36:46.365396Z","shell.execute_reply":"2022-01-21T05:36:46.729286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train, x_test, y_train, y_test, x_valid, y_valid","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:46:09.000376Z","iopub.execute_input":"2022-01-21T05:46:09.000663Z","iopub.status.idle":"2022-01-21T05:46:09.013397Z","shell.execute_reply.started":"2022-01-21T05:46:09.000633Z","shell.execute_reply":"2022-01-21T05:46:09.012379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def softmax(x):\n    return x/x.sum()\n\ndef relu(x):\n    return np.maximum(0, x)\n\ndef forward(x, w, activation):\n    return activation(np.matmul(x, w))\n\ndef accuracy_fn(y, y_hat):\n    return (np.where(y == y_hat)[0].size / y_hat.size)\n\ndef predict(x, y_hat, weights, activation):\n    predictions = np.zeros(shape=(x.shape[0]))\n    \n    for idx in range(x.shape[0]):\n        r1 = x[idx, :]\n        for curr_weights in weights:\n            r1 = forward(r1, curr_weights, activation)\n        predictions[idx] = np.where(r1 == np.max(r1))[0][0]\n\n    accuracy = accuracy_fn(predictions, y_hat)\n    return accuracy, predictions\n    \ndef fitness(x, y_hat, weights, activation):\n    accuracy = np.empty(shape=(weights.shape[0]))\n    for idx in range(weights.shape[0]):\n        accuracy[idx], _ = predict(x, y_hat, weights[idx, :], activation)\n    return accuracy","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.749986Z","iopub.execute_input":"2022-01-21T05:36:46.750331Z","iopub.status.idle":"2022-01-21T05:36:46.761915Z","shell.execute_reply.started":"2022-01-21T05:36:46.750293Z","shell.execute_reply":"2022-01-21T05:36:46.7611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mat_to_vector(mat_pop_weights):\n    weights_vector = []\n    for idx in range(mat_pop_weights.shape[0]):\n        curr_vector = []\n        for layer_idx in range(mat_pop_weights.shape[1]):\n            vector_weights = np.reshape(mat_pop_weights[idx, layer_idx], newshape=(mat_pop_weights[idx, layer_idx].size))\n            curr_vector.extend(vector_weights)\n        weights_vector.append(curr_vector)\n    return np.array(weights_vector)\n\n\ndef vector_to_mat(vector_weights, mat_pop_weights):\n    mat_weights = []\n    for idx in range(mat_pop_weights.shape[0]):\n        start = 0\n        end = 0\n        for layer_idx in range(mat_pop_weights.shape[1]):\n            end = end + mat_pop_weights[idx, layer_idx].size\n            curr_vector = vector_weights[idx, start:end]\n            mat_layer_weights = np.reshape(curr_vector, newshape=(mat_pop_weights[idx, layer_idx].shape))\n            mat_weights.append(mat_layer_weights)\n            start = end\n    return np.reshape(mat_weights, newshape=mat_pop_weights.shape)","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.763439Z","iopub.execute_input":"2022-01-21T05:36:46.763767Z","iopub.status.idle":"2022-01-21T05:36:46.77467Z","shell.execute_reply.started":"2022-01-21T05:36:46.763732Z","shell.execute_reply":"2022-01-21T05:36:46.77397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mating_pool(pop, fitness, num_parents):\n    parents = np.empty((num_parents, pop.shape[1]))\n    for parent_num in range(num_parents):\n        max_fitness_idx = np.where(fitness == np.max(fitness))\n        max_fitness_idx = max_fitness_idx[0][0]\n        parents[parent_num, :] = pop[max_fitness_idx, :]\n        fitness[max_fitness_idx] = -75\n    return parents\n\ndef crossover(parents, offspring_size):\n    offspring = np.empty(offspring_size)\n    crossover_point = np.uint32(offspring_size[1]/2)\n\n    for k in range(offspring_size[0]):\n        \n        parent1_idx = k%parents.shape[0]\n        parent2_idx = (k+1)%parents.shape[0]\n        \n        offspring[k, 0:crossover_point] = parents[parent1_idx, 0:crossover_point]\n        offspring[k, crossover_point:] = parents[parent2_idx, crossover_point:]\n        \n    return offspring\n\ndef mutation(offspring_crossover, mutation_percent):\n    num_mutations = np.uint32((mutation_percent*offspring_crossover.shape[1]))\n    mutation_indices = np.array(random.sample(range(0, offspring_crossover.shape[1]), num_mutations))\n    \n    for idx in range(offspring_crossover.shape[0]):\n        random_value = np.random.uniform(-1.0, 1.0, 1)\n        offspring_crossover[idx, mutation_indices] = offspring_crossover[idx, mutation_indices] + random_value\n    \n    return offspring_crossover","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.776322Z","iopub.execute_input":"2022-01-21T05:36:46.776521Z","iopub.status.idle":"2022-01-21T05:36:46.789833Z","shell.execute_reply.started":"2022-01-21T05:36:46.776497Z","shell.execute_reply":"2022-01-21T05:36:46.789083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The below params can be changed","metadata":{}},{"cell_type":"code","source":"solution_per_population = 24\nnum_parents_mating = 12\nnum_generations = 100\nmutation_percent = 0.1","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.791112Z","iopub.execute_input":"2022-01-21T05:36:46.791508Z","iopub.status.idle":"2022-01-21T05:36:46.800692Z","shell.execute_reply.started":"2022-01-21T05:36:46.791472Z","shell.execute_reply":"2022-01-21T05:36:46.799933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.802224Z","iopub.execute_input":"2022-01-21T05:36:46.802578Z","iopub.status.idle":"2022-01-21T05:36:46.811585Z","shell.execute_reply.started":"2022-01-21T05:36:46.802539Z","shell.execute_reply":"2022-01-21T05:36:46.810896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(y_train)","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.812683Z","iopub.execute_input":"2022-01-21T05:36:46.813169Z","iopub.status.idle":"2022-01-21T05:36:46.820253Z","shell.execute_reply.started":"2022-01-21T05:36:46.812984Z","shell.execute_reply":"2022-01-21T05:36:46.819051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(y_train, return_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.821556Z","iopub.execute_input":"2022-01-21T05:36:46.822494Z","iopub.status.idle":"2022-01-21T05:36:46.829409Z","shell.execute_reply.started":"2022-01-21T05:36:46.822458Z","shell.execute_reply":"2022-01-21T05:36:46.828356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(dataset['diagnosis'], return_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.833078Z","iopub.execute_input":"2022-01-21T05:36:46.83349Z","iopub.status.idle":"2022-01-21T05:36:46.840884Z","shell.execute_reply.started":"2022-01-21T05:36:46.833463Z","shell.execute_reply":"2022-01-21T05:36:46.839535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_shape = x_train.shape[1]\noutput_shape = len(np.unique(y_train))\nprint(input_shape,output_shape)","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:46.842424Z","iopub.execute_input":"2022-01-21T05:36:46.844265Z","iopub.status.idle":"2022-01-21T05:36:46.850003Z","shell.execute_reply.started":"2022-01-21T05:36:46.844227Z","shell.execute_reply":"2022-01-21T05:36:46.849247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initial_weights = []\nfor curr_sol in np.arange(0, solution_per_population):\n    \n    w1 = np.random.uniform(low=-0.9, high=0.9, size=(input_shape, 128))\n    w2 = np.random.uniform(low=-0.9, high=0.9, size=(128, 64))\n    w3 = np.random.uniform(low=-0.9, high=0.9,size=(64, output_shape))\n\n    initial_weights.append(np.array([w1, w2, w3]))\nprint(initial_weights)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights_mat = np.array(initial_weights)\nweights_vector = mat_to_vector(weights_mat)\n\nbest_outputs = []\naccuracies = np.empty(shape=(num_generations))","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:47.104303Z","iopub.execute_input":"2022-01-21T05:36:47.104544Z","iopub.status.idle":"2022-01-21T05:36:51.057448Z","shell.execute_reply.started":"2022-01-21T05:36:47.104501Z","shell.execute_reply":"2022-01-21T05:36:51.056679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for generation in tqdm(range(num_generations)):\n    weights_mat = vector_to_mat(weights_vector, weights_mat)\n    fit = fitness(x_train, y_train, weights_mat, activation=softmax)\n    accuracies[generation] = fit[0]\n    parents = mating_pool(weights_vector, fit.copy(), num_parents_mating)\n    \n    offspring_crossover = crossover(parents, offspring_size=(weights_vector.shape[0]-parents.shape[0], weights_vector.shape[1]))\n    offspring_mutation = mutation(offspring_crossover, mutation_percent=mutation_percent)\n    \n    weights_vector[0:parents.shape[0], :] = parents\n    weights_vector[parents.shape[0]:, :] = offspring_mutation","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:36:51.058699Z","iopub.execute_input":"2022-01-21T05:36:51.058997Z","iopub.status.idle":"2022-01-21T05:37:10.256835Z","shell.execute_reply.started":"2022-01-21T05:36:51.058959Z","shell.execute_reply":"2022-01-21T05:37:10.256047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights_mat = vector_to_mat(weights_vector, weights_mat)\nbest_weights = weights_mat [0, :]\nacc, predictions = predict(x_train, y_train, best_weights, softmax)\nprint(\"Accuracy of the best solution is : \", acc)","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:37:10.258268Z","iopub.execute_input":"2022-01-21T05:37:10.258721Z","iopub.status.idle":"2022-01-21T05:37:10.97475Z","shell.execute_reply.started":"2022-01-21T05:37:10.258683Z","shell.execute_reply":"2022-01-21T05:37:10.97397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(accuracies)\nplt.xlabel(\"Iteration\", fontsize=15)\nplt.ylabel(\"Fitness\", fontsize=15)\nplt.xticks(np.arange(0, num_generations+1, 100))\nplt.yticks(np.arange(0, 1, 0.1))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:37:10.979072Z","iopub.execute_input":"2022-01-21T05:37:10.979709Z","iopub.status.idle":"2022-01-21T05:37:11.133829Z","shell.execute_reply.started":"2022-01-21T05:37:10.979665Z","shell.execute_reply":"2022-01-21T05:37:11.13312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"confusion_matrix(y_train,y_valid)","metadata":{"execution":{"iopub.status.busy":"2022-01-21T05:37:11.136175Z","iopub.execute_input":"2022-01-21T05:37:11.136628Z","iopub.status.idle":"2022-01-21T05:37:11.409169Z","shell.execute_reply.started":"2022-01-21T05:37:11.136587Z","shell.execute_reply":"2022-01-21T05:37:11.40799Z"},"trusted":true},"execution_count":null,"outputs":[]}]}