{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11403143,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-20T17:07:12.401125Z","iopub.execute_input":"2025-03-20T17:07:12.401491Z","iopub.status.idle":"2025-03-20T17:07:12.951329Z","shell.execute_reply.started":"2025-03-20T17:07:12.401461Z","shell.execute_reply":"2025-03-20T17:07:12.949871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport math\nimport os\nfrom IPython.display import display, HTML\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom sklearn.metrics import pairwise_distances\n\n# Genetic Algorithm imports\nimport random\nimport ipywidgets as widgets\nfrom ipywidgets import interact, interactive, fixed, interact_manual\nimport seaborn as sns\n\n# --- Constants and Parameters ---\nPOPULATION_SIZE = 50\nNUM_GENERATIONS = 20\nMUTATION_RATE = 0.1\n\n# --- Ambulância simulation parameters ---\nBASE_SPEED = 60  # km/h\nMAX_SPEED = 100  # km/h\nGRAVE_PROBLEM_AVOIDANCE_IMPACT = 0.8  # Reduction if a severe issue avoided\n\n# --- Constants ---\nTARGET_EFFICIENCY = 100  # Target efficiency for calculating K\nFIXED_COST = 500\n\ndef calculate_lref_d0(df):\n    df['Lref'] = df['sequence'].apply(len)\n    df['d0'] = df['Lref'].apply(lambda lref: 0.6 * math.sqrt(lref - 0.5) - 2.5)\n    return df\n\n# --- Load Data ---\n# Simulate generic data instead of loading from files\nnum_sequences = 10\ndata = {\n    'target_id': [f'seq_{i}' for i in range(num_sequences)],\n    'sequence': ['AUCG' * (i + 1) for i in range(num_sequences)], # Different sequence lengths\n    'description': ['Generic sequence' for _ in range(num_sequences)]\n}\n\ntest_df = pd.DataFrame(data)\ntest_df = calculate_lref_d0(test_df)\n\n# --- GENETIC ALGORITHM FOR AMBULANCE PARAMETERS ---\ndef create_individual():\n    \"\"\"Creates a random individual (ambulance parameters).\"\"\"\n    return {\n        'speed_factor': random.uniform(0.8, 1.2), # Factor for speed adjustment\n        'route_preference': random.uniform(0.0, 1.0)  # Preference for faster vs. shorter route\n    }\n\ndef calculate_efficiency(individual, distance, patients, time, grave_problem_avoided, fixed_cost):\n    \"\"\"Calculates the efficiency based on the provided equation.\"\"\"\n    # Adjust parameters based on the individual\n    effective_speed = BASE_SPEED * individual['speed_factor']\n    if effective_speed > MAX_SPEED:\n        effective_speed = MAX_SPEED\n\n    # Route adjustment based on 'route_preference' - Simplified here\n    adjusted_distance = distance * (1 - individual['route_preference'] * 0.1)  # Shorter route preference reduces distance\n    adjusted_time = time * (1 + individual['route_preference'] * 0.1)  # Shorter route preference might increase time slightly\n\n    # Calculate the penalization\n    penalization_simulation = GRAVE_PROBLEM_AVOIDANCE_IMPACT if grave_problem_avoided else 0.0\n\n    # Calculate efficiency\n    efficiency = ((patients / adjusted_time) / (adjusted_distance * (1 + penalization_simulation) + fixed_cost))\n\n    return efficiency\n\ndef evaluate_fitness(individual, distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost):\n    \"\"\"Evaluates the fitness of an individual over a set of scenarios.\"\"\"\n    total_efficiency = 0\n    num_scenarios = len(distances)\n\n    for i in range(num_scenarios):\n        distance = distances[i]\n        patients = patients_list[i]\n        time = times_list[i]\n        grave_problem_avoided = grave_problems_avoided_list[i]\n\n        efficiency = calculate_efficiency(individual, distance, patients, time, grave_problem_avoided, fixed_cost)\n        total_efficiency += efficiency\n\n    # Return the average efficiency\n    return total_efficiency / num_scenarios\n\ndef crossover(parent1, parent2):\n    \"\"\"Performs crossover between two parents.\"\"\"\n    child = {}\n    for key in parent1.keys():\n        if random.random() < 0.5:\n            child[key] = parent1[key]\n        else:\n            child[key] = parent2[key]\n    return child\n\ndef mutate(individual):\n    \"\"\"Mutates an individual.\"\"\"\n    for key in individual.keys():\n        if random.random() < MUTATION_RATE:\n            # Adjust the mutation based on the key. For example, we should use a small mutation on factors and larger on route selection.\n            if key == 'speed_factor':\n                individual[key] += random.uniform(-0.1, 0.1) # Small adjustments\n                individual[key] = max(0.8, min(1.2, individual[key]))  # Clamp value.\n            else: # route_preference.\n                individual[key] += random.uniform(-0.2, 0.2)  # Larger adjustment\n                individual[key] = max(0.0, min(1.0, individual[key])) # Clamp value.\n    return individual\n\ndef genetic_algorithm(distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost):\n    \"\"\"Runs the genetic algorithm.\"\"\"\n    # 1. Initialization\n    population = [create_individual() for _ in range(POPULATION_SIZE)]\n    fitness_history = []\n    best_individual_history = []\n\n    # 2. Evolution\n    for generation in range(NUM_GENERATIONS):\n        # a. Evaluate Fitness\n        fitness_scores = [evaluate_fitness(individual, distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost) for individual in population]\n\n        # Check if fitness_scores is empty\n        if not fitness_scores:\n            print(\"Warning: All fitness scores are zero. Stopping the GA.\")\n            return create_individual(), fitness_history, best_individual_history  # Return a default individual\n\n        # Find best individual only if fitness_scores is not empty\n        if fitness_scores:\n            best_individual_index = np.argmax(fitness_scores)\n            best_individual = population[best_individual_index]\n            best_fitness = fitness_scores[best_individual_index]\n            fitness_history.append(best_fitness)\n            best_individual_history.append(best_individual)\n        else:\n            # If fitness_scores is empty, append a default value\n            best_individual = create_individual()\n            best_fitness = 0.0  # Set to a default value\n            fitness_history.append(best_fitness)\n            best_individual_history.append(best_individual)\n\n        # b. Selection (Tournament Selection)\n        selected_indices = random.choices(range(POPULATION_SIZE), weights=fitness_scores, k=POPULATION_SIZE)\n        selected_population = [population[i] for i in selected_indices]\n\n        # c. Crossover\n        offspring = []\n        for i in range(0, POPULATION_SIZE, 2):\n            parent1 = selected_population[i]\n            parent2 = selected_population[i+1] if i+1 < POPULATION_SIZE else selected_population[i]\n            child = crossover(parent1, parent2)\n            offspring.append(child)\n\n        # d. Mutation\n        mutated_offspring = [mutate(child) for child in offspring]\n\n        # Replace the old population with the new offspring\n        population = mutated_offspring\n\n        print(f\"Generation {generation}: Best individual = {best_individual}, Fitness = {best_fitness}\")\n\n    # 3. Return the Best Individual\n    if fitness_scores:\n        best_individual_index = np.argmax(fitness_scores)\n        return population[best_individual_index], fitness_history, best_individual_history\n    else:\n        return create_individual(), fitness_history, best_individual_history\n\n# --- Simulation Scenarios ---\n# Here, we define different scenarios for the ambulance to operate in. These scenarios include\n# the distance traveled, the number of patients attended, the time taken, and whether a grave\n# problem was avoided. This allows us to test the ambulance in different situations.\ndistances = [100, 150, 200, 120]  # Distances in km\npatients_list = [5, 8, 6, 7]  # Number of patients\ntimes_list = [2, 3, 2.5, 2.2]  # Times in hours\ngrave_problems_avoided_list = [True, False, True, False]  # Whether a grave problem was avoided\n\n# --- Run Genetic Algorithm ---\ntry:\n    best_params, fitness_history, best_individual_history = genetic_algorithm(distances, patients_list, times_list, grave_problems_avoided_list, FIXED_COST)\n    print(\"\\nBest Ambulance Parameters:\", best_params)\nexcept ValueError as e:\n    print(f\"Error during genetic algorithm execution: {e}\")\n    best_params = create_individual()  # Use a default individual\n    fitness_history = []\n    best_individual_history = []\n\n\n# --- Calculate K factor ---\n# To calculate the K factor, let's assume an 'average' scenario:\navg_distance = np.mean(distances)\navg_patients = np.mean(patients_list)\navg_time = np.mean(times_list)\navg_grave_problem_avoided = any(grave_problems_avoided_list) # Assume there's at least one.\n\n# Calculate efficiency with the best parameters\navg_efficiency_with_best_params = calculate_efficiency(\n    best_params, avg_distance, avg_patients, avg_time, avg_grave_problem_avoided, FIXED_COST\n)\n\n# Calculate K based on the target efficiency\nK = TARGET_EFFICIENCY / avg_efficiency_with_best_params\n\nprint(\"Calculated K factor:\", K)\n\n# --- PLOT FITNESS HISTORY ---\nif fitness_history:\n    plt.figure(figsize=(10, 6))\n    plt.plot(fitness_history)\n    plt.xlabel(\"Generation\")\n    plt.ylabel(\"Best Fitness (Average Efficiency)\")\n    plt.title(\"Genetic Algorithm: Fitness Over Generations\")\n    plt.grid(True)\n    plt.show()\nelse:\n    print(\"No fitness history to plot (GA might have failed).\")\n\n# --- DYNAMIC SCENARIO VISUALIZATION ---\ndef visualize_scenario(scenario_index):\n    distance = distances[scenario_index]\n    patients = patients_list[scenario_index]\n    time = times_list[scenario_index]\n    grave_problem_avoided = grave_problems_avoided_list[scenario_index]\n\n    efficiency = calculate_efficiency(best_params, distance, patients, time, grave_problem_avoided, FIXED_COST)\n    scaled_efficiency = efficiency * K\n\n    print(f\"Scenario {scenario_index + 1}:\")\n    print(f\"  Distance: {distance} km\")\n    print(f\"  Patients: {patients}\")\n    print(f\"  Time: {time} hours\")\n    print(f\"  Grave problem avoided: {grave_problem_avoided}\")\n\n    print(f\"\\nBest Parameters:\")\n    print(f\"  Speed Factor: {best_params['speed_factor']:.2f}\")\n    print(f\"  Route Preference: {best_params['route_preference']:.2f}\")\n\n    print(f\"\\nEfficiency (Unscaled): {efficiency:.4f}\")\n    print(f\"Efficiency (Scaled - K={K:.2f}): {scaled_efficiency:.2f}\")\n\n    # Create a bar chart for efficiency\n    plt.figure(figsize=(6, 4))\n    plt.bar(['Efficiency'], [scaled_efficiency], color='green')\n    plt.ylabel('Efficiency (Scaled)')\n    plt.title(f'Scenario {scenario_index + 1} Efficiency')\n    plt.ylim(0, 1.2 * TARGET_EFFICIENCY)  # Set y-axis limit slightly above target\n    plt.show()\n\n# Create an interactive widget for scenario selection\nscenario_slider = widgets.IntSlider(\n    value=0,\n    min=0,\n    max=len(distances) - 1,\n    step=1,\n    description='Scenario:',\n    continuous_update=False\n)\n\ninteractive_plot = interactive(visualize_scenario, scenario_index=scenario_slider)\ndisplay(interactive_plot)\n\n# --- TABLE OF BEST INDIVIDUALS OVER GENERATIONS ---\nif best_individual_history:\n    best_individuals_df = pd.DataFrame(best_individual_history)\n    best_individuals_df['Generation'] = range(len(best_individuals_df))\n    best_individuals_df = best_individuals_df[['Generation', 'speed_factor', 'route_preference']]  # Reorder columns\n\n    print(\"\\nTable: Best Individuals Over Generations\")\n    display(HTML(best_individuals_df.to_html(index=False)))\nelse:\n     print(\"No best individuals history to display (GA might have failed).\")\n\n# --- ANALYTICAL EQUATION DISPLAY ---\nprint(\"\\nAnalytical Equation:\")\nequation_text = \"\"\"\nEfficiency = K * [ (Patients / Adjusted Time) / (Adjusted Distance * (1 + Penalization Simulação) + Custo Fixo) ]\nWhere:\n  Adjusted Distance = Distance * (1 - Route Preference * 0.1)\n  Adjusted Time = Time * (1 + Route Preference * 0.1)\n  Penalization Simulação = 0.8 if Grave Problem Avoided else 0.0\n  K = Scaling Factor (to bring efficiency to a meaningful scale)\n\"\"\"\nprint(equation_text)\n\n# --- PREDICTION CODE (Using the optimized parameters from the GA) ---\n# Remaining structure prediction code is unchanged and executed (as per user request).  If\n# this code caused errors or was not desired, you would comment it out here.\nnum_structures = 2\nnp.random.seed(42)\nhelix_radius = 5.0\nhelix_pitch = 3.0\nz_scaling_factor = 1.0\n\ndef generate_helical_structure(sequence_length, structure_number, phase_shift=0):\n    angles = np.linspace(0, 4 * np.pi, sequence_length) + phase_shift\n    random_offsets = np.random.rand(sequence_length) * 1.0\n    x_coords = helix_radius * np.cos(angles) + random_offsets\n    y_coords = helix_radius * np.sin(angles) + random_offsets\n    z_coords = helix_pitch * angles  + z_scaling_factor * structure_number\n    return x_coords, y_coords, z_coords\n\nall_structure_data = []\nresults_data = []\n\nfig = plt.figure(figsize=(12, 8))\nax = fig.add_subplot(111, projection='3d')\nax.set_xlabel(\"X\")\nax.set_ylabel(\"Y\")\nax.set_zlabel(\"Z\")\nax.set_title(\"3D Structures of RNA Sequences\")\n\nfor index, row in test_df.iterrows():  # Use the simulated test_df\n    target_id = row['target_id']\n    sequence = row['sequence']\n    sequence_length = len(sequence)\n\n    x_coords1, y_coords1, z_coords1 = generate_helical_structure(sequence_length, 1)\n    x_coords2, y_coords2, z_coords2 = generate_helical_structure(sequence_length, 2, phase_shift=np.pi)\n\n    ax.plot(x_coords1, y_coords1, z_coords1, c='red', label=f\"{target_id} Helix 1\")\n    ax.plot(x_coords2, y_coords2, z_coords2, c='blue', label=f\"{target_id} Helix 2\")\n\n    all_coords = np.column_stack((np.concatenate([x_coords1, x_coords2]),\n                                   np.concatenate([y_coords1, y_coords2]),\n                                   np.concatenate([z_coords1, z_coords2])))\n\n    distance_matrix = pairwise_distances(all_coords)\n    avg_distance = np.mean(distance_matrix)\n\n    results_data.append([target_id, sequence_length, avg_distance])\n\n    for residue_index in range(sequence_length):\n        resname = sequence[residue_index]\n        resid = residue_index + 1\n        structure_id = f\"{target_id}_1\"\n        all_structure_data.append([structure_id, resname, resid, x_coords1[residue_index], y_coords1[residue_index], z_coords1[residue_index]])\n    for residue_index in range(sequence_length):\n        resname = sequence[residue_index]\n        resid = residue_index + 1\n        structure_id = f\"{target_id}_2\"\n        all_structure_data.append([structure_id, resname, resid, x_coords2[residue_index], y_coords2[residue_index], z_coords2[residue_index]])\n\nax.legend()\nplt.show()\n\nresults_df = pd.DataFrame(results_data, columns=['target_id', 'sequence_length', 'average_distance'])\n\nplt.figure(figsize=(10, 6))\nplt.bar(results_df['target_id'], results_df['average_distance'])\nplt.xlabel(\"Target ID\")\nplt.ylabel(\"Average Distance (Singularity Measure)\")\nplt.title(\"Comparison of Singularity Measures\")\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()\n\ncol_names = ['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1']\n\nsubmission_df = pd.DataFrame(all_structure_data, columns=col_names)\n\nprint(\"Code execution complete.  No submission file created.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T17:07:12.952901Z","iopub.execute_input":"2025-03-20T17:07:12.953354Z","iopub.status.idle":"2025-03-20T17:07:13.946586Z","shell.execute_reply.started":"2025-03-20T17:07:12.953323Z","shell.execute_reply":"2025-03-20T17:07:13.945267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport math\nimport os\nfrom IPython.display import display, HTML\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom sklearn.metrics import pairwise_distances\n\n# Genetic Algorithm imports\nimport random\nimport ipywidgets as widgets\nfrom ipywidgets import interact, interactive, fixed, interact_manual\nimport seaborn as sns\nimport folium\n\n# --- Constants and Parameters ---\nPOPULATION_SIZE = 50\nNUM_GENERATIONS = 20\nMUTATION_RATE = 0.1\n\n# --- Ambulância simulation parameters ---\nBASE_SPEED = 60  # km/h\nMAX_SPEED = 100  # km/h\nGRAVE_PROBLEM_AVOIDANCE_IMPACT = 0.8  # Reduction if a severe issue avoided\n\n# --- Constants ---\nTARGET_EFFICIENCY = 100  # Target efficiency for calculating K\nFIXED_COST = 500\n\n# --- Generic patient data ---\n# Function to generate random patient coordinates and severity\ndef generate_patient_data(num_patients):\n    patient_locations = []\n    for i in range(num_patients):\n        latitude = random.uniform(-90, 90)  # Simulate latitudes\n        longitude = random.uniform(-180, 180)  # Simulate longitudes\n        severity = random.randint(1, 10)  # Severity from 1 to 10\n        patient_locations.append((latitude, longitude, severity))\n    return patient_locations\n\n# --- Functions (unchanged from previous version) ---\ndef calculate_lref_d0(df):\n    df['Lref'] = df['sequence'].apply(len)\n    df['d0'] = df['Lref'].apply(lambda lref: 0.6 * math.sqrt(lref - 0.5) - 2.5)\n    return df\n\ndef create_individual():\n    \"\"\"Creates a random individual (ambulance parameters).\"\"\"\n    return {\n        'speed_factor': random.uniform(0.8, 1.2), # Factor for speed adjustment\n        'route_preference': random.uniform(0.0, 1.0)  # Preference for faster vs. shorter route\n    }\n\ndef calculate_efficiency(individual, distance, patients, time, grave_problem_avoided, fixed_cost):\n    \"\"\"Calculates the efficiency based on the provided equation.\"\"\"\n    # Adjust parameters based on the individual\n    effective_speed = BASE_SPEED * individual['speed_factor']\n    if effective_speed > MAX_SPEED:\n        effective_speed = MAX_SPEED\n\n    # Route adjustment based on 'route_preference' - Simplified here\n    adjusted_distance = distance * (1 - individual['route_preference'] * 0.1)  # Shorter route preference reduces distance\n    adjusted_time = time * (1 + individual['route_preference'] * 0.1)  # Shorter route preference might increase time slightly\n\n    # Calculate the penalization\n    penalization_simulation = GRAVE_PROBLEM_AVOIDANCE_IMPACT if grave_problem_avoided else 0.0\n\n    # Calculate efficiency\n    efficiency = ((patients / adjusted_time) / (adjusted_distance * (1 + penalization_simulation) + fixed_cost))\n\n    return efficiency\n\ndef evaluate_fitness(individual, distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost):\n    \"\"\"Evaluates the fitness of an individual over a set of scenarios.\"\"\"\n    total_efficiency = 0\n    num_scenarios = len(distances)\n\n    for i in range(num_scenarios):\n        distance = distances[i]\n        patients = patients_list[i]\n        time = times_list[i]\n        grave_problem_avoided = grave_problems_avoided_list[i]\n\n        efficiency = calculate_efficiency(individual, distance, patients, time, grave_problem_avoided, fixed_cost)\n        total_efficiency += efficiency\n\n    # Return the average efficiency\n    return total_efficiency / num_scenarios\n\ndef crossover(parent1, parent2):\n    \"\"\"Performs crossover between two parents.\"\"\"\n    child = {}\n    for key in parent1.keys():\n        if random.random() < 0.5:\n            child[key] = parent1[key]\n        else:\n            child[key] = parent2[key]\n    return child\n\ndef mutate(individual):\n    \"\"\"Mutates an individual.\"\"\"\n    for key in individual.keys():\n        if random.random() < MUTATION_RATE:\n            # Adjust the mutation based on the key. For example, we should use a small mutation on factors and larger on route selection.\n            if key == 'speed_factor':\n                individual[key] += random.uniform(-0.1, 0.1) # Small adjustments\n                individual[key] = max(0.8, min(1.2, individual[key]))  # Clamp value.\n            else: # route_preference.\n                individual[key] += random.uniform(-0.2, 0.2)  # Larger adjustment\n                individual[key] = max(0.0, min(1.0, individual[key])) # Clamp value.\n    return individual\n\ndef genetic_algorithm(distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost):\n    \"\"\"Runs the genetic algorithm.\"\"\"\n    # 1. Initialization\n    population = [create_individual() for _ in range(POPULATION_SIZE)]\n    fitness_history = []\n    best_individual_history = []\n\n    # 2. Evolution\n    for generation in range(NUM_GENERATIONS):\n        # a. Evaluate Fitness\n        fitness_scores = [evaluate_fitness(individual, distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost) for individual in population]\n\n        # Check if fitness_scores is empty\n        if not fitness_scores:\n            print(\"Warning: All fitness scores are zero. Stopping the GA.\")\n            return create_individual(), fitness_history, best_individual_history  # Return a default individual\n\n        # Find best individual only if fitness_scores is not empty\n        if fitness_scores:\n            best_individual_index = np.argmax(fitness_scores)\n            best_individual = population[best_individual_index]\n            best_fitness = fitness_scores[best_individual_index]\n            fitness_history.append(best_fitness)\n            best_individual_history.append(best_individual)\n        else:\n            # If fitness_scores is empty, append a default value\n            best_individual = create_individual()\n            best_fitness = 0.0  # Set to a default value\n            fitness_history.append(best_fitness)\n            best_individual_history.append(best_individual)\n\n        # b. Selection (Tournament Selection)\n        selected_indices = random.choices(range(POPULATION_SIZE), weights=fitness_scores, k=POPULATION_SIZE)\n        selected_population = [population[i] for i in selected_indices]\n\n        # c. Crossover\n        offspring = []\n        for i in range(0, POPULATION_SIZE, 2):\n            parent1 = selected_population[i]\n            parent2 = selected_population[i+1] if i+1 < POPULATION_SIZE else selected_population[i]\n            child = crossover(parent1, parent2)\n            offspring.append(child)\n\n        # d. Mutation\n        mutated_offspring = [mutate(child) for child in offspring]\n\n        # Replace the old population with the new offspring\n        population = mutated_offspring\n\n        print(f\"Generation {generation}: Best individual = {best_individual}, Fitness = {best_fitness}\")\n\n    # 3. Return the Best Individual\n    if fitness_scores:\n        best_individual_index = np.argmax(fitness_scores)\n        return population[best_individual_index], fitness_history, best_individual_history\n    else:\n        return create_individual(), fitness_history, best_individual_history\n\n# --- MAIN SCRIPT ---\n\n# --- Simulated Data and Load/Create it\n# Simulate generic data instead of loading from files\nnum_sequences = 10\ndata = {\n    'target_id': [f'seq_{i}' for i in range(num_sequences)],\n    'sequence': ['AUCG' * (i + 1) for i in range(num_sequences)], # Different sequence lengths\n    'description': ['Generic sequence' for _ in range(num_sequences)]\n}\ntest_df = pd.DataFrame(data)\ntest_df = calculate_lref_d0(test_df)\n\n# --- Create Ambulance Scenario Data: Number and Severity of Cases\nnum_patients_per_scenario = [random.randint(3, 10) for _ in range(4)] # Random number of patients per scenario\npatient_data_per_scenario = [generate_patient_data(num_patients) for num_patients in num_patients_per_scenario]\n\n# --- Assign Scenario - Data - Patients, Time to Solve the Grave Problem or not:\n# Make a table and define parameters for each Scenario\nscenarios = [\n    {\"distance\": 100, \"grave_problem_avoided\": True, \"time\": 2},\n    {\"distance\": 150, \"grave_problem_avoided\": False, \"time\": 3},\n    {\"distance\": 200, \"grave_problem_avoided\": True, \"time\": 2.5},\n    {\"distance\": 120, \"grave_problem_avoided\": False, \"time\": 2.2}\n]\ndistances = [scenario[\"distance\"] for scenario in scenarios]\ngrave_problems_avoided_list = [scenario[\"grave_problem_avoided\"] for scenario in scenarios]\ntimes_list = [scenario[\"time\"] for scenario in scenarios]\npatients_list = num_patients_per_scenario\n\n# --- Run Genetic Algorithm ---\ntry:\n    best_params, fitness_history, best_individual_history = genetic_algorithm(distances, patients_list, times_list, grave_problems_avoided_list, FIXED_COST)\n    print(\"\\nBest Ambulance Parameters:\", best_params)\nexcept ValueError as e:\n    print(f\"Error during genetic algorithm execution: {e}\")\n    best_params = create_individual()  # Use a default individual\n    fitness_history = []\n    best_individual_history = []\n\n# --- Calculate K factor ---\n# To calculate the K factor, let's assume an 'average' scenario:\navg_distance = np.mean(distances)\navg_patients = np.mean(patients_list)\navg_time = np.mean(times_list)\navg_grave_problem_avoided = any(grave_problems_avoided_list) # Assume there's at least one.\n\n# Calculate efficiency with the best parameters\navg_efficiency_with_best_params = calculate_efficiency(\n    best_params, avg_distance, avg_patients, avg_time, avg_grave_problem_avoided, FIXED_COST\n)\n\n# Calculate K based on the target efficiency\nK = TARGET_EFFICIENCY / avg_efficiency_with_best_params\n\nprint(\"Calculated K factor:\", K)\n\n# --- Create base map (centered on the mean location) ---\nall_latitudes = [loc[0] for scenario_data in patient_data_per_scenario for loc in scenario_data]\nall_longitudes = [loc[1] for scenario_data in patient_data_per_scenario for loc in scenario_data]\n\nif all_latitudes and all_longitudes:  # Only create map if patient data exists\n    mean_latitude = sum(all_latitudes) / len(all_latitudes)\n    mean_longitude = sum(all_longitudes) / len(all_longitudes)\n\n    m = folium.Map(location=[mean_latitude, mean_longitude], zoom_start=6)\n\n    # Add markers for each patient location for each scenario with color corresponding to the simulation\n    colors = ['red', 'blue', 'green', 'purple']  # Colors for different scenarios\n\n    for i, patient_data in enumerate(patient_data_per_scenario):\n        color = colors[i % len(colors)]  # Get a unique color per scenario\n        for latitude, longitude, severity in patient_data:\n            folium.CircleMarker(\n                location=[latitude, longitude],\n                radius=severity * 2, # Circle size corresponding to severity\n                color=color,\n                fill=True,\n                fill_color=color,\n                fill_opacity=0.4,\n                popup=f'Severity: {severity} - Scenario: {i+1}' # Tooltip with scenario\n            ).add_to(m)\n\n    # Display the map\n    print(\"\\nPatient Location Map:\")\n    display(m)\nelse:\n    print(\"No patient data available to display on the map.\")\n\n# --- Plot Fitness History ---\nif fitness_history:\n    plt.figure(figsize=(10, 6))\n    plt.plot(fitness_history)\n    plt.xlabel(\"Generation\")\n    plt.ylabel(\"Best Fitness (Average Efficiency)\")\n    plt.title(\"Genetic Algorithm: Fitness Over Generations\")\n    plt.grid(True)\n    plt.show()\nelse:\n    print(\"No fitness history to plot (GA might have failed).\")\n\n# --- Create Simulation Table ---\nscenario_data = []\nfor i in range(len(scenarios)):\n    scenario = scenarios[i]\n    efficiency = calculate_efficiency(best_params, distances[i], patients_list[i], times_list[i], grave_problems_avoided_list[i], FIXED_COST)\n    scaled_efficiency = efficiency * K\n    scenario_data.append({\n        \"Scenario\": i + 1,\n        \"Distance (km)\": scenario[\"distance\"],\n        \"Patients\": patients_list[i],\n        \"Time (hours)\": scenario[\"time\"],\n        \"Grave Problem Avoided\": scenario[\"grave_problem_avoided\"],\n        \"Unscaled Efficiency\": efficiency,\n        \"Scaled Efficiency\": scaled_efficiency\n    })\nscenario_df = pd.DataFrame(scenario_data)\n\nprint(\"\\nSimulation Scenario Table:\")\ndisplay(HTML(scenario_df.to_html(index=False)))\n\n# --- ANALYTICAL EQUATION DISPLAY ---\nprint(\"\\nAnalytical Equation:\")\nequation_text = \"\"\"\nEfficiency = K * [ (Patients / Adjusted Time) / (Adjusted Distance * (1 + Penalization Simulação) + Custo Fixo) ]\nWhere:\n  Adjusted Distance = Distance * (1 - Route Preference * 0.1)\n  Adjusted Time = Time * (1 + Route Preference * 0.1)\n  Penalization Simulação = 0.8 if Grave Problem Avoided else 0.0\n  K = Scaling Factor (to bring efficiency to a meaningful scale)\n\"\"\"\nprint(equation_text)\n\n# --- PREDICTION CODE (Using the optimized parameters from the GA) ---\n# Remaining structure prediction code is unchanged and executed (as per user request).  If\n# this code caused errors or was not desired, you would comment it out here.\nnum_structures = 2\nnp.random.seed(42)\nhelix_radius = 5.0\nhelix_pitch = 3.0\nz_scaling_factor = 1.0\n\ndef generate_helical_structure(sequence_length, structure_number, phase_shift=0):\n    angles = np.linspace(0, 4 * np.pi, sequence_length) + phase_shift\n    random_offsets = np.random.rand(sequence_length) * 1.0\n    x_coords = helix_radius * np.cos(angles) + random_offsets\n    y_coords = helix_radius * np.sin(angles) + random_offsets\n    z_coords = helix_pitch * angles  + z_scaling_factor * structure_number\n    return x_coords, y_coords, z_coords\n\nall_structure_data = []\nresults_data = []\n\nfig = plt.figure(figsize=(12, 8))\nax = fig.add_subplot(111, projection='3d')\nax.set_xlabel(\"X\")\nax.set_ylabel(\"Y\")\nax.set_zlabel(\"Z\")\nax.set_title(\"3D Structures of RNA Sequences\")\n\nfor index, row in test_df.iterrows():  # Use the simulated test_df\n    target_id = row['target_id']\n    sequence = row['sequence']\n    sequence_length = len(sequence)\n\n    x_coords1, y_coords1, z_coords1 = generate_helical_structure(sequence_length, 1)\n    x_coords2, y_coords2, z_coords2 = generate_helical_structure(sequence_length, 2, phase_shift=np.pi)\n\n    ax.plot(x_coords1, y_coords1, z_coords1, c='red', label=f\"{target_id} Helix 1\")\n    ax.plot(x_coords2, y_coords2, z_coords2, c='blue', label=f\"{target_id} Helix 2\")\n\n    all_coords = np.column_stack((np.concatenate([x_coords1, x_coords2]),\n                                   np.concatenate([y_coords1, y_coords2]),\n                                   np.concatenate([z_coords1, z_coords2])))\n\n    distance_matrix = pairwise_distances(all_coords)\n    avg_distance = np.mean(distance_matrix)\n\n    results_data.append([target_id, sequence_length, avg_distance])\n\n    for residue_index in range(sequence_length):\n        resname = sequence[residue_index]\n        resid = residue_index + 1\n        structure_id = f\"{target_id}_1\"\n        all_structure_data.append([structure_id, resname, resid, x_coords1[residue_index], y_coords1[residue_index], z_coords1[residue_index]])\n    for residue_index in range(sequence_length):\n        resname = sequence[residue_index]\n        resid = residue_index + 1\n        structure_id = f\"{target_id}_2\"\n        all_structure_data.append([structure_id, resname, resid, x_coords2[residue_index], y_coords2[residue_index], z_coords2[residue_index]])\n\nax.legend()\nplt.show()\n\nresults_df = pd.DataFrame(results_data, columns=['target_id', 'sequence_length', 'average_distance'])\n\nplt.figure(figsize=(10, 6))\nplt.bar(results_df['target_id'], results_df['average_distance'])\nplt.xlabel(\"Target ID\")\nplt.ylabel(\"Average Distance (Singularity Measure)\")\nplt.title(\"Comparison of Singularity Measures\")\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()\n\ncol_names = ['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1']\n\nsubmission_df = pd.DataFrame(all_structure_data, columns=col_names)\n\nprint(\"Code execution complete.  No submission file created.\")\n\n# --- Explanation of Simulation ---\nprint(\"\\nExplanation of Simulation:\")\nexplanation_text = \"\"\"\nThis simulation uses a Genetic Algorithm to optimize the parameters of an ambulance service. The goal is to maximize efficiency, which is defined as the number of patients served per unit of time, divided by the total cost of operation.\n\nThe parameters that are optimized by the Genetic Algorithm are:\n  - Speed Factor: A multiplier for the ambulance's base speed. This represents the ambulance's ability to drive faster.\n  - Route Preference: A value between 0 and 1 that represents the ambulance's preference for taking a faster route vs. a shorter route.\n\nThe scenarios that are simulated represent different situations the ambulance might encounter. These scenarios vary in terms of:\n  - Distance to patients.\n  - Number of patients.\n  - Time to solve a grave problem or not.\n  - Whether a grave problem, such as a traffic jam, was avoided.\n\nThe Genetic Algorithm works by creating a population of random individuals. Each individual represents a set of ambulance parameters. The fitness of each individual is evaluated by simulating the ambulance service's operation with those parameters. The individuals with the highest fitness are selected to reproduce and create the next generation. This process is repeated for a number of generations, until the population converges on a set of parameters that maximizes efficiency.\n\nThe map is a visualization of patients locations. The simulation table shows the parameters of the ambulance service that were found to be optimal, as well as the efficiency that was achieved with those parameters.\n\"\"\"\nprint(explanation_text)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T17:07:13.948939Z","iopub.execute_input":"2025-03-20T17:07:13.949485Z","iopub.status.idle":"2025-03-20T17:07:15.003605Z","shell.execute_reply.started":"2025-03-20T17:07:13.949441Z","shell.execute_reply":"2025-03-20T17:07:15.001615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport math\nimport os\nfrom IPython.display import display, HTML\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom sklearn.metrics import pairwise_distances\n\n# Genetic Algorithm imports\nimport random\nimport ipywidgets as widgets\nfrom ipywidgets import interact, interactive, fixed, interact_manual\nimport seaborn as sns\nimport folium\n\n# --- Constants and Parameters ---\nPOPULATION_SIZE = 50\nNUM_GENERATIONS = 20\nMUTATION_RATE = 0.1\n\n# --- Ambulância simulation parameters ---\nBASE_SPEED = 60  # km/h\nMAX_SPEED = 100  # km/h\nGRAVE_PROBLEM_AVOIDANCE_IMPACT = 0.8  # Reduction if a severe issue avoided\n\n# --- Constants ---\nTARGET_EFFICIENCY = 100  # Target efficiency for calculating K\nFIXED_COST = 500\n\n# --- Hydrogen Propulsion Constants ---\nHYDROGEN_EFFICIENCY_FACTOR = 2  # Multiplier for the hydrogen propulsion calculation\n\n# --- Generic patient data ---\n# Function to generate random patient coordinates and severity\ndef generate_patient_data(num_patients):\n    patient_locations = []\n    for i in range(num_patients):\n        latitude = random.uniform(-90, 90)  # Simulate latitudes\n        longitude = random.uniform(-180, 180)  # Simulate longitudes\n        severity = random.randint(1, 10)  # Severity from 1 to 10\n        patient_locations.append((latitude, longitude, severity))\n    return patient_locations\n\n# --- Functions (unchanged from previous version) ---\ndef calculate_lref_d0(df):\n    df['Lref'] = df['sequence'].apply(len)\n    df['d0'] = df['Lref'].apply(lambda lref: 0.6 * math.sqrt(lref - 0.5) - 2.5)\n    return df\n\ndef create_individual():\n    \"\"\"Creates a random individual (ambulance parameters).\"\"\"\n    return {\n        'speed_factor': random.uniform(0.8, 1.2), # Factor for speed adjustment\n        'route_preference': random.uniform(0.0, 1.0),  # Preference for faster vs. shorter route\n        'hydrogen_efficiency': random.uniform(0.8, 1.2),  # Factor for hydrogen propulsion efficiency\n        'hydrogen_station_distance_preference': random.uniform(0.1,100.0) # The R parameter, for distance to the hydrogen station\n    }\n\ndef calculate_hydrogen_propulsion(speed, R,individual):\n    \"\"\"Calculates hydrogen propulsion efficiency.\"\"\"\n    x = speed  # use speed of ambulance as x for efficiency optimization calculation\n    eltze = 10 # Simulate the fixed value for variable eltze\n\n    hydrogen_propulsion = HYDROGEN_EFFICIENCY_FACTOR * (2 * (3 * (x**2)) + ((-21**9) + 3) / (R**2)) / (eltze + 36)\n    return hydrogen_propulsion\n\ndef calculate_efficiency(individual, distance, patients, time, grave_problem_avoided, fixed_cost):\n    \"\"\"Calculates the efficiency, incorporating hydrogen propulsion.\"\"\"\n    # Adjust parameters based on the individual\n    effective_speed = BASE_SPEED * individual['speed_factor']\n    if effective_speed > MAX_SPEED:\n        effective_speed = MAX_SPEED\n\n    # Route adjustment based on 'route_preference' - Simplified here\n    adjusted_distance = distance * (1 - individual['route_preference'] * 0.1)  # Shorter route preference reduces distance\n    adjusted_time = time * (1 + individual['route_preference'] * 0.1)  # Shorter route preference might increase time slightly\n\n    # Calculate the penalization\n    penalization_simulation = GRAVE_PROBLEM_AVOIDANCE_IMPACT if grave_problem_avoided else 0.0\n\n    # Calculate hydrogen propulsion\n    hydrogen_propulsion = calculate_hydrogen_propulsion(effective_speed, individual['hydrogen_station_distance_preference'],individual) * individual['hydrogen_efficiency']\n\n    # Incorporate hydrogen propulsion into efficiency calculation (Example: Additively)\n    efficiency = ((patients / adjusted_time) / (adjusted_distance * (1 + penalization_simulation) + fixed_cost)) + hydrogen_propulsion\n\n    return efficiency\n\ndef evaluate_fitness(individual, distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost):\n    \"\"\"Evaluates the fitness of an individual over a set of scenarios.\"\"\"\n    total_efficiency = 0\n    num_scenarios = len(distances)\n\n    for i in range(num_scenarios):\n        distance = distances[i]\n        patients = patients_list[i]\n        time = times_list[i]\n        grave_problem_avoided = grave_problems_avoided_list[i]\n\n        efficiency = calculate_efficiency(individual, distance, patients, time, grave_problem_avoided, fixed_cost)\n        total_efficiency += efficiency\n\n    # Return the average efficiency\n    return total_efficiency / num_scenarios\n\ndef crossover(parent1, parent2):\n    \"\"\"Performs crossover between two parents.\"\"\"\n    child = {}\n    for key in parent1.keys():\n        if random.random() < 0.5:\n            child[key] = parent1[key]\n        else:\n            child[key] = parent2[key]\n    return child\n\ndef mutate(individual):\n    \"\"\"Mutates an individual.\"\"\"\n    for key in individual.keys():\n        if random.random() < MUTATION_RATE:\n            # Adjust the mutation based on the key. For example, we should use a small mutation on factors and larger on route selection.\n            if key == 'speed_factor':\n                individual[key] += random.uniform(-0.1, 0.1) # Small adjustments\n                individual[key] = max(0.8, min(1.2, individual[key]))  # Clamp value.\n            elif key == 'route_preference':\n                individual[key] += random.uniform(-0.2, 0.2)  # Larger adjustment\n                individual[key] = max(0.0, min(1.0, individual[key])) # Clamp value.\n            elif key == 'hydrogen_efficiency':\n                individual[key] += random.uniform(-0.1, 0.1) # Small adjustments\n                individual[key] = max(0.8, min(1.2, individual[key]))\n            else: # hydrogen_station_distance_preference, R parameter in propulsion calculation\n                individual[key] += random.uniform(-10, 10) # Adjust R for the hydrogen propulsion calculation\n                individual[key] = max(0.1, min(100.0, individual[key]))  #R must be positive.\n    return individual\n\ndef genetic_algorithm(distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost):\n    \"\"\"Runs the genetic algorithm.\"\"\"\n    # 1. Initialization\n    population = [create_individual() for _ in range(POPULATION_SIZE)]\n    fitness_history = []\n    best_individual_history = []\n\n    # 2. Evolution\n    for generation in range(NUM_GENERATIONS):\n        # a. Evaluate Fitness\n        fitness_scores = [evaluate_fitness(individual, distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost) for individual in population]\n\n        # Check if fitness_scores is empty\n        if not fitness_scores:\n            print(\"Warning: All fitness scores are zero. Stopping the GA.\")\n            return create_individual(), fitness_history, best_individual_history  # Return a default individual\n\n        # Find best individual only if fitness_scores is not empty\n        if fitness_scores:\n            best_individual_index = np.argmax(fitness_scores)\n            best_individual = population[best_individual_index]\n            best_fitness = fitness_scores[best_individual_index]\n            fitness_history.append(best_fitness)\n            best_individual_history.append(best_individual)\n        else:\n            # If fitness_scores is empty, append a default value\n            best_individual = create_individual()\n            best_fitness = 0.0  # Set to a default value\n            fitness_history.append(best_fitness)\n            best_individual_history.append(best_individual)\n\n        # b. Selection (Tournament Selection)\n        selected_indices = random.choices(range(POPULATION_SIZE), weights=fitness_scores, k=POPULATION_SIZE)\n        selected_population = [population[i] for i in selected_indices]\n\n        # c. Crossover\n        offspring = []\n        for i in range(0, POPULATION_SIZE, 2):\n            parent1 = selected_population[i]\n            parent2 = selected_population[i+1] if i+1 < POPULATION_SIZE else selected_population[i]\n            child = crossover(parent1, parent2)\n            offspring.append(child)\n\n        # d. Mutation\n        mutated_offspring = [mutate(child) for child in offspring]\n\n        # Replace the old population with the new offspring\n        population = mutated_offspring\n\n        print(f\"Generation {generation}: Best individual = {best_individual}, Fitness = {best_fitness}\")\n\n    # 3. Return the Best Individual\n    if fitness_scores:\n        best_individual_index = np.argmax(fitness_scores)\n        return population[best_individual_index], fitness_history, best_individual_history\n    else:\n        return create_individual(), fitness_history, best_individual_history\n\n# --- MAIN SCRIPT ---\n\n# --- Simulated Data and Load/Create it\n# Simulate generic data instead of loading from files\nnum_sequences = 10\ndata = {\n    'target_id': [f'seq_{i}' for i in range(num_sequences)],\n    'sequence': ['AUCG' * (i + 1) for i in range(num_sequences)], # Different sequence lengths\n    'description': ['Generic sequence' for _ in range(num_sequences)]\n}\ntest_df = pd.DataFrame(data)\ntest_df = calculate_lref_d0(test_df)\n\n# --- Create Ambulance Scenario Data: Number and Severity of Cases\nnum_patients_per_scenario = [random.randint(3, 10) for _ in range(4)] # Random number of patients per scenario\npatient_data_per_scenario = [generate_patient_data(num_patients) for num_patients in num_patients_per_scenario]\n\n# --- Assign Scenario - Data - Patients, Time to Solve the Grave Problem or not:\n# Make a table and define parameters for each Scenario\nscenarios = [\n    {\"distance\": 100, \"grave_problem_avoided\": True, \"time\": 2},\n    {\"distance\": 150, \"grave_problem_avoided\": False, \"time\": 3},\n    {\"distance\": 200, \"grave_problem_avoided\": True, \"time\": 2.5},\n    {\"distance\": 120, \"grave_problem_avoided\": False, \"time\": 2.2}\n]\ndistances = [scenario[\"distance\"] for scenario in scenarios]\ngrave_problems_avoided_list = [scenario[\"grave_problem_avoided\"] for scenario in scenarios]\ntimes_list = [scenario[\"time\"] for scenario in scenarios]\npatients_list = num_patients_per_scenario\n\n# --- Run Genetic Algorithm ---\ntry:\n    best_params, fitness_history, best_individual_history = genetic_algorithm(distances, patients_list, times_list, grave_problems_avoided_list, FIXED_COST)\n    print(\"\\nBest Ambulance Parameters:\", best_params)\nexcept ValueError as e:\n    print(f\"Error during genetic algorithm execution: {e}\")\n    best_params = create_individual()  # Use a default individual\n    fitness_history = []\n    best_individual_history = []\n\n# --- Calculate K factor ---\n# To calculate the K factor, let's assume an 'average' scenario:\navg_distance = np.mean(distances)\navg_patients = np.mean(patients_list)\navg_time = np.mean(times_list)\navg_grave_problem_avoided = any(grave_problems_avoided_list) # Assume there's at least one.\n\n# Calculate efficiency with the best parameters\navg_efficiency_with_best_params = calculate_efficiency(\n    best_params, avg_distance, avg_patients, avg_time, avg_grave_problem_avoided, FIXED_COST\n)\n\n# Calculate K based on the target efficiency\nK = TARGET_EFFICIENCY / avg_efficiency_with_best_params\n\nprint(\"Calculated K factor:\", K)\n\n# --- Create base map (centered on the world) ---\nm = folium.Map(location=[0, 0], zoom_start=2)  # Center on 0,0 and zoom out\n\n# Add markers for each patient location for each scenario with color corresponding to the simulation\ncolors = ['red', 'blue', 'green', 'purple']  # Colors for different scenarios\n\nfor i, patient_data in enumerate(patient_data_per_scenario):\n    color = colors[i % len(colors)]  # Get a unique color per scenario\n    for latitude, longitude, severity in patient_data:\n        folium.CircleMarker(\n            location=[latitude, longitude],\n            radius=severity * 2, # Circle size corresponding to severity\n            color=color,\n            fill=True,\n            fill_color=color,\n            fill_opacity=0.4,\n            popup=f'Severity: {severity} - Scenario: {i+1}' # Tooltip with scenario\n        ).add_to(m)\n\n# Display the map\nprint(\"\\nPatient Location Map:\")\ndisplay(m)\n\n# --- Plot Fitness History ---\nif fitness_history:\n    plt.figure(figsize=(10, 6))\n    plt.plot(fitness_history)\n    plt.xlabel(\"Generation\")\n    plt.ylabel(\"Best Fitness (Average Efficiency)\")\n    plt.title(\"Genetic Algorithm: Fitness Over Generations\")\n    plt.grid(True)\n    plt.show()\nelse:\n    print(\"No fitness history to plot (GA might have failed).\")\n\n# --- Create Simulation Table ---\nscenario_data = []\nfor i in range(len(scenarios)):\n    scenario = scenarios[i]\n    efficiency = calculate_efficiency(best_params, distances[i], patients_list[i], times_list[i], grave_problems_avoided_list[i], FIXED_COST)\n    scaled_efficiency = efficiency * K\n    scenario_data.append({\n        \"Scenario\": i + 1,\n        \"Distance (km)\": scenario[\"distance\"],\n        \"Patients\": patients_list[i],\n        \"Time (hours)\": scenario[\"time\"],\n        \"Grave Problem Avoided\": scenario[\"grave_problem_avoided\"],\n        \"Unscaled Efficiency\": efficiency,\n        \"Scaled Efficiency\": scaled_efficiency\n    })\nscenario_df = pd.DataFrame(scenario_data)\n\nprint(\"\\nSimulation Scenario Table:\")\ndisplay(HTML(scenario_df.to_html(index=False)))\n\n# --- ANALYTICAL EQUATION DISPLAY ---\nprint(\"\\nAnalytical Equation:\")\nequation_text = \"\"\"\nOverall Efficiency = Scaled Service Efficiency + Hydrogen Propulsion Efficiency\nWhere:\n\nService Efficiency = K * [ (Patients / Adjusted Time) / (Adjusted Distance * (1 + Penalization Simulação) + Custo Fixo) ]\n  Adjusted Distance = Distance * (1 - Route Preference * 0.1)\n  Adjusted Time = Time * (1 + Route Preference * 0.1)\n  Penalization Simulação = 0.8 if Grave Problem Avoided else 0.0\n  K = Scaling Factor (to bring efficiency to a meaningful scale)\n\nHydrogen Propulsion Efficiency = HYDROGEN_EFFICIENCY_FACTOR * (2 * (3 * (x**2)) + ((-21**9) + 3) / (R**2)) / (eltze + 36)\n  Where:\n    x = Speed of the ambulance\n    R = Individual value for distance to station\n    eltze = a fixed parameter (10, by default in this simulation)\n\"\"\"\nprint(equation_text)\n\n# --- PREDICTION CODE (Using the optimized parameters from the GA) ---\n# Remaining structure prediction code is unchanged and executed (as per user request).  If\n# this code caused errors or was not desired, you would comment it out here.\nnum_structures = 2\nnp.random.seed(42)\nhelix_radius = 5.0\nhelix_pitch = 3.0\nz_scaling_factor = 1.0\n\ndef generate_helical_structure(sequence_length, structure_number, phase_shift=0):\n    angles = np.linspace(0, 4 * np.pi, sequence_length) + phase_shift\n    random_offsets = np.random.rand(sequence_length) * 1.0\n    x_coords = helix_radius * np.cos(angles) + random_offsets\n    y_coords = helix_radius * np.sin(angles) + random_offsets\n    z_coords = helix_pitch * angles  + z_scaling_factor * structure_number\n    return x_coords, y_coords, z_coords\n\nall_structure_data = []\nresults_data = []\n\nfig = plt.figure(figsize=(12, 8))\nax = fig.add_subplot(111, projection='3d')\nax.set_xlabel(\"X\")\nax.set_ylabel(\"Y\")\nax.set_zlabel(\"Z\")\nax.set_title(\"3D Structures of RNA Sequences\")\n\nfor index, row in test_df.iterrows():  # Use the simulated test_df\n    target_id = row['target_id']\n    sequence = row['sequence']\n    sequence_length = len(sequence)\n\n    x_coords1, y_coords1, z_coords1 = generate_helical_structure(sequence_length, 1)\n    x_coords2, y_coords2, z_coords2 = generate_helical_structure(sequence_length, 2, phase_shift=np.pi)\n\n    ax.plot(x_coords1, y_coords1, z_coords1, c='red', label=f\"{target_id} Helix 1\")\n    ax.plot(x_coords2, y_coords2, z_coords2, c='blue', label=f\"{target_id} Helix 2\")\n\n    all_coords = np.column_stack((np.concatenate([x_coords1, x_coords2]),\n                                   np.concatenate([y_coords1, y_coords2]),\n                                   np.concatenate([z_coords1, z_coords2])))\n\n    distance_matrix = pairwise_distances(all_coords)\n    avg_distance = np.mean(distance_matrix)\n\n    results_data.append([target_id, sequence_length, avg_distance])\n\n    for residue_index in range(sequence_length):\n        resname = sequence[residue_index]\n        resid = residue_index + 1\n        structure_id = f\"{target_id}_1\"\n        all_structure_data.append([structure_id, resname, resid, x_coords1[residue_index], y_coords1[residue_index], z_coords1[residue_index]])\n    for residue_index in range(sequence_length):\n        resname = sequence[residue_index]\n        resid = residue_index + 1\n        structure_id = f\"{target_id}_2\"\n        all_structure_data.append([structure_id, resname, resid, x_coords2[residue_index], y_coords2[residue_index], z_coords2[residue_index]])\n\nax.legend()\nplt.show()\n\nresults_df = pd.DataFrame(results_data, columns=['target_id', 'sequence_length', 'average_distance'])\n\nplt.figure(figsize=(10, 6))\nplt.bar(results_df['target_id'], results_df['average_distance'])\nplt.xlabel(\"Target ID\")\nplt.ylabel(\"Average Distance (Singularity Measure)\")\nplt.title(\"Comparison of Singularity Measures\")\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.show()\n\ncol_names = ['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1']\n\nsubmission_df = pd.DataFrame(all_structure_data, columns=col_names)\n\nprint(\"Code execution complete.  No submission file created.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T17:07:15.005626Z","iopub.execute_input":"2025-03-20T17:07:15.006126Z","iopub.status.idle":"2025-03-20T17:07:16.037153Z","shell.execute_reply.started":"2025-03-20T17:07:15.006077Z","shell.execute_reply":"2025-03-20T17:07:16.035024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport math\nimport os\nfrom IPython.display import display, HTML\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom sklearn.metrics import pairwise_distances\n\n# Genetic Algorithm imports\nimport random\nimport ipywidgets as widgets\nfrom ipywidgets import interact, interactive, fixed, interact_manual\nimport seaborn as sns\nimport folium\n\n# --- Constants and Parameters ---\nPOPULATION_SIZE = 50\nNUM_GENERATIONS = 20\nMUTATION_RATE = 0.1\n\n# --- Ambulância simulation parameters ---\nBASE_SPEED = 60  # km/h\nMAX_SPEED = 100  # km/h\nGRAVE_PROBLEM_AVOIDANCE_IMPACT = 0.8  # Reduction if a severe issue avoided\n\n# --- Constants ---\nTARGET_EFFICIENCY = 100  # Target efficiency for calculating K\nFIXED_COST = 500\n\n# --- Hydrogen Propulsion Constants ---\nHYDROGEN_EFFICIENCY_FACTOR = 2  # Multiplier for the hydrogen propulsion calculation\n\n# --- Generic patient data ---\n# Function to generate random patient coordinates and severity\ndef generate_patient_data(num_patients):\n    patient_locations = []\n    for i in range(num_patients):\n        latitude = random.uniform(-90, 90)  # Simulate latitudes\n        longitude = random.uniform(-180, 180)  # Simulate longitudes\n        severity = random.randint(1, 10)  # Severity from 1 to 10\n        patient_locations.append((latitude, longitude, severity))\n    return patient_locations\n\n# --- Functions (unchanged from previous version) ---\ndef calculate_lref_d0(df):\n    df['Lref'] = df['sequence'].apply(len)\n    df['d0'] = df['Lref'].apply(lambda lref: 0.6 * math.sqrt(lref - 0.5) - 2.5)\n    return df\n\ndef create_individual():\n    \"\"\"Creates a random individual (ambulance parameters).\"\"\"\n    return {\n        'speed_factor': random.uniform(0.8, 1.2), # Factor for speed adjustment\n        'route_preference': random.uniform(0.0, 1.0),  # Preference for faster vs. shorter route\n        'hydrogen_efficiency': random.uniform(0.8, 1.2),  # Factor for hydrogen propulsion efficiency\n        'hydrogen_station_distance_preference': random.uniform(0.1,100.0) # The R parameter, for distance to the hydrogen station\n    }\n\ndef calculate_hydrogen_propulsion(speed, R,individual):\n    \"\"\"Calculates hydrogen propulsion efficiency.\"\"\"\n    x = speed  # use speed of ambulance as x for efficiency optimization calculation\n    eltze = 10 # Simulate the fixed value for variable eltze\n\n    hydrogen_propulsion = HYDROGEN_EFFICIENCY_FACTOR * (2 * (3 * (x**2)) + ((-21**9) + 3) / (R**2)) / (eltze + 36)\n    return hydrogen_propulsion\n\ndef calculate_efficiency(individual, distance, patients, time, grave_problem_avoided, fixed_cost):\n    \"\"\"Calculates the efficiency, incorporating hydrogen propulsion.\"\"\"\n    # Adjust parameters based on the individual\n    effective_speed = BASE_SPEED * individual['speed_factor']\n    if effective_speed > MAX_SPEED:\n        effective_speed = MAX_SPEED\n\n    # Route adjustment based on 'route_preference' - Simplified here\n    adjusted_distance = distance * (1 - individual['route_preference'] * 0.1)  # Shorter route preference reduces distance\n    adjusted_time = time * (1 + individual['route_preference'] * 0.1)  # Shorter route preference might increase time slightly\n\n    # Calculate the penalization\n    penalization_simulation = GRAVE_PROBLEM_AVOIDANCE_IMPACT if grave_problem_avoided else 0.0\n\n    # Calculate hydrogen propulsion\n    hydrogen_propulsion = calculate_hydrogen_propulsion(effective_speed, individual['hydrogen_station_distance_preference'],individual) * individual['hydrogen_efficiency']\n\n    # Incorporate hydrogen propulsion into efficiency calculation (Example: Additively)\n    efficiency = ((patients / adjusted_time) / (adjusted_distance * (1 + penalization_simulation) + fixed_cost)) + hydrogen_propulsion\n\n    return efficiency\n\ndef evaluate_fitness(individual, distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost):\n    \"\"\"Evaluates the fitness of an individual over a set of scenarios.\"\"\"\n    total_efficiency = 0\n    num_scenarios = len(distances)\n\n    for i in range(num_scenarios):\n        distance = distances[i]\n        patients = patients_list[i]\n        time = times_list[i]\n        grave_problem_avoided = grave_problems_avoided_list[i]\n\n        efficiency = calculate_efficiency(individual, distance, patients, time, grave_problem_avoided, fixed_cost)\n        total_efficiency += efficiency\n\n    # Return the average efficiency\n    return total_efficiency / num_scenarios\n\ndef crossover(parent1, parent2):\n    \"\"\"Performs crossover between two parents.\"\"\"\n    child = {}\n    for key in parent1.keys():\n        if random.random() < 0.5:\n            child[key] = parent1[key]\n        else:\n            child[key] = parent2[key]\n    return child\n\ndef mutate(individual):\n    \"\"\"Mutates an individual.\"\"\"\n    for key in individual.keys():\n        if random.random() < MUTATION_RATE:\n            # Adjust the mutation based on the key. For example, we should use a small mutation on factors and larger on route selection.\n            if key == 'speed_factor':\n                individual[key] += random.uniform(-0.1, 0.1) # Small adjustments\n                individual[key] = max(0.8, min(1.2, individual[key]))  # Clamp value.\n            elif key == 'route_preference':\n                individual[key] += random.uniform(-0.2, 0.2)  # Larger adjustment\n                individual[key] = max(0.0, min(1.0, individual[key])) # Clamp value.\n            elif key == 'hydrogen_efficiency':\n                individual[key] += random.uniform(-0.1, 0.1) # Small adjustments\n                individual[key] = max(0.8, min(1.2, individual[key]))\n            else: # hydrogen_station_distance_preference, R parameter in propulsion calculation\n                individual[key] += random.uniform(-10, 10) # Adjust R for the hydrogen propulsion calculation\n                individual[key] = max(0.1, min(100.0, individual[key]))  #R must be positive.\n    return individual\n\ndef genetic_algorithm(distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost):\n    \"\"\"Runs the genetic algorithm.\"\"\"\n    # 1. Initialization\n    population = [create_individual() for _ in range(POPULATION_SIZE)]\n    fitness_history = []\n    best_individual_history = []\n\n    # 2. Evolution\n    for generation in range(NUM_GENERATIONS):\n        # a. Evaluate Fitness\n        fitness_scores = [evaluate_fitness(individual, distances, patients_list, times_list, grave_problems_avoided_list, fixed_cost) for individual in population]\n\n        # Check if fitness_scores is empty\n        if not fitness_scores:\n            print(\"Warning: All fitness scores are zero. Stopping the GA.\")\n            return create_individual(), fitness_history, best_individual_history  # Return a default individual\n\n        # Find best individual only if fitness_scores is not empty\n        if fitness_scores:\n            best_individual_index = np.argmax(fitness_scores)\n            best_individual = population[best_individual_index]\n            best_fitness = fitness_scores[best_individual_index]\n            fitness_history.append(best_fitness)\n            best_individual_history.append(best_individual)\n        else:\n            # If fitness_scores is empty, append a default value\n            best_individual = create_individual()\n            best_fitness = 0.0  # Set to a default value\n            fitness_history.append(best_fitness)\n            best_individual_history.append(best_individual)\n\n        # b. Selection (Tournament Selection)\n        selected_indices = random.choices(range(POPULATION_SIZE), weights=fitness_scores, k=POPULATION_SIZE)\n        selected_population = [population[i] for i in selected_indices]\n\n        # c. Crossover\n        offspring = []\n        for i in range(0, POPULATION_SIZE, 2):\n            parent1 = selected_population[i]\n            parent2 = selected_population[i+1] if i+1 < POPULATION_SIZE else selected_population[i]\n            child = crossover(parent1, parent2)\n            offspring.append(child)\n\n        # d. Mutation\n        mutated_offspring = [mutate(child) for child in offspring]\n\n        # Replace the old population with the new offspring\n        population = mutated_offspring\n\n        print(f\"Generation {generation}: Best individual = {best_individual}, Fitness = {best_fitness}\")\n\n    # 3. Return the Best Individual\n    if fitness_scores:\n        best_individual_index = np.argmax(fitness_scores)\n        return population[best_individual_index], fitness_history, best_individual_history\n    else:\n        return create_individual(), fitness_history, best_individual_history\n\n# --- MAIN SCRIPT ---\n\n# --- Simulated Data and Load/Create it\n# Simulate generic data instead of loading from files\nnum_sequences = 10\ndata = {\n    'target_id': [f'seq_{i}' for i in range(num_sequences)],\n    'sequence': ['AUCG' * (i + 1) for i in range(num_sequences)], # Different sequence lengths\n    'description': ['Generic sequence' for _ in range(num_sequences)]\n}\ntest_df = pd.DataFrame(data)\ntest_df = calculate_lref_d0(test_df)\n\n# --- Create Ambulance Scenario Data: Number and Severity of Cases\nnum_patients_per_scenario = [random.randint(3, 10) for _ in range(4)] # Random number of patients per scenario\npatient_data_per_scenario = [generate_patient_data(num_patients) for num_patients in num_patients_per_scenario]\n\n# --- Assign Scenario - Data - Patients, Time to Solve the Grave Problem or not:\n# Make a table and define parameters for each Scenario\nscenarios = [\n    {\"distance\": 100, \"grave_problem_avoided\": True, \"time\": 2},\n    {\"distance\": 150, \"grave_problem_avoided\": False, \"time\": 3},\n    {\"distance\": 200, \"grave_problem_avoided\": True, \"time\": 2.5},\n    {\"distance\": 120, \"grave_problem_avoided\": False, \"time\": 2.2}\n]\ndistances = [scenario[\"distance\"] for scenario in scenarios]\ngrave_problems_avoided_list = [scenario[\"grave_problem_avoided\"] for scenario in scenarios]\ntimes_list = [scenario[\"time\"] for scenario in scenarios]\npatients_list = num_patients_per_scenario\n\n# --- Run Genetic Algorithm ---\ntry:\n    best_params, fitness_history, best_individual_history = genetic_algorithm(distances, patients_list, times_list, grave_problems_avoided_list, FIXED_COST)\n    print(\"\\nBest Ambulance Parameters:\", best_params)\nexcept ValueError as e:\n    print(f\"Error during genetic algorithm execution: {e}\")\n    best_params = create_individual()  # Use a default individual\n    fitness_history = []\n    best_individual_history = []\n\n# --- Calculate K factor ---\n# To calculate the K factor, let's assume an 'average' scenario:\navg_distance = np.mean(distances)\navg_patients = np.mean(patients_list)\navg_time = np.mean(times_list)\navg_grave_problem_avoided = any(grave_problems_avoided_list) # Assume there's at least one.\n\n# Calculate efficiency with the best parameters\navg_efficiency_with_best_params = calculate_efficiency(\n    best_params, avg_distance, avg_patients, avg_time, avg_grave_problem_avoided, FIXED_COST\n)\n\n# Calculate K based on the target efficiency\nK = TARGET_EFFICIENCY / avg_efficiency_with_best_params\n\nprint(\"Calculated K factor:\", K)\n\n# --- Prepare Data for Output File ---\noutput_data = {\n    \"best_params\": best_params,\n    \"fitness_history\": fitness_history,\n    \"distances\": distances,\n    \"patients_list\": patients_list,\n    \"times_list\": times_list,\n    \"grave_problems_avoided_list\": grave_problems_avoided_list,\n    \"K\": K,\n    \"best_RNA\" : test_df.to_dict(),\n    \"best_test\" : test_df.to_dict(),\n    \"ex\" : patient_data_per_scenario\n}\n\n# --- Output File Creation ---\noutput_filename = \"ambulance_simulation_data.world\"\noutput_path = os.path.join(\"/kaggle/working\", output_filename)\n\n# --- Write the world file\n# Open the file for writing\nwith open(output_path, 'w') as f:\n    # Write header\n    f.write(\"// Ambulance Simulation Data\\n\\n\")\n\n    # Write each of the data\n    f.write(\"// Best Parameters:\\n\")\n    for k, v in output_data[\"best_params\"].items():\n        f.write(f\"//   {k}: {v}\\n\")\n    f.write(\"\\n\")\n\n    f.write(\"// Fitness History:\\n\")\n    f.write(f\"//   {str(output_data['fitness_history'])}\\n\")\n    f.write(\"\\n\")\n\n    f.write(\"// Distances:\\n\")\n    f.write(f\"//   {str(output_data['distances'])}\\n\")\n    f.write(\"\\n\")\n\n    f.write(\"// Patients List:\\n\")\n    f.write(f\"//   {str(output_data['patients_list'])}\\n\")\n    f.write(\"\\n\")\n\n    f.write(\"// Times List:\\n\")\n    f.write(f\"//   {str(output_data['times_list'])}\\n\")\n    f.write(\"\\n\")\n\n    f.write(\"// Grave Problems Avoided List:\\n\")\n    f.write(f\"//   {str(output_data['grave_problems_avoided_list'])}\\n\")\n    f.write(\"\\n\")\n\n    f.write(\"// K Value:\\n\")\n    f.write(f\"//   {str(output_data['K'])}\\n\")\n    f.write(\"\\n\")\n\n    f.write(\"// Best RNA Data:\\n\")\n    f.write(f\"//   {str(output_data['best_RNA'])}\\n\")\n    f.write(\"\\n\")\n\n    f.write(\"// Best Test Data:\\n\")\n    f.write(f\"//   {str(output_data['best_test'])}\\n\")\n    f.write(\"\\n\")\n\n    f.write(\"// Patient Data per Scenario:\\n\")\n    f.write(f\"//   {str(output_data['ex'])}\\n\")\n    f.write(\"\\n\")\n\n# --- Verify File Creation ---\nif os.path.exists(output_path):\n    print(f\"Successfully created the file: {output_path}\")\nelse:\n    print(\"Failed to create the file.\")\n\n# --- Clear any leftover data ---\nall_structure_data = []\nresults_data = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-20T17:08:15.207740Z","iopub.execute_input":"2025-03-20T17:08:15.208166Z","iopub.status.idle":"2025-03-20T17:08:15.257699Z","shell.execute_reply.started":"2025-03-20T17:08:15.208131Z","shell.execute_reply":"2025-03-20T17:08:15.256419Z"}},"outputs":[],"execution_count":null}]}