{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is basically originated from the [notebook](https://www.kaggle.com/code/lesamu/reinforcement-learning-baseline-in-python/notebook).","metadata":{}},{"cell_type":"code","source":"!pip install --target=lib --no-deps stable-baselines3 gym","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:20:49.995194Z","iopub.execute_input":"2022-07-12T05:20:49.995892Z","iopub.status.idle":"2022-07-12T05:21:07.966096Z","shell.execute_reply.started":"2022-07-12T05:20:49.995853Z","shell.execute_reply":"2022-07-12T05:21:07.964904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nKAGGLE_AGENT_PATH = \"/kaggle_simulations/agent/\"\nif os.path.exists(KAGGLE_AGENT_PATH):\n    sys.path.insert(0, os.path.join(KAGGLE_AGENT_PATH, 'lib'))\nelse:\n    sys.path.insert(0, os.path.join(os.getcwd(), 'lib'))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T05:21:07.968414Z","iopub.execute_input":"2022-07-12T05:21:07.969172Z","iopub.status.idle":"2022-07-12T05:21:07.977708Z","shell.execute_reply.started":"2022-07-12T05:21:07.969122Z","shell.execute_reply":"2022-07-12T05:21:07.976209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile config.py\nimport numpy as np\nfrom kaggle_environments import make\n\n# Read env specification\nENV_SPECIFICATION = make('kore_fleets').specification\nSHIP_COST = ENV_SPECIFICATION.configuration.spawnCost.default\nSHIPYARD_COST = ENV_SPECIFICATION.configuration.convertCost.default\nGAME_CONFIG = {\n    'episodeSteps':  ENV_SPECIFICATION.configuration.episodeSteps.default,  # You might want to start with smaller values\n    'size': ENV_SPECIFICATION.configuration.size.default,\n    'maxLogLength': None\n}\n\n# Define your opponent. We'll use the starter bot in the notebook environment for this baseline.\nOPPONENT = 'opponent.py'\nGAME_AGENTS = [None, OPPONENT]\n\n# Define our parameters\nN_FEATURES = 4\nACTION_SIZE = (2,)\nDTYPE = np.float64\nMAX_OBSERVABLE_KORE = 500\nMAX_OBSERVABLE_SHIPS = 200\nMAX_ACTION_FLEET_SIZE = 150\nMAX_KORE_IN_RESERVE = 40000\nWIN_REWARD = 1000","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:07.979985Z","iopub.execute_input":"2022-07-12T05:21:07.980590Z","iopub.status.idle":"2022-07-12T05:21:07.990601Z","shell.execute_reply.started":"2022-07-12T05:21:07.980554Z","shell.execute_reply":"2022-07-12T05:21:07.989539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile opponent.py\nfrom kaggle_environments.envs.kore_fleets.helpers import *\n\n\n# This is just the starter bot. Change this with the agent of your choice.\ndef agent(obs, config):\n    board = Board(obs, config)\n\n    me = board.current_player\n    turn = board.step\n    spawn_cost = board.configuration.spawn_cost\n    kore_left = me.kore\n\n    for shipyard in me.shipyards:\n        if shipyard.ship_count > 10:\n            direction = Direction.from_index(turn % 4)\n            action = ShipyardAction.launch_fleet_with_flight_plan(2, direction.to_char())\n            shipyard.next_action = action\n        elif kore_left > spawn_cost * shipyard.max_spawn:\n            action = ShipyardAction.spawn_ships(shipyard.max_spawn)\n            shipyard.next_action = action\n            kore_left -= spawn_cost * shipyard.max_spawn\n        elif kore_left > spawn_cost:\n            action = ShipyardAction.spawn_ships(1)\n            shipyard.next_action = action\n            kore_left -= spawn_cost\n\n    return me.next_actions\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:07.993722Z","iopub.execute_input":"2022-07-12T05:21:07.994147Z","iopub.status.idle":"2022-07-12T05:21:08.002431Z","shell.execute_reply.started":"2022-07-12T05:21:07.994105Z","shell.execute_reply":"2022-07-12T05:21:08.001425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile reward_utils.py\nfrom config import GAME_CONFIG, SHIP_COST, SHIPYARD_COST\nfrom kaggle_environments.envs.kore_fleets.helpers import Board\nimport numpy as np\nfrom math import floor\n\n# Compute weight constants -- See get_board_value's docstring\n_max_steps = GAME_CONFIG['episodeSteps']\n_end_of_asset_value = floor(.5 * _max_steps)\n_weights_assets = np.linspace(start=1, stop=0, num=_end_of_asset_value)\n_weights_kore = np.linspace(start=0, stop=1, num=_end_of_asset_value)\nWEIGHTS_ASSETS = np.append(_weights_assets, np.zeros(_max_steps - _end_of_asset_value))\nWEIGHTS_KORE = np.append(_weights_kore, np.ones(_max_steps - _end_of_asset_value))\nWEIGHTS_MAX_SPAWN = {x: (x+3)/4 for x in range(1, 11)}  # Value multiplier of a shipyard as a function of its max spawn\nWEIGHTS_KORE_IN_FLEETS = WEIGHTS_KORE * WEIGHTS_ASSETS/2  # Always equal or smaller than either, almost always smaller\n\n\ndef get_board_value(board: Board) -> float:\n    \"\"\"Computes the board value for the current player.\n\n    The board value captures how are we currently performing, compared to the opponent. Each player's partial board\n    value assesses the player's situation, taking into account their current kore, ship count, shipyard count\n    (including their max spawn) and kore carried by fleets. We then define the board value as the difference between\n    player's partial board values.\n    Flight plans and the positioning of fleet and shipyards do not flow into the board value (yet).\n\n    To keep things simple, we'll take a weighted sum as the partial board value. We need weighting since\n    the importance of each item changes over time. We don't need to have the most kore at the beginning of the game,\n    but we do at the end. Ship count won't help us win games in the latter stages, but it is crucial in the beginning.\n    Fleets and shipyards will be accounted for proportionally to their kore cost.\n\n    For efficiency, the weight factors are pre-computed at module level. Here is the logic behind the weighting:\n    WEIGHTS_KORE: Applied to the player's kore count. Increases linearly from 0 to 1. It reaches one before\n        the maximum game length is reached.\n    WEIGHTS_ASSETS: Applied to fleets and shipyards. Decreases linearly from 1 to 0 and reaches zero before the maximum\n        length. It emphasizes the need of having ships over kore at the beginning of the game.\n    WEIGHTS_MAX_SPAWN: Shipyard value is multiplied by its max spawn. This captures the idea that long-held shipyards\n        are more valuable.\n    WEIGHTS_KORE_IN_FLEETS: Kore in fleets should be valued, too. But its value must be upper-bounded by WEIGHTS_KORE\n        (it can never be better to have kore in cargo than home) and it must decrease in time, since it doesn't\n        count towards the end kore count.\n\n    Args:\n        board: The board for which we want to compute the value.\n\n    Returns:\n        The value of the board.\n    \"\"\"\n    board_value = 0\n    if not board:\n        return board_value\n\n    # Get the weights as a function of the current game step\n    step = board.step\n    weight_kore, weight_assets, weight_cargo = WEIGHTS_KORE[step], WEIGHTS_ASSETS[step], WEIGHTS_KORE_IN_FLEETS[step]\n\n    # Compute the partial board values\n    for player in board.players.values():\n        player_fleets, player_shipyards = list(player.fleets), list(player.shipyards)\n\n        value_kore = weight_kore * player.kore\n\n        value_fleets = weight_assets * SHIP_COST * (\n                sum(fleet.ship_count for fleet in player_fleets)\n                + sum(shipyard.ship_count for shipyard in player_shipyards)\n        )\n\n        value_shipyards = weight_assets * SHIPYARD_COST * (\n            sum(shipyard.max_spawn * WEIGHTS_MAX_SPAWN[shipyard.max_spawn] for shipyard in player_shipyards)\n        )\n\n        value_kore_in_cargo = weight_cargo * sum(fleet.kore for fleet in player_fleets)\n\n        # Add (or subtract) the partial values to the total board value. The current player is always us.\n        modifier = 1 if player.is_current_player else -1\n        board_value += modifier * (value_kore + value_fleets + value_shipyards + value_kore_in_cargo)\n\n    return board_value","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:08.003972Z","iopub.execute_input":"2022-07-12T05:21:08.004583Z","iopub.status.idle":"2022-07-12T05:21:08.015370Z","shell.execute_reply.started":"2022-07-12T05:21:08.004548Z","shell.execute_reply":"2022-07-12T05:21:08.014490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile environment.py\nimport gym\nimport numpy as np\nfrom gym import spaces\nfrom math import floor\nfrom kaggle_environments import make\nfrom kaggle_environments.envs.kore_fleets.helpers import ShipyardAction, Board, Direction\nfrom typing import Union, Tuple, Dict\nfrom reward_utils import get_board_value\nfrom config import (\n    N_FEATURES,\n    ACTION_SIZE,\n    GAME_AGENTS,\n    GAME_CONFIG,\n    DTYPE,\n    MAX_OBSERVABLE_KORE,\n    MAX_OBSERVABLE_SHIPS,\n    MAX_ACTION_FLEET_SIZE,\n    MAX_KORE_IN_RESERVE,\n    WIN_REWARD,\n)\n\n\nclass KoreGymEnv(gym.Env):\n    \"\"\"An openAI-gym env wrapper for kaggle's kore environment. Can be used with stable-baselines3.\n\n    There are three fundamental components to this class which you would want to customize for your own agents:\n        The action space is defined by `action_space` and `gym_to_kore_action()`\n        The state space (observations) is defined by `state_space` and `obs_as_gym_state()`\n        The reward is computed with `compute_reward()`\n\n    Note that the action and state spaces define the inputs and outputs to your model *as numpy arrays*. Use the\n    functions mentioned above to translate these arrays into actual kore environment observations and actions.\n\n    The rest is basically boilerplate and makes sure that the kaggle environment plays nicely with stable-baselines3.\n\n    Usage:\n        >>> from stable_baselines3 import PPO\n        >>>\n        >>> kore_env = KoreGymEnv()\n        >>> model = PPO('MlpPolicy', kore_env, verbose=1)\n        >>> model.learn(total_timesteps=100000)\n    \"\"\"\n\n    def __init__(self, config=None, agents=None, debug=None):\n        super(KoreGymEnv, self).__init__()\n\n        if not config:\n            config = GAME_CONFIG\n        if not agents:\n            agents = GAME_AGENTS\n        if not debug:\n            debug = True\n\n        self.agents = agents\n        self.env = make(\"kore_fleets\", configuration=config, debug=debug)\n        self.config = self.env.configuration\n        self.trainer = None\n        self.raw_obs = None\n        self.previous_obs = None\n\n        # Define the action and state space\n        # Change these to match your needs. Normalization to the [-1, 1] interval is recommended. See:\n        # https://araffin.github.io/slides/rlvs-tips-tricks/#/13/0/0\n        # See https://www.gymlibrary.ml/content/spaces/ for more info on OpenAI-gym spaces.\n        self.action_space = spaces.Box(\n            low=-1,\n            high=1,\n            shape=ACTION_SIZE,\n            dtype=DTYPE\n        )\n\n        self.observation_space = spaces.Box(\n            low=-1,\n            high=1,\n            shape=(self.config.size ** 2 * N_FEATURES + 3,),\n            dtype=DTYPE\n        )\n\n        self.strict_reward = config.get('strict', False)\n\n        # Debugging info - Enable or disable as needed\n        self.reward = 0\n        self.n_steps = 0\n        self.n_resets = 0\n        self.n_dones = 0\n        self.last_action = None\n        self.last_done = False\n\n    def reset(self) -> np.ndarray:\n        \"\"\"Resets the trainer and returns the initial observation in state space.\n\n        Returns:\n            self.obs_as_gym_state: the current observation encoded as a state in state space\n        \"\"\"\n        # agents = self.agents if np.random.rand() > .5 else self.agents[::-1]  # Randomize starting position\n        self.trainer = self.env.train(self.agents)\n        self.raw_obs = self.trainer.reset()\n        self.n_resets += 1\n        return self.obs_as_gym_state\n\n    def step(self, action: np.ndarray) -> Tuple[np.ndarray, float, bool, Dict]:\n        \"\"\"Execute action in the trainer and return the results.\n\n        Args:\n            action: The action in action space, i.e. the output of the stable-baselines3 agent\n\n        Returns:\n            self.obs_as_gym_state: the current observation encoded as a state in state space\n            reward: The agent's reward\n            done: If True, the episode is over\n            info: A dictionary with additional debugging information\n        \"\"\"\n        kore_action = self.gym_to_kore_action(action)\n        self.previous_obs = self.raw_obs\n        self.raw_obs, _, done, info = self.trainer.step(kore_action)  # Ignore trainer reward, which is just delta kore\n        self.reward = self.compute_reward(done)\n\n        # Debugging info\n        # with open('logs/tmp.log', 'a') as log:\n        #    print(kore_action.action_type, kore_action.num_ships, kore_action.flight_plan, file=log)\n        #    if done:\n        #        print('done', file=log)\n        #    if info:\n        #        print('info', file=log)\n        self.n_steps += 1\n        self.last_done = done\n        self.last_action = kore_action\n        self.n_dones += 1 if done else 0\n\n        return self.obs_as_gym_state, self.reward, done, info\n\n    def render(self, **kwargs):\n        self.env.render(**kwargs)\n\n    def close(self):\n        pass\n\n    @property\n    def board(self):\n        return Board(self.raw_obs, self.config)\n\n    @property\n    def previous_board(self):\n        return Board(self.previous_obs, self.config)\n\n    def gym_to_kore_action(self, gym_action: np.ndarray) -> Dict[str, str]:\n        \"\"\"Decode an action in action space as a kore action.\n\n        In other words, transform a stable-baselines3 action into an action compatible with the kore environment.\n\n        This method is central - It defines how the agent output is mapped to kore actions.\n        You can modify it to suit your needs.\n\n        Let's start with an übereasy mapping. Our gym_action is a 1-dimensional vector of size 2 (as defined in\n        self.action_space). We will interpret the values as follows:\n        if gym_action[0] > 0 launch a fleet, elif < 0 build ships, else wait.\n        abs(gym_action[0]) encodes the number of ships to build/launch.\n        gym_action[1] represents the direction in which to launch the fleet.\n\n        Notes: The same action is sent to all shipyards, though we make sure that the actions are valid.\n\n        Args:\n            gym_action: The action produces by our stable-baselines3 agent.\n\n        Returns:\n            The corresponding kore environment actions or None if the agent wants to wait.\n\n        \"\"\"\n        action_launch = gym_action[0] > 0\n        action_build = gym_action[0] < 0\n        # Mapping the number of ships is an interesting exercise. Here we chose a linear mapping to the interval\n        # [1, MAX_ACTION_FLEET_SIZE], but you could use something else. With a linear mapping, all values are\n        # evenly spaced. An exponential mapping, however, would space out lower values, making them easier for the agent\n        # to distinguish and choose, at the cost of needing more precision to accurately select higher values.\n        number_of_ships = int(\n            clip_normalize(\n                x=abs(gym_action[0]),\n                low_in=0,\n                high_in=1,\n                low_out=1,\n                high_out=MAX_ACTION_FLEET_SIZE\n            )\n        )\n\n        # Broadcast the same action to all shipyards\n        board = self.board\n        me = board.current_player\n        for shipyard in me.shipyards:\n            action = None\n            if action_build:\n                # Limit the number of ships to the maximum that can be actually built\n                max_spawn = shipyard.max_spawn\n                max_purchasable = floor(me.kore / self.config[\"spawnCost\"])\n                number_of_ships = min(number_of_ships, max_spawn, max_purchasable)\n                if number_of_ships:\n                    action = ShipyardAction.spawn_ships(number_ships=number_of_ships)\n\n            elif action_launch:\n                # Limit the number of ships to the amount that is actually present in the shipyard\n                shipyard_count = shipyard.ship_count\n                number_of_ships = min(number_of_ships, shipyard_count)\n                if number_of_ships:\n                    direction = round((gym_action[1] + 1) * 1.5)  # int between 0 (North) and 3 (West)\n                    action = ShipyardAction.launch_fleet_in_direction(number_ships=number_of_ships,\n                                                                      direction=Direction.from_index(direction))\n            shipyard.next_action = action\n\n        return me.next_actions\n\n    @property\n    def obs_as_gym_state(self) -> np.ndarray:\n        \"\"\"Return the current observation encoded as a state in state space.\n\n        In other words, transform a kore observation into a stable-baselines3-compatible np.ndarray.\n\n        This property is central - It defines how the kore board is mapped to our state space.\n        You can modify it to include as many features as you see convenient.\n\n        Let's keep start with something easy: Define a 21x21x4+3 state (size x size x n_features and 3 extra features).\n        # Feature 0: How much kore there is in a cell\n        # Feature 1: How many ships there are in a cell (>0: friendly, <0: enemy)\n        # Feature 2: Fleet direction\n        # Feature 3: Is a shipyard present? (1: friendly, -1: enemy, 0: no)\n        # Feature 4: Progress - What turn is it?\n        # Feature 5: How much kore do I have?\n        # Feature 6: How much kore does the opponent have?\n\n        We'll make sure that all features are in the range [-1, 1] and as close to a normal distribution as possible.\n\n        Note: This mapping doesn't tackle a critical issue in kore: How to encode (full) flight plans?\n        \"\"\"\n        # Init output state\n        gym_state = np.ndarray(shape=(self.config.size, self.config.size, N_FEATURES))\n\n        # Get our player ID\n        board = self.board\n        our_id = board.current_player_id\n\n        for point, cell in board.cells.items():\n            # Feature 0: How much kore\n            gym_state[point.y, point.x, 0] = cell.kore\n\n            # Feature 1: How many ships (>0: friendly, <0: enemy)\n            # Feature 2: Fleet direction\n            fleet = cell.fleet\n            if fleet:\n                modifier = 1 if fleet.player_id == our_id else -1\n                gym_state[point.y, point.x, 1] = modifier * fleet.ship_count\n                gym_state[point.y, point.x, 2] = fleet.direction.value\n            else:\n                # The current cell has no fleet\n                gym_state[point.y, point.x, 1] = gym_state[point.y, point.x, 2] = 0\n\n            # Feature 3: Shipyard present (1: friendly, -1: enemy)\n            shipyard = cell.shipyard\n            if shipyard:\n                gym_state[point.y, point.x, 3] = 1 if shipyard.player_id == our_id else -1\n            else:\n                # The current cell has no shipyard\n                gym_state[point.y, point.x, 3] = 0\n\n        # Normalize features to interval [-1, 1]\n        # Feature 0: Logarithmic scale, kore in range [0, MAX_OBSERVABLE_KORE]\n        gym_state[:, :, 0] = clip_normalize(\n            x=np.log2(gym_state[:, :, 0] + 1),\n            low_in=0,\n            high_in=np.log2(MAX_OBSERVABLE_KORE)\n        )\n\n        # Feature 1: Ships in range [-MAX_OBSERVABLE_SHIPS, MAX_OBSERVABLE_SHIPS]\n        gym_state[:, :, 1] = clip_normalize(\n            x=gym_state[:, :, 1],\n            low_in=-MAX_OBSERVABLE_SHIPS,\n            high_in=MAX_OBSERVABLE_SHIPS\n        )\n\n        # Feature 2: Fleet direction in range (1, 4)\n        gym_state[:, :, 2] = clip_normalize(\n            x=gym_state[:, :, 2],\n            low_in=1,\n            high_in=4\n        )\n\n        # Feature 3 is already as normal as it gets\n\n        # Flatten the input (recommended by stable_baselines3.common.env_checker.check_env)\n        output_state = gym_state.flatten()\n\n        # Extra Features: Progress, how much kore do I have, how much kore does opponent have\n        player = board.current_player\n        opponent = board.opponents[0]\n        progress = clip_normalize(board.step, low_in=0, high_in=GAME_CONFIG['episodeSteps'])\n        my_kore = clip_normalize(np.log2(player.kore+1), low_in=0, high_in=np.log2(MAX_KORE_IN_RESERVE))\n        opponent_kore = clip_normalize(np.log2(opponent.kore+1), low_in=0, high_in=np.log2(MAX_KORE_IN_RESERVE))\n\n        return np.append(output_state, [progress, my_kore, opponent_kore])\n\n    def compute_reward(self, done: bool, strict=False) -> float:\n        \"\"\"Compute the agent reward. Welcome to the fine art of RL.\n\n         We'll compute the reward as the current board value and a final bonus if the episode is over. If the player\n          wins the episode, we'll add a final bonus that increases with shorter time-to-victory.\n        If the player loses, we'll subtract that bonus.\n\n        Args:\n            done: True if the episode is over\n            strict: If True, count only wins/loses (Useful for evaluating a trained agent)\n\n        Returns:\n            The agent's reward\n        \"\"\"\n        board = self.board\n        previous_board = self.previous_board\n\n        if strict:\n            if done:\n                # Who won?\n                # Ugly but 99% sure correct, see https://www.kaggle.com/competitions/kore-2022/discussion/324150#1789804\n                agent_reward = self.raw_obs.players[0][0]\n                opponent_reward = self.raw_obs.players[1][0]\n                return int(agent_reward > opponent_reward)\n            else:\n                return 0\n        else:\n            if done:\n                # Who won?\n                agent_reward = self.raw_obs.players[0][0]\n                opponent_reward = self.raw_obs.players[1][0]\n                if agent_reward is None or opponent_reward is None:\n                    we_won = -1\n                else:\n                    we_won = 1 if agent_reward > opponent_reward else -1\n                win_reward = we_won * (WIN_REWARD + 5 * (GAME_CONFIG['episodeSteps'] - board.step))\n            else:\n                win_reward = 0\n\n            return get_board_value(board) - get_board_value(previous_board) + win_reward\n\n\ndef clip_normalize(x: Union[np.ndarray, float],\n                   low_in: float,\n                   high_in: float,\n                   low_out=-1.,\n                   high_out=1.) -> Union[np.ndarray, float]:\n    \"\"\"Clip values in x to the interval [low_in, high_in] and then MinMax-normalize to [low_out, high_out].\n\n    Args:\n        x: The array of float to clip and normalize\n        low_in: The lowest possible value in x\n        high_in: The highest possible value in x\n        low_out: The lowest possible value in the output\n        high_out: The highest possible value in the output\n\n    Returns:\n        The clipped and normalized version of x\n\n    Raises:\n        AssertionError if the limits are not consistent\n\n    Examples:\n        >>> clip_normalize(50, low_in=0, high_in=100)\n        0.0\n\n        >>> clip_normalize(np.array([-1, .5, 99]), low_in=-1, high_in=1, low_out=0, high_out=2)\n        array([0., 1.5, 2.])\n    \"\"\"\n    assert high_in > low_in and high_out > low_out, \"Wrong limits\"\n\n    # Clip outliers\n    try:\n        x[x > high_in] = high_in\n        x[x < low_in] = low_in\n    except TypeError:\n        x = high_in if x > high_in else x\n        x = low_in if x < low_in else x\n\n    # y = ax + b\n    a = (high_out - low_out) / (high_in - low_in)\n    b = high_out - high_in * a\n\n    return a * x + b\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:08.018672Z","iopub.execute_input":"2022-07-12T05:21:08.018925Z","iopub.status.idle":"2022-07-12T05:21:08.036666Z","shell.execute_reply.started":"2022-07-12T05:21:08.018895Z","shell.execute_reply":"2022-07-12T05:21:08.035567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from stable_baselines3 import PPO\nfrom stable_baselines3.common.evaluation import evaluate_policy\nfrom stable_baselines3.common.monitor import Monitor\nfrom environment import KoreGymEnv","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:08.038213Z","iopub.execute_input":"2022-07-12T05:21:08.038605Z","iopub.status.idle":"2022-07-12T05:21:10.283894Z","shell.execute_reply.started":"2022-07-12T05:21:08.038570Z","shell.execute_reply":"2022-07-12T05:21:10.282374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kore_env = KoreGymEnv(config=dict(randomSeed=997269658))  # TODO: This seed is not enough. Seed everything!\nmonitored_env = Monitor(env=kore_env)\nmodel = PPO('MlpPolicy', monitored_env, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:10.286367Z","iopub.execute_input":"2022-07-12T05:21:10.287627Z","iopub.status.idle":"2022-07-12T05:21:13.382379Z","shell.execute_reply.started":"2022-07-12T05:21:10.287518Z","shell.execute_reply":"2022-07-12T05:21:13.381242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# # For serious training, likely many more iterations will be needed, as well as hyperparameter tuning!\n# # Even so, sometimes training will still fail. RL is like that. Try a couple times with the same config before giving up!\nmodel.learn(total_timesteps=50000)  \n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:45:16.843689Z","iopub.execute_input":"2022-07-12T00:45:16.844433Z","iopub.status.idle":"2022-07-12T00:46:52.591241Z","shell.execute_reply.started":"2022-07-12T00:45:16.844386Z","shell.execute_reply":"2022-07-12T00:46:52.589322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kore_env.render(mode=\"ipython\", width=1000, height=800)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:46:52.592233Z","iopub.status.idle":"2022-07-12T00:46:52.593258Z","shell.execute_reply.started":"2022-07-12T00:46:52.592973Z","shell.execute_reply":"2022-07-12T00:46:52.593001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import optuna\nfrom optuna.pruners import MedianPruner\nfrom optuna.samplers import TPESampler\nfrom optuna.visualization import plot_optimization_history, plot_param_importances","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:13.384117Z","iopub.execute_input":"2022-07-12T05:21:13.384543Z","iopub.status.idle":"2022-07-12T05:21:13.391560Z","shell.execute_reply.started":"2022-07-12T05:21:13.384507Z","shell.execute_reply":"2022-07-12T05:21:13.388835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_TRIALS = 100  # Maximum number of trials\nN_JOBS = 1 # Number of jobs to run in parallel\nN_STARTUP_TRIALS = 5  # Stop random sampling after N_STARTUP_TRIALS\nN_EVALUATIONS = 2  # Number of evaluations during the training\nN_TIMESTEPS = int(2e4)  # Training budget\nEVAL_FREQ = int(N_TIMESTEPS / N_EVALUATIONS)\nN_EVAL_ENVS = 5\nN_EVAL_EPISODES = 10\nTIMEOUT = int(60 * 15)  # 15 minutes\n\n\nDEFAULT_HYPERPARAMS = {\n    \"policy\": \"MlpPolicy\",\n    \"env\": kore_env,\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:13.396322Z","iopub.execute_input":"2022-07-12T05:21:13.396594Z","iopub.status.idle":"2022-07-12T05:21:13.403180Z","shell.execute_reply.started":"2022-07-12T05:21:13.396570Z","shell.execute_reply":"2022-07-12T05:21:13.402121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from typing import Any, Dict\nimport torch\nimport torch.nn as nn\n\ndef sample_ppo_params(trial: optuna.Trial) -> Dict[str, Any]:\n    \"\"\"\n    Sampler for ppo hyperparameters.\n\n    :param trial: Optuna trial object\n    :return: The sampled hyperparameters for the given trial.\n    \"\"\"\n    # Discount factor between 0.9 and 0.9999\n    gamma = 1.0 - trial.suggest_float(\"gamma\", 0.0001, 0.1, log=True)\n    max_grad_norm = trial.suggest_float(\"max_grad_norm\", 0.3, 5.0, log=True)\n    # 8, 16, 32, ... 1024\n    n_steps = 2 ** trial.suggest_int(\"exponent_n_steps\", 3, 10)\n\n    ### YOUR CODE HERE\n    # TODO:\n    # - define the learning rate search space [1e-5, 1] (log) -> `suggest_float`\n    # - define the network architecture search space [\"tiny\", \"small\"] -> `suggest_categorical`\n    # - define the activation function search space [\"tanh\", \"relu\"]\n    learning_rate = trial.suggest_float(\"lr\", 1e-5, 1, log=True)\n    net_arch = trial.suggest_categorical(\"net_arch\", [\"tiny\", \"small\"])\n    activation_fn = trial.suggest_categorical(\"activation_fn\", [\"tanh\", \"relu\"])\n\n    ### END OF YOUR CODE\n\n    # Display true values\n    trial.set_user_attr(\"gamma_\", gamma)\n    trial.set_user_attr(\"n_steps\", n_steps)\n\n    net_arch = [\n        {\"pi\": [64], \"vf\": [64]}\n        if net_arch == \"tiny\"\n        else {\"pi\": [64, 64], \"vf\": [64, 64]}\n    ]\n\n    activation_fn = {\"tanh\": nn.Tanh, \"relu\": nn.ReLU}[activation_fn]\n\n    return {\n        \"n_steps\": n_steps,\n        \"gamma\": gamma,\n        \"learning_rate\": learning_rate,\n        \"max_grad_norm\": max_grad_norm,\n        \"policy_kwargs\": {\n            \"net_arch\": net_arch,\n            \"activation_fn\": activation_fn,\n        },\n    }","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:13.404734Z","iopub.execute_input":"2022-07-12T05:21:13.405092Z","iopub.status.idle":"2022-07-12T05:21:13.418523Z","shell.execute_reply.started":"2022-07-12T05:21:13.405055Z","shell.execute_reply":"2022-07-12T05:21:13.417668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from stable_baselines3.common.callbacks import EvalCallback\nimport gym\n\nclass TrialEvalCallback(EvalCallback):\n    \"\"\"\n    Callback used for evaluating and reporting a trial.\n    \n    :param eval_env: Evaluation environement\n    :param trial: Optuna trial object\n    :param n_eval_episodes: Number of evaluation episodes\n    :param eval_freq:   Evaluate the agent every ``eval_freq`` call of the callback.\n    :param deterministic: Whether the evaluation should\n        use a stochastic or deterministic policy.\n    :param verbose:\n    \"\"\"\n\n    def __init__(\n        self,\n        eval_env: gym.Env,\n        trial: optuna.Trial,\n        n_eval_episodes: int = 5,\n        eval_freq: int = 10000,\n        deterministic: bool = True,\n        verbose: int = 0,\n    ):\n\n        super().__init__(\n            eval_env=eval_env,\n            n_eval_episodes=n_eval_episodes,\n            eval_freq=eval_freq,\n            deterministic=deterministic,\n            verbose=verbose,\n        )\n        self.trial = trial\n        self.eval_idx = 0\n        self.is_pruned = False\n\n    def _on_step(self) -> bool:\n        if self.eval_freq > 0 and self.n_calls % self.eval_freq == 0:\n            # Evaluate policy (done in the parent class)\n            super()._on_step()\n            self.eval_idx += 1\n            # Send report to Optuna\n            self.trial.report(self.last_mean_reward, self.eval_idx)\n            # Prune trial if need\n            if self.trial.should_prune():\n                self.is_pruned = True\n                return False\n        return True","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:13.419911Z","iopub.execute_input":"2022-07-12T05:21:13.420490Z","iopub.status.idle":"2022-07-12T05:21:13.431708Z","shell.execute_reply.started":"2022-07-12T05:21:13.420446Z","shell.execute_reply":"2022-07-12T05:21:13.430750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from stable_baselines3.common.env_util import make_vec_env\ndef objective(trial: optuna.Trial) -> float:\n    \"\"\"\n    Objective function using by Optuna to evaluate\n    one configuration (i.e., one set of hyperparameters).\n\n    Given a trial object, it will sample hyperparameters,\n    evaluate it and report the result (mean episodic reward after training)\n\n    :param trial: Optuna trial object\n    :return: Mean episodic reward after training\n    \"\"\"\n\n    kwargs = DEFAULT_HYPERPARAMS.copy()\n    kwargs.update(sample_ppo_params(trial))\n    model = PPO(**kwargs, batch_size=8)\n\n    \n    eval_envs = make_vec_env(lambda: kore_env, N_EVAL_ENVS)\n\n    \n    eval_callback = TrialEvalCallback(eval_envs, trial, N_EVAL_EPISODES, EVAL_FREQ, deterministic=True)\n\n \n\n    nan_encountered = False\n    try:\n        # Train the model\n        model.learn(N_TIMESTEPS, callback=eval_callback)\n    except AssertionError as e:\n        # Sometimes, random hyperparams can generate NaN\n        print(e)\n        nan_encountered = True\n    finally:\n        # Free memory\n        model.env.close()\n        eval_envs.close()\n\n    # Tell the optimizer that the trial failed\n    if nan_encountered:\n        return float(\"nan\")\n\n    if eval_callback.is_pruned:\n        raise optuna.exceptions.TrialPruned()\n        \n    return eval_callback.last_mean_reward","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:13.433274Z","iopub.execute_input":"2022-07-12T05:21:13.433857Z","iopub.status.idle":"2022-07-12T05:21:13.444701Z","shell.execute_reply.started":"2022-07-12T05:21:13.433820Z","shell.execute_reply":"2022-07-12T05:21:13.443776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch as th\n\n# Set pytorch num threads to 1 for faster training\nth.set_num_threads(1)\n# Select the sampler, can be random, TPESampler, CMAES, ...\nsampler = TPESampler(n_startup_trials=N_STARTUP_TRIALS)\n# Do not prune before 1/3 of the max budget is used\npruner = MedianPruner(\n    n_startup_trials=N_STARTUP_TRIALS, n_warmup_steps=N_EVALUATIONS // 3\n)\n# Create the study and start the hyperparameter optimization\nstudy = optuna.create_study(sampler=sampler, pruner=pruner, direction=\"maximize\")\n\ntry:\n    study.optimize(objective, n_trials=N_TRIALS, n_jobs=N_JOBS, timeout=TIMEOUT)\nexcept KeyboardInterrupt:\n    pass\n\nprint(\"Number of finished trials: \", len(study.trials))\n\nprint(\"Best trial:\")\ntrial = study.best_trial\n\nprint(f\"  Value: {trial.value}\")\n\nprint(\"  Params: \")\nfor key, value in trial.params.items():\n    print(f\"    {key}: {value}\")\n\nprint(\"  User attrs:\")\nfor key, value in trial.user_attrs.items():\n    print(f\"    {key}: {value}\")\n\n# # Write report\n# study.trials_dataframe().to_csv(\"hyperparameter-tuning.csv\")\n\n# fig1 = plot_optimization_history(study)\n# fig2 = plot_param_importances(study)\n\n# fig1.show()\n# fig2.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:21:13.446043Z","iopub.execute_input":"2022-07-12T05:21:13.446527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}