{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Writing classes for three different enemies","metadata":{}},{"cell_type":"markdown","source":"First enemy is just making ships","metadata":{}},{"cell_type":"code","source":"%%writefile opponent_1.py\n\nfrom kaggle_environments.envs.kore_fleets.helpers import *\n\ndef agent(obs, config):\n    board = Board(obs, config)\n\n    me = board.current_player\n    turn = board.step\n    spawn_cost = board.configuration.spawn_cost\n    kore_left = me.kore\n\n    for shipyard in me.shipyards:\n        if shipyard.ship_count > 10:\n            direction = Direction.from_index(turn % 4)\n            action = ShipyardAction.launch_fleet_with_flight_plan(2, direction.to_char())\n            shipyard.next_action = action\n        elif kore_left > spawn_cost * shipyard.max_spawn:\n            action = ShipyardAction.spawn_ships(shipyard.max_spawn)\n            shipyard.next_action = action\n            kore_left -= spawn_cost * shipyard.max_spawn\n        elif kore_left > spawn_cost:\n            action = ShipyardAction.spawn_ships(1)\n            shipyard.next_action = action\n            kore_left -= spawn_cost\n\n    return me.next_actions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Second enemy acts randomly","metadata":{}},{"cell_type":"code","source":"%%writefile opponent_2.py\nfrom custom_kore_env import CustomKoreEnv\n\nfrom kaggle_environments.envs.kore_fleets.helpers import *\nimport numpy as np\n\ncustom_env = CustomKoreEnv(agent2='random')\n\ndef agent(obs: Observation, config: Configuration):\n    model_observation = np.reshape(custom_env.build_observation(obs), [1, 4, 21, 21])\n    action = np.random.randint(0, 4)\n    board = Board(obs, config)\n    custom_env.board = board\n    if board.current_player.shipyards:\n        board.current_player.shipyards[0].next_action = custom_env.match_action(action)\n    return board.current_player.next_actions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Trird enemy is our previous model","metadata":{}},{"cell_type":"code","source":"%%writefile opponent_3.py\nfrom kaggle_environments.envs.kore_fleets.helpers import *\nfrom custom_kore_env import CustomKoreEnv\nimport agent_class\nimport torch\nimport numpy as np\nfrom kaggle_environments.envs.kore_fleets.helpers import *\nagent_path = 'checkpoint_local.pth'\nagent_main = agent_class.Agent(state_size=4, action_size=4, seed=42)\nagent_main.qnetwork_local.load_state_dict(torch.load(agent_path, map_location='cpu'))\ncustom_env = CustomKoreEnv(agent2='random')\n\ndef agent(obs: Observation, config: Configuration):\n    model_observation = np.reshape(custom_env.build_observation(obs), [1, 4, 21, 21])\n    action = agent_main.act(model_observation)\n    board = Board(obs, config)\n    custom_env.board = board\n    if board.current_player.shipyards:\n        board.current_player.shipyards[0].next_action = custom_env.match_action(action)\n    return board.current_player.next_actions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Writing the environment class","metadata":{}},{"cell_type":"markdown","source":"The code for this environment was taken from https://www.kaggle.com/code/gabrielmilan/drl-ga-starter and was changed a bit.","metadata":{}},{"cell_type":"code","source":"%%writefile custom_kore_env.py\n\nimport gym\nimport torch\nimport torch.nn as nn\nfrom random import choice\nfrom typing import Union\n\nfrom gym import Env, spaces\nfrom kaggle_environments import make\nfrom kaggle_environments.envs.kore_fleets.helpers import (\n    Board,\n    Configuration,\n    Observation,\n    ShipyardAction,\n)\nimport numpy as np\n\nclass CustomKoreEnv(Env):\n    \"\"\"\n    This is a custom Kore environment, adapted for use with\n    OpenAI Gym. We don't actually need to use OpenAI Gym in\n    this example, but I've chosen to do it for the sake of\n    compatibility for those who already have code for it.\n    \"\"\"\n\n    metadata = {\"render.modes\": [\"human\"]}\n    \n    def __init__(self, agent2):\n        super().__init__()\n        # Initialize the actual environment\n        kore_env = make(\"kore_fleets\", debug=True)\n        self.env = kore_env.train([None, agent2])\n        self.env_configuration: Configuration = kore_env.configuration\n        map_size = self.env_configuration.size\n        self.board: Board = None\n\n        # Our observation space will be a matrix of size\n        # (map_size, map_size, 4).\n        #\n        # - The first layer is the kore count, which has\n        #   its values mapped to [0, 1]\n        #\n        # - The second layer is the fleet size, which has\n        #   its values mapped to [-1, 1] (negative values\n        #   are used to represent enemy fleets)\n        #\n        # - The third layer represents possible places that\n        #   the enemy fleets can be at the next turn. All\n        #   values are either -1 (enemy can be there) or\n        #   0 (enemy can't be there).\n        #\n        # - The fourth layer represents the amount of kore\n        #   that each fleet is carrying.\n        #\n        self.observation_space = spaces.Box(\n            low=-1, high=1, shape=(map_size, map_size, 4), dtype=np.float64\n        )\n\n        # Our action space will be an array of shape (4,).\n        #\n        # - The following combinations will map to the\n        #   respective actions:\n        #\n        #   - 0 -> Do nothing\n        #   - 1 -> Build a new ship\n        #   - 2 -> Launch a fleet with a simple flight\n        #     plan (e.g. N, E, S, W). All ships will be\n        #     used to launch the fleet.\n        #   - 3 -> Same as 2, but only half of the\n        #     ships will be used to launch the fleet.\n        #\n        self.action_space = spaces.Box(low=0, high=1, shape=(4,), dtype=np.float64)\n\n    def map_value(self, value: Union[int, float], enemy: bool = False,) -> float:\n        \"\"\"\n        Helper function for build_observation. For this to work, we must\n        assume the following:\n            - The maximum value of Kore in a single square is expected to be\n                500.\n            - The maximum fleet size is expected to be 1000.\n            - The maximum amount of kore a fleet can carry is expected to be\n                5000.\n        Maps a value to a range of [-1, 1] if enemy, or [0, 1] otherwise.\n        \"\"\"\n        MAX_NATURAL_KORE_IN_SQUARE = 500\n        MAX_ASSUMED_FLEET_SIZE = 1000\n        MAX_ASSUMED_KORE_IN_FLEET = 5000\n        max_value = float(\n            max(\n                MAX_NATURAL_KORE_IN_SQUARE,\n                MAX_ASSUMED_FLEET_SIZE,\n                MAX_ASSUMED_KORE_IN_FLEET,\n            )\n        )\n        val = value / max_value\n        if enemy:\n            return -val\n        return val\n\n    def build_observation(self, raw_observation: Observation) -> np.ndarray:\n        \"\"\"\n        Our observation space will be a matrix of size\n        (map_size, map_size, 4).\n        \n        - The first layer is the kore count, which has\n            its values mapped to [0, 1]\n        \n        - The second layer is the fleet size, which has\n            its values mapped to [-1, 1] (negative values\n            are used to represent enemy fleets)\n        \n        - The third layer represents possible places that\n            the enemy fleets can be at the next turn. All\n            values are either -1 (enemy can be there) or\n            0 (enemy can't be there).\n        \n        - The fourth layer represents the amount of kore\n            that each fleet is carrying.\n        \"\"\"\n        # Build the Board object that will help us build the layers\n        board = Board(raw_observation, self.env_configuration)\n\n        # Building the kore layer\n        kore_layer = np.array(raw_observation.kore).reshape(\n            self.env_configuration.size, self.env_configuration.size\n        )\n\n        # Building the fleet layer\n        fleet_layer = np.zeros(\n            (self.env_configuration.size, self.env_configuration.size)\n        )\n        # - Get fleets and shipyards on the map\n        fleets = [fleet for _, fleet in board.fleets.items()]\n        shipyards = [shipyard for _, shipyard in board.shipyards.items()]\n        # - Iterate over fleets, getting its position and size\n        for fleet in fleets:\n            # - Get the position of the fleet\n            position = fleet.position\n            x, y = position.x, position.y\n            # - Get the size of the fleet\n            size = fleet.ship_count\n            # - Check if the fleet is an enemy fleet\n            if fleet.player != board.current_player:\n                multilpier = -1\n            else:\n                multilpier = 1\n            # - Set the fleet layer to the size of the fleet\n            fleet_layer[x, y] = multilpier * self.map_value(size)\n        # - Iterate over shipyards, getting its position and size\n        for shipyard in shipyards:\n            # - Get the position of the shipyard\n            position = shipyard.position\n            x, y = position.x, position.y\n            # - Get the size of the shipyard\n            size = shipyard.ship_count\n            # - Check if the shipyard is an enemy shipyard\n            if shipyard.player != board.current_player:\n                multilpier = -1\n            else:\n                multilpier = 1\n            # - Set the fleet layer to the size of the shipyard\n            fleet_layer[x, y] = multilpier * self.map_value(size)\n\n        # Building the enemy positions layer\n        enemy_positions_layer = np.zeros(\n            (self.env_configuration.size, self.env_configuration.size)\n        )\n        # - Iterate over fleets\n        for fleet in fleets:\n            # If fleet is ours, skip it\n            if fleet.player == board.current_player:\n                continue\n            # - Get the position of the fleet\n            position = fleet.position\n            x, y = position.x, position.y\n            # - Set the enemy positions layer to -1\n            enemy_positions_layer[x, y] = -1\n            enemy_positions_layer[x - 1, y] = -1\n            if x + 1 >= self.env_configuration.size:\n                enemy_positions_layer[0, y]\n            else:\n                enemy_positions_layer[x + 1, y] = -1\n            enemy_positions_layer[x, y - 1] = -1\n            if y + 1 >= self.env_configuration.size:\n                enemy_positions_layer[x, 0]\n            else:\n                enemy_positions_layer[x, y + 1] = -1\n\n        # Building the kore layer\n        kore_layer = np.zeros(\n            (self.env_configuration.size, self.env_configuration.size)\n        )\n        # - Iterate over fleets\n        for fleet in fleets:\n            # - Get the position of the fleet\n            position = fleet.position\n            x, y = position.x, position.y\n            # - Get the amount of kore the fleet is carrying\n            kore = fleet.kore\n            # - Set the kore layer to the amount of kore\n            kore_layer[x, y] = kore\n\n        # Building our observation box\n        observation = np.zeros(\n            (self.env_configuration.size, self.env_configuration.size, 4)\n        )\n        observation[:, :, 0] = kore_layer\n        observation[:, :, 1] = fleet_layer\n        observation[:, :, 2] = enemy_positions_layer\n        observation[:, :, 3] = kore_layer\n\n        return observation\n\n    def map_reward(self, old_reward: Union[int, float]):\n        \"\"\"\n        If you want to modify the reward, you can do it here.\n        \"\"\"\n        return old_reward\n    def match_action(self, action_space: np.ndarray) -> ShipyardAction:\n        \"\"\"\n        This function will match the action space to a\n        ShipyardAction.\n        \"\"\" \n        # If there are no shipyards, return None\n        if len(self.board.current_player.shipyards) == 0:\n            return None\n        # - Check if the action space is 0\n        if action_space == 0:\n            return None\n        # - Check if the action space is 1\n        elif action_space == 1:\n            return ShipyardAction.spawn_ships(1)\n        # - Check if the action space is 2\n        elif action_space == 2:\n            ships_in_fleet = self.board.current_player.shipyards[0].ship_count\n            if ships_in_fleet == 0:\n                return None\n            return ShipyardAction.launch_fleet_with_flight_plan(\n                self.board.current_player.shipyards[0].ship_count,\n                choice([\"N\", \"E\", \"S\", \"W\"])\n            )\n        # - Check if the action space is 3\n        elif action_space == 3:\n            ships_in_fleet = int(self.board.current_player.shipyards[0].ship_count / 2)\n            if ships_in_fleet == 0 and self.board.current_player.shipyards[0].ship_count > 0:\n                ships_in_fleet = 1\n            else:\n                return None\n            return ShipyardAction.launch_fleet_with_flight_plan(\n                ships_in_fleet,\n                choice([\"N\", \"E\", \"S\", \"W\"])\n            )\n        else:\n            raise ValueError(f\"Invalid action space: {action_space}\")\n\n    def reset(self):\n        \"\"\"\n        Resets the environment.\n        \"\"\"\n        self.raw_observation = self.env.reset()\n        obs = self.build_observation(self.raw_observation)\n        return obs\n\n    def step(self, action_space: np.ndarray):\n        \"\"\"\n        Performs an action in the environment.\n        \"\"\"\n        # Get the Board object and update it\n        self.board = Board(self.raw_observation, self.env_configuration)\n        # Sets done if no shipyards are left\n        if len(self.board.current_player.shipyards) == 0:\n            return np.zeros((21, 21, 4)), 0, True, {}\n        # Get the action for the shipyard\n        action = self.match_action(action_space)\n        self.board.current_player.shipyards[0].next_action = action\n        self.raw_observation, old_reward, done, info = self.env.step(\n            self.board.current_player.next_actions\n        )\n        observation = self.build_observation(self.raw_observation)\n        reward = self.map_reward(old_reward)\n        return observation, reward, done, info","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Qnetwork","metadata":{}},{"cell_type":"code","source":"%%writefile qnetwork.py\n\nimport torch\nimport torch.nn as nn\n\nclass QNetwork(nn.Module):\n    \"\"\" Actor (Policy) Model.\"\"\"\n    def __init__(self, state_size, action_size, seed):\n        super(QNetwork, self).__init__() ## calls __init__ method of nn.Module class\n        self.conv = nn.Sequential(\n            nn.Conv2d(in_channels=4, out_channels=64, kernel_size=8, stride=1),\n            #nn.ReLU(),\n            nn.BatchNorm2d(64),\n            #nn.Dropout(p=0.1),\n            nn.Conv2d(in_channels=64, out_channels=128, kernel_size=10, stride=1),\n            #nn.ReLU(),\n            nn.BatchNorm2d(128),\n            nn.Conv2d(in_channels=128, out_channels=256, kernel_size=3, stride=1),\n            #nn.ReLU(),\n            nn.BatchNorm2d(256),\n            #nn.Dropout(p=0.2),\n        )\n        self.linear = nn.Sequential(\n            nn.Flatten(),\n            nn.Linear(3*3*256, 256),\n            nn.Dropout(p=0.2),\n            nn.Linear(256, 32),\n            nn.Linear(32, 4),\n            nn.Softmax(dim=-1),\n        )\n        \n    def forward(self,x):\n        x = self.conv(x)\n        #x = x.view(x.size(0), -1)\n        x = self.linear(x)\n        return x","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Writing the class for our model","metadata":{}},{"cell_type":"code","source":"%%writefile agent_class.py\n\nimport numpy as np\nimport random \nfrom collections import namedtuple, deque\n\nimport torch\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom qnetwork import QNetwork\n\nBUFFER_SIZE = int(0.9e4)  #replay buffer size\nBATCH_SIZE = 5750         # minibatch size\nGAMMA = 0.99          # discount factor\nTAU = 0.2              # for soft update of target parameters\nLR = 1e-4             # learning rate\nUPDATE_EVERY = 3200       # how often to update the network\n\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n\nclass Agent():\n    \"\"\"Interacts with and learns form environment.\"\"\"\n    \n    def __init__(self, state_size, action_size, seed):\n        \"\"\"Initialize an Agent object.\n        \n        Params\n        =======\n            state_size (int): dimension of each state\n            action_size (int): dimension of each action\n            seed (int): random seed\n        \"\"\"\n        \n        self.state_size = state_size\n        self.action_size = action_size\n        self.seed = random.seed(seed)\n\n        #seeding the network\n        torch.manual_seed(seed)\n        if torch.cuda.is_available():\n            torch.cuda.manual_seed_all(seed)\n\n        #Q- Network\n        self.qnetwork_local = QNetwork(state_size, action_size, seed).to(device)\n        self.qnetwork_target = QNetwork(state_size, action_size, seed).to(device)\n        \n        self.optimizer = optim.Adam(self.qnetwork_local.parameters(),lr=LR,)\n        \n        # Replay memory \n        self.memory = ReplayBuffer(action_size, BUFFER_SIZE, BATCH_SIZE, seed)\n        # Initialize time step (for updating every UPDATE_EVERY steps)\n        self.t_step = 0\n        \n    def step(self, state, action, reward, next_step, done, episode):\n        # Save experience in replay memory\n        self.memory.add(state, action, reward, next_step, done)\n        \n        # Learn every UPDATE_EVERY time steps.\n        self.t_step = (self.t_step+1) % UPDATE_EVERY\n        if self.t_step == 0:\n             # If enough samples are available in memory, get random subset and learn\n            if len(self.memory)>BATCH_SIZE:\n                experience = self.memory.sample()\n                self.learn(experience, GAMMA)\n    def act(self, state, eps = 0):\n        \"\"\"Returns action for given state as per current policy\n        Params\n        =======\n            state (array_like): current state\n            eps (float): epsilon, for epsilon-greedy action selection\n        \"\"\"\n        state = torch.from_numpy(state).float().to(device)\n        self.qnetwork_local.eval()\n        with torch.no_grad():\n            action_values = self.qnetwork_local(state).squeeze()\n        self.qnetwork_local.train()\n        #r = random.random()\n        #Epsilon -greedy action selection\n        #if r > eps:\n        if random.random() > eps:\n            return np.argmax(action_values.cpu().data.numpy())\n        else:\n            return random.choice(np.arange(self.action_size))\n        return action_values\n            \n    def learn(self, experiences, gamma):\n        \"\"\"Update value parameters using given batch of experience tuples.\n        Params\n        =======\n            experiences (Tuple[torch.Variable]): tuple of (s, a, r, s', done) tuples\n            gamma (float): discount factor\n        \"\"\"\n        states, actions, rewards, next_states, dones = experiences\n        ## TODO: compute and minimize the loss\n        criterion = torch.nn.MSELoss()\n        # Local model is one which we need to train so it's in training mode\n        self.qnetwork_local.train()\n        # Target model is one with which we need to get our target so it's in evaluation mode\n        # So that when we do a forward pass with target model it does not calculate gradient.\n        # We will update target model weights with soft_update function\n        self.qnetwork_target.eval()\n        #shape of output from the model (batch_size,action_dim)\n        #print(states.shape, 'states', states)\n        #print(next_states.shape, 'next_states', next_states)\n        #print(actions.shape, 'actions', actions)\n        predicted_targets = self.qnetwork_local(states).gather(1, actions)\n        #print(predicted_targets.shape, 'predicted_targets', predicted_targets)\n    \n        with torch.no_grad():\n            labels_next = self.qnetwork_target(next_states).detach().max(1)[0].unsqueeze(1)\n            #print(labels_next.shape, 'labels_next', labels_next)\n\n        # .detach() ->  Returns a new Tensor, detached from the current graph.\n        labels = rewards + (gamma * labels_next)\n        #print(labels.shape, 'labels', labels)\n        loss = criterion(predicted_targets,labels).to(device)\n        print(loss)\n        self.optimizer.zero_grad()\n        loss.backward()\n        self.optimizer.step()\n\n        # ------------------- update target network ------------------- #\n        self.soft_update(self.qnetwork_local,self.qnetwork_target,TAU)\n\n    def soft_update(self, local_model, target_model, tau):\n        \"\"\"Soft update model parameters.\n        θ_target = τ*θ_local + (1 - τ)*θ_target\n        Params\n        =======\n            local model (PyTorch model): weights will be copied from\n            target model (PyTorch model): weights will be copied to\n            tau (float): interpolation parameter\n        \"\"\"\n        for target_param, local_param in zip(target_model.parameters(),\n                                           local_model.parameters()):\n            target_param.data.copy_(tau*local_param.data + (1-tau)*target_param.data)\n\n            \n\nclass ReplayBuffer:\n    \"\"\"Fixed -size buffe to store experience tuples.\"\"\"\n    \n    def __init__(self, action_size, buffer_size, batch_size, seed):\n        \"\"\"Initialize a ReplayBuffer object.\n        \n        Params\n        ======\n            action_size (int): dimension of each action\n            buffer_size (int): maximum size of buffer\n            batch_size (int): size of each training batch\n            seed (int): random seed\n        \"\"\"\n        \n        self.action_size = action_size\n        self.buffer_size = buffer_size\n        self.memory = deque(maxlen=buffer_size)\n        self.batch_size = batch_size\n        self.experiences = namedtuple(\"Experience\", field_names=[\"state\",\n                                                               \"action\",\n                                                               \"reward\",\n                                                               \"next_state\",\n                                                               \"done\"])\n        self.seed = random.seed(seed)\n        \n    def add(self,state, action, reward, next_state,done):\n        \"\"\"Add a new experience to memory.\"\"\"\n        e = self.experiences(state,action,reward,next_state,done)\n        self.memory.append(e)\n        \n    def sample(self):\n        \"\"\"Randomly sample a batch of experiences from memory\"\"\"\n        experiences = random.sample(self.memory,k=self.batch_size)\n        \n        states = torch.from_numpy(np.vstack([e.state for e in experiences if e is not None])).float().to(device)\n        actions = torch.from_numpy(np.vstack([e.action for e in experiences if e is not None])).long().to(device)\n        rewards = torch.from_numpy(np.vstack([e.reward for e in experiences if e is not None])).float().to(device)\n        next_states = torch.from_numpy(np.vstack([e.next_state for e in experiences if e is not None])).float().to(device)\n        dones = torch.from_numpy(np.vstack([e.done for e in experiences if e is not None]).astype(np.uint8)).float().to(device)\n        return (states,actions,rewards,next_states,dones)\n\n    def __len__(self):\n        \"\"\"Return the current size of internal memory.\"\"\"\n        return len(self.memory)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom agent_class import Agent\nfrom custom_kore_env import CustomKoreEnv\nfrom collections import namedtuple, deque\nimport numpy as np\nimport os\n\ndef dqn(round, n_episodes= 250, max_t = 400, eps_start=1.0, \n        eps_end = 0.04, eps_decay=0.995,):\n    \"\"\"Deep Q-Learning\n    \n    Params\n    ======\n        n_episodes (int): maximum number of training epsiodes\n        max_t (int): maximum number of timesteps per episode\n        eps_start (float): starting value of epsilon, for epsilon-greedy action selection\n        eps_end (float): minimum value of epsilon \n        eps_decay (float): mutiplicative factor (per episode) for decreasing epsilon\n        \n    \"\"\"\n    opponents = ['opponent_1.py', 'opponent_2.py', 'opponent_3.py']\n    env = CustomKoreEnv(agent2=opponents[round])\n    agent = Agent(state_size=4, action_size=4, seed=42)\n\n    if os.path.exists('checkpoint_local.pth') and os.path.exists('checkpoint_target.pth'):\n        agent.qnetwork_local.load_state_dict(torch.load('checkpoint_local.pth', map_location='cpu'))\n        agent.qnetwork_target.load_state_dict(torch.load('checkpoint_target.pth', map_location='cpu'))\n    scores = [] # list containing score from each episode\n    scores_window = deque(maxlen=100) # last 100 scores\n    eps = eps_start\n    for i_episode in range(1, n_episodes+1):\n        state = env.reset()\n        score = 0\n        t = 0\n        done = False\n        while done == False and t < max_t:\n            t += 1\n            state = np.reshape(state, [1, 4, 21, 21])\n            action = agent.act(state, eps)\n            next_state, reward, done, _ = env.step(action)\n            next_state = np.reshape(next_state, [1, 4, 21, 21])\n            agent.step(state, action, reward, next_state, done, i_episode)\n            ## above step decides whether we will train(learn) the network\n            ## actor (local_qnetwork) or we will fill the replay buffer\n            ## if len replay buffer is equal to the batch size then we will\n            ## train the network or otherwise we will add experience tuple in our \n            ## replay buffer.\n            state = next_state\n            score += reward\n            if done:\n                break\n            scores_window.append(score) ## save the most recent score\n            scores.append(score) ## sae the most recent score\n            eps = max(eps*eps_decay,eps_end)## decrease the epsilon\n            print('\\rEpisode {}\\tAverage Score {:.2f}'.format(i_episode,np.mean(scores_window)), end=\"\")\n                \n            if np.mean(scores_window)>5000.0:\n                #print('\\nEnvironment solve in {:d} episodes!\\tAverage score: {:.2f}'.format(i_episode,\n                 #                                                                          np.mean(scores_window)))\n                torch.save(agent.qnetwork_local.state_dict(),'checkpoint_local.pth')\n                torch.save(agent.qnetwork_target.state_dict(),'checkpoint_target.pth')\n                #return scores\n    return scores\n\nfor epoch in range(3):                    # will be training for three rounds\n    scores= dqn(round=epoch)\n    #plot the scores\n    fig = plt.figure()\n    ax = fig.add_subplot(111)\n    plt.plot(np.arange(len(scores)),scores)\n    plt.ylabel('Score')\n    plt.xlabel('Epsiode #')\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Writing the main function","metadata":{}},{"cell_type":"code","source":"%%writefile main.py\n\nimport os\nimport sys\nKAGGLE_AGENT_PATH = \"/kaggle_simulations/agent/\"\nif os.path.exists(KAGGLE_AGENT_PATH):\n    # We're in the kaggle target system\n    sys.path.insert(0, os.path.join(KAGGLE_AGENT_PATH))\n    agent_path = os.path.join(KAGGLE_AGENT_PATH, 'checkpoint_local.pth')\nelse:\n    # We're somewhere else\n    sys.path.insert(0, os.path.join(os.getcwd()))\n    agent_path = 'checkpoint_local.pth'\n\nimport gym\nimport torch\nimport torch.nn as nn\nfrom random import choice\nfrom typing import Union\n\nfrom gym import Env, spaces\nfrom kaggle_environments import make\nfrom kaggle_environments.envs.kore_fleets.helpers import (\n    Board,\n    Configuration,\n    Observation,\n    ShipyardAction,\n)\nimport numpy as np\nimport agent_class\nimport torch\nfrom custom_kore_env import CustomKoreEnv\n\n    \nagent_main = agent_class.Agent(state_size=4, action_size=4, seed=42)\nagent_main.qnetwork_local.load_state_dict(torch.load(agent_path, map_location='cpu'))\ncustom_env = CustomKoreEnv(agent2='random')\n\ndef agent(obs: Observation, config: Configuration):\n    model_observation = np.reshape(custom_env.build_observation(obs), [1, 4, 21, 21])\n    action = agent_main.act(model_observation)\n    board = Board(obs, config)\n    custom_env.board = board\n    if board.current_player.shipyards:\n        board.current_player.shipyards[0].next_action = custom_env.match_action(action)\n    return board.current_player.next_actions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking the main function","metadata":{}},{"cell_type":"code","source":"from custom_kore_env import CustomKoreEnv\nenv_test = CustomKoreEnv(agent2='random')\n\nenv_test = make(\"kore_fleets\", debug=True)\nenv_test.run(['main.py', 'random'])\nenv_test.render(mode=\"ipython\", width=1000, height=800)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making a submission","metadata":{}},{"cell_type":"code","source":"!tar -czf submission_3.tar.gz main.py checkpoint_local.pth custom_kore_env.py qnetwork.py agent_class.py","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}