{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install stable-baselines3","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"execution":{"iopub.status.busy":"2022-08-11T13:45:16.360553Z","iopub.execute_input":"2022-08-11T13:45:16.360918Z","iopub.status.idle":"2022-08-11T13:45:22.677728Z","shell.execute_reply.started":"2022-08-11T13:45:16.360870Z","shell.execute_reply":"2022-08-11T13:45:22.676862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gym\nfrom kaggle_environments import make, evaluate\n\nimport os\nimport numpy as np\nimport torch as th\nfrom torch import nn as nn\nimport torch.nn.functional as F\nfrom stable_baselines3 import DQN\nfrom stable_baselines3 import PPO\nfrom stable_baselines3.common.monitor import Monitor\nfrom stable_baselines3.common.vec_env import DummyVecEnv\nfrom stable_baselines3.common.monitor import load_results\nfrom stable_baselines3.common.torch_layers import NatureCNN\nfrom stable_baselines3.common.policies import ActorCriticPolicy, ActorCriticCnnPolicy\nfrom stable_baselines3.common.torch_layers import BaseFeaturesExtractor","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2022-08-11T13:45:22.680128Z","iopub.execute_input":"2022-08-11T13:45:22.680485Z","iopub.status.idle":"2022-08-11T13:45:22.687723Z","shell.execute_reply.started":"2022-08-11T13:45:22.680446Z","shell.execute_reply":"2022-08-11T13:45:22.686750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Doku: https://www.gymlibrary.ml/content/environment_creation/","metadata":{}},{"cell_type":"code","source":"class ConnectFourGym(gym.Env):\n    def __init__(self, agent2=\"random\"):\n        ks_env = make(\"connectx\", debug=True)\n        self.env = ks_env.train([None, agent2])\n        self.rows = ks_env.configuration.rows\n        self.columns = ks_env.configuration.columns\n        # action_space = Auswahlmöglichkeiten des agents\n        # observation_space = Struktur der Umgebung \n        self.action_space = gym.spaces.Discrete(self.columns)\n        self.observation_space = gym.spaces.Box(low=0, high=1, \n                                            shape=(1,self.rows,self.columns), dtype=np.float)\n        # Max und Min Belohnung\n        self.reward_range = (-10 ,1)\n        # StableBaselines Fehler, wenn nicht definiert\n        self.spec = None\n        self.metadata = None\n        # zurücksetzen nach spiel\n    def reset(self):\n        self.obs = self.env.reset()\n        return np.array(self.obs['board']).reshape(1,self.rows,self.columns)/2\n    def change_reward(self, old_reward, done):\n        if old_reward == 1: # Unser Agent hat gewonnen\n            return 1\n        elif done: # Gegner gewinnt\n            return -1\n        else: # Belohnung 1/42\n            return 1/(self.rows*self.columns)\n    # setzen einer Marke\n    def step(self, action):\n        is_valid = (self.obs['board'][int(action)] == 0)\n        if is_valid: # spiel beginnt\n            self.obs, old_reward, done, _ = self.env.step(int(action))\n            reward = self.change_reward(old_reward, done)\n        else: # spiel wird beendet und Agent wird bestraft\n            reward, done, _ = -10, True, {}\n        return np.array(self.obs['board']).reshape(1,self.rows,self.columns)/2, reward, done, _","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:22.688919Z","iopub.execute_input":"2022-08-11T13:45:22.689273Z","iopub.status.idle":"2022-08-11T13:45:22.706574Z","shell.execute_reply.started":"2022-08-11T13:45:22.689236Z","shell.execute_reply":"2022-08-11T13:45:22.705795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"env = ConnectFourGym()\nenv","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:22.708484Z","iopub.execute_input":"2022-08-11T13:45:22.708738Z","iopub.status.idle":"2022-08-11T13:45:22.735889Z","shell.execute_reply.started":"2022-08-11T13:45:22.708713Z","shell.execute_reply":"2022-08-11T13:45:22.735003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loggging Infos\nlog_dir = \"log/\"\nos.makedirs(log_dir, exist_ok=True)\n\nenv = Monitor(env, log_dir, allow_early_resets=True)\nenv","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:22.739429Z","iopub.execute_input":"2022-08-11T13:45:22.739847Z","iopub.status.idle":"2022-08-11T13:45:22.748044Z","shell.execute_reply.started":"2022-08-11T13:45:22.739713Z","shell.execute_reply":"2022-08-11T13:45:22.747060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nenv = DummyVecEnv([lambda: env])\nenv","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:22.750238Z","iopub.execute_input":"2022-08-11T13:45:22.750943Z","iopub.status.idle":"2022-08-11T13:45:22.758039Z","shell.execute_reply.started":"2022-08-11T13:45:22.750852Z","shell.execute_reply":"2022-08-11T13:45:22.756997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"env.observation_space.sample()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:22.759450Z","iopub.execute_input":"2022-08-11T13:45:22.759953Z","iopub.status.idle":"2022-08-11T13:45:22.769653Z","shell.execute_reply.started":"2022-08-11T13:45:22.759891Z","shell.execute_reply":"2022-08-11T13:45:22.768636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learner = PPO('MlpPolicy', env)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:22.772629Z","iopub.execute_input":"2022-08-11T13:45:22.773106Z","iopub.status.idle":"2022-08-11T13:45:22.786519Z","shell.execute_reply.started":"2022-08-11T13:45:22.773076Z","shell.execute_reply":"2022-08-11T13:45:22.785848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nlearner.learn(total_timesteps=140000)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:22.788852Z","iopub.execute_input":"2022-08-11T13:45:22.789212Z","iopub.status.idle":"2022-08-11T13:45:34.579951Z","shell.execute_reply.started":"2022-08-11T13:45:22.789178Z","shell.execute_reply":"2022-08-11T13:45:34.579009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = load_results(log_dir)['r']\ndf.rolling(window=10000).mean().plot()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:34.581266Z","iopub.execute_input":"2022-08-11T13:45:34.581603Z","iopub.status.idle":"2022-08-11T13:45:34.712950Z","shell.execute_reply.started":"2022-08-11T13:45:34.581565Z","shell.execute_reply":"2022-08-11T13:45:34.712113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learner.predict(env.reset())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:34.714487Z","iopub.execute_input":"2022-08-11T13:45:34.714869Z","iopub.status.idle":"2022-08-11T13:45:34.728875Z","shell.execute_reply.started":"2022-08-11T13:45:34.714831Z","shell.execute_reply":"2022-08-11T13:45:34.727775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def testagent(obs, config):\n    import numpy as np\n    #print(obs)\n    obs = np.array(obs['board']).reshape(1, config.rows, config.columns)/2\n    #print(obs)\n    action, _ = learner.predict(obs)\n    return int(action)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:34.730880Z","iopub.execute_input":"2022-08-11T13:45:34.731365Z","iopub.status.idle":"2022-08-11T13:45:34.738144Z","shell.execute_reply.started":"2022-08-11T13:45:34.731329Z","shell.execute_reply":"2022-08-11T13:45:34.736983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_win_percentages(agent1, agent2, n_rounds=100):\n    config = {'rows': 6, 'columns': 7, 'inarow': 4}        \n    outcomes = evaluate(\"connectx\", [agent1, agent2], config, [], n_rounds//2)     \n    outcomes += [[b,a] for [a,b] in evaluate(\"connectx\", [agent2, agent1], config, [], n_rounds-n_rounds//2)]\n    print(\"Agent 1 Win Percentage:\", np.round(outcomes.count([1,-1])/len(outcomes), 2))\n    print(\"Agent 2 Win Percentage:\", np.round(outcomes.count([-1,1])/len(outcomes), 2))\n    print(\"Number of Invalid Plays by Agent 1:\", outcomes.count([None, 0]))\n    print(\"Number of Invalid Plays by Agent 2:\", outcomes.count([0, None]))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:34.739827Z","iopub.execute_input":"2022-08-11T13:45:34.740432Z","iopub.status.idle":"2022-08-11T13:45:34.750839Z","shell.execute_reply.started":"2022-08-11T13:45:34.740396Z","shell.execute_reply":"2022-08-11T13:45:34.750103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_win_percentages(agent1=testagent, agent2=\"random\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:34.752335Z","iopub.execute_input":"2022-08-11T13:45:34.752711Z","iopub.status.idle":"2022-08-11T13:45:38.051226Z","shell.execute_reply.started":"2022-08-11T13:45:34.752675Z","shell.execute_reply":"2022-08-11T13:45:38.049707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#env = make(\"connectx\", debug=True)\n#env.run([testagent, \"random\"])\n\n# Show the game\n#env.render(mode=\"ipython\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:38.054611Z","iopub.execute_input":"2022-08-11T13:45:38.054874Z","iopub.status.idle":"2022-08-11T13:45:38.310878Z","shell.execute_reply.started":"2022-08-11T13:45:38.054845Z","shell.execute_reply":"2022-08-11T13:45:38.310069Z"},"trusted":true},"execution_count":null,"outputs":[]}]}