Tianshou/test/discrete/test_ppo.py

import argparse
import os
import pprint

import gym
import numpy as np
import torch
from torch.utils.tensorboard import SummaryWriter

from tianshou.data import Collector, VectorReplayBuffer
from tianshou.env import DummyVectorEnv
from tianshou.policy import PPOPolicy
from tianshou.trainer import onpolicy_trainer
from tianshou.utils import TensorboardLogger
from tianshou.utils.net.common import Net
from tianshou.utils.net.discrete import Actor, Critic


def get_args():
    parser = argparse.ArgumentParser()
    parser.add_argument('--task', type=str, default='CartPole-v0')
    parser.add_argument('--seed', type=int, default=0)
    parser.add_argument('--buffer-size', type=int, default=20000)
    parser.add_argument('--lr', type=float, default=3e-4)
    parser.add_argument('--gamma', type=float, default=0.99)
    parser.add_argument('--epoch', type=int, default=10)
    parser.add_argument('--step-per-epoch', type=int, default=50000)
    parser.add_argument('--step-per-collect', type=int, default=2000)
    parser.add_argument('--repeat-per-collect', type=int, default=10)
    parser.add_argument('--batch-size', type=int, default=64)
    parser.add_argument('--hidden-sizes', type=int, nargs='*', default=[64, 64])
    parser.add_argument('--training-num', type=int, default=20)
    parser.add_argument('--test-num', type=int, default=100)
    parser.add_argument('--logdir', type=str, default='log')
    parser.add_argument('--render', type=float, default=0.)
    parser.add_argument(
        '--device', type=str, default='cuda' if torch.cuda.is_available() else 'cpu'
    )
    # ppo special
    parser.add_argument('--vf-coef', type=float, default=0.5)
    parser.add_argument('--ent-coef', type=float, default=0.0)
    parser.add_argument('--eps-clip', type=float, default=0.2)
    parser.add_argument('--max-grad-norm', type=float, default=0.5)
    parser.add_argument('--gae-lambda', type=float, default=0.95)
    parser.add_argument('--rew-norm', type=int, default=0)
    parser.add_argument('--norm-adv', type=int, default=0)
    parser.add_argument('--recompute-adv', type=int, default=0)
    parser.add_argument('--dual-clip', type=float, default=None)
    parser.add_argument('--value-clip', type=int, default=0)
    args = parser.parse_known_args()[0]
    return args


def test_ppo(args=get_args()):
    env = gym.make(args.task)
    args.state_shape = env.observation_space.shape or env.observation_space.n
    args.action_shape = env.action_space.shape or env.action_space.n
    # train_envs = gym.make(args.task)
    # you can also use tianshou.env.SubprocVectorEnv
    train_envs = DummyVectorEnv(
        [lambda: gym.make(args.task) for _ in range(args.training_num)]
    )
    # test_envs = gym.make(args.task)
    test_envs = DummyVectorEnv(
        [lambda: gym.make(args.task) for _ in range(args.test_num)]
    )
    # seed
    np.random.seed(args.seed)
    torch.manual_seed(args.seed)
    train_envs.seed(args.seed)
    test_envs.seed(args.seed)
    # model
    net = Net(args.state_shape, hidden_sizes=args.hidden_sizes, device=args.device)
    actor = Actor(net, args.action_shape, device=args.device).to(args.device)
    critic = Critic(net, device=args.device).to(args.device)
    # orthogonal initialization
    for m in set(actor.modules()).union(critic.modules()):
        if isinstance(m, torch.nn.Linear):
            torch.nn.init.orthogonal_(m.weight)
            torch.nn.init.zeros_(m.bias)
    optim = torch.optim.Adam(
        set(actor.parameters()).union(critic.parameters()), lr=args.lr
    )
    dist = torch.distributions.Categorical
    policy = PPOPolicy(
        actor,
        critic,
        optim,
        dist,
        discount_factor=args.gamma,
        max_grad_norm=args.max_grad_norm,
        eps_clip=args.eps_clip,
        vf_coef=args.vf_coef,
        ent_coef=args.ent_coef,
        gae_lambda=args.gae_lambda,
        reward_normalization=args.rew_norm,
        dual_clip=args.dual_clip,
        value_clip=args.value_clip,
        action_space=env.action_space,
        deterministic_eval=True,
        advantage_normalization=args.norm_adv,
        recompute_advantage=args.recompute_adv
    )
    # collector
    train_collector = Collector(
        policy, train_envs, VectorReplayBuffer(args.buffer_size, len(train_envs))
    )
    test_collector = Collector(policy, test_envs)
    # log
    log_path = os.path.join(args.logdir, args.task, 'ppo')
    writer = SummaryWriter(log_path)
    logger = TensorboardLogger(writer)

    def save_fn(policy):
        torch.save(policy.state_dict(), os.path.join(log_path, 'policy.pth'))

    def stop_fn(mean_rewards):
        return mean_rewards >= env.spec.reward_threshold

    # trainer
    result = onpolicy_trainer(
        policy,
        train_collector,
        test_collector,
        args.epoch,
        args.step_per_epoch,
        args.repeat_per_collect,
        args.test_num,
        args.batch_size,
        step_per_collect=args.step_per_collect,
        stop_fn=stop_fn,
        save_fn=save_fn,
        logger=logger
    )
    assert stop_fn(result['best_reward'])

    if __name__ == '__main__':
        pprint.pprint(result)
        # Let's watch its performance!
        env = gym.make(args.task)
        policy.eval()
        collector = Collector(policy, env)
        result = collector.collect(n_episode=1, render=args.render)
        rews, lens = result["rews"], result["lens"]
        print(f"Final reward: {rews.mean()}, length: {lens.mean()}")


if __name__ == '__main__':
    test_ppo()
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`import argparse`
save_fn 2020-04-11 16:54:27 +08:00			`import os`
ppo and early stop 2020-03-20 19:52:29 +08:00			`import pprint`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00
			`import gym`
ppo and early stop 2020-03-20 19:52:29 +08:00			`import numpy as np`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`import torch`
ppo and early stop 2020-03-20 19:52:29 +08:00			`from torch.utils.tensorboard import SummaryWriter`

bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`from tianshou.data import Collector, VectorReplayBuffer`
			`from tianshou.env import DummyVectorEnv`
ppo and early stop 2020-03-20 19:52:29 +08:00			`from tianshou.policy import PPOPolicy`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`from tianshou.trainer import onpolicy_trainer`
Add Weights and Biases Logger (#427) - rename BasicLogger to TensorboardLogger - refactor logger code - add WandbLogger Co-authored-by: Jiayi Weng <trinkle23897@gmail.com> 2021-08-30 10:35:02 -04:00			`from tianshou.utils import TensorboardLogger`
Numba acceleration (#193) Training FPS improvement (base commit is 94bfb32): test_pdqn: 1660 (without numba) -> 1930 discrete/test_ppo: 5100 -> 5170 since nstep has little impact on overall performance, the unit test result is: GAE: 4.1s -> 0.057s nstep: 0.3s -> 0.15s (little improvement) Others: - fix a bug in ttt set_eps - keep only sumtree in segment tree implementation - dirty fix for asyncVenv check_id test 2020-09-02 13:03:32 +08:00			`from tianshou.utils.net.common import Net`
Remove dummy net code (#123) * remove dummy net; delete two files * split code to have backbone and head * rename class * change torch.float to torch.float32 * use flatten(1) instead of view(batch, -1) * remove dummy net in docs * bugfix for rnn * fix cuda error * minor fix of docs * do not change the example code in dqn tutorial, since it is for demonstration Co-authored-by: Trinkle23897 <463003665@qq.com> 2020-07-09 22:57:01 +08:00			`from tianshou.utils.net.discrete import Actor, Critic`
ppo and early stop 2020-03-20 19:52:29 +08:00

			`def get_args():`
			`parser = argparse.ArgumentParser()`
			`parser.add_argument('--task', type=str, default='CartPole-v0')`
fix ppo 2020-04-19 14:30:42 +08:00			`parser.add_argument('--seed', type=int, default=0)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`parser.add_argument('--buffer-size', type=int, default=20000)`
make ppo discrete test script more general (#418) 2021-08-15 21:37:37 +08:00			`parser.add_argument('--lr', type=float, default=3e-4)`
fix ppo 2020-04-19 14:30:42 +08:00			`parser.add_argument('--gamma', type=float, default=0.99)`
			`parser.add_argument('--epoch', type=int, default=10)`
Trainer refactor : some definition change (#293) This PR focus on some definition change of trainer to make it more friendly to use and be consistent with typical usage in research papers, typically change `collect-per-step` to `step-per-collect`, add `update-per-step` / `episode-per-collect` accordingly, and modify the documentation. 2021-02-21 13:06:02 +08:00			`parser.add_argument('--step-per-epoch', type=int, default=50000)`
make ppo discrete test script more general (#418) 2021-08-15 21:37:37 +08:00			`parser.add_argument('--step-per-collect', type=int, default=2000)`
			`parser.add_argument('--repeat-per-collect', type=int, default=10)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`parser.add_argument('--batch-size', type=int, default=64)`
ppo benchmark (#330) 2021-03-30 11:50:35 +08:00			`parser.add_argument('--hidden-sizes', type=int, nargs='*', default=[64, 64])`
fix ppo 2020-04-19 14:30:42 +08:00			`parser.add_argument('--training-num', type=int, default=20)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`parser.add_argument('--test-num', type=int, default=100)`
			`parser.add_argument('--logdir', type=str, default='log')`
add examples, fix some bugs (#5) * update atari.py * fix setup.py pass the pytest * fix setup.py pass the pytest * add args "render" * change the tensorboard writter * change the tensorboard writter * change device, render, tensorboard log location * change device, render, tensorboard log location * remove some wrong local files * fix some tab mistakes and the envs name in continuous/test_xx.py * add examples and point robot maze environment * fix some bugs during testing examples * add dqn network and fix some args * change back the tensorboard writter's frequency to ensure ppo and a2c can write things normally * add a warning to collector * rm some unrelated files * reformat * fix a bug in test_dqn due to the model wrong selection 2020-03-28 07:27:18 +08:00			`parser.add_argument('--render', type=float, default=0.)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`parser.add_argument(`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`'--device', type=str, default='cuda' if torch.cuda.is_available() else 'cpu'`
			`)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`# ppo special`
			`parser.add_argument('--vf-coef', type=float, default=0.5)`
			`parser.add_argument('--ent-coef', type=float, default=0.0)`
			`parser.add_argument('--eps-clip', type=float, default=0.2)`
			`parser.add_argument('--max-grad-norm', type=float, default=0.5)`
ppo benchmark (#330) 2021-03-30 11:50:35 +08:00			`parser.add_argument('--gae-lambda', type=float, default=0.95)`
make ppo discrete test script more general (#418) 2021-08-15 21:37:37 +08:00			`parser.add_argument('--rew-norm', type=int, default=0)`
			`parser.add_argument('--norm-adv', type=int, default=0)`
			`parser.add_argument('--recompute-adv', type=int, default=0)`
fix ppo 2020-04-19 14:30:42 +08:00			`parser.add_argument('--dual-clip', type=float, default=None)`
make ppo discrete test script more general (#418) 2021-08-15 21:37:37 +08:00			`parser.add_argument('--value-clip', type=int, default=0)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`args = parser.parse_known_args()[0]`
			`return args`


			`def test_ppo(args=get_args()):`
			`env = gym.make(args.task)`
			`args.state_shape = env.observation_space.shape or env.observation_space.n`
			`args.action_shape = env.action_space.shape or env.action_space.n`
			`# train_envs = gym.make(args.task)`
add some docs 2020-04-03 21:28:12 +08:00			`# you can also use tianshou.env.SubprocVectorEnv`
code refactor for venv (#179) - Refacor code to remove duplicate code - Enable async simulation for all vector envs - Remove `collector.close` and rename `VectorEnv` to `DummyVectorEnv` The abstraction of vector env changed. Prior to this pr, each vector env is almost independent. After this pr, each env is wrapped into a worker, and vector envs differ with their worker type. In fact, users can just use `BaseVectorEnv` with different workers, I keep `SubprocVectorEnv`, `ShmemVectorEnv` for backward compatibility. Co-authored-by: n+e <463003665@qq.com> Co-authored-by: magicly <magicly007@gmail.com> 2020-08-19 15:00:24 +08:00			`train_envs = DummyVectorEnv(`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`[lambda: gym.make(args.task) for _ in range(args.training_num)]`
			`)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`# test_envs = gym.make(args.task)`
code refactor for venv (#179) - Refacor code to remove duplicate code - Enable async simulation for all vector envs - Remove `collector.close` and rename `VectorEnv` to `DummyVectorEnv` The abstraction of vector env changed. Prior to this pr, each vector env is almost independent. After this pr, each env is wrapped into a worker, and vector envs differ with their worker type. In fact, users can just use `BaseVectorEnv` with different workers, I keep `SubprocVectorEnv`, `ShmemVectorEnv` for backward compatibility. Co-authored-by: n+e <463003665@qq.com> Co-authored-by: magicly <magicly007@gmail.com> 2020-08-19 15:00:24 +08:00			`test_envs = DummyVectorEnv(`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`[lambda: gym.make(args.task) for _ in range(args.test_num)]`
			`)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`# seed`
			`np.random.seed(args.seed)`
			`torch.manual_seed(args.seed)`
			`train_envs.seed(args.seed)`
			`test_envs.seed(args.seed)`
			`# model`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`net = Net(args.state_shape, hidden_sizes=args.hidden_sizes, device=args.device)`
hotfix：fix test failure in cuda environment (#289) 2021-02-09 17:13:40 +08:00			`actor = Actor(net, args.action_shape, device=args.device).to(args.device)`
			`critic = Critic(net, device=args.device).to(args.device)`
orthogonal init for ppo in test script 2020-05-16 20:27:01 +08:00			`# orthogonal initialization`
fix docs build failure and a bug in a2c/ppo optimizer (#428) * fix rtfd build * list + list -> set.union * change seed of test_qrdqn * add py39 test 2021-08-30 02:07:03 +08:00			`for m in set(actor.modules()).union(critic.modules()):`
orthogonal init for ppo in test script 2020-05-16 20:27:01 +08:00			`if isinstance(m, torch.nn.Linear):`
			`torch.nn.init.orthogonal_(m.weight)`
oinit with 0 bias 2020-05-17 17:06:20 +08:00			`torch.nn.init.zeros_(m.bias)`
Make trainer resumable (#350) - specify tensorboard >= 2.5.0 - add `save_checkpoint_fn` and `resume_from_log` in trainer Co-authored-by: Trinkle23897 <trinkle23897@gmail.com> 2021-05-06 08:53:53 +08:00			`optim = torch.optim.Adam(`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`set(actor.parameters()).union(critic.parameters()), lr=args.lr`
			`)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`dist = torch.distributions.Categorical`
			`policy = PPOPolicy(`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`actor,`
			`critic,`
			`optim,`
			`dist,`
Refactor PG algorithm and change behavior of `compute_episodic_return` (#319) - simplify code - apply value normalization (global) and adv norm (per-batch) in on-policy algorithms 2021-03-23 22:05:48 +08:00			`discount_factor=args.gamma,`
ppo and early stop 2020-03-20 19:52:29 +08:00			`max_grad_norm=args.max_grad_norm,`
			`eps_clip=args.eps_clip,`
			`vf_coef=args.vf_coef,`
			`ent_coef=args.ent_coef,`
fix ppo 2020-04-19 14:30:42 +08:00			`gae_lambda=args.gae_lambda,`
			`reward_normalization=args.rew_norm,`
			`dual_clip=args.dual_clip,`
Remap action to fit gym's action space (#313) Co-authored-by: Trinkle23897 <trinkle23897@gmail.com> 2021-03-21 16:45:50 +08:00			`value_clip=args.value_clip,`
Support deterministic evaluation for onpolicy algorithms (#354) 2021-04-27 21:22:39 +08:00			`action_space=env.action_space,`
make ppo discrete test script more general (#418) 2021-08-15 21:37:37 +08:00			`deterministic_eval=True,`
			`advantage_normalization=args.norm_adv,`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`recompute_advantage=args.recompute_adv`
			`)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`# collector`
			`train_collector = Collector(`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`policy, train_envs, VectorReplayBuffer(args.buffer_size, len(train_envs))`
			`)`
td3 2020-03-23 11:34:52 +08:00			`test_collector = Collector(policy, test_envs)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`# log`
save_fn 2020-04-11 16:54:27 +08:00			`log_path = os.path.join(args.logdir, args.task, 'ppo')`
			`writer = SummaryWriter(log_path)`
Add Weights and Biases Logger (#427) - rename BasicLogger to TensorboardLogger - refactor logger code - add WandbLogger Co-authored-by: Jiayi Weng <trinkle23897@gmail.com> 2021-08-30 10:35:02 -04:00			`logger = TensorboardLogger(writer)`
save_fn 2020-04-11 16:54:27 +08:00
			`def save_fn(policy):`
			`torch.save(policy.state_dict(), os.path.join(log_path, 'policy.pth'))`
ppo and early stop 2020-03-20 19:52:29 +08:00
change API of train_fn and test_fn (#229) train_fn(epoch) -> train_fn(epoch, num_env_step) test_fn(epoch) -> test_fn(epoch, num_env_step) 2020-09-26 16:35:37 +08:00			`def stop_fn(mean_rewards):`
			`return mean_rewards >= env.spec.reward_threshold`
ppo and early stop 2020-03-20 19:52:29 +08:00
			`# trainer`
			`result = onpolicy_trainer(`
bump to v0.4.3 (#432) * add makefile * bump version * add isort and yapf * update contributing.md * update PR template * spelling check 2021-09-03 05:05:04 +08:00			`policy,`
			`train_collector,`
			`test_collector,`
			`args.epoch,`
			`args.step_per_epoch,`
			`args.repeat_per_collect,`
			`args.test_num,`
			`args.batch_size,`
			`step_per_collect=args.step_per_collect,`
			`stop_fn=stop_fn,`
			`save_fn=save_fn,`
			`logger=logger`
			`)`
ppo and early stop 2020-03-20 19:52:29 +08:00			`assert stop_fn(result['best_reward'])`
Make trainer resumable (#350) - specify tensorboard >= 2.5.0 - add `save_checkpoint_fn` and `resume_from_log` in trainer Co-authored-by: Trinkle23897 <trinkle23897@gmail.com> 2021-05-06 08:53:53 +08:00
ppo and early stop 2020-03-20 19:52:29 +08:00			`if __name__ == '__main__':`
			`pprint.pprint(result)`
			`# Let's watch its performance!`
			`env = gym.make(args.task)`
optimize training procedure and improve code coverage (#189) 1. add policy.eval() in all test scripts' "watch performance" 2. remove dict return support for collector preprocess_fn 3. add `__contains__` and `pop` in batch: `key in batch`, `batch.pop(key, deft)` 4. exact n_episode for a list of n_episode limitation and save fake data in cache_buffer when self.buffer is None (#184) 5. fix tensorboard logging: h-axis stands for env step instead of gradient step; add test results into tensorboard 6. add test_returns (both GAE and nstep) 7. change the type-checking order in batch.py and converter.py in order to meet the most often case first 8. fix shape inconsistency for torch.Tensor in replay buffer 9. remove `**kwargs` in ReplayBuffer 10. remove default value in batch.split() and add merge_last argument (#185) 11. improve nstep efficiency 12. add max_batchsize in onpolicy algorithms 13. potential bugfix for subproc.wait 14. fix RecurrentActorProb 15. improve the code-coverage (from 90% to 95%) and remove the dead code 16. fix some incorrect type annotation The above improvement also increases the training FPS: on my computer, the previous version is only ~1800 FPS and after that, it can reach ~2050 (faster than v0.2.4.post1). 2020-08-27 12:15:18 +08:00			`policy.eval()`
ppo and early stop 2020-03-20 19:52:29 +08:00			`collector = Collector(policy, env)`
add examples, fix some bugs (#5) * update atari.py * fix setup.py pass the pytest * fix setup.py pass the pytest * add args "render" * change the tensorboard writter * change the tensorboard writter * change device, render, tensorboard log location * change device, render, tensorboard log location * remove some wrong local files * fix some tab mistakes and the envs name in continuous/test_xx.py * add examples and point robot maze environment * fix some bugs during testing examples * add dqn network and fix some args * change back the tensorboard writter's frequency to ensure ppo and a2c can write things normally * add a warning to collector * rm some unrelated files * reformat * fix a bug in test_dqn due to the model wrong selection 2020-03-28 07:27:18 +08:00			`result = collector.collect(n_episode=1, render=args.render)`
Step collector implementation (#280) This is the third PR of 6 commits mentioned in #274, which features refactor of Collector to fix #245. You can check #274 for more detail. Things changed in this PR: 1. refactor collector to be more cleaner, split AsyncCollector to support asyncvenv; 2. change buffer.add api to add(batch, bffer_ids); add several types of buffer (VectorReplayBuffer, PrioritizedVectorReplayBuffer, etc.) 3. add policy.exploration_noise(act, batch) -> act 4. small change in BasePolicy.compute_*_returns 5. move reward_metric from collector to trainer 6. fix np.asanyarray issue (different version's numpy will result in different output) 7. flake8 maxlength=88 8. polish docs and fix test Co-authored-by: n+e <trinkle23897@gmail.com> 2021-02-19 10:33:49 +08:00			`rews, lens = result["rews"], result["lens"]`
			`print(f"Final reward: {rews.mean()}, length: {lens.mean()}")`
ppo and early stop 2020-03-20 19:52:29 +08:00

			`if __name__ == '__main__':`
			`test_ppo()`