chore: import upstream snapshot with attribution
This commit is contained in:
@@ -0,0 +1,229 @@
|
||||
"""Example of customizing the evaluation procedure for an RLlib Algorithm.
|
||||
|
||||
Note, that you should only choose to provide a custom eval function, in case the already
|
||||
built-in eval options are not sufficient. Normally, though, RLlib's eval utilities
|
||||
that come with each Algorithm are enough to properly evaluate the learning progress
|
||||
of your Algorithm.
|
||||
|
||||
This script uses the SimpleCorridor environment, a simple 1D gridworld, in which
|
||||
the agent can only walk left (action=0) or right (action=1). The goal state is located
|
||||
at the end of the (1D) corridor. The env exposes an API to change the length of the
|
||||
corridor on-the-fly. We use this API here to extend the size of the corridor for the
|
||||
evaluation runs.
|
||||
|
||||
For demonstration purposes only, we define a simple custom evaluation method that does
|
||||
the following:
|
||||
- It changes the corridor length of all environments used on the evaluation EnvRunners.
|
||||
- It runs a defined number of episodes for evaluation purposes.
|
||||
- It collects the metrics from those runs, summarizes these metrics and returns them.
|
||||
|
||||
|
||||
How to run this script
|
||||
----------------------
|
||||
`python [script file name].py
|
||||
|
||||
You can switch off custom evaluation (and use RLlib's default evaluation procedure)
|
||||
with the `--no-custom-eval` flag.
|
||||
|
||||
You can switch on parallel evaluation to training using the
|
||||
`--evaluation-parallel-to-training` flag. See this example script here:
|
||||
https://github.com/ray-project/ray/blob/master/rllib/examples/evaluation/evaluation_parallel_to_training.py # noqa
|
||||
for more details on running evaluation parallel to training.
|
||||
|
||||
For debugging, use the following additional command line options
|
||||
`--no-tune --num-env-runners=0`
|
||||
which should allow you to set breakpoints anywhere in the RLlib code and
|
||||
have the execution stop there for inspection and debugging.
|
||||
|
||||
For logging to your WandB account, use:
|
||||
`--wandb-key=[your WandB API key] --wandb-project=[some project name]
|
||||
--wandb-run-name=[optional: WandB run name (within the defined project)]`
|
||||
|
||||
|
||||
Results to expect
|
||||
-----------------
|
||||
You should see the following (or very similar) console output when running this script.
|
||||
Note that for each iteration, due to the definition of our custom evaluation function,
|
||||
we run 3 evaluation rounds per single training round.
|
||||
|
||||
...
|
||||
Training iteration 1 -> evaluation round 0
|
||||
Training iteration 1 -> evaluation round 1
|
||||
Training iteration 1 -> evaluation round 2
|
||||
...
|
||||
...
|
||||
+--------------------------------+------------+-----------------+--------+
|
||||
| Trial name | status | loc | iter |
|
||||
|--------------------------------+------------+-----------------+--------+
|
||||
| PPO_SimpleCorridor_06582_00000 | TERMINATED | 127.0.0.1:69905 | 4 |
|
||||
+--------------------------------+------------+-----------------+--------+
|
||||
+------------------+-------+----------+--------------------+
|
||||
| total time (s) | ts | reward | episode_len_mean |
|
||||
|------------------+-------+----------+--------------------|
|
||||
| 26.1973 | 16000 | 0.872034 | 13.7966 |
|
||||
+------------------+-------+----------+--------------------+
|
||||
"""
|
||||
from typing import Tuple
|
||||
|
||||
from ray.rllib.algorithms.algorithm import Algorithm
|
||||
from ray.rllib.algorithms.algorithm_config import AlgorithmConfig
|
||||
from ray.rllib.env.env_runner_group import EnvRunnerGroup
|
||||
from ray.rllib.examples.envs.classes.simple_corridor import SimpleCorridor
|
||||
from ray.rllib.examples.utils import (
|
||||
add_rllib_example_script_args,
|
||||
run_rllib_example_script_experiment,
|
||||
)
|
||||
from ray.rllib.utils.metrics import (
|
||||
ENV_RUNNER_RESULTS,
|
||||
EPISODE_RETURN_MEAN,
|
||||
EVALUATION_RESULTS,
|
||||
NUM_ENV_STEPS_SAMPLED_LIFETIME,
|
||||
)
|
||||
from ray.rllib.utils.typing import ResultDict
|
||||
from ray.tune.registry import get_trainable_cls
|
||||
from ray.tune.result import TRAINING_ITERATION
|
||||
|
||||
parser = add_rllib_example_script_args(
|
||||
default_iters=50,
|
||||
default_reward=0.7,
|
||||
default_timesteps=50000,
|
||||
)
|
||||
parser.add_argument("--no-custom-eval", action="store_true")
|
||||
parser.add_argument("--corridor-length-training", type=int, default=10)
|
||||
parser.add_argument("--corridor-length-eval-worker-1", type=int, default=20)
|
||||
parser.add_argument("--corridor-length-eval-worker-2", type=int, default=30)
|
||||
|
||||
|
||||
def custom_eval_function(
|
||||
algorithm: Algorithm,
|
||||
eval_workers: EnvRunnerGroup,
|
||||
) -> Tuple[ResultDict, int, int]:
|
||||
"""Example of a custom evaluation function.
|
||||
|
||||
Args:
|
||||
algorithm: Algorithm class to evaluate.
|
||||
eval_workers: Evaluation EnvRunnerGroup.
|
||||
|
||||
Returns:
|
||||
metrics: Evaluation metrics dict.
|
||||
"""
|
||||
# Set different env settings for each (eval) EnvRunner. Here we use the EnvRunner's
|
||||
# `worker_index` property to figure out the actual length.
|
||||
# Loop through all workers and all sub-envs (gym.Env) on each worker and call the
|
||||
# `set_corridor_length` method on these.
|
||||
eval_workers.foreach_env_runner(
|
||||
func=lambda worker: (
|
||||
env.unwrapped.set_corridor_length(
|
||||
args.corridor_length_eval_worker_1
|
||||
if worker.worker_index == 1
|
||||
else args.corridor_length_eval_worker_2
|
||||
)
|
||||
for env in worker.env.unwrapped.envs
|
||||
)
|
||||
)
|
||||
|
||||
# Collect metrics results collected by eval workers in this list for later
|
||||
# processing.
|
||||
env_runner_metrics = []
|
||||
sampled_episodes = []
|
||||
# For demonstration purposes, run through some number of evaluation
|
||||
# rounds within this one call. Note that this function is called once per
|
||||
# training iteration (`Algorithm.train()` call) OR once per `Algorithm.evaluate()`
|
||||
# (which can be called manually by the user).
|
||||
for i in range(3):
|
||||
print(f"Training iteration {algorithm.iteration} -> evaluation round {i}")
|
||||
# Sample episodes from the EnvRunners AND have them return only the thus
|
||||
# collected metrics.
|
||||
episodes_and_metrics_all_env_runners = eval_workers.foreach_env_runner(
|
||||
# Return only the metrics, NOT the sampled episodes (we don't need them
|
||||
# anymore).
|
||||
func=lambda worker: (worker.sample(), worker.get_metrics()),
|
||||
local_env_runner=False,
|
||||
)
|
||||
sampled_episodes.extend(
|
||||
eps
|
||||
for eps_and_mtrcs in episodes_and_metrics_all_env_runners
|
||||
for eps in eps_and_mtrcs[0]
|
||||
)
|
||||
env_runner_metrics.extend(
|
||||
eps_and_mtrcs[1] for eps_and_mtrcs in episodes_and_metrics_all_env_runners
|
||||
)
|
||||
|
||||
# You can compute metrics from the episodes manually, or use the Algorithm's
|
||||
# convenient MetricsLogger to store all evaluation metrics inside the main
|
||||
# algo.
|
||||
algorithm.metrics.aggregate(
|
||||
env_runner_metrics, key=(EVALUATION_RESULTS, ENV_RUNNER_RESULTS)
|
||||
)
|
||||
eval_results = algorithm.metrics.peek((EVALUATION_RESULTS, ENV_RUNNER_RESULTS))
|
||||
# Alternatively, you could manually reduce over the n returned `env_runner_metrics`
|
||||
# dicts, but this would be much harder as you might not know, which metrics
|
||||
# to sum up, which ones to average over, etc..
|
||||
|
||||
# Compute env and agent steps from sampled episodes.
|
||||
env_steps = sum(eps.env_steps() for eps in sampled_episodes)
|
||||
agent_steps = sum(eps.agent_steps() for eps in sampled_episodes)
|
||||
|
||||
return eval_results, env_steps, agent_steps
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
args = parser.parse_args()
|
||||
base_config = (
|
||||
get_trainable_cls(args.algo)
|
||||
.get_default_config()
|
||||
# For training, we use a corridor length of n. For evaluation, we use different
|
||||
# values, depending on the eval worker index (1 or 2).
|
||||
.environment(
|
||||
SimpleCorridor,
|
||||
env_config={"corridor_length": args.corridor_length_training},
|
||||
)
|
||||
.env_runners(create_env_on_local_worker=True)
|
||||
.evaluation(
|
||||
# Do we use the custom eval function defined above?
|
||||
custom_evaluation_function=(
|
||||
None if args.no_custom_eval else custom_eval_function
|
||||
),
|
||||
# Number of eval EnvRunners to use.
|
||||
evaluation_num_env_runners=2,
|
||||
# Enable evaluation, once per training iteration.
|
||||
evaluation_interval=1,
|
||||
# Run 10 episodes each time evaluation runs (OR "auto" if parallel to
|
||||
# training).
|
||||
evaluation_duration="auto" if args.evaluation_parallel_to_training else 10,
|
||||
# Evaluate parallelly to training?
|
||||
evaluation_parallel_to_training=args.evaluation_parallel_to_training,
|
||||
# Override the env settings for the eval workers.
|
||||
# Note, though, that this setting here is only used in case --no-custom-eval
|
||||
# is set, b/c in case the custom eval function IS used, we override the
|
||||
# length of the eval environments in that custom function, so this setting
|
||||
# here is simply ignored.
|
||||
evaluation_config=AlgorithmConfig.overrides(
|
||||
env_config={"corridor_length": args.corridor_length_training * 2},
|
||||
# TODO (sven): Add support for window=float(inf) and reduce=mean for
|
||||
# evaluation episode_return_mean reductions (identical to old stack
|
||||
# behavior, which does NOT use a window (100 by default) to reduce
|
||||
# eval episode returns.
|
||||
metrics_num_episodes_for_smoothing=5,
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
stop = {
|
||||
TRAINING_ITERATION: args.stop_iters,
|
||||
f"{EVALUATION_RESULTS}/{ENV_RUNNER_RESULTS}/{EPISODE_RETURN_MEAN}": (
|
||||
args.stop_reward
|
||||
),
|
||||
NUM_ENV_STEPS_SAMPLED_LIFETIME: args.stop_timesteps,
|
||||
}
|
||||
|
||||
run_rllib_example_script_experiment(
|
||||
base_config,
|
||||
args,
|
||||
stop=stop,
|
||||
success_metric={
|
||||
f"{EVALUATION_RESULTS}/{ENV_RUNNER_RESULTS}/{EPISODE_RETURN_MEAN}": (
|
||||
args.stop_reward
|
||||
),
|
||||
},
|
||||
)
|
||||
@@ -0,0 +1,256 @@
|
||||
"""Example showing how one can set up evaluation running in parallel to training.
|
||||
|
||||
Such a setup saves a considerable amount of time during RL Algorithm training, b/c
|
||||
the next training step does NOT have to wait for the previous evaluation procedure to
|
||||
finish, but can already start running (in parallel).
|
||||
|
||||
See RLlib's documentation for more details on the effect of the different supported
|
||||
evaluation configuration options:
|
||||
https://docs.ray.io/en/latest/rllib/rllib-advanced-api.html#customized-evaluation-during-training # noqa
|
||||
|
||||
For an example of how to write a fully customized evaluation function (which normally
|
||||
is not necessary as the config options are sufficient and offer maximum flexibility),
|
||||
see this example script here:
|
||||
|
||||
https://github.com/ray-project/ray/blob/master/rllib/examples/evaluation/custom_evaluation.py # noqa
|
||||
|
||||
|
||||
How to run this script
|
||||
----------------------
|
||||
`python [script file name].py`
|
||||
|
||||
Use the `--evaluation-num-workers` option to scale up the evaluation workers. Note
|
||||
that the requested evaluation duration (`--evaluation-duration` measured in
|
||||
`--evaluation-duration-unit`, which is either "timesteps" (default) or "episodes") is
|
||||
shared between all configured evaluation workers. For example, if the evaluation
|
||||
duration is 10 and the unit is "episodes" and you configured 5 workers, then each of the
|
||||
evaluation workers will run exactly 2 episodes.
|
||||
|
||||
For debugging, use the following additional command line options
|
||||
`--no-tune --num-env-runners=0`
|
||||
which should allow you to set breakpoints anywhere in the RLlib code and
|
||||
have the execution stop there for inspection and debugging.
|
||||
|
||||
For logging to your WandB account, use:
|
||||
`--wandb-key=[your WandB API key] --wandb-project=[some project name]
|
||||
--wandb-run-name=[optional: WandB run name (within the defined project)]`
|
||||
|
||||
|
||||
Results to expect
|
||||
-----------------
|
||||
You should see the following output (at the end of the experiment) in your console when
|
||||
running with a fixed number of 100k training timesteps
|
||||
(`--evaluation-duration=auto --stop-timesteps=100000
|
||||
--stop-reward=100000`):
|
||||
+-----------------------------+------------+-----------------+--------+
|
||||
| Trial name | status | loc | iter |
|
||||
|-----------------------------+------------+-----------------+--------+
|
||||
| PPO_CartPole-v1_1377a_00000 | TERMINATED | 127.0.0.1:73330 | 25 |
|
||||
+-----------------------------+------------+-----------------+--------+
|
||||
+------------------+--------+----------+--------------------+
|
||||
| total time (s) | ts | reward | episode_len_mean |
|
||||
|------------------+--------+----------+--------------------|
|
||||
| 71.7485 | 100000 | 476.51 | 476.51 |
|
||||
+------------------+--------+----------+--------------------+
|
||||
|
||||
When running without parallel evaluation (no `--evaluation-parallel-to-training` flag),
|
||||
the experiment takes considerably longer (~70sec vs ~80sec):
|
||||
+-----------------------------+------------+-----------------+--------+
|
||||
| Trial name | status | loc | iter |
|
||||
|-----------------------------+------------+-----------------+--------+
|
||||
| PPO_CartPole-v1_f1788_00000 | TERMINATED | 127.0.0.1:75135 | 25 |
|
||||
+-----------------------------+------------+-----------------+--------+
|
||||
+------------------+--------+----------+--------------------+
|
||||
| total time (s) | ts | reward | episode_len_mean |
|
||||
|------------------+--------+----------+--------------------|
|
||||
| 81.7371 | 100000 | 494.68 | 494.68 |
|
||||
+------------------+--------+----------+--------------------+
|
||||
"""
|
||||
from typing import Optional
|
||||
|
||||
from ray.rllib.algorithms.algorithm import Algorithm
|
||||
from ray.rllib.callbacks.callbacks import RLlibCallback
|
||||
from ray.rllib.examples.envs.classes.multi_agent import MultiAgentCartPole
|
||||
from ray.rllib.examples.utils import (
|
||||
add_rllib_example_script_args,
|
||||
run_rllib_example_script_experiment,
|
||||
)
|
||||
from ray.rllib.utils.metrics import (
|
||||
ENV_RUNNER_RESULTS,
|
||||
EPISODE_RETURN_MEAN,
|
||||
EVALUATION_RESULTS,
|
||||
NUM_ENV_STEPS_SAMPLED,
|
||||
NUM_ENV_STEPS_SAMPLED_LIFETIME,
|
||||
NUM_EPISODES,
|
||||
)
|
||||
from ray.rllib.utils.metrics.metrics_logger import MetricsLogger
|
||||
from ray.rllib.utils.typing import ResultDict
|
||||
from ray.tune.registry import get_trainable_cls, register_env
|
||||
from ray.tune.result import TRAINING_ITERATION
|
||||
|
||||
parser = add_rllib_example_script_args(
|
||||
default_timesteps=200000,
|
||||
default_reward=500.0,
|
||||
)
|
||||
parser.set_defaults(
|
||||
evaluation_num_env_runners=2,
|
||||
evaluation_interval=1,
|
||||
)
|
||||
|
||||
|
||||
class AssertEvalCallback(RLlibCallback):
|
||||
def on_train_result(
|
||||
self,
|
||||
*,
|
||||
algorithm: Algorithm,
|
||||
metrics_logger: Optional[MetricsLogger] = None,
|
||||
result: ResultDict,
|
||||
**kwargs,
|
||||
):
|
||||
# The eval results can be found inside the main `result` dict
|
||||
# (old API stack: "evaluation").
|
||||
eval_results = result.get(EVALUATION_RESULTS, {})
|
||||
# In there, there is a sub-key: ENV_RUNNER_RESULTS.
|
||||
eval_env_runner_results = eval_results.get(ENV_RUNNER_RESULTS)
|
||||
# Make sure we always run exactly the given evaluation duration,
|
||||
# no matter what the other settings are (such as
|
||||
# `evaluation_num_env_runners` or `evaluation_parallel_to_training`).
|
||||
if eval_env_runner_results and NUM_EPISODES in eval_env_runner_results:
|
||||
num_episodes_done = eval_env_runner_results[NUM_EPISODES]
|
||||
if algorithm.config.enable_env_runner_and_connector_v2:
|
||||
num_timesteps_reported = eval_env_runner_results[NUM_ENV_STEPS_SAMPLED]
|
||||
else:
|
||||
num_timesteps_reported = eval_results["timesteps_this_iter"]
|
||||
|
||||
# We run for automatic duration (as long as training takes).
|
||||
if algorithm.config.evaluation_duration == "auto":
|
||||
# If duration=auto: Expect at least as many timesteps as workers
|
||||
# (each worker's `sample()` is at least called once).
|
||||
# UNLESS: All eval workers were completely busy during the auto-time
|
||||
# with older (async) requests and did NOT return anything from the async
|
||||
# fetch.
|
||||
assert (
|
||||
num_timesteps_reported == 0
|
||||
or num_timesteps_reported
|
||||
>= algorithm.config.evaluation_num_env_runners
|
||||
)
|
||||
# We count in episodes.
|
||||
elif algorithm.config.evaluation_duration_unit == "episodes":
|
||||
# Compare number of entries in episode_lengths (this is the
|
||||
# number of episodes actually run) with desired number of
|
||||
# episodes from the config.
|
||||
assert (
|
||||
algorithm.iteration + 1 % algorithm.config.evaluation_interval != 0
|
||||
or num_episodes_done == algorithm.config.evaluation_duration
|
||||
), (num_episodes_done, algorithm.config.evaluation_duration)
|
||||
print(
|
||||
"Number of run evaluation episodes: " f"{num_episodes_done} (ok)!"
|
||||
)
|
||||
# We count in timesteps.
|
||||
else:
|
||||
# TODO (sven): This assertion works perfectly fine locally, but breaks
|
||||
# the CI for no reason. The observed collected timesteps is +500 more
|
||||
# than desired (~2500 instead of 2011 and ~1250 vs 1011).
|
||||
# num_timesteps_wanted = algorithm.config.evaluation_duration
|
||||
# delta = num_timesteps_wanted - num_timesteps_reported
|
||||
# Expect roughly the same (desired // num-eval-workers).
|
||||
# assert abs(delta) < 20, (
|
||||
# delta,
|
||||
# num_timesteps_wanted,
|
||||
# num_timesteps_reported,
|
||||
# )
|
||||
print(
|
||||
"Number of run evaluation timesteps: "
|
||||
f"{num_timesteps_reported} (ok?)!"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
args = parser.parse_args()
|
||||
|
||||
# Register our environment with tune.
|
||||
if args.num_agents > 0:
|
||||
register_env(
|
||||
"env",
|
||||
lambda _: MultiAgentCartPole(config={"num_agents": args.num_agents}),
|
||||
)
|
||||
|
||||
base_config = (
|
||||
get_trainable_cls(args.algo)
|
||||
.get_default_config()
|
||||
.environment("env" if args.num_agents > 0 else "CartPole-v1")
|
||||
.env_runners(create_env_on_local_worker=True)
|
||||
# Use a custom callback that asserts that we are running the
|
||||
# configured exact number of episodes per evaluation OR - in auto
|
||||
# mode - run at least as many episodes as we have eval workers.
|
||||
.callbacks(AssertEvalCallback)
|
||||
.evaluation(
|
||||
# Parallel evaluation+training config.
|
||||
# Switch on evaluation in parallel with training.
|
||||
evaluation_parallel_to_training=args.evaluation_parallel_to_training,
|
||||
# Use two evaluation workers. Must be >0, otherwise,
|
||||
# evaluation will run on a local worker and block (no parallelism).
|
||||
evaluation_num_env_runners=args.evaluation_num_env_runners,
|
||||
# Evaluate every other training iteration (together
|
||||
# with every other call to Algorithm.train()).
|
||||
evaluation_interval=args.evaluation_interval,
|
||||
# Run for n episodes/timesteps (properly distribute load amongst
|
||||
# all eval workers). The longer it takes to evaluate, the more sense
|
||||
# it makes to use `evaluation_parallel_to_training=True`.
|
||||
# Use "auto" to run evaluation for roughly as long as the training
|
||||
# step takes.
|
||||
evaluation_duration=args.evaluation_duration,
|
||||
# "episodes" or "timesteps".
|
||||
evaluation_duration_unit=args.evaluation_duration_unit,
|
||||
# Switch off exploratory behavior for better (greedy) results.
|
||||
evaluation_config={
|
||||
"explore": False,
|
||||
# TODO (sven): Add support for window=float(inf) and reduce=mean for
|
||||
# evaluation episode_return_mean reductions (identical to old stack
|
||||
# behavior, which does NOT use a window (100 by default) to reduce
|
||||
# eval episode returns.
|
||||
"metrics_num_episodes_for_smoothing": 5,
|
||||
},
|
||||
)
|
||||
)
|
||||
|
||||
# Set the minimum time for an iteration to 10sec, even for algorithms like PPO
|
||||
# that naturally limit their iteration times to exactly one `training_step`
|
||||
# call. This provides enough time for the eval EnvRunners in the
|
||||
# "evaluation_duration=auto" setting to sample at least one complete episode.
|
||||
if args.evaluation_duration == "auto":
|
||||
base_config.reporting(min_time_s_per_iteration=10)
|
||||
|
||||
# Add a simple multi-agent setup.
|
||||
if args.num_agents > 0:
|
||||
base_config.multi_agent(
|
||||
policies={f"p{i}" for i in range(args.num_agents)},
|
||||
policy_mapping_fn=lambda aid, *a, **kw: f"p{aid}",
|
||||
)
|
||||
# Set some PPO-specific tuning settings to learn better in the env (assumed to be
|
||||
# CartPole-v1).
|
||||
if args.algo == "PPO":
|
||||
base_config.training(
|
||||
lr=0.0003,
|
||||
num_epochs=6,
|
||||
vf_loss_coeff=0.01,
|
||||
)
|
||||
|
||||
stop = {
|
||||
TRAINING_ITERATION: args.stop_iters,
|
||||
f"{EVALUATION_RESULTS}/{ENV_RUNNER_RESULTS}/{EPISODE_RETURN_MEAN}": (
|
||||
args.stop_reward
|
||||
),
|
||||
NUM_ENV_STEPS_SAMPLED_LIFETIME: args.stop_timesteps,
|
||||
}
|
||||
|
||||
run_rllib_example_script_experiment(
|
||||
base_config,
|
||||
args,
|
||||
stop=stop,
|
||||
success_metric={
|
||||
f"{EVALUATION_RESULTS}/{ENV_RUNNER_RESULTS}/{EPISODE_RETURN_MEAN}": (
|
||||
args.stop_reward
|
||||
),
|
||||
},
|
||||
)
|
||||
Reference in New Issue
Block a user