def setUp(self) -> None: self.temp_dir = tempfile.TemporaryDirectory() episodes = 80 seeds = [0, 1, 3, 4, 5] experiment_name = "test_env" logger = Logger( output_path=Path(self.temp_dir.name), experiment_name=experiment_name, step_write_frequency=None, episode_write_frequency=None, ) benchmark = SigmoidBenchmark() env = benchmark.get_benchmark() agent = RandomAgent(env) logger.set_env(env) env_logger = logger.add_module(env) for seed in seeds: env.seed(seed) logger.set_additional_info(seed=seed) logger.reset_episode() for episode in range(episodes): state = env.reset() done = False reward = 0 step = 0 while not done: action = agent.act(state, reward) env_logger.log( "logged_step", step, ) env_logger.log( "logged_episode", episode, ) next_state, reward, done, _ = env.step(action) env_logger.log( "reward", reward, ) env_logger.log( "done", done, ) agent.train(next_state, reward) state = next_state logger.next_step() step += 1 agent.end_episode(state, reward) logger.next_episode() env.close() logger.close() self.log_file = env_logger.log_file.name
def run_policy(results_path, benchmark_name, num_episodes, policy, seeds=np.arange(10)): bench = getattr(benchmarks, benchmark_name)() for s in seeds: if benchmark_name == "CMAESBenchmark": experiment_name = f"csa_{s}" else: experiment_name = f"optimal_{s}" logger = Logger(experiment_name=experiment_name, output_path=results_path / benchmark_name) env = bench.get_benchmark(seed=s) env = PerformanceTrackingWrapper( env, logger=logger.add_module(PerformanceTrackingWrapper)) agent = GenericAgent(env, policy) logger.add_agent(agent) logger.add_benchmark(bench) logger.set_env(env) run_benchmark(env, agent, num_episodes, logger) logger.close()
def run_dacbench(results_path, agent_method, num_episodes, bench=None, seeds=None): """ Run all benchmarks for 10 seeds for a given number of episodes with a given agent and save result Parameters ------- bench results_path : str Path to where results should be saved agent_method : function Method that takes an env as input and returns an agent num_episodes : int Number of episodes to run for each benchmark seeds : list[int] List of seeds to runs all benchmarks for. If None (default) seeds [1, ..., 10] are used. """ if bench is None: bench = map(benchmarks.__dict__.get, benchmarks.__all__) else: bench = [getattr(benchmarks, b) for b in bench] seeds = seeds if seeds is not None else range(10) for b in bench: print(f"Evaluating {b.__name__}") for i in seeds: print(f"Seed {i}/10") bench = b() try: env = bench.get_benchmark(seed=i) except: continue logger = Logger( experiment_name=f"seed_{i}", output_path=Path(results_path) / f"{b.__name__}", ) perf_logger = logger.add_module(PerformanceTrackingWrapper) logger.add_benchmark(bench) logger.set_env(env) env = PerformanceTrackingWrapper(env, logger=perf_logger) agent = agent_method(env) logger.add_agent(agent) run_benchmark(env, agent, num_episodes, logger) logger.close()
def run_random(results_path, benchmark_name, num_episodes, seeds, fixed): bench = getattr(benchmarks, benchmark_name)() for s in seeds: if fixed > 1: experiment_name = f"random_fixed{fixed}_{s}" else: experiment_name = f"random_{s}" logger = Logger(experiment_name=experiment_name, output_path=results_path / benchmark_name) env = bench.get_benchmark(seed=s) env = PerformanceTrackingWrapper( env, logger=logger.add_module(PerformanceTrackingWrapper)) agent = DynamicRandomAgent(env, fixed) logger.add_agent(agent) logger.add_benchmark(bench) logger.set_env(env) run_benchmark(env, agent, num_episodes, logger) logger.close()
def run_optimal(results_path, benchmark_name, num_episodes, seeds=np.arange(10)): bench = getattr(benchmarks, benchmark_name)() if benchmark_name == "LubyBenchmark": policy = optimal_luby elif benchmark_name == "SigmoidBenchmark": policy = optimal_sigmoid elif benchmark_name == "FastDownwardBenchmark": policy = optimal_fd elif benchmark_name == "CMAESBenchmark": policy = csa else: print("No comparison policy found for this benchmark") return for s in seeds: if benchmark_name == "CMAESBenchmark": experiment_name = f"csa_{s}" else: experiment_name = f"optimal_{s}" logger = Logger(experiment_name=experiment_name, output_path=results_path / benchmark_name) env = bench.get_benchmark(seed=s) env = PerformanceTrackingWrapper( env, logger=logger.add_module(PerformanceTrackingWrapper)) agent = GenericAgent(env, policy) logger.add_agent(agent) logger.add_benchmark(bench) logger.set_env(env) logger.set_additional_info(seed=s) run_benchmark(env, agent, num_episodes, logger) logger.close()
def run_dacbench(results_path, agent_method, num_episodes): """ Run all benchmarks for 10 seeds for a given number of episodes with a given agent and save result Parameters ------- results_path : str Path to where results should be saved agent_method : function Method that takes an env as input and returns an agent num_episodes : int Number of episodes to run for each benchmark """ for b in map(benchmarks.__dict__.get, benchmarks.__all__): print(f"Evaluating {b.__name__}") for i in range(10): print(f"Seed {i}/10") bench = b() env = bench.get_benchmark(seed=i) logger = Logger( experiment_name=f"seed_{i}", output_path=Path(results_path) / f"{b.__name__}", ) perf_logger = logger.add_module(PerformanceTrackingWrapper) logger.add_benchmark(bench) logger.set_env(env) logger.set_additional_info(seed=i) env = PerformanceTrackingWrapper(env, logger=perf_logger) agent = agent_method(env) logger.add_agent(agent) run_benchmark(env, agent, num_episodes, logger) logger.close()
def run_static(results_path, benchmark_name, action, num_episodes, seeds=np.arange(10)): bench = getattr(benchmarks, benchmark_name)() for s in seeds: logger = Logger( experiment_name=f"static_{action}_{s}", output_path=results_path / benchmark_name, ) env = bench.get_benchmark(seed=s) env = PerformanceTrackingWrapper( env, logger=logger.add_module(PerformanceTrackingWrapper)) agent = StaticAgent(env, action) logger.add_agent(agent) logger.add_benchmark(bench) logger.set_env(env) logger.set_additional_info(seed=s, action=action) run_benchmark(env, agent, num_episodes, logger) logger.close()
def test_logging(self): temp_dir = tempfile.TemporaryDirectory() episodes = 5 logger = Logger( output_path=Path(temp_dir.name), experiment_name="test_logging", ) bench = LubyBenchmark() env = bench.get_environment() time_logger = logger.add_module(EpisodeTimeWrapper) wrapped = EpisodeTimeWrapper(env, logger=time_logger) agent = StaticAgent(env=env, action=1) run_benchmark(wrapped, agent, episodes, logger) logger.close() logs = load_logs(time_logger.get_logfile()) dataframe = log2dataframe(logs, wide=True) # all steps must have logged time self.assertTrue((~dataframe.step_duration.isna()).all()) # each episode has a recored time episodes = dataframe.groupby("episode") last_steps_per_episode = dataframe.iloc[episodes.step.idxmax()] self.assertTrue( (~last_steps_per_episode.episode_duration.isna()).all()) # episode time equals the sum of the steps in episode calculated_episode_times = episodes.step_duration.sum() recorded_episode_times = last_steps_per_episode.episode_duration self.assertListEqual(calculated_episode_times.tolist(), recorded_episode_times.tolist()) temp_dir.cleanup()
from pathlib import Path from dacbench.agents import RandomAgent from dacbench.logger import Logger from dacbench.runner import run_benchmark from dacbench.benchmarks import CMAESBenchmark from dacbench.wrappers import StateTrackingWrapper # Make CMAESBenchmark environment bench = CMAESBenchmark() env = bench.get_environment() # Make Logger object to track state information logger = Logger(experiment_name=type(bench).__name__, output_path=Path("../plotting/data")) logger.set_env(env) # Wrap env with StateTrackingWrapper env = StateTrackingWrapper(env, logger=logger.add_module(StateTrackingWrapper)) # Run random agent for 5 episodes and log state information to file # You can plot these results with the plotting examples agent = RandomAgent(env) run_benchmark(env, agent, 5, logger=logger) logger.close()