From 1ef3bc01049697ca9553e44204add7942dd08bc4 Mon Sep 17 00:00:00 2001 From: tactino <18781106300@163.com> Date: Sun, 27 Sep 2026 17:35:18 -0400 Subject: [PATCH 1/8] gaussian-policy and ppo: CleanRL's continuous-action PPO The coverage figure pairs expressive policies with algorithms built for them and has no row for the baseline they are measured against. This adds it, written to CleanRL's ppo_continuous_action.py with every default: gaussian-policy tanh MLPs of 64 for the mean and the value, a log std independent of the observation, orthogonal init (gain sqrt 2; 0.01 on the mean's head, 1.0 on the value's). ppo 2048-step rollouts, 10 epochs of 64-sample minibatches, clip 0.2 on the ratio and the value, per-minibatch advantage normalisation, one Adam at eps 1e-5, gradient norm 0.5, lr 3e-4 annealed linearly over 488 iterations. CleanRL clips actions and normalises observations and rewards in gymnasium wrappers, and PlugRL's client wraps nothing, so they happen on the server: the env receives the clipped sample and the buffer keeps the drawn one; observation statistics are buffers updated after each learn, so collection and learning share one normalisation; PPOBuffer divides rewards by the running deviation of the discounted return and clips at 10, restarting the return with each episode. tests/test_gaussian_ppo.py checks each piece and that the pair learns the bandit FPO's test uses and keeps it on five seeds (last/best at most 1.12). --- README.md | 26 + src/plugrl_server/algorithm/__init__.py | 2 + src/plugrl_server/algorithm/ppo/__init__.py | 0 src/plugrl_server/algorithm/ppo/ppo.py | 296 +++++++++++ src/plugrl_server/algorithm/ppo/ppo_buffer.py | 92 ++++ src/plugrl_server/algorithm/ppo/ppo_config.py | 47 ++ src/plugrl_server/policy/__init__.py | 2 + src/plugrl_server/policy/gaussian/__init__.py | 7 + .../policy/gaussian/gaussian_policy.py | 188 +++++++ tests/test_gaussian_ppo.py | 463 ++++++++++++++++++ 10 files changed, 1123 insertions(+) create mode 100644 src/plugrl_server/algorithm/ppo/__init__.py create mode 100644 src/plugrl_server/algorithm/ppo/ppo.py create mode 100644 src/plugrl_server/algorithm/ppo/ppo_buffer.py create mode 100644 src/plugrl_server/algorithm/ppo/ppo_config.py create mode 100644 src/plugrl_server/policy/gaussian/__init__.py create mode 100644 src/plugrl_server/policy/gaussian/gaussian_policy.py create mode 100644 tests/test_gaussian_ppo.py diff --git a/README.md b/README.md index c262871..4214016 100644 --- a/README.md +++ b/README.md @@ -126,10 +126,15 @@ We use `uv` to manage dependencies and development environments. - `dppo` - DPPO (Diffusion Policy Policy Optimization). No extras needed, and it runs against `fpo-policy`. - `dppo-dist` - Distributed DPPO (experimental) +- `ppo` - PPO as CleanRL's `ppo_continuous_action.py` runs it, every + default included. Drives `gaussian-policy`; no extras needed. - `eval` - Evaluation only, no learning **Policies:** - `fpo-policy` - Flow policy. Defaults to `obs_dim=17`, `action_dim=6` +- `gaussian-policy` - CleanRL's Gaussian MLP: tanh layers of 64, a log std + that does not depend on the observation. Defaults to `obs_dim=17`, + `action_dim=6` - `dummy-policy` - Outputs random actions (for testing) - `dppo-policy` - DPPO policy (requires `plugrl-server[dppo]` and a checkpoint) - `pi0-policy` - PI0 policy (OpenPI). Needs more than a checkpoint. The @@ -213,6 +218,27 @@ million steps therefore learns exactly once, at the very end - producing a single point rather than a curve. 4096 gives one update per 4096 environment steps. +#### The baseline: a Gaussian policy with PPO + +The pair every other one is measured against, written to CleanRL's +`ppo_continuous_action.py`: a rollout of 2048 steps, ten epochs of 32 +minibatches, clip 0.2, a learning rate of 3e-4 annealed to zero over 488 +iterations (a million steps), observations and rewards normalised. CleanRL +does the normalising and the action clipping in gymnasium wrappers; PlugRL's +client does not wrap, so the policy and the algorithm do it on the server. + +```bash +# Terminal 1: Hopper-v5 has an 11-dimensional observation and 3 actions +python -m plugrl_server.cli gaussian-policy default ppo default \ + --port 8000 --policy.device cpu \ + --policy.obs-dim 11 --policy.action-dim 3 + +# Terminal 2 +python -m plugrl_env_client.cli mujoco-v1 \ + --server-port 8000 --num-envs 1 --num-episodes 100000 \ + --env.name Hopper-v5 --runner.replan-steps 1 --runner.seed 0 +``` + #### Quick Start: Testing with Dummy Components To test the agent-server connection with dummy algorithm and policy: diff --git a/src/plugrl_server/algorithm/__init__.py b/src/plugrl_server/algorithm/__init__.py index 076c413..b520883 100644 --- a/src/plugrl_server/algorithm/__init__.py +++ b/src/plugrl_server/algorithm/__init__.py @@ -16,3 +16,5 @@ import plugrl_server.algorithm.dppo.dppo_config # noqa: F401,E402 import plugrl_server.algorithm.dppo.dppo_dist # noqa: F401,E402 import plugrl_server.algorithm.dppo.dppo_dist_config # noqa: F401,E402 +import plugrl_server.algorithm.ppo.ppo # noqa: F401,E402 +import plugrl_server.algorithm.ppo.ppo_config # noqa: F401,E402 diff --git a/src/plugrl_server/algorithm/ppo/__init__.py b/src/plugrl_server/algorithm/ppo/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/plugrl_server/algorithm/ppo/ppo.py b/src/plugrl_server/algorithm/ppo/ppo.py new file mode 100644 index 0000000..0bc1432 --- /dev/null +++ b/src/plugrl_server/algorithm/ppo/ppo.py @@ -0,0 +1,296 @@ +"""PPO, as CleanRL's `ppo_continuous_action.py` runs it. + +Clipped surrogate at 0.2, a clipped value loss, advantages normalised per +minibatch, one Adam (eps 1e-5) over the actor and the critic together, the +whole gradient clipped at norm 0.5, the learning rate annealed linearly to +zero. It drives `gaussian-policy`, and any policy that offers the same four +things: `get_action_and_runtime_state`, `evaluate_actions`, `get_value` and, +optionally, `update_obs_stats`. +""" + +import math + +import numpy as np +import torch +import torch.nn as nn + +from plugrl_server.algorithm.base_algorithm import BaseAlgorithm +from plugrl_server.algorithm.registration import register_algo +from plugrl_server.algorithm.train_utils import move_batch_to_device +from plugrl_server.buffer.rollout_buffer import ROLLOUT_BUFFER_SCHEMA_VERSION +from plugrl_server.common.checkpoint_manager import Checkpoint +from plugrl_server.common.logging_utils import get_logger +from plugrl_server.policy.gaussian.gaussian_policy import GaussianPolicy +from plugrl_server.policy.state import ( + PolicyRuntimeState, + PolicyTrainState, + to_numpy_state, +) + +from .ppo_buffer import PPOBuffer +from .ppo_config import UID, PPOAlgoConfig + +logger = get_logger(__name__) + + +@register_algo(UID) +class PPOAlgorithm(BaseAlgorithm): + config: PPOAlgoConfig + policy: GaussianPolicy + optimizer: torch.optim.Optimizer + + def __init__(self, config: PPOAlgoConfig, policy: GaussianPolicy): + super().__init__(config, policy) + if not hasattr(policy, "evaluate_actions"): + raise TypeError( + f"ppo needs a policy with evaluate_actions, such as " + f"gaussian-policy; {type(policy).__name__} has none" + ) + self.rollout_buffer = PPOBuffer( + buffer_size=config.buffer_size, + example_train_state=self.example_train_state(batch_size=1), + gamma=config.gamma, + gae_lambda=config.gae_lambda, + normalize_rewards=config.normalize_rewards, + reward_clip=config.reward_clip, + ) + self.global_step = 0 + self.curr_train_itrs = 0 + self.last_saved_itr = 0 + + def init_optimizers(self) -> None: + self.optimizer = torch.optim.Adam( + self.policy.parameters(), lr=self.config.learning_rate, eps=1e-5 + ) + + def infer(self, obs: dict) -> tuple[np.ndarray, PolicyRuntimeState]: + with torch.inference_mode(): + return self.policy.get_action_and_runtime_state(obs) + + def derive_train_state(self, runtime_state: PolicyRuntimeState) -> PolicyTrainState: + numpy_state = to_numpy_state(runtime_state) + if not isinstance(numpy_state, dict): + raise TypeError("ppo requires a mapping-like train_state export.") + return numpy_state + + def feedback( + self, + *, + obs: dict, + runtime_state: PolicyRuntimeState, + train_state: PolicyTrainState = None, + terminated: bool, + truncated: bool, + next_obs: dict, + reward: float, + next_terminated: bool, + next_truncated: bool, + info: dict, + prev_node: tuple, + ) -> tuple[tuple, int, dict]: + assert train_state is not None, "ppo requires train_state for rollout storage." + current_node = self.rollout_buffer.add_frame( + prev_node=prev_node, + train_state=train_state, + reward=reward, + terminated=terminated, + truncated=truncated, + last_value=None, + next_terminated=next_terminated, + next_truncated=next_truncated, + ) + self.rollout_buffer.add_next_obs_value_request( + obs=next_obs, end_node=current_node + ) + if next_terminated or next_truncated: + if "episode" in info and bool(info["episode"].get("mask", True)): + self.record_episode_metrics(info["episode"]) + self.rollout_buffer.finish_rollout(info=info) + self.global_step += 1 + return current_node, self.global_step, {} + + def pre_learn(self) -> None: + self.rollout_buffer.compute_advantages_and_returns( + policy=self.policy, batch_size=self.config.batch_size + ) + + def get_collect_progress_total(self) -> int | None: + return self.config.buffer_size + + def get_collect_progress_completed(self) -> int | None: + return len(self.rollout_buffer) + + def get_learn_progress_total(self) -> int | None: + return self.config.update_epochs * math.ceil( + self.config.buffer_size / self.config.batch_size + ) + + def _learning_rate(self) -> float: + """CleanRL's: (1 - (iteration - 1) / num_iterations) * learning_rate.""" + if not self.config.anneal_lr: + return self.config.learning_rate + frac = 1.0 - self.curr_train_itrs / self.config.train_itrs + return frac * self.config.learning_rate + + def _loss(self, obs, action, oldlogprob, value, advantage, ret): + config = self.config + newlogprob, entropy, newvalue = self.policy.evaluate_actions(obs, action) + logratio = newlogprob - oldlogprob + ratio = logratio.exp() + + if config.norm_adv and advantage.numel() > 1: + advantage = (advantage - advantage.mean()) / (advantage.std() + 1e-8) + pg_loss = torch.max( + -advantage * ratio, + -advantage * torch.clamp(ratio, 1 - config.clip_coef, 1 + config.clip_coef), + ).mean() + + if config.clip_vloss: + v_clipped = value + torch.clamp( + newvalue - value, -config.clip_coef, config.clip_coef + ) + v_loss = ( + 0.5 * torch.max((newvalue - ret) ** 2, (v_clipped - ret) ** 2).mean() + ) + else: + v_loss = 0.5 * ((newvalue - ret) ** 2).mean() + + return pg_loss, v_loss, entropy.mean(), logratio, ratio + + def learn_impl(self) -> tuple[int, dict]: + config = self.config + lr = self._learning_rate() + for group in self.optimizer.param_groups: + group["lr"] = lr + + dataloader = torch.utils.data.DataLoader( + self.rollout_buffer, + batch_size=config.batch_size, + shuffle=True, + drop_last=False, + num_workers=0, + collate_fn=self.rollout_buffer.collate_fn, + ) + rollout_summary = self.rollout_buffer.description() + pg_loss = v_loss = entropy_loss = torch.tensor(0.0) + old_approx_kl = approx_kl = torch.tensor(0.0) + clipfracs: list[float] = [] + grad_norms: list[float] = [] + progress_total = self.get_learn_progress_total() + progress = 0 + + for epoch in range(config.update_epochs): + for batch in dataloader: + obs, action, oldlogprob, _reward, value, advantage, ret = ( + move_batch_to_device(batch, device=self.policy.device) + ) + pg_loss, v_loss, entropy_loss, logratio, ratio = self._loss( + obs, action, oldlogprob, value, advantage, ret + ) + with torch.no_grad(): + old_approx_kl = (-logratio).mean() + approx_kl = ((ratio - 1) - logratio).mean() + clipfracs.append( + ((ratio - 1.0).abs() > config.clip_coef).float().mean().item() + ) + + loss = ( + pg_loss - config.ent_coef * entropy_loss + config.vf_coef * v_loss + ) + self.optimizer.zero_grad() + loss.backward() + grad_norms.append( + float( + nn.utils.clip_grad_norm_( + self.policy.parameters(), config.max_grad_norm + ) + ) + ) + self.optimizer.step() + progress += 1 + self.report_learn_progress(progress, progress_total) + + if config.target_kl is not None and approx_kl > config.target_kl: + logger.info("Stopping after epoch %d: approx_kl %.4f", epoch, approx_kl) + break + + self.curr_train_itrs += 1 + return self.global_step, dict( + models=dict(learning_rate=lr), + losses=dict( + value_loss=v_loss.item(), + policy_loss=pg_loss.item(), + entropy=entropy_loss.item(), + old_approx_kl=old_approx_kl.item(), + approx_kl=approx_kl.item(), + clipfrac=float(np.mean(clipfracs)) if clipfracs else 0.0, + ), + train=dict( + max_grad_norm=max(grad_norms) if grad_norms else 0.0, + train_itrs=float(self.curr_train_itrs), + ), + rollout=rollout_summary, + ) + + def post_learn(self) -> None: + # After learning and before the reset, as DPPO does for fpo-policy: + # this iteration collected and learned under one normalisation, and + # the next collects under the new one. + update = getattr(self.policy, "update_obs_stats", None) + filled = len(self.rollout_buffer) + if update is not None and filled > 0: + stored = self.rollout_buffer.train_state_storage.get_item(slice(0, filled)) + update(torch.as_tensor(stored, device=self.policy.device)) + self.rollout_buffer.reset() + super().post_learn() + + def should_learn(self) -> bool: + return self.rollout_buffer.full() + + def should_stop(self) -> bool: + return self.curr_train_itrs >= self.config.train_itrs + + def should_save(self) -> bool: + return (self.curr_train_itrs % self.config.save_interval == 0) and ( + self.curr_train_itrs > self.last_saved_itr + ) + + def create_checkpoint(self) -> Checkpoint: + self.last_saved_itr = self.curr_train_itrs + rms = self.rollout_buffer.ret_rms + return Checkpoint( + step=self.global_step, + model=self.policy.state_dict(), + optimizer={"adam": self.optimizer.state_dict()}, + meta={ + "train_itrs": self.curr_train_itrs, + "last_saved_itr": self.last_saved_itr, + # The reward scale, which a resume would otherwise relearn. + "return_rms": { + "mean": float(rms.mean), + "var": float(rms.var), + "count": float(rms.count), + }, + "rollout_buffer_schema_version": ROLLOUT_BUFFER_SCHEMA_VERSION, + }, + ) + + def load_checkpoint(self, checkpoint: Checkpoint) -> None: + self.global_step = checkpoint.step + if checkpoint.model is not None: + self.policy.load_state_dict(checkpoint.model) + if checkpoint.optimizer is not None: + self.optimizer.load_state_dict(checkpoint.optimizer["adam"]) + meta = checkpoint.meta + self.curr_train_itrs = meta.get("train_itrs", self.curr_train_itrs) + self.last_saved_itr = meta.get("last_saved_itr", self.last_saved_itr) + if "return_rms" in meta: + rms = self.rollout_buffer.ret_rms + rms.mean = np.float64(meta["return_rms"]["mean"]) + rms.var = np.float64(meta["return_rms"]["var"]) + rms.count = meta["return_rms"]["count"] + logger.info( + "Loaded checkpoint at step %d, train_itrs %d", + self.global_step, + self.curr_train_itrs, + ) diff --git a/src/plugrl_server/algorithm/ppo/ppo_buffer.py b/src/plugrl_server/algorithm/ppo/ppo_buffer.py new file mode 100644 index 0000000..9cfe268 --- /dev/null +++ b/src/plugrl_server/algorithm/ppo/ppo_buffer.py @@ -0,0 +1,92 @@ +import uuid + +import numpy as np + +from plugrl_server.algorithm.dppo.third_party.reward_scaling import RunningMeanStd +from plugrl_server.buffer.rollout_buffer import GAEBuffer +from plugrl_server.policy.base_policy import BasePolicy +from plugrl_server.policy.state import PolicyTrainState + + +class PPOBuffer(GAEBuffer): + """GAE, with rewards scaled the way gymnasium's NormalizeReward scales them. + + Each frame's discounted return is kept as it arrives; at learning time the + running variance takes in the buffer's returns and every reward is divided + by the running deviation, then clipped. NormalizeReward divides each reward + by the deviation as it stood at that step instead; over the same returns + the two differ only in how recent the estimate is. + + The discounted return restarts with each episode, as it does in + NormalizeReward and in DPPO's RunningRewardScaler. `DPPOBuffer` follows + the server's chain of frames, which runs on across episodes, so its return + carries over from one episode into the next. + """ + + def __init__( + self, + buffer_size, + example_train_state: PolicyTrainState, + gamma: float = 0.99, + gae_lambda: float = 0.95, + normalize_rewards: bool = True, + reward_clip: float = 10.0, + epsilon: float = 1e-8, + ): + super().__init__(buffer_size, example_train_state, gamma, gae_lambda) + self.normalize_rewards = normalize_rewards + self.reward_clip = reward_clip + self.epsilon = epsilon + self.ret_rms = RunningMeanStd(shape=()) + self.rets = np.zeros(buffer_size, dtype=np.float64) + + def add_frame( + self, + *, + prev_node: tuple[int, uuid.UUID], + train_state: PolicyTrainState, + reward: float, + terminated: bool, + truncated: bool, + last_value: np.ndarray | None, + next_terminated: bool, + next_truncated: bool, + ) -> tuple[int, uuid.UUID]: + node = super().add_frame( + prev_node=prev_node, + train_state=train_state, + reward=reward, + terminated=terminated, + truncated=truncated, + last_value=last_value, + next_terminated=next_terminated, + next_truncated=next_truncated, + ) + current_idx = node[0] + if current_idx == -1: + return node + prev_idx, prev_signature = prev_node + # `terminated` or `truncated` on a frame means the step before it + # ended an episode, so this frame begins one. + continues = ( + prev_idx != -1 + and prev_signature == self.buffer_signature + and not (terminated or truncated) + ) + self.rets[current_idx] = float(reward) + ( + self.gamma * self.rets[prev_idx] if continues else 0.0 + ) + return node + + def compute_advantages_and_returns( + self, policy: BasePolicy | None = None, batch_size: int = 1 + ): + if self.normalize_rewards and self.idx > 0: + self.ret_rms.update(self.rets[: self.idx]) + scale = np.float32(np.sqrt(self.ret_rms.var + self.epsilon)) + self.rewards[: self.idx] = np.clip( + self.rewards[: self.idx] / scale, -self.reward_clip, self.reward_clip + ) + return super().compute_advantages_and_returns( + policy=policy, batch_size=batch_size + ) diff --git a/src/plugrl_server/algorithm/ppo/ppo_config.py b/src/plugrl_server/algorithm/ppo/ppo_config.py new file mode 100644 index 0000000..4bf0cd3 --- /dev/null +++ b/src/plugrl_server/algorithm/ppo/ppo_config.py @@ -0,0 +1,47 @@ +import dataclasses + +from plugrl_server.algorithm.base_algorithm import BaseAlgoConfig +from plugrl_server.algorithm.registration import register_algo_config + +UID = "ppo" + + +@register_algo_config(UID) +@dataclasses.dataclass +class PPOAlgoConfig(BaseAlgoConfig): + """CleanRL `ppo_continuous_action.py`'s defaults, which it runs on every + MuJoCo task unchanged. + + CleanRL's rollout is `num_envs * num_steps` = 1 x 2048 transitions, split + into `num_minibatches` = 32; here that is `buffer_size` 2048 and + `batch_size` 64, however many environments the client runs. Its + 1,000,000 steps are `train_itrs` 488 iterations of 2048. + """ + + learning_rate: float = 3e-4 + # Linearly to zero across `train_itrs`, as CleanRL's `anneal_lr`. + anneal_lr: bool = True + buffer_size: int = 2048 + batch_size: int = 64 + update_epochs: int = 10 + gamma: float = 0.99 + gae_lambda: float = 0.95 + norm_adv: bool = True + clip_coef: float = 0.2 + clip_vloss: bool = True + ent_coef: float = 0.0 + vf_coef: float = 0.5 + max_grad_norm: float = 0.5 + # Stop an iteration's epochs once the approximate KL passes this. None, + # CleanRL's default, never stops. + target_kl: float | None = None + # gymnasium's NormalizeReward and TransformReward, which CleanRL wraps its + # environments in: rewards divided by the running deviation of the + # discounted return, then clipped. + normalize_rewards: bool = True + reward_clip: float = 10.0 + train_itrs: int = 488 + save_interval: int = 50 + + def __post_init__(self): + self.global_steps = self.train_itrs * self.buffer_size diff --git a/src/plugrl_server/policy/__init__.py b/src/plugrl_server/policy/__init__.py index 687c3b0..3000c9d 100644 --- a/src/plugrl_server/policy/__init__.py +++ b/src/plugrl_server/policy/__init__.py @@ -1,11 +1,13 @@ from . import dummy_policy as dummy_policy from . import dppo as dppo from . import fpo as fpo +from . import gaussian as gaussian from . import openpi as openpi __all__ = [ "dummy_policy", "dppo", "fpo", + "gaussian", "openpi", ] diff --git a/src/plugrl_server/policy/gaussian/__init__.py b/src/plugrl_server/policy/gaussian/__init__.py new file mode 100644 index 0000000..8451a6a --- /dev/null +++ b/src/plugrl_server/policy/gaussian/__init__.py @@ -0,0 +1,7 @@ +from .gaussian_policy import GaussianPolicy as GaussianPolicy +from .gaussian_policy import GaussianPolicyConfig as GaussianPolicyConfig + +__all__ = [ + "GaussianPolicy", + "GaussianPolicyConfig", +] diff --git a/src/plugrl_server/policy/gaussian/gaussian_policy.py b/src/plugrl_server/policy/gaussian/gaussian_policy.py new file mode 100644 index 0000000..32d95d3 --- /dev/null +++ b/src/plugrl_server/policy/gaussian/gaussian_policy.py @@ -0,0 +1,188 @@ +"""A Gaussian MLP policy, as CleanRL's `ppo_continuous_action.py` builds it. + +The baseline the expressive policies are measured against: a tanh MLP for the +mean, a log standard deviation that does not depend on the observation, and a +separate tanh MLP for the value. Layers are initialised orthogonally at gain +sqrt(2), except the mean's last layer at 0.01 - so the first policy is centred +on zero with unit deviation in every dimension - and the value's at 1.0. + +CleanRL does three things in gymnasium wrappers that PlugRL's environment +client does not do, so they are done here instead: + + ClipAction the environment receives the sample clipped to + [-action_clip, action_clip]; the runtime state keeps + the unclipped sample, whose density PPO's ratio is. + NormalizeObservation running mean and variance, stored as buffers and + updated by the algorithm between iterations rather + than at every step, so that an iteration's collection + and learning see the same normalisation. + TransformObservation the normalised observation clipped to +/-obs_clip. + +Rewards are normalised by `ppo`'s buffer, which has them. +""" + +from __future__ import annotations + +import dataclasses +from typing import Any + +import numpy as np +import torch +import torch.nn as nn + +from ..base_torch_policy import BaseTorchPolicy, BaseTorchPolicyConfig +from ..fpo.fpo_policy import _update_running_stats +from ..registration import register_policy, register_policy_config + +UID = "gaussian-policy" + + +def _layer(in_dim: int, out_dim: int, gain: float = float(np.sqrt(2))) -> nn.Linear: + layer = nn.Linear(in_dim, out_dim) + nn.init.orthogonal_(layer.weight, gain) + nn.init.constant_(layer.bias, 0.0) + return layer + + +def _tanh_mlp( + in_dim: int, hidden_dims: tuple[int, ...], out_dim: int, *, out_gain: float +) -> nn.Sequential: + layers: list[nn.Module] = [] + for hidden in hidden_dims: + layers += [_layer(in_dim, hidden), nn.Tanh()] + in_dim = hidden + layers.append(_layer(in_dim, out_dim, out_gain)) + return nn.Sequential(*layers) + + +@dataclasses.dataclass +class GaussianRuntimeState: + obs: torch.Tensor # (B, obs_dim), as the environment sent it + action: torch.Tensor # (B, action_dim), the sample before clipping + logprob: torch.Tensor # (B,), summed over the action's dimensions + value: torch.Tensor # (B,) + + +@register_policy_config(UID) +@dataclasses.dataclass +class GaussianPolicyConfig(BaseTorchPolicyConfig): + obs_dim: int = 17 + action_dim: int = 6 + hidden_dims: tuple[int, ...] = (64, 64) + # The observation's state keys, concatenated in this order, as for + # fpo-policy. MuJoCo sends one, "obs". + state_keys: tuple[str, ...] = ("obs",) + normalize_observations: bool = True + obs_clip: float = 10.0 + # Keep the observation statistics as loaded, as fpo-policy can. + freeze_obs_stats: bool = False + # HalfCheetah, Hopper and Walker2d all act in [-1, 1]. + action_clip: float = 1.0 + + +@register_policy(UID) +class GaussianPolicy(BaseTorchPolicy): + config: GaussianPolicyConfig + + def __init__(self, config: GaussianPolicyConfig): + super().__init__(config) + self.action_dim = config.action_dim + self.action_horizon = 1 + self.actor_mean = _tanh_mlp( + config.obs_dim, config.hidden_dims, config.action_dim, out_gain=0.01 + ) + self.actor_logstd = nn.Parameter(torch.zeros(1, config.action_dim)) + self.critic = _tanh_mlp(config.obs_dim, config.hidden_dims, 1, out_gain=1.0) + self.register_buffer("obs_stats_count", torch.zeros((), dtype=torch.float32)) + self.register_buffer("obs_stats_mean", torch.zeros(config.obs_dim)) + self.register_buffer("obs_stats_var_sum", torch.zeros(config.obs_dim)) + self.register_buffer("obs_stats_std", torch.ones(config.obs_dim)) + self.to(self.device) + + def extract_model_obs_tensor(self, _obs: dict[str, Any]) -> torch.Tensor: + states = _obs["states"] + keys = self.config.state_keys + missing = [k for k in keys if k not in states] + if missing: + raise KeyError( + f"state keys {missing} are not in the observation, which has " + f"{sorted(states)}; set GaussianPolicyConfig.state_keys" + ) + parts = [ + torch.as_tensor(states[k], dtype=torch.float32, device=self.device) + for k in keys + ] + x = parts[0] if len(parts) == 1 else torch.cat(parts, dim=-1) + if x.shape[-1] != self.config.obs_dim: + raise ValueError( + f"state keys {list(keys)} give {x.shape[-1]} values per " + f"observation, but obs_dim is {self.config.obs_dim}" + ) + return x + + def normalize_obs(self, x: torch.Tensor) -> torch.Tensor: + if not self.config.normalize_observations: + return x + z = (x - self.obs_stats_mean) / self.obs_stats_std + return z.clamp(-self.config.obs_clip, self.config.obs_clip) + + @torch.no_grad() + def update_obs_stats(self, x: torch.Tensor) -> None: + if not self.config.normalize_observations or self.config.freeze_obs_stats: + return + count, mean, var_sum, _ = _update_running_stats( + x=x.to(self.device), + count=self.obs_stats_count, + mean=self.obs_stats_mean, + var_sum=self.obs_stats_var_sum, + ) + self.obs_stats_count = count + self.obs_stats_mean = mean + self.obs_stats_var_sum = var_sum + # NormalizeObservation's divisor, sqrt(var + 1e-8). + self.obs_stats_std = torch.sqrt(var_sum / count + 1e-8) + + def _distribution(self, z: torch.Tensor) -> torch.distributions.Normal: + mean = self.actor_mean(z) + return torch.distributions.Normal(mean, self.actor_logstd.expand_as(mean).exp()) + + def get_action_and_runtime_state( + self, _obs: dict[str, Any] + ) -> tuple[np.ndarray, GaussianRuntimeState]: + x = self.extract_model_obs_tensor(_obs) + z = self.normalize_obs(x) + dist = self._distribution(z) + sample = dist.sample() + state = GaussianRuntimeState( + obs=x, + action=sample, + logprob=dist.log_prob(sample).sum(-1), + value=self.critic(z).squeeze(-1), + ) + clip = self.config.action_clip + action = sample.clamp(-clip, clip).cpu().numpy().astype(np.float32) + return action[:, None, :], state + + def evaluate_actions( + self, obs: torch.Tensor, action: torch.Tensor + ) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]: + """Log-probability, entropy and value, each (B,), for PPO's loss.""" + z = self.normalize_obs(obs) + dist = self._distribution(z) + return ( + dist.log_prob(action).sum(-1), + dist.entropy().sum(-1), + self.critic(z).squeeze(-1), + ) + + def get_value(self, _obs: dict[str, Any]) -> torch.Tensor: + z = self.normalize_obs(self.extract_model_obs_tensor(_obs)) + return self.critic(z).squeeze(-1).cpu() + + def fake_runtime_state(self, batch_size: int) -> GaussianRuntimeState: + return GaussianRuntimeState( + obs=torch.zeros(batch_size, self.config.obs_dim), + action=torch.zeros(batch_size, self.action_dim), + logprob=torch.zeros(batch_size), + value=torch.zeros(batch_size), + ) diff --git a/tests/test_gaussian_ppo.py b/tests/test_gaussian_ppo.py new file mode 100644 index 0000000..416c470 --- /dev/null +++ b/tests/test_gaussian_ppo.py @@ -0,0 +1,463 @@ +"""`gaussian-policy` and `ppo`: CleanRL's continuous-action PPO, through PlugRL. + +Every other policy-algorithm pair on the coverage figure trains an expressive +policy - a flow or a diffusion model - with an algorithm built for it. None is +the baseline those are measured against: a Gaussian MLP trained by PPO. These +two are that, written to CleanRL's `ppo_continuous_action.py` (Huang et al., +"The 37 Implementation Details of Proximal Policy Optimization", 2022), so +that if the pair fails to learn MuJoCo it is the platform that failed and not +an unfamiliar algorithm. + +CleanRL does part of its work in gymnasium wrappers on the environment - +clipping actions, normalising observations and rewards. PlugRL's environment +client does none of that, so here it is done on the server, and these tests +check each piece where PlugRL's plumbing could get it wrong: the sample the +log-probability was taken of is the one learned from, the ratio is one before +any update, the statistics do not move between collecting and learning, the +schedule anneals as CleanRL's does, and the whole thing improves a policy on a +task with a known answer. +""" + +from __future__ import annotations + +import math +import subprocess +import sys + +import numpy as np +import pytest +import torch + +from plugrl_server.algorithm.ppo import ppo as ppo_module +from plugrl_server.algorithm.ppo.ppo import PPOAlgorithm +from plugrl_server.algorithm.ppo.ppo_buffer import PPOBuffer +from plugrl_server.algorithm.ppo.ppo_config import PPOAlgoConfig +from plugrl_server.common.data_utils import unbatch_aggregate +from plugrl_server.policy.gaussian.gaussian_policy import ( + GaussianPolicy, + GaussianPolicyConfig, +) +from plugrl_server.policy.state import slice_policy_step_state + +OBS_DIM = 5 +ACTION_DIM = 3 +ENVS = 2 +TARGET = 0.5 + + +def _policy(seed: int = 0, **overrides) -> GaussianPolicy: + torch.manual_seed(seed) + return GaussianPolicy( + GaussianPolicyConfig( + obs_dim=OBS_DIM, action_dim=ACTION_DIM, device="cpu", **overrides + ) + ) + + +def _algo(seed: int = 0, **overrides) -> PPOAlgorithm: + fields = dict(buffer_size=128, batch_size=32, update_epochs=2, train_itrs=10) + fields.update(overrides) + algo = PPOAlgorithm(PPOAlgoConfig(**fields), _policy(seed)) + algo.init_optimizers() + return algo + + +def _obs(envs: int, rng: np.random.Generator) -> dict: + return {"states": {"obs": rng.normal(size=(envs, OBS_DIM)).astype(np.float32)}} + + +def _collect(algo, rng, *, reward_fn=lambda action: 1.0, episode_length=5): + """Fill the buffer the way websocket_agent_server does. Returns the rewards.""" + prev_node = {env: (-1, "") for env in range(ENVS)} + terminated = {env: False for env in range(ENVS)} + obs = _obs(ENVS, rng) + rewards: list[float] = [] + for round_index in range(10_000): + action, runtime_state = algo.infer(obs) + step_state = algo.build_step_state_from_runtime_state( + runtime_state, include_train_state=True + ) + next_obs = _obs(ENVS, rng) + obs_list = unbatch_aggregate(obs, aggregate_method="concat") + next_obs_list = unbatch_aggregate(next_obs, aggregate_method="concat") + done = round_index % episode_length == episode_length - 1 + for env in range(ENVS): + reward = float(reward_fn(action[env])) + rewards.append(reward) + one = slice_policy_step_state(step_state, slice(env, env + 1)) + prev_node[env], _, _ = algo.feedback( + obs=obs_list[env], + runtime_state=one.runtime_state, + train_state=one.train_state, + terminated=terminated[env], + truncated=False, + next_obs=next_obs_list[env], + reward=reward, + info={}, + next_terminated=done, + next_truncated=False, + prev_node=prev_node[env], + ) + terminated[env] = done + obs = next_obs + if algo.should_learn(): + return rewards + pytest.fail("the buffer never filled") + + +def _iteration(algo, rng, **collect_kwargs) -> tuple[float, dict]: + rewards = _collect(algo, rng, **collect_kwargs) + algo.pre_learn() + _, metrics = algo.learn() + algo.post_learn() + return float(np.mean(rewards)), metrics + + +def _bandit_reward(action: np.ndarray) -> float: + return -float(((action - TARGET) ** 2).mean()) + + +class TestThePolicy: + def test_it_declares_its_action_shape(self): + """The server's metadata message publishes both.""" + policy = _policy() + + assert (policy.action_dim, policy.action_horizon) == (ACTION_DIM, 1) + + def test_the_env_gets_a_clipped_sample_and_the_state_keeps_the_sample(self): + """ClipAction clips what the environment receives, not what was drawn. + + The log-probability is the density of the unclipped sample, and PPO's + ratio is only a ratio of densities if learning scores that same sample. + """ + policy = _policy() + policy.actor_logstd.data.fill_(1.0) # std e: many samples land outside + with torch.inference_mode(): + action, state = policy.get_action_and_runtime_state( + _obs(64, np.random.default_rng(0)) + ) + + assert action.shape == (64, 1, ACTION_DIM) + assert action.dtype == np.float32 + assert np.abs(action).max() <= 1.0 + assert state.action.shape == (64, ACTION_DIM) + assert (state.action.abs() > 1).any() + np.testing.assert_array_equal(action[:, 0], state.action.clamp(-1, 1).numpy()) + assert state.logprob.shape == state.value.shape == (64,) + + def test_learning_scores_exactly_what_was_sampled(self): + policy = _policy() + with torch.inference_mode(): + _, state = policy.get_action_and_runtime_state( + _obs(16, np.random.default_rng(0)) + ) + + logprob, entropy, value = policy.evaluate_actions(state.obs, state.action) + + torch.testing.assert_close(logprob, state.logprob) + torch.testing.assert_close(value, state.value) + # A diagonal Gaussian's entropy, summed over the action's dimensions. + per_dim = 0.5 + 0.5 * math.log(2 * math.pi) + policy.actor_logstd.detach() + torch.testing.assert_close(entropy, per_dim.sum().expand(16)) + + def test_it_is_initialised_the_way_cleanrl_initialises_it(self): + """Orthogonal weights at gain sqrt(2), 0.01 on the mean's last layer and + 1.0 on the value's, zero biases, a log std of zero, tanh between.""" + policy = _policy() + + assert torch.equal(policy.actor_logstd.detach(), torch.zeros(1, ACTION_DIM)) + for net, out, last_gain in ( + (policy.actor_mean, ACTION_DIM, 0.01), + (policy.critic, 1, 1.0), + ): + layers = [m for m in net if isinstance(m, torch.nn.Linear)] + assert [layer.out_features for layer in layers] == [64, 64, out] + assert sum(isinstance(m, torch.nn.Tanh) for m in net) == 2 + for layer in layers: + assert torch.equal(layer.bias.detach(), torch.zeros_like(layer.bias)) + first, hidden, last = (layer.weight.detach() for layer in layers) + # Orthogonal: a tall matrix's columns and a wide one's rows are + # orthonormal, times the gain. + torch.testing.assert_close(first.T @ first, 2.0 * torch.eye(OBS_DIM)) + torch.testing.assert_close(hidden @ hidden.T, 2.0 * torch.eye(64)) + torch.testing.assert_close( + last @ last.T, last_gain**2 * torch.eye(out), atol=1e-6, rtol=1e-5 + ) + + def test_observations_are_normalised_then_clipped_at_ten(self): + """NormalizeObservation, then TransformObservation's clip to [-10, 10].""" + policy = _policy() + rng = np.random.default_rng(0) + scales = np.array([0.1, 1.0, 10.0, 100.0, 1000.0]) + data = torch.as_tensor( + rng.normal(3.0, scales, size=(4096, OBS_DIM)), dtype=torch.float32 + ) + + policy.update_obs_stats(data) + z = policy.normalize_obs(data) + + torch.testing.assert_close(z.mean(0), torch.zeros(OBS_DIM), atol=1e-3, rtol=0) + torch.testing.assert_close(z.std(0), torch.ones(OBS_DIM), atol=1e-3, rtol=0) + far = policy.normalize_obs(torch.full((1, OBS_DIM), 1e6)) + assert far.max().item() == 10.0 + + def test_frozen_statistics_do_not_move(self): + policy = _policy(freeze_obs_stats=True) + + policy.update_obs_stats(torch.randn(64, OBS_DIM) * 5) + + assert policy.obs_stats_count.item() == 0.0 + assert torch.equal(policy.obs_stats_std, torch.ones(OBS_DIM)) + + def test_state_keys_are_concatenated_in_the_order_given(self): + policy = _policy(state_keys=("a", "b")) + obs = { + "states": { + "b": np.ones((2, 3), dtype=np.float32), + "a": np.zeros((2, 2), dtype=np.float32), + } + } + + x = policy.extract_model_obs_tensor(obs) + + assert torch.equal(x, torch.tensor([[0.0, 0, 1, 1, 1]] * 2)) + + +class TestTheBuffer: + def _buffer(self, **kwargs) -> tuple[PPOBuffer, dict]: + algo = _algo() + example = algo.example_train_state(batch_size=1) + return PPOBuffer(buffer_size=8, example_train_state=example, **kwargs), example + + def test_the_discounted_return_restarts_with_each_episode(self): + """As gymnasium's NormalizeReward and DPPO's RunningRewardScaler keep it. + + A frame's `terminated` says the step before it ended an episode, so a + frame carrying it is the first of a new one. + """ + buffer, one = self._buffer(gamma=0.5) + node = (-1, "") + for starts_episode in (False, False, True, False): + node = buffer.add_frame( + prev_node=node, + train_state=one, + reward=1.0, + terminated=starts_episode, + truncated=False, + last_value=None, + next_terminated=False, + next_truncated=False, + ) + + np.testing.assert_allclose(buffer.rets[:4], [1.0, 1.5, 1.0, 1.5]) + + @pytest.mark.parametrize("normalize", [True, False]) + def test_rewards_are_scaled_by_the_returns_deviation_and_clipped(self, normalize): + buffer, one = self._buffer(gamma=0.9, normalize_rewards=normalize) + raw = np.array([0.1, 50.0, -3.0, 2.0, 0.0, 400.0, 1.0, -1.0], np.float32) + node = (-1, "") + for reward in raw: + node = buffer.add_frame( + prev_node=node, + train_state=one, + reward=float(reward), + terminated=False, + truncated=False, + last_value=None, + next_terminated=False, + next_truncated=False, + ) + buffer.add_next_obs_value_request( + obs=_obs(1, np.random.default_rng(0)), end_node=node + ) + + buffer.compute_advantages_and_returns(policy=_policy(), batch_size=8) + + if not normalize: + np.testing.assert_array_equal(buffer.rewards[:8], raw) + return + scale = np.sqrt(buffer.ret_rms.var + 1e-8) + expected = np.clip(raw / scale, -10.0, 10.0) + np.testing.assert_allclose(buffer.rewards[:8], expected, rtol=1e-6) + assert np.abs(buffer.rewards[:8]).max() < np.abs(raw).max() + + +class TestTheAlgorithm: + def test_one_iteration_moves_every_parameter_and_reports_finite_losses(self): + algo = _algo() + before = {k: v.detach().clone() for k, v in algo.policy.named_parameters()} + + _, metrics = _iteration( + algo, np.random.default_rng(0), reward_fn=_bandit_reward + ) + + for name, parameter in algo.policy.named_parameters(): + assert not torch.equal(parameter.detach(), before[name]), name + for key in ( + "policy_loss", + "value_loss", + "entropy", + "old_approx_kl", + "approx_kl", + "clipfrac", + ): + assert math.isfinite(metrics["losses"][key]), key + assert algo.curr_train_itrs == 1 + assert len(algo.rollout_buffer) == 0 + + def test_before_any_update_every_ratio_is_one(self): + algo = _algo() + _collect(algo, np.random.default_rng(0)) + algo.pre_learn() + n = len(algo.rollout_buffer) + obs = torch.as_tensor( + algo.rollout_buffer.train_state_storage.get_item(slice(0, n)) + ) + + logprob, _, _ = algo.policy.evaluate_actions( + obs, torch.as_tensor(algo.rollout_buffer.actions[:n]) + ) + + torch.testing.assert_close( + logprob, torch.as_tensor(algo.rollout_buffer.logprobs[:n]) + ) + + def test_statistics_are_updated_after_learning_from_the_buffer(self): + """After, as DPPO does for fpo-policy: one iteration collects and learns + under one normalisation, and the next collects under the new one.""" + algo = _algo() + rng = np.random.default_rng(0) + _collect(algo, rng) + algo.pre_learn() + algo.learn() + assert algo.policy.obs_stats_count.item() == 0.0 + + algo.post_learn() + + assert algo.policy.obs_stats_count.item() == 128.0 + + def test_the_learning_rate_anneals_linearly_as_cleanrl_anneals_it(self): + """lr = (1 - (iteration - 1) / num_iterations) * learning_rate.""" + algo = _algo(train_itrs=4, learning_rate=1e-3) + rng = np.random.default_rng(0) + + seen = [_iteration(algo, rng)[1]["models"]["learning_rate"] for _ in range(4)] + + np.testing.assert_allclose(seen, [1e-3, 0.75e-3, 0.5e-3, 0.25e-3]) + assert algo.should_stop() + + def test_without_annealing_the_rate_stays_put(self): + algo = _algo(train_itrs=4, learning_rate=1e-3, anneal_lr=False) + rng = np.random.default_rng(0) + + seen = [_iteration(algo, rng)[1]["models"]["learning_rate"] for _ in range(2)] + + np.testing.assert_allclose(seen, [1e-3, 1e-3]) + + def test_one_adam_over_every_parameter_with_cleanrl_epsilon(self): + algo = _algo() + + (group,) = algo.optimizer.param_groups + assert isinstance(algo.optimizer, torch.optim.Adam) + assert group["eps"] == 1e-5 + assert {id(p) for p in group["params"]} == { + id(p) for p in algo.policy.parameters() + } + + def test_every_step_clips_the_whole_gradient_at_half(self, monkeypatch): + calls = [] + real = torch.nn.utils.clip_grad_norm_ + + def recording(parameters, max_norm, *args, **kwargs): + parameters = list(parameters) + calls.append((len(parameters), max_norm)) + return real(parameters, max_norm, *args, **kwargs) + + monkeypatch.setattr(ppo_module.nn.utils, "clip_grad_norm_", recording) + algo = _algo() + _iteration(algo, np.random.default_rng(0)) + + steps = algo.config.update_epochs * algo.config.buffer_size // 32 + assert calls == [(len(list(algo.policy.parameters())), 0.5)] * steps + + def test_a_checkpoint_resumes_where_it_left_off(self): + algo = _algo() + rng = np.random.default_rng(0) + _iteration(algo, rng, reward_fn=_bandit_reward) + checkpoint = algo.create_checkpoint() + + resumed = _algo(seed=1) + resumed.load_checkpoint(checkpoint) + + for (name, a), b in zip( + algo.policy.state_dict().items(), resumed.policy.state_dict().values() + ): + assert torch.equal(a, b), name + assert resumed.curr_train_itrs == 1 + assert resumed.global_step == algo.global_step + assert resumed.rollout_buffer.ret_rms.var == algo.rollout_buffer.ret_rms.var + assert resumed.rollout_buffer.ret_rms.count == algo.rollout_buffer.ret_rms.count + state, other = algo.optimizer.state_dict(), resumed.optimizer.state_dict() + for key in state["state"]: + assert torch.equal( + state["state"][key]["exp_avg"], other["state"][key]["exp_avg"] + ) + # The schedule is a function of the iteration, so it resumes with it. + _, metrics = _iteration(resumed, rng) + assert metrics["models"]["learning_rate"] == pytest.approx(3e-4 * 0.9) + + +@pytest.mark.parametrize("seed", (0, 1, 2, 3, 4)) +def test_ppo_learns_a_bandit_with_a_known_answer_and_keeps_it(seed: int): + """The bandit `test_fpo_learns_anything` gives FPO: reward is minus the + squared distance from the action to 0.5 in every dimension. + + A Gaussian at mean zero with unit deviation scores about -0.78 here. It + has to end at a quarter of that, and no worse than twice its best - the + bar four of FPO's ten cases fail. Measured, last over best: 1.00, 1.12, + 1.00, 1.00, 1.02, ending between -0.115 and -0.145. + + The rate is 3e-3, not CleanRL's 3e-4, because the floor on this reward is + set by how far the deviation narrows, and at 3e-4, annealed over thirty + iterations of this size, it cannot narrow far: seed 0 goes from -0.78 to + -0.23, still rising at the last iteration. + """ + algo = _algo( + seed=seed, + buffer_size=256, + batch_size=64, + update_epochs=10, + train_itrs=30, + learning_rate=3e-3, + ) + rng = np.random.default_rng(seed) + + history = [_iteration(algo, rng, reward_fn=_bandit_reward)[0] for _ in range(30)] + + best, last = max(history), history[-1] + assert np.isfinite(history).all(), history + assert last > history[0] / 4, ( + f"first {history[0]:+.4f}, best {best:+.4f}, last {last:+.4f}" + ) + assert last > 2 * best, f"best {best:+.4f}, last {last:+.4f}" + + +def test_both_are_on_the_cli_of_a_plain_install(): + """In a fresh interpreter, so it is the packages' own imports that register + them and not this file's.""" + result = subprocess.run( + [ + sys.executable, + "-m", + "plugrl_server.cli", + "gaussian-policy", + "default", + "--help", + ], + capture_output=True, + text=True, + timeout=180, + ) + + assert result.returncode == 0, result.stderr[-2000:] + assert "ppo" in result.stdout From ab6e32f6608d74d7f1a20885adf1811a5e460f4a Mon Sep 17 00:00:00 2001 From: tactino <18781106300@163.com> Date: Sun, 27 Sep 2026 17:38:43 -0400 Subject: [PATCH 2/8] exp: E38 scripts - gaussian-policy under ppo on three MuJoCo tasks (before the pilot) --- experiments/e38-gaussian-ppo/run.sh | 58 +++++ experiments/e38-gaussian-ppo/summarise.py | 244 ++++++++++++++++++++++ 2 files changed, 302 insertions(+) create mode 100644 experiments/e38-gaussian-ppo/run.sh create mode 100644 experiments/e38-gaussian-ppo/summarise.py diff --git a/experiments/e38-gaussian-ppo/run.sh b/experiments/e38-gaussian-ppo/run.sh new file mode 100644 index 0000000..a5764ab --- /dev/null +++ b/experiments/e38-gaussian-ppo/run.sh @@ -0,0 +1,58 @@ +#!/usr/bin/env bash +# E38: gaussian-policy under ppo - CleanRL's ppo_continuous_action.py - on the +# three MuJoCo tasks, 488 iterations. See PROTOCOL.md. +# +# bash run.sh +# +# Every cell is CleanRL's file with its defaults: a rollout of 2,048 steps, +# minibatches of 64, 488 iterations (999,424 steps: CleanRL runs +# 1,000,000 // 2,048 = 488). One environment per seed, as CleanRL's +# num_envs=1. E30's run_cell.sh, unchanged, given the task's +# dimensions through EXTRA_POLICY_ARGS. The server runs this checkout's code +# through PYTHONPATH. +set -uo pipefail + +HERE="$(cd "$(dirname "$0")" && pwd)" +ROOT="$(cd "$HERE/../.." && pwd)" +R="$HERE/results" +CELL="$ROOT/experiments/e30-fpo-dppo-factors/run_cell.sh" +ITERS="${ITERS:-488}" +SEEDS="${SEEDS:-0 1 2}" +mkdir -p "$R" + +export OMP_NUM_THREADS=1 +export PYTHONPATH="$ROOT/src" +export PLUGRL_ENV_CLIENT="${PLUGRL_ENV_CLIENT:-$ROOT/../plugrl-env-client}" +export PLUGRL_SERVER_PYTHON="${PLUGRL_SERVER_PYTHON:-$ROOT/../plugrl-server/.venv/bin/python}" +export PLUGRL_CLIENT_PYTHON="${PLUGRL_CLIENT_PYTHON:-$PLUGRL_ENV_CLIENT/.venv/bin/python}" +for py in "$PLUGRL_SERVER_PYTHON" "$PLUGRL_CLIENT_PYTHON"; do + [ -x "$py" ] || { echo "error: no Python at $py" >&2; exit 1; } +done + +date '+start %F %T' +echo "code: $(git -C "$ROOT" rev-parse --short HEAD) (src $PYTHONPATH)" +echo "client: $(git -C "$PLUGRL_ENV_CLIENT" rev-parse --short HEAD)" +echo "iters: $ITERS seeds: $SEEDS" + +cell() { # NAME TASK OBS_DIM ACTION_DIM PORT_BASE + SEEDS="$SEEDS" BUFFER=2048 BATCH=64 PORT_BASE="$5" \ + EXTRA_POLICY_ARGS="--policy.obs-dim $3 --policy.action-dim $4" \ + bash "$CELL" "$1" gaussian-policy default ppo default "$2" "$ITERS" "$R/$1" \ + > "$R/$1.out" 2>&1 +} + +cell ppo-cheetah HalfCheetah-v5 17 6 9940 & +A=$! +sleep 20 +cell ppo-hopper Hopper-v5 11 3 9950 & +B=$! +sleep 20 +cell ppo-walker Walker2d-v5 17 6 9960 & +C=$! +wait "$A" "$B" "$C" + +date '+end %F %T' +for c in ppo-cheetah ppo-hopper ppo-walker; do + printf '%-12s %s\n' "$c" "$(grep -E 'failed seeds' "$R/$c.out" 2>/dev/null)" +done +echo E38_DONE diff --git a/experiments/e38-gaussian-ppo/summarise.py b/experiments/e38-gaussian-ppo/summarise.py new file mode 100644 index 0000000..ef9745b --- /dev/null +++ b/experiments/e38-gaussian-ppo/summarise.py @@ -0,0 +1,244 @@ +"""Read E38's verdicts from each cell's logs and tensorboards. + + python summarise.py + +P1 cell by cell, V1, the status rule, then P2, P3 and P4 - PROTOCOL.md's +reading order - and the reported returns beside CleanRL's. E33's +summarise.py with E38's cells, 488 iterations, and the learning-rate schedule +as the check that the run was the configuration it claims. Writes +`summary.tsv` in E24's columns. +""" + +import datetime +import pathlib +import re +import sys + +from tensorboard.backend.event_processing.event_accumulator import EventAccumulator + +HERE = pathlib.Path(__file__).resolve().parent +RESULTS = HERE / "results" +ITERS = 488 +SEEDS = (0, 1, 2) + +# cell: (task, CleanRL's reported return for the -v4 task, three seeds) +CELLS = { + "ppo-cheetah": ("HalfCheetah-v5", "1442.64 +/- 46.03"), + "ppo-hopper": ("Hopper-v5", "2382.86 +/- 271.74"), + "ppo-walker": ("Walker2d-v5", "2287.95 +/- 571.78"), +} +ABSOLUTE = 500.0 # Hopper, Walker2d: mean of iterations 479-488 +GAIN = 200.0 # HalfCheetah: the mean of the last ten minus the first +LEARNING_RATE = 3e-4 + + +def run_dir(cell: str, seed: int) -> pathlib.Path | None: + hits = sorted((RESULTS / cell).glob(f"*/*/{cell}-seed{seed}")) + return hits[0] if hits else None + + +def curve(run: pathlib.Path | None, tag: str) -> list[float]: + if run is None: + return [] + events = sorted((run / "tensorboard").glob("events.*")) + if not events: + return [] + acc = EventAccumulator(str(events[0]), size_guidance={"scalars": 0}) + acc.Reload() + if tag not in acc.Tags()["scalars"]: + return [] + return [e.value for e in acc.Scalars(tag)] + + +def wall_minutes(out: str) -> float | None: + stamps = dict(re.findall(r"^(start|end)\s+(\S+ \S+)$", out, flags=re.M)) + if "start" not in stamps or "end" not in stamps: + return None + fmt = "%Y-%m-%d %H:%M:%S" + delta = datetime.datetime.strptime(stamps["end"], fmt) - datetime.datetime.strptime( + stamps["start"], fmt + ) + return delta.total_seconds() / 60 + + +def mean(xs: list[float]) -> float: + return sum(xs) / len(xs) if xs else float("nan") + + +def figure(task: str, reward: list[float]) -> float: + """The number the status rule reads: the gain for HalfCheetah, the level otherwise.""" + if len(reward) < ITERS: + return float("nan") + last10 = mean(reward[ITERS - 10 : ITERS]) + return last10 - reward[0] if task == "HalfCheetah-v5" else last10 + + +def passes(task: str, reward: list[float]) -> bool: + bar = GAIN if task == "HalfCheetah-v5" else ABSOLUTE + return figure(task, reward) >= bar + + +def schedule_took(rates: list[float]) -> bool: + """CleanRL's anneal: iteration i (from 1) learns at (1 - (i-1)/488) * 3e-4.""" + if len(rates) < ITERS: + return False + expected = [(1 - i / ITERS) * LEARNING_RATE for i in range(ITERS)] + return all(abs(r - e) <= 1e-6 * LEARNING_RATE for r, e in zip(rates, expected)) + + +def main() -> int: + rows = {} + runs_ok = {} + print("P1 every cell runs end to end") + for cell, (task, _) in CELLS.items(): + out_path = RESULTS / f"{cell}.out" + out = ( + out_path.read_text(encoding="utf-8", errors="replace") + if out_path.exists() + else "" + ) + exits = { + int(s): int(rc) for s, rc in re.findall(r"seed (\d) finished rc=(\d+)", out) + } + ok_cell = True + for seed in SEEDS: + run = run_dir(cell, seed) + reward = curve(run, "rollout/reward") + saves = [p for p in run.iterdir() if p.name.isdigit()] if run else [] + log_path = RESULTS / cell / f"server-seed{seed}.log" + log = ( + log_path.read_text(encoding="utf-8", errors="replace") + if log_path.exists() + else "" + ) + # The checkpoint directory's step can be off by one (E16), so + # saves are counted, not named. + ok = ( + len(reward) >= ITERS + and len(saves) >= ITERS // 20 + and bool(log) + and "Traceback" not in log + and exits.get(seed) == 0 + ) + ok_cell &= ok + rows[cell, seed] = dict( + ok=ok, + reward=reward, + length=curve(run, "rollout/length"), + rate=curve(run, "models/learning_rate"), + ) + runs_ok[cell] = ok_cell + minutes = wall_minutes(out) + print( + f" {cell:12s} {task:15s} {'HOLDS' if ok_cell else 'FAILS'}" + + (f" ({minutes:.0f} min)" if minutes is not None else "") + ) + + print("\nV1 every iteration learned at CleanRL's annealed rate") + v1 = {} + for cell in CELLS: + v1[cell] = all(schedule_took(rows[cell, s]["rate"]) for s in SEEDS) + rates = rows[cell, SEEDS[0]]["rate"] + span = f"{rates[0]:.3g} to {rates[-1]:.3g}" if rates else "not logged" + print(f" {cell:12s} {span} {'PASS' if v1[cell] else 'FAIL'}") + + print( + f"\nstatus learns: Hopper, Walker2d 479-488 >= {ABSOLUTE:.0f}; " + f"HalfCheetah 479-488 minus the first >= +{GAIN:.0f}; on 2 of 3" + ) + status = {} + for cell, (task, _) in CELLS.items(): + passed = sum(passes(task, rows[cell, s]["reward"]) for s in SEEDS) + if not (runs_ok[cell] and v1[cell]): + status[cell] = "not read" + elif passed >= 2: + status[cell] = "learns" + else: + status[cell] = "did not learn in 999,424 steps" + shown = ", ".join( + f"{figure(task, rows[cell, s]['reward']):+.1f}" + if task == "HalfCheetah-v5" + else f"{figure(task, rows[cell, s]['reward']):.1f}" + for s in SEEDS + ) + print(f" {cell:12s} {shown:28s} {passed} of 3 -> {status[cell]}") + + def verdict(name: str, text: str, cell: str) -> None: + s = status[cell] + mark = ( + "NOT READ" + if s == "not read" + else ("HOLDS" if s == "learns" else "FALSIFIED") + ) + print(f"\n{name} {text}\n {mark}") + + verdict("P2", "gaussian-policy learns Hopper under ppo", "ppo-hopper") + verdict("P3", "gaussian-policy learns Walker2d under ppo", "ppo-walker") + verdict("P4", "gaussian-policy learns HalfCheetah under ppo", "ppo-cheetah") + + print( + "\nreported: mean return over iterations 91-100, 241-250 and 479-488," + " beside CleanRL's on the -v4 task" + ) + for cell, (task, cleanrl) in CELLS.items(): + for s in SEEDS: + r = rows[cell, s]["reward"] + spans = [ + mean(r[a:b]) if len(r) >= b else float("nan") + for a, b in ((90, 100), (240, 250), (478, 488)) + ] + print(f" {cell:12s} seed {s} " + " ".join(f"{v:8.1f}" for v in spans)) + print(f" {cell:12s} CleanRL {cleanrl}") + + print( + "\nreported: first iteration and mean of the last ten, return and episode length" + ) + lines = [ + "cell\tpolicy\talgorithm\ttask\tclaim\tseed\titerations\truns\tfirst_return\tlast10_return\tfirst_length\tlast10_length" + ] + for cell, (task, _) in CELLS.items(): + for s in SEEDS: + r = rows[cell, s] + reward, length = r["reward"], r["length"] + first = reward[0] if reward else float("nan") + last10 = ( + mean(reward[ITERS - 10 : ITERS]) + if len(reward) >= ITERS + else float("nan") + ) + first_len = length[0] if length else float("nan") + last10_len = ( + mean(length[ITERS - 10 : ITERS]) + if len(length) >= ITERS + else float("nan") + ) + print( + f" {cell:12s} seed {s} return {first:8.1f} -> {last10:8.1f} " + f"length {first_len:6.1f} -> {last10_len:6.1f}" + ) + claim = "learns" if status[cell] == "learns" else "runs" + lines.append( + "\t".join( + [ + cell, + "gaussian-policy", + "ppo", + task, + claim, + str(s), + str(len(reward)), + str(r["ok"]).lower(), + f"{first:.1f}", + f"{last10:.1f}", + f"{first_len:.1f}", + f"{last10_len:.1f}", + ] + ) + ) + (HERE / "summary.tsv").write_text("\n".join(lines) + "\n", encoding="utf-8") + print(f"\nwrote {HERE / 'summary.tsv'}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From 4ccf245801be2d230de0c77e6229e8c1cc8ae670 Mon Sep 17 00:00:00 2001 From: tactino <18781106300@163.com> Date: Sun, 27 Sep 2026 17:49:22 -0400 Subject: [PATCH 3/8] fix: GAE ends an episode at the step that ended it, not one later Both servers call feedback for step t with terminated/truncated set to step t-1's outcome and next_terminated/next_truncated to step t's, and chain an episode's first frame to the last frame of the episode before. A frame carrying `terminated` therefore begins an episode. Until 48042e5 the GAE recursion cut a linked frame with the successor's dones[next_idx] - step t's own outcome. 48042e5 made it read the frame's own dones[step] (terminated[step]/truncated[step]) - step t-1's - so since then the step that ended an episode bootstrapped from the next episode's first state and kept accumulating its advantages, and each episode's first step was cut off from the rest. It also made a chain end carrying `dones` count as terminal and drop its bootstrap value. Both branches now read whether a frame's own transition ended from next_*. Three 3-step episodes of reward 1 at zero value, gamma = lambda = 1: main gave 4 3 2 1 3 2 1 2 1; the answer is 3 2 1 three times. --- src/plugrl_server/buffer/rollout_buffer.py | 32 ++--- tests/test_gae_episode_boundary.py | 151 +++++++++++++++++++++ 2 files changed, 164 insertions(+), 19 deletions(-) create mode 100644 tests/test_gae_episode_boundary.py diff --git a/src/plugrl_server/buffer/rollout_buffer.py b/src/plugrl_server/buffer/rollout_buffer.py index 0e73c9e..ffae014 100644 --- a/src/plugrl_server/buffer/rollout_buffer.py +++ b/src/plugrl_server/buffer/rollout_buffer.py @@ -340,22 +340,22 @@ def compute_advantages_and_returns( for step in reversed(range(self.idx)): next_idx = self.next_indices[step] + # Whether this frame's own transition ended its episode. The + # servers pass a step's `terminated`/`truncated` as the step + # before it's outcome - a frame carrying them begins an episode - + # and `next_*` as its own, and they chain the first frame of an + # episode to the last of the one before. Reading the frame's own + # `dones` here, as this did from 48042e5 on, ended every episode + # one step late. + if self.treat_truncated_as_done: + terminated = float(self.next_done[step]) + truncated = 0.0 + else: + terminated = float(self.next_terminated[step]) + truncated = float(self.next_truncated[step]) if next_idx == 0: next_values = self.last_values[step] next_gae_lam = 0 - if self.treat_truncated_as_done: - if self.dones[step]: - terminated = float(self.dones[step]) - else: - terminated = float(self.next_done[step]) - truncated = 0.0 - else: - if self.dones[step]: - terminated = float(self.terminated[step]) - truncated = float(self.truncated[step]) - else: - terminated = float(self.next_terminated[step]) - truncated = float(self.next_truncated[step]) # check last values not overflow or abs extreme large assert abs(next_values).max() < 1e6, ( f"last_values overflow: {next_values}" @@ -363,12 +363,6 @@ def compute_advantages_and_returns( else: next_values = self.values[next_idx] next_gae_lam = self.advantages[next_idx] - if self.treat_truncated_as_done: - terminated = float(self.dones[step]) - truncated = 0.0 - else: - terminated = float(self.terminated[step]) - truncated = float(self.truncated[step]) next_non_terminal = 1.0 - terminated trunc_mask = 1.0 - truncated delta = ( diff --git a/tests/test_gae_episode_boundary.py b/tests/test_gae_episode_boundary.py new file mode 100644 index 0000000..36476d1 --- /dev/null +++ b/tests/test_gae_episode_boundary.py @@ -0,0 +1,151 @@ +"""GAE has to end an episode at the step that ended it, not one step later. + +Both servers (`websocket_agent_server`, `ray_agent_server`) call `feedback` +for step t with + + terminated, truncated = step t-1's outcome + next_terminated, next_truncated = step t's outcome + prev_node = the node returned for step t-1 + +and never reset `prev_node` at an episode boundary, so the first frame of an +episode is linked to the last frame of the one before it. A frame carrying +`terminated` therefore *begins* an episode; whether its own transition ended +one is in `next_*`. + +`GAEBuffer` read the wrong pair for linked frames. Until 48042e5 it cut the +recursion with the successor's `dones[next_idx]`, which under this +convention is step t's own outcome. 48042e5 changed that to the frame's own +`dones[step]` - step t-1's - so since then every episode has been cut one +step late: the step that fell over in Hopper bootstrapped from the value of +the next episode's first state, and each episode's first step was cut off +from everything after it. FPO, DPPO and anything else on `GAEBuffer` all +computed advantages this way. + +These feed the buffer exactly as the servers do and check the advantages +against ones worked out by hand. +""" + +from __future__ import annotations + +import numpy as np +import pytest +import torch + +from plugrl_server.buffer.rollout_buffer import GAEBuffer + + +def _train_state(value: float = 0.0) -> dict: + return dict( + obs=np.zeros((1, 2), np.float32), + action=np.zeros((1, 1), np.float32), + logprob=np.zeros((1,), np.float32), + value=np.full((1,), value, np.float32), + ) + + +class _ConstantValue: + def __init__(self, value: float) -> None: + self.value = value + + def get_value(self, obs: dict) -> torch.Tensor: + return torch.full((np.asarray(obs["v"]).shape[0],), self.value) + + +def _as_the_server_feeds_it( + outcomes: list[str], + *, + treat_truncated_as_done: bool, + bootstrap: float = 0.0, +) -> GAEBuffer: + """One env, reward 1 at every step, every stored value 0, gamma = lambda = 1. + + `outcomes[t]` is how step t ended: "" (it did not), "terminated" or + "truncated". The client resets after either, and the server keeps the + chain going across the reset. + """ + buffer = GAEBuffer( + len(outcomes), + _train_state(), + gamma=1.0, + gae_lambda=1.0, + treat_truncated_as_done=treat_truncated_as_done, + ) + prev_node: tuple = (-1, "") + previous = "" + for outcome in outcomes: + prev_node = buffer.add_frame( + prev_node=prev_node, + train_state=_train_state(), + reward=1.0, + terminated=previous == "terminated", + truncated=previous == "truncated", + last_value=None, + next_terminated=outcome == "terminated", + next_truncated=outcome == "truncated", + ) + buffer.add_next_obs_value_request( + obs={"v": np.zeros((1, 2), np.float32)}, end_node=prev_node + ) + previous = outcome + buffer.compute_advantages_and_returns( + policy=_ConstantValue(bootstrap), batch_size=4 + ) + return buffer + + +THREE_EPISODES = ["", "", "terminated"] * 3 + + +@pytest.mark.parametrize("treat_truncated_as_done", [True, False]) +def test_each_terminated_episode_is_its_own(treat_truncated_as_done): + """Three steps of reward 1 and a fall: 3, 2, 1 in every episode. + + Before the fix: 4, 3, 2, 1, 3, 2, 1, 2, 1 - the first fall carried on + into the second episode, and the last episode lost its first step. + """ + buffer = _as_the_server_feeds_it( + THREE_EPISODES, treat_truncated_as_done=treat_truncated_as_done + ) + + np.testing.assert_array_equal(buffer.advantages[:9], [3, 2, 1] * 3) + + +def test_a_truncated_episode_ends_like_a_terminated_one_when_told_to(): + buffer = _as_the_server_feeds_it( + ["", "", "truncated"] * 3, treat_truncated_as_done=True + ) + + np.testing.assert_array_equal(buffer.advantages[:9], [3, 2, 1] * 3) + + +def test_otherwise_the_truncated_step_is_left_out_and_nothing_crosses_it(): + """With treat_truncated_as_done off, the step that was truncated has its + delta masked to zero - GAEBuffer's treatment - and the episode before it + still does not reach into the next.""" + buffer = _as_the_server_feeds_it( + ["", "", "truncated"] * 3, treat_truncated_as_done=False + ) + + np.testing.assert_array_equal(buffer.advantages[:9], [2, 1, 0] * 3) + + +def test_a_chain_that_runs_off_the_buffer_bootstraps_even_if_it_just_began(): + """The buffer fills on the first step of a new episode. That step did not + end anything, so it bootstraps from the value of what came after it. + + Before the fix a frame carrying `terminated` at the end of a chain was + taken as terminal itself, and its bootstrap value was dropped. + """ + buffer = _as_the_server_feeds_it( + ["", "", "terminated", ""], treat_truncated_as_done=True, bootstrap=5.0 + ) + + np.testing.assert_array_equal(buffer.advantages[:4], [3, 2, 1, 6]) + + +def test_a_chain_that_runs_off_the_buffer_mid_episode_bootstraps(): + buffer = _as_the_server_feeds_it( + ["", "", "terminated", "", ""], treat_truncated_as_done=False, bootstrap=5.0 + ) + + np.testing.assert_array_equal(buffer.advantages[:5], [3, 2, 1, 7, 6]) From cfceaae6270ae15e04621f3256190c71b9ebe762 Mon Sep 17 00:00:00 2001 From: tactino <18781106300@163.com> Date: Sun, 27 Sep 2026 17:56:07 -0400 Subject: [PATCH 4/8] test: the PPO bandit's measured numbers, with the GAE fix merged in --- tests/test_gaussian_ppo.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/tests/test_gaussian_ppo.py b/tests/test_gaussian_ppo.py index 416c470..999739f 100644 --- a/tests/test_gaussian_ppo.py +++ b/tests/test_gaussian_ppo.py @@ -414,8 +414,9 @@ def test_ppo_learns_a_bandit_with_a_known_answer_and_keeps_it(seed: int): A Gaussian at mean zero with unit deviation scores about -0.78 here. It has to end at a quarter of that, and no worse than twice its best - the - bar four of FPO's ten cases fail. Measured, last over best: 1.00, 1.12, - 1.00, 1.00, 1.02, ending between -0.115 and -0.145. + bar `test_fpo_holds_what_it_learns` sets FPO, and which some of FPO's + cases fail. Measured, last over best: 1.00, 1.14, 1.00, 1.00, 1.03, + ending between -0.120 and -0.160. The rate is 3e-3, not CleanRL's 3e-4, because the floor on this reward is set by how far the deviation narrows, and at 3e-4, annealed over thirty From 2f70733f39d5355e5b3ca9ce016431326031f5c1 Mon Sep 17 00:00:00 2001 From: tactino <18781106300@163.com> Date: Sun, 27 Sep 2026 18:05:35 -0400 Subject: [PATCH 5/8] exp: pre-register E38 - gaussian-policy under ppo, CleanRL's defaults, on HalfCheetah, Hopper and Walker2d --- experiments/e38-gaussian-ppo/PROTOCOL.md | 128 ++++++++++++++++++ .../e38-gaussian-ppo/results/pilot.txt | 28 ++++ 2 files changed, 156 insertions(+) create mode 100644 experiments/e38-gaussian-ppo/PROTOCOL.md create mode 100644 experiments/e38-gaussian-ppo/results/pilot.txt diff --git a/experiments/e38-gaussian-ppo/PROTOCOL.md b/experiments/e38-gaussian-ppo/PROTOCOL.md new file mode 100644 index 0000000..46556cd --- /dev/null +++ b/experiments/e38-gaussian-ppo/PROTOCOL.md @@ -0,0 +1,128 @@ +# E38 measurement protocol (pre-registered) + +**Written 2026-09-27, before any E38 run. The three-iteration pilot and the +GAE diagnostic below came first and are recorded in `results/pilot.txt`.** + +This file must not be edited after the first registered data point. Anything +learned afterwards goes in `AMENDMENT.md`, dated. + +--- + +## The question + +The coverage figure's rows are expressive policies with the algorithms built +for them. It has no row for the baseline they are measured against. #85 adds +one: `gaussian-policy` under `ppo`, written to CleanRL's +`ppo_continuous_action.py` with every default. + +**Run as CleanRL runs it, does it learn HalfCheetah, Hopper and Walker2d +through PlugRL?** + +### What this cannot settle + +* Whether PlugRL reproduces CleanRL's returns. The tasks are -v5, not -v4, + and three things happen differently (declared below). The returns are + reported beside CleanRL's, not tested against them. +* Anything about the other rows. + +--- + +## Declared in advance: what was already known + +1. **CleanRL's own results** for this file, from its documentation (MuJoCo + -v4; the benchmark command passes no `--total-timesteps`, so its default + of 1,000,000; three seeds): HalfCheetah 1442.64 +/- 46.03, Hopper + 2382.86 +/- 271.74, Walker2d 2287.95 +/- 571.78. +2. **The pair on a bandit** (#85's tests, with #86 merged): from about + -0.78 to between -0.120 and -0.160 on five seeds, final over best + 1.00-1.14. +3. **The pilot**: three iterations on each task, seed 0, at ab6e32f. Every + run ended with exit 0 and no traceback, at 528-584 environment steps a + second including learning - about 30 minutes for 488 iterations alone. +4. **GAE ended every episode one step late** on every branch since 48042e5 + (#86); E38 runs with the fix. A diagnostic of this pair on Hopper-v5, 100 + iterations (annealed over 100), seeds 0-2, at ab6e32f and 274929c: + iterations 41-50 at 267 / 589 / 367 before the fix and 674 / 673 / 454 + after; iterations 91-100 at 432 / 1512 / 910 before and 1005 / 868 / 949 + after. +5. **Runs are not reproducible bit for bit** after the first update (E20). + +### Where this differs from CleanRL, declared + +* Observation statistics are updated after each iteration's learning, from + that iteration's observations, not at every step; the first iteration runs + unnormalised. CleanRL's NormalizeObservation updates at every step from + the first. +* Rewards are scaled by the return's running deviation once per iteration, + over the iteration's returns, not step by step as NormalizeReward does. +* The environments are -v5. + +--- + +## Design + +| cell | task | seeds | iterations | +| --- | --- | --- | --- | +| `ppo-cheetah` | HalfCheetah-v5 | 0, 1, 2 | 488 | +| `ppo-hopper` | Hopper-v5 | 0, 1, 2 | 488 | +| `ppo-walker` | Walker2d-v5 | 0, 1, 2 | 488 | + +`gaussian-policy default ppo default` with the task's dimensions and nothing +else: rollouts of 2,048 steps, minibatches of 64, ten epochs, clip 0.2, +learning rate 3e-4 annealed to zero over the 488 iterations (999,424 steps, +CleanRL's 1,000,000 // 2,048). One environment per seed, the client +replanning every step; checkpoints every 20 iterations. E30's +`run_cell.sh`, unchanged, through `run.sh`. + +Machine `guangzhao`, CPU, one thread per process, the three cells at once +beside E37; code this branch (#85 and #86 merged, plus this directory), +client plugrl-env-client at 931ab56. + +--- + +## Checks + +* **V1 - the configuration took**: every iteration's logged + `models/learning_rate` equal to (1 - (i-1)/488) x 3e-4 for iteration i, + to one part in a million. A cell that fails is not read. + +--- + +## The status rule + +As the coverage figure has it, on the mean return of iterations 479-488, on +at least **2 of 3** seeds: Hopper and Walker2d at least **500**; HalfCheetah +at least **+200** over its first iteration. A cell that passes **learns**; +one that does not **did not learn in 999,424 steps**. + +--- + +## Predictions, and what falsifies each + +**P1 - every cell runs end to end** (488 iterations logged, at least 24 +checkpoints, no traceback, client exit 0, on every seed). + +**P2 - it learns Hopper.** **P3 - it learns Walker2d.** **P4 - it learns +HalfCheetah.** + +> Grounds for all three: known item 1 - CleanRL's returns with this +> configuration are four to five times the bars on Hopper and Walker2d, and +> 1443 on HalfCheetah, where an untrained policy starts near -300 (E6). +> For Hopper also known item 4: with the fix, 868 to 1005 by iteration 100. +> Each is falsified if fewer than two seeds clear its bar. + +**Reported, not predicted:** the mean return over iterations 91-100, 241-250 +and 479-488 for every seed, beside CleanRL's; wall clock. + +--- + +## Declared deviations allowed in advance + +1. One restart of any run that dies for a reason outside the experiment, + recorded in `AMENDMENT.md`. + +--- + +## Reading order + +P1, V1, the status rule, P2, P3, P4, then the reported figures. diff --git a/experiments/e38-gaussian-ppo/results/pilot.txt b/experiments/e38-gaussian-ppo/results/pilot.txt new file mode 100644 index 0000000..98a85bd --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/pilot.txt @@ -0,0 +1,28 @@ +E38 pilot, 2026-09-27, guangzhao, code ab6e32f (#85 plus E38's scripts, before +the GAE fix), client 931ab56. + + ITERS=3 SEEDS=0 bash experiments/e38-gaussian-ppo/run.sh + + start 17:38:10, end 17:39:04; failed seeds: 0 in every cell. + + cell seed 0 rc traceback env steps client effective_fps + ppo-cheetah 0 none 6,145 528.19 + ppo-hopper 0 none 6,145 563.74 + ppo-walker 0 none 6,145 584.20 + +Each run took about 12 s for 3 iterations of 2,048 steps including learning; +488 iterations alone would take about 30 minutes. Returns were not read. + +GAE diagnostic, 2026-09-27 17:55-18:03, guangzhao. gaussian-policy under ppo +on Hopper-v5, 100 iterations of 2,048 (the schedule annealed over 100), +seeds 0-2, through E30's run_cell.sh, once in a checkout before the GAE +episode-boundary fix (ab6e32f) and once after it (274929c, #85 with #86 +merged). Every run rc=0, no traceback, 100 iterations logged. + + mean return seed first 41-50 91-100 length 91-100 + before 0 12.1 266.5 431.5 150.4 + 1 15.2 588.7 1512.1 486.6 + 2 14.2 366.5 909.9 295.1 + after 0 12.1 674.3 1005.1 312.7 + 1 15.2 673.1 868.0 271.4 + 2 14.2 453.6 949.3 296.0 From 124c04506e18141413fbea8a640e40437b0d9f34 Mon Sep 17 00:00:00 2001 From: tactino <18781106300@163.com> Date: Sun, 27 Sep 2026 18:10:41 -0400 Subject: [PATCH 6/8] gaussian-policy: --policy.deterministic acts with the mean, for evaluation; ppo refuses it --- src/plugrl_server/algorithm/ppo/ppo.py | 5 +++++ .../policy/gaussian/gaussian_policy.py | 5 ++++- tests/test_gaussian_ppo.py | 19 +++++++++++++++++++ 3 files changed, 28 insertions(+), 1 deletion(-) diff --git a/src/plugrl_server/algorithm/ppo/ppo.py b/src/plugrl_server/algorithm/ppo/ppo.py index 0bc1432..e67f688 100644 --- a/src/plugrl_server/algorithm/ppo/ppo.py +++ b/src/plugrl_server/algorithm/ppo/ppo.py @@ -46,6 +46,11 @@ def __init__(self, config: PPOAlgoConfig, policy: GaussianPolicy): f"ppo needs a policy with evaluate_actions, such as " f"gaussian-policy; {type(policy).__name__} has none" ) + if getattr(policy.config, "deterministic", False): + raise ValueError( + "ppo needs a policy that samples: --policy.deterministic is for " + "evaluation, and a mean action has no density to form a ratio from" + ) self.rollout_buffer = PPOBuffer( buffer_size=config.buffer_size, example_train_state=self.example_train_state(batch_size=1), diff --git a/src/plugrl_server/policy/gaussian/gaussian_policy.py b/src/plugrl_server/policy/gaussian/gaussian_policy.py index 32d95d3..a964438 100644 --- a/src/plugrl_server/policy/gaussian/gaussian_policy.py +++ b/src/plugrl_server/policy/gaussian/gaussian_policy.py @@ -78,6 +78,9 @@ class GaussianPolicyConfig(BaseTorchPolicyConfig): freeze_obs_stats: bool = False # HalfCheetah, Hopper and Walker2d all act in [-1, 1]. action_clip: float = 1.0 + # Act with the mean instead of a sample: for evaluation, which acts + # without training's sampling noise. `ppo` refuses it. + deterministic: bool = False @register_policy(UID) @@ -152,7 +155,7 @@ def get_action_and_runtime_state( x = self.extract_model_obs_tensor(_obs) z = self.normalize_obs(x) dist = self._distribution(z) - sample = dist.sample() + sample = dist.mean if self.config.deterministic else dist.sample() state = GaussianRuntimeState( obs=x, action=sample, diff --git a/tests/test_gaussian_ppo.py b/tests/test_gaussian_ppo.py index 999739f..9bb4fa4 100644 --- a/tests/test_gaussian_ppo.py +++ b/tests/test_gaussian_ppo.py @@ -201,6 +201,20 @@ def test_observations_are_normalised_then_clipped_at_ten(self): far = policy.normalize_obs(torch.full((1, OBS_DIM), 1e6)) assert far.max().item() == 10.0 + def test_deterministic_acts_with_the_mean(self): + """For evaluation, which acts without training's sampling noise.""" + policy = _policy(deterministic=True) + obs = _obs(8, np.random.default_rng(0)) + with torch.inference_mode(): + first, _ = policy.get_action_and_runtime_state(obs) + second, _ = policy.get_action_and_runtime_state(obs) + mean = policy.actor_mean( + policy.normalize_obs(policy.extract_model_obs_tensor(obs)) + ) + + np.testing.assert_array_equal(first, second) + np.testing.assert_allclose(first[:, 0], mean.clamp(-1, 1).numpy()) + def test_frozen_statistics_do_not_move(self): policy = _policy(freeze_obs_stats=True) @@ -354,6 +368,11 @@ def test_without_annealing_the_rate_stays_put(self): np.testing.assert_allclose(seen, [1e-3, 1e-3]) + def test_it_refuses_a_policy_that_does_not_sample(self): + """The ratio is of the density of what was done; a mean has none.""" + with pytest.raises(ValueError, match="deterministic"): + PPOAlgorithm(PPOAlgoConfig(), _policy(deterministic=True)) + def test_one_adam_over_every_parameter_with_cleanrl_epsilon(self): algo = _algo() From afa9537ea296b57fcd05a8e753fe0f313ce147f9 Mon Sep 17 00:00:00 2001 From: tactino <18781106300@163.com> Date: Sun, 27 Sep 2026 18:38:13 -0400 Subject: [PATCH 7/8] exp: E38 results - gaussian-policy under ppo learns HalfCheetah, Hopper and Walker2d, 3 of 3 each, at CleanRL's returns --- experiments/e38-gaussian-ppo/FINDINGS.md | 83 +++++++++++++++++++ .../e38-gaussian-ppo/results/ppo-cheetah.out | 11 +++ .../results/ppo-cheetah/client-seed0.log | 65 +++++++++++++++ .../results/ppo-cheetah/client-seed1.log | 66 +++++++++++++++ .../results/ppo-cheetah/client-seed2.log | 66 +++++++++++++++ .../results/ppo-cheetah/server-seed0.log | 43 ++++++++++ .../results/ppo-cheetah/server-seed1.log | 43 ++++++++++ .../results/ppo-cheetah/server-seed2.log | 43 ++++++++++ .../e38-gaussian-ppo/results/ppo-hopper.out | 11 +++ .../results/ppo-hopper/client-seed0.log | 65 +++++++++++++++ .../results/ppo-hopper/client-seed1.log | 65 +++++++++++++++ .../results/ppo-hopper/client-seed2.log | 64 ++++++++++++++ .../results/ppo-hopper/server-seed0.log | 43 ++++++++++ .../results/ppo-hopper/server-seed1.log | 43 ++++++++++ .../results/ppo-hopper/server-seed2.log | 43 ++++++++++ .../e38-gaussian-ppo/results/ppo-walker.out | 11 +++ .../results/ppo-walker/client-seed0.log | 66 +++++++++++++++ .../results/ppo-walker/client-seed1.log | 65 +++++++++++++++ .../results/ppo-walker/client-seed2.log | 66 +++++++++++++++ .../results/ppo-walker/server-seed0.log | 43 ++++++++++ .../results/ppo-walker/server-seed1.log | 43 ++++++++++ .../results/ppo-walker/server-seed2.log | 43 ++++++++++ experiments/e38-gaussian-ppo/results/run.out | 9 ++ .../e38-gaussian-ppo/results/verdicts.txt | 50 +++++++++++ experiments/e38-gaussian-ppo/summary.tsv | 10 +++ 25 files changed, 1160 insertions(+) create mode 100644 experiments/e38-gaussian-ppo/FINDINGS.md create mode 100644 experiments/e38-gaussian-ppo/results/ppo-cheetah.out create mode 100644 experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed0.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed1.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed2.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed0.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed1.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed2.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-hopper.out create mode 100644 experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed0.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed1.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed2.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed0.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed1.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed2.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-walker.out create mode 100644 experiments/e38-gaussian-ppo/results/ppo-walker/client-seed0.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-walker/client-seed1.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-walker/client-seed2.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-walker/server-seed0.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-walker/server-seed1.log create mode 100644 experiments/e38-gaussian-ppo/results/ppo-walker/server-seed2.log create mode 100644 experiments/e38-gaussian-ppo/results/run.out create mode 100644 experiments/e38-gaussian-ppo/results/verdicts.txt create mode 100644 experiments/e38-gaussian-ppo/summary.tsv diff --git a/experiments/e38-gaussian-ppo/FINDINGS.md b/experiments/e38-gaussian-ppo/FINDINGS.md new file mode 100644 index 0000000..de9b722 --- /dev/null +++ b/experiments/e38-gaussian-ppo/FINDINGS.md @@ -0,0 +1,83 @@ +# E38: a Gaussian policy with PPO, run as CleanRL runs it, learns all three MuJoCo tasks through PlugRL - at CleanRL's returns + +2026-09-27 · Linux workstation (`guangzhao`), CPU only · three cells, three +seeds, 488 iterations of 2,048 steps · protocol: [`PROTOCOL.md`](PROTOCOL.md) +(`2f70733`, after the pilot and the GAE diagnostic in `results/pilot.txt`, +before the registered run) + +--- + +## The result + +`gaussian-policy` under `ppo` (#85) is CleanRL's `ppo_continuous_action.py` +with every default, on code with the GAE episode-boundary fix (#86). The +coverage figure's rule, read on iterations 479-488: + +| cell | seed | first iteration | iterations 479-488 | CleanRL (-v4) | +| --- | --- | --- | --- | --- | +| HalfCheetah | 0 | -335.2 | 1512.9 (**+1848**) | 1442.64 +/- 46.03 | +| | 1 | -384.6 | 1442.5 (**+1827**) | | +| | 2 | -393.9 | 1557.8 (**+1952**) | | +| Hopper | 0 | 12.1 | **2298.7** | 2382.86 +/- 271.74 | +| | 1 | 15.2 | **2207.4** | | +| | 2 | 14.2 | **2181.3** | | +| Walker2d | 0 | -0.7 | **2957.2** | 2287.95 +/- 571.78 | +| | 1 | -1.8 | **3260.0** | | +| | 2 | -0.6 | **3154.2** | | + +* **P1 holds**: nine runs, each with 488 iterations logged, 24 checkpoints, + no traceback and a client that exited 0. The three cells ran at once + beside E37 and took 28 minutes. +* **V1 passes**: every iteration of every run learned at CleanRL's annealed + rate, from 3e-4 down to 6.15e-7 (3e-4 / 488) at the last. +* **P2, P3, P4 hold**: it learns Hopper, Walker2d and HalfCheetah, each on + 3 of 3 seeds. The smallest figure is 4.4 times its bar. + +The new row, `gaussian-policy` · PPO, **learns** on all three tasks. + +--- + +## Beside CleanRL + +Mean return over three windows: + +| cell | seed | 91-100 | 241-250 | 479-488 | +| --- | --- | --- | --- | --- | +| HalfCheetah | 0 | 744.4 | 1322.5 | 1512.9 | +| | 1 | 999.0 | 1348.1 | 1442.5 | +| | 2 | 925.5 | 1428.5 | 1557.8 | +| Hopper | 0 | 1623.2 | 2242.8 | 2298.7 | +| | 1 | 1297.1 | 2486.9 | 2207.4 | +| | 2 | 920.3 | 2440.6 | 2181.3 | +| Walker2d | 0 | 514.9 | 2747.0 | 2957.2 | +| | 1 | 476.6 | 2037.8 | 3260.0 | +| | 2 | 652.6 | 3344.2 | 3154.2 | + +HalfCheetah ends at CleanRL's number or a little above it, and Hopper +within its spread. Walker2d ends above it, by more than CleanRL's own +spread; E38 does not explain that. The protocol registered these as reported +figures, not a test. The tasks are -v5, not -v4, and three things are done +differently (declared there). What they do show is that the one pair here +whose behaviour is known well outside this project comes out of PlugRL +about where it comes out of CleanRL. Here the server has no simulator +installed, and the environment runs in a separate client with only the +protocol between them. + +Episodes grew from 18-20 steps to 586-636 on Hopper and 686-792 on Walker2d. +HalfCheetah's are always 1,000. + +--- + +## What E38 does not show + +* **That the platform was right before #86.** E38 ran with the GAE fix. The + diagnostic in `results/pilot.txt` ran this pair on Hopper for 100 + iterations on each side of the fix. By iteration 100 the two were + indistinguishable on three seeds (951 against 941 on average), so the + defect cost this task little. That diagnostic was not registered. +* **The square column.** `gaussian-policy` was not run on robomimic square. + From random weights, square's sparse reward gives it nothing to learn from + (E27), and no behaviour-cloned Gaussian start was built. + +Checkpoints and tensorboards stay on `guangzhao`. The logs, `verdicts.txt` +and `summary.tsv` are here. diff --git a/experiments/e38-gaussian-ppo/results/ppo-cheetah.out b/experiments/e38-gaussian-ppo/results/ppo-cheetah.out new file mode 100644 index 0000000..2d98a2e --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-cheetah.out @@ -0,0 +1,11 @@ +cell: ppo-cheetah = gaussian-policy/default x ppo/default x HalfCheetah-v5 +iters: 488 x 2048 batch: 64 replan: 1 seeds: 0 1 2 +extra: algo none; policy --policy.obs-dim 17 --policy.action-dim 6 +out: /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah +start 2026-09-27 18:04:58 +seed 0 finished rc=0 at 18:32:53 +seed 2 finished rc=0 at 18:33:15 +seed 1 finished rc=0 at 18:33:16 +end 2026-09-27 18:33:16 +failed seeds: 0 +CELL_DONE ppo-cheetah diff --git a/experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed0.log b/experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed0.log new file mode 100644 index 0000000..9cd3a1d --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed0.log @@ -0,0 +1,65 @@ +2026-09-27 18:05:02.015 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.classic.classic_env: pygame is not installed. Please install it with pip install "plugrl-env-client[classic]". +2026-09-27 18:05:02.016 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.atari.atari_env: Atari is not installed. Please install it with the 'atari' extra, e.g. 'pip install plugrl-env-client[atari]' +2026-09-27 18:05:02.016 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.d4rl.d4rl_env: d4rl is not installed. Please install it with pip install "plugrl-env-client[d4rl]". +2026-09-27 18:05:02.016 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.libero.libero_env: libero is not installed. Please install it with pip install "plugrl-env-client[libero]". +2026-09-27 18:05:02.017 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.robomimic.robomimic_env: Robomimic is not installed. Please install it with the 'robomimic' extra, e.g. 'pip install plugrl-env-client[robomimic]' +2026-09-27 18:05:02.059 | INFO | __main__:main:83 - Starting env client exp_name=mujoco-v1-nenv1-20260927-180502-0e66a5ba output_dir=runs/mujoco-v1-nenv1-20260927-180502-0e66a5ba recorder={'episode_freq': 0, 'thread0_only': True, 'record_video': False, 'video_fps': 30.0, 'record_full_rollout': False, 'record_obs_stats': True, 'record_episode_metrics': True, 'metric_window': 100} +2026-09-27 18:05:02.099 | INFO | plugrl_env_client.agent.websocket_env_client_agent:_wait_for_server:67 - Waiting for server at ws://127.0.0.1:9940... +2026-09-27 18:05:02.101 | INFO | plugrl_env_client.runner.run:report_server_metadata:56 - Server metadata: {'protocol_version': 1, 'server': 'plugrl-server', 'server_version': '0.1.0', 'algorithm': 'PPOAlgorithm', 'policy': 'GaussianPolicy', 'action_dim': 6, 'action_horizon': 1} +2026-09-27 18:05:32.103 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=16008 infer_calls=16008 feedback_calls=16008 infer_wait=26.245s infer_obs_pack=0.255s env_step=1.539s feedback_total=1.201s feedback_obs_pack=0.312s feedback_info_pack=0.053s effective_fps=547.47 +2026-09-27 18:06:02.103 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=34942 infer_calls=34942 feedback_calls=34942 infer_wait=51.377s infer_obs_pack=0.573s env_step=3.565s feedback_total=2.772s feedback_obs_pack=0.729s feedback_info_pack=0.130s effective_fps=599.48 +2026-09-27 18:06:32.106 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=54989 infer_calls=54989 feedback_calls=54989 infer_wait=76.111s infer_obs_pack=0.915s env_step=5.778s feedback_total=4.469s feedback_obs_pack=1.171s feedback_info_pack=0.217s effective_fps=630.07 +2026-09-27 18:07:02.108 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=74534 infer_calls=74534 feedback_calls=74534 infer_wait=101.021s infer_obs_pack=1.248s env_step=7.904s feedback_total=6.120s feedback_obs_pack=1.606s feedback_info_pack=0.297s effective_fps=640.92 +2026-09-27 18:07:32.109 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=92428 infer_calls=92428 feedback_calls=92428 infer_wait=126.416s infer_obs_pack=1.544s env_step=9.831s feedback_total=7.588s feedback_obs_pack=2.006s feedback_info_pack=0.369s effective_fps=635.77 +2026-09-27 18:08:02.109 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=110605 infer_calls=110605 feedback_calls=110605 infer_wait=151.768s infer_obs_pack=1.850s env_step=11.767s feedback_total=9.077s feedback_obs_pack=2.401s feedback_info_pack=0.439s effective_fps=633.97 +2026-09-27 18:08:32.109 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=128773 infer_calls=128773 feedback_calls=128773 infer_wait=177.107s infer_obs_pack=2.153s env_step=13.708s feedback_total=10.576s feedback_obs_pack=2.805s feedback_info_pack=0.514s effective_fps=632.65 +2026-09-27 18:09:02.111 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=145189 infer_calls=145189 feedback_calls=145189 infer_wait=203.122s infer_obs_pack=2.419s env_step=15.376s feedback_total=11.842s feedback_obs_pack=3.138s feedback_info_pack=0.574s effective_fps=623.78 +2026-09-27 18:09:32.112 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=160552 infer_calls=160552 feedback_calls=160552 infer_wait=229.557s infer_obs_pack=2.651s env_step=16.860s feedback_total=12.978s feedback_obs_pack=3.442s feedback_info_pack=0.630s effective_fps=612.69 +2026-09-27 18:10:02.113 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=176123 infer_calls=176123 feedback_calls=176123 infer_wait=255.907s infer_obs_pack=2.894s env_step=18.375s feedback_total=14.140s feedback_obs_pack=3.746s feedback_info_pack=0.694s effective_fps=604.58 +2026-09-27 18:10:32.113 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=191607 infer_calls=191607 feedback_calls=191607 infer_wait=282.267s infer_obs_pack=3.136s env_step=19.895s feedback_total=15.281s feedback_obs_pack=4.058s feedback_info_pack=0.758s effective_fps=597.69 +2026-09-27 18:11:02.115 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=207077 infer_calls=207077 feedback_calls=207077 infer_wait=308.603s infer_obs_pack=3.381s env_step=21.417s feedback_total=16.448s feedback_obs_pack=4.371s feedback_info_pack=0.823s effective_fps=591.91 +2026-09-27 18:11:32.117 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=222777 infer_calls=222777 feedback_calls=222777 infer_wait=334.918s infer_obs_pack=3.627s env_step=22.944s feedback_total=17.621s feedback_obs_pack=4.685s feedback_info_pack=0.887s effective_fps=587.63 +2026-09-27 18:12:02.118 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=238305 infer_calls=238305 feedback_calls=238305 infer_wait=361.277s infer_obs_pack=3.872s env_step=24.441s feedback_total=18.797s feedback_obs_pack=5.000s feedback_info_pack=0.950s effective_fps=583.53 +2026-09-27 18:12:32.119 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=253926 infer_calls=253926 feedback_calls=253926 infer_wait=387.615s infer_obs_pack=4.116s env_step=25.953s feedback_total=19.970s feedback_obs_pack=5.302s feedback_info_pack=1.014s effective_fps=580.20 +2026-09-27 18:13:02.119 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=271160 infer_calls=271160 feedback_calls=271160 infer_wait=413.339s infer_obs_pack=4.397s env_step=27.715s feedback_total=21.352s feedback_obs_pack=5.667s feedback_info_pack=1.091s effective_fps=580.89 +2026-09-27 18:13:32.120 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=289277 infer_calls=289277 feedback_calls=289277 infer_wait=438.649s infer_obs_pack=4.704s env_step=29.680s feedback_total=22.850s feedback_obs_pack=6.061s feedback_info_pack=1.171s effective_fps=583.36 +2026-09-27 18:14:02.121 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=307323 infer_calls=307323 feedback_calls=307323 infer_wait=463.998s infer_obs_pack=5.004s env_step=31.634s feedback_total=24.325s feedback_obs_pack=6.455s feedback_info_pack=1.254s effective_fps=585.42 +2026-09-27 18:14:32.121 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=325497 infer_calls=325497 feedback_calls=325497 infer_wait=489.340s infer_obs_pack=5.309s env_step=33.588s feedback_total=25.814s feedback_obs_pack=6.856s feedback_info_pack=1.338s effective_fps=587.49 +2026-09-27 18:15:02.121 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=343656 infer_calls=343656 feedback_calls=343656 infer_wait=514.665s infer_obs_pack=5.614s env_step=35.540s feedback_total=27.313s feedback_obs_pack=7.261s feedback_info_pack=1.423s effective_fps=589.33 +2026-09-27 18:15:32.121 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=363158 infer_calls=363158 feedback_calls=363158 infer_wait=539.550s infer_obs_pack=5.946s env_step=37.676s feedback_total=28.966s feedback_obs_pack=7.702s feedback_info_pack=1.514s effective_fps=593.26 +2026-09-27 18:16:02.391 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=382978 infer_calls=382978 feedback_calls=382978 infer_wait=564.639s infer_obs_pack=6.285s env_step=39.831s feedback_total=30.648s feedback_obs_pack=8.143s feedback_info_pack=1.605s effective_fps=597.09 +2026-09-27 18:16:32.393 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=403027 infer_calls=403027 feedback_calls=403027 infer_wait=589.376s infer_obs_pack=6.633s env_step=42.012s feedback_total=32.374s feedback_obs_pack=8.586s feedback_info_pack=1.702s effective_fps=601.18 +2026-09-27 18:17:02.394 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=422262 infer_calls=422262 feedback_calls=422262 infer_wait=614.215s infer_obs_pack=6.968s env_step=44.155s feedback_total=34.050s feedback_obs_pack=9.025s feedback_info_pack=1.798s effective_fps=603.76 +2026-09-27 18:17:32.394 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=442268 infer_calls=442268 feedback_calls=442268 infer_wait=638.755s infer_obs_pack=7.320s env_step=46.414s feedback_total=35.844s feedback_obs_pack=9.486s feedback_info_pack=1.899s effective_fps=607.23 +2026-09-27 18:18:02.395 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=462316 infer_calls=462316 feedback_calls=462316 infer_wait=663.488s infer_obs_pack=7.658s env_step=48.619s feedback_total=37.551s feedback_obs_pack=9.932s feedback_info_pack=1.994s effective_fps=610.47 +2026-09-27 18:18:32.397 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=481496 infer_calls=481496 feedback_calls=481496 infer_wait=688.389s infer_obs_pack=7.995s env_step=50.759s feedback_total=39.194s feedback_obs_pack=10.369s feedback_info_pack=2.090s effective_fps=612.33 +2026-09-27 18:19:02.398 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=500894 infer_calls=500894 feedback_calls=500894 infer_wait=713.236s infer_obs_pack=8.340s env_step=52.907s feedback_total=40.854s feedback_obs_pack=10.811s feedback_info_pack=2.182s effective_fps=614.34 +2026-09-27 18:19:32.519 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=520194 infer_calls=520194 feedback_calls=520194 infer_wait=738.259s infer_obs_pack=8.678s env_step=55.032s feedback_total=42.492s feedback_obs_pack=11.245s feedback_info_pack=2.275s effective_fps=616.01 +2026-09-27 18:20:02.521 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=539552 infer_calls=539552 feedback_calls=539552 infer_wait=763.126s infer_obs_pack=9.014s env_step=57.167s feedback_total=44.157s feedback_obs_pack=11.681s feedback_info_pack=2.369s effective_fps=617.72 +2026-09-27 18:20:32.522 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=558141 infer_calls=558141 feedback_calls=558141 infer_wait=788.257s infer_obs_pack=9.338s env_step=59.197s feedback_total=45.728s feedback_obs_pack=12.096s feedback_info_pack=2.460s effective_fps=618.43 +2026-09-27 18:21:02.523 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=576371 infer_calls=576371 feedback_calls=576371 infer_wait=813.569s infer_obs_pack=9.643s env_step=61.141s feedback_total=47.255s feedback_obs_pack=12.498s feedback_info_pack=2.545s effective_fps=618.68 +2026-09-27 18:21:32.524 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=594525 infer_calls=594525 feedback_calls=594525 infer_wait=838.933s infer_obs_pack=9.948s env_step=63.065s feedback_total=48.747s feedback_obs_pack=12.899s feedback_info_pack=2.628s effective_fps=618.85 +2026-09-27 18:22:02.524 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=611885 infer_calls=611885 feedback_calls=611885 infer_wait=864.590s infer_obs_pack=10.236s env_step=64.862s feedback_total=50.142s feedback_obs_pack=13.275s feedback_info_pack=2.702s effective_fps=618.17 +2026-09-27 18:22:32.526 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=627407 infer_calls=627407 feedback_calls=627407 infer_wait=890.933s infer_obs_pack=10.480s env_step=66.369s feedback_total=51.310s feedback_obs_pack=13.586s feedback_info_pack=2.765s effective_fps=615.65 +2026-09-27 18:23:02.528 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=643073 infer_calls=643073 feedback_calls=643073 infer_wait=917.257s infer_obs_pack=10.727s env_step=67.904s feedback_total=52.474s feedback_obs_pack=13.896s feedback_info_pack=2.828s effective_fps=613.41 +2026-09-27 18:23:32.529 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=658611 infer_calls=658611 feedback_calls=658611 infer_wait=943.586s infer_obs_pack=10.974s env_step=69.414s feedback_total=53.649s feedback_obs_pack=14.204s feedback_info_pack=2.893s effective_fps=611.17 +2026-09-27 18:24:02.529 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=673997 infer_calls=673997 feedback_calls=673997 infer_wait=969.960s infer_obs_pack=11.217s env_step=70.912s feedback_total=54.803s feedback_obs_pack=14.512s feedback_info_pack=2.954s effective_fps=608.91 +2026-09-27 18:24:32.530 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=689709 infer_calls=689709 feedback_calls=689709 infer_wait=996.244s infer_obs_pack=11.464s env_step=72.454s feedback_total=55.990s feedback_obs_pack=14.832s feedback_info_pack=3.019s effective_fps=607.06 +2026-09-27 18:25:02.530 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=705174 infer_calls=705174 feedback_calls=705174 infer_wait=1022.613s infer_obs_pack=11.711s env_step=73.947s feedback_total=57.154s feedback_obs_pack=15.142s feedback_info_pack=3.081s effective_fps=605.08 +2026-09-27 18:25:32.532 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=720883 infer_calls=720883 feedback_calls=720883 infer_wait=1048.925s infer_obs_pack=11.958s env_step=75.470s feedback_total=58.335s feedback_obs_pack=15.453s feedback_info_pack=3.146s effective_fps=603.41 +2026-09-27 18:26:02.533 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=737966 infer_calls=737966 feedback_calls=737966 infer_wait=1074.671s infer_obs_pack=12.235s env_step=77.243s feedback_total=59.696s feedback_obs_pack=15.816s feedback_info_pack=3.218s effective_fps=602.99 +2026-09-27 18:26:32.533 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=756023 infer_calls=756023 feedback_calls=756023 infer_wait=1099.993s infer_obs_pack=12.538s env_step=79.195s feedback_total=61.201s feedback_obs_pack=16.226s feedback_info_pack=3.301s effective_fps=603.40 +2026-09-27 18:27:02.652 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=774146 infer_calls=774146 feedback_calls=774146 infer_wait=1125.440s infer_obs_pack=12.850s env_step=81.132s feedback_total=62.709s feedback_obs_pack=16.627s feedback_info_pack=3.385s effective_fps=603.80 +2026-09-27 18:27:32.652 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=792450 infer_calls=792450 feedback_calls=792450 infer_wait=1150.695s infer_obs_pack=13.156s env_step=83.116s feedback_total=64.233s feedback_obs_pack=17.030s feedback_info_pack=3.468s effective_fps=604.37 +2026-09-27 18:28:02.653 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=810647 infer_calls=810647 feedback_calls=810647 infer_wait=1176.014s infer_obs_pack=13.471s env_step=85.058s feedback_total=65.741s feedback_obs_pack=17.435s feedback_info_pack=3.552s effective_fps=604.83 +2026-09-27 18:28:32.653 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=830085 infer_calls=830085 feedback_calls=830085 infer_wait=1200.912s infer_obs_pack=13.813s env_step=87.177s feedback_total=67.398s feedback_obs_pack=17.872s feedback_info_pack=3.641s effective_fps=606.21 +2026-09-27 18:29:02.653 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=849768 infer_calls=849768 feedback_calls=849768 infer_wait=1225.661s infer_obs_pack=14.162s env_step=89.364s feedback_total=69.100s feedback_obs_pack=18.318s feedback_info_pack=3.734s effective_fps=607.72 +2026-09-27 18:29:32.653 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=869412 infer_calls=869412 feedback_calls=869412 infer_wait=1250.531s infer_obs_pack=14.495s env_step=91.505s feedback_total=70.770s feedback_obs_pack=18.756s feedback_info_pack=3.827s effective_fps=609.13 +2026-09-27 18:30:02.653 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=888801 infer_calls=888801 feedback_calls=888801 infer_wait=1275.336s infer_obs_pack=14.836s env_step=93.681s feedback_total=72.451s feedback_obs_pack=19.192s feedback_info_pack=3.922s effective_fps=610.31 +2026-09-27 18:30:32.654 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=908443 infer_calls=908443 feedback_calls=908443 infer_wait=1300.016s infer_obs_pack=15.186s env_step=95.908s feedback_total=74.170s feedback_obs_pack=19.640s feedback_info_pack=4.018s effective_fps=611.63 +2026-09-27 18:31:02.655 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=927895 infer_calls=927895 feedback_calls=927895 infer_wait=1324.846s infer_obs_pack=15.530s env_step=98.058s feedback_total=75.847s feedback_obs_pack=20.077s feedback_info_pack=4.112s effective_fps=612.76 +2026-09-27 18:31:32.656 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=947675 infer_calls=947675 feedback_calls=947675 infer_wait=1349.651s infer_obs_pack=15.876s env_step=100.219s feedback_total=77.528s feedback_obs_pack=20.518s feedback_info_pack=4.207s effective_fps=614.07 +2026-09-27 18:32:02.657 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=966921 infer_calls=966921 feedback_calls=966921 infer_wait=1374.481s infer_obs_pack=16.220s env_step=102.364s feedback_total=79.207s feedback_obs_pack=20.964s feedback_info_pack=4.302s effective_fps=614.98 +2026-09-27 18:32:32.657 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=986249 infer_calls=986249 feedback_calls=986249 infer_wait=1399.408s infer_obs_pack=16.554s env_step=104.475s feedback_total=80.864s feedback_obs_pack=21.401s feedback_info_pack=4.393s effective_fps=615.90 +2026-09-27 18:32:53.333 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Final rollout timing summary: env_steps=999425 infer_calls=999425 feedback_calls=999425 infer_wait=1416.263s infer_obs_pack=16.784s env_step=105.923s feedback_total=81.994s feedback_obs_pack=21.707s feedback_info_pack=4.455s effective_fps=616.56 +2026-09-27 18:32:53.334 | INFO | plugrl_env_client.runner.run:run:156 - Server signalled the end of the run; collection stopped after 999/3997696 episodes. diff --git a/experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed1.log b/experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed1.log new file mode 100644 index 0000000..bbda1d7 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed1.log @@ -0,0 +1,66 @@ +2026-09-27 18:05:07.017 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.classic.classic_env: pygame is not installed. Please install it with pip install "plugrl-env-client[classic]". +2026-09-27 18:05:07.018 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.atari.atari_env: Atari is not installed. Please install it with the 'atari' extra, e.g. 'pip install plugrl-env-client[atari]' +2026-09-27 18:05:07.018 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.d4rl.d4rl_env: d4rl is not installed. Please install it with pip install "plugrl-env-client[d4rl]". +2026-09-27 18:05:07.018 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.libero.libero_env: libero is not installed. Please install it with pip install "plugrl-env-client[libero]". +2026-09-27 18:05:07.019 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.robomimic.robomimic_env: Robomimic is not installed. Please install it with the 'robomimic' extra, e.g. 'pip install plugrl-env-client[robomimic]' +2026-09-27 18:05:07.062 | INFO | __main__:main:83 - Starting env client exp_name=mujoco-v1-nenv1-20260927-180507-6b8e164f output_dir=runs/mujoco-v1-nenv1-20260927-180507-6b8e164f recorder={'episode_freq': 0, 'thread0_only': True, 'record_video': False, 'video_fps': 30.0, 'record_full_rollout': False, 'record_obs_stats': True, 'record_episode_metrics': True, 'metric_window': 100} +2026-09-27 18:05:07.102 | INFO | plugrl_env_client.agent.websocket_env_client_agent:_wait_for_server:67 - Waiting for server at ws://127.0.0.1:9941... +2026-09-27 18:05:07.104 | INFO | plugrl_env_client.runner.run:report_server_metadata:56 - Server metadata: {'protocol_version': 1, 'server': 'plugrl-server', 'server_version': '0.1.0', 'algorithm': 'PPOAlgorithm', 'policy': 'GaussianPolicy', 'action_dim': 6, 'action_horizon': 1} +2026-09-27 18:05:37.107 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=16430 infer_calls=16430 feedback_calls=16430 infer_wait=26.113s infer_obs_pack=0.268s env_step=1.586s feedback_total=1.263s feedback_obs_pack=0.328s feedback_info_pack=0.064s effective_fps=562.09 +2026-09-27 18:06:07.108 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=35904 infer_calls=35904 feedback_calls=35904 infer_wait=51.122s infer_obs_pack=0.597s env_step=3.671s feedback_total=2.874s feedback_obs_pack=0.757s feedback_info_pack=0.137s effective_fps=616.22 +2026-09-27 18:06:37.194 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=55298 infer_calls=55298 feedback_calls=55298 infer_wait=76.162s infer_obs_pack=0.933s env_step=5.766s feedback_total=4.509s feedback_obs_pack=1.189s feedback_info_pack=0.212s effective_fps=632.92 +2026-09-27 18:07:07.195 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=74093 infer_calls=74093 feedback_calls=74093 infer_wait=101.278s infer_obs_pack=1.254s env_step=7.809s feedback_total=6.074s feedback_obs_pack=1.610s feedback_info_pack=0.284s effective_fps=636.45 +2026-09-27 18:07:37.392 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=92162 infer_calls=92162 feedback_calls=92162 infer_wait=126.881s infer_obs_pack=1.559s env_step=9.723s feedback_total=7.544s feedback_obs_pack=2.010s feedback_info_pack=0.356s effective_fps=632.52 +2026-09-27 18:08:07.393 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=110321 infer_calls=110321 feedback_calls=110321 infer_wait=152.209s infer_obs_pack=1.865s env_step=11.660s feedback_total=9.051s feedback_obs_pack=2.423s feedback_info_pack=0.427s effective_fps=631.18 +2026-09-27 18:08:37.395 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=128236 infer_calls=128236 feedback_calls=128236 infer_wait=177.609s infer_obs_pack=2.165s env_step=13.570s feedback_total=10.543s feedback_obs_pack=2.828s feedback_info_pack=0.497s effective_fps=628.96 +2026-09-27 18:09:07.396 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=143793 infer_calls=143793 feedback_calls=143793 infer_wait=203.879s infer_obs_pack=2.409s env_step=15.116s feedback_total=11.745s feedback_obs_pack=3.150s feedback_info_pack=0.553s effective_fps=616.74 +2026-09-27 18:09:37.396 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=159053 infer_calls=159053 feedback_calls=159053 infer_wait=230.309s infer_obs_pack=2.644s env_step=16.609s feedback_total=12.878s feedback_obs_pack=3.460s feedback_info_pack=0.604s effective_fps=606.05 +2026-09-27 18:10:07.397 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=174195 infer_calls=174195 feedback_calls=174195 infer_wait=256.766s infer_obs_pack=2.878s env_step=18.090s feedback_total=13.999s feedback_obs_pack=3.766s feedback_info_pack=0.654s effective_fps=597.10 +2026-09-27 18:10:37.399 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=189539 infer_calls=189539 feedback_calls=189539 infer_wait=283.169s infer_obs_pack=3.119s env_step=19.583s feedback_total=15.149s feedback_obs_pack=4.082s feedback_info_pack=0.706s effective_fps=590.43 +2026-09-27 18:11:07.476 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=204802 infer_calls=204802 feedback_calls=204802 infer_wait=309.723s infer_obs_pack=3.351s env_step=21.046s feedback_total=16.273s feedback_obs_pack=4.387s feedback_info_pack=0.757s effective_fps=584.49 +2026-09-27 18:11:37.477 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=220246 infer_calls=220246 feedback_calls=220246 infer_wait=336.124s infer_obs_pack=3.588s env_step=22.548s feedback_total=17.411s feedback_obs_pack=4.697s feedback_info_pack=0.808s effective_fps=580.10 +2026-09-27 18:12:07.600 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=235522 infer_calls=235522 feedback_calls=235522 infer_wait=362.709s infer_obs_pack=3.817s env_step=24.017s feedback_total=18.543s feedback_obs_pack=5.008s feedback_info_pack=0.861s effective_fps=575.73 +2026-09-27 18:12:37.601 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=250932 infer_calls=250932 feedback_calls=250932 infer_wait=389.113s infer_obs_pack=4.055s env_step=25.512s feedback_total=19.694s feedback_obs_pack=5.320s feedback_info_pack=0.913s effective_fps=572.42 +2026-09-27 18:13:07.602 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=268760 infer_calls=268760 feedback_calls=268760 infer_wait=414.615s infer_obs_pack=4.352s env_step=27.376s feedback_total=21.145s feedback_obs_pack=5.710s feedback_info_pack=0.981s effective_fps=574.90 +2026-09-27 18:13:37.603 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=286790 infer_calls=286790 feedback_calls=286790 infer_wait=440.028s infer_obs_pack=4.655s env_step=29.270s feedback_total=22.635s feedback_obs_pack=6.109s feedback_info_pack=1.049s effective_fps=577.52 +2026-09-27 18:14:07.604 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=305097 infer_calls=305097 feedback_calls=305097 infer_wait=465.410s infer_obs_pack=4.961s env_step=31.182s feedback_total=24.132s feedback_obs_pack=6.512s feedback_info_pack=1.115s effective_fps=580.38 +2026-09-27 18:14:37.605 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=322973 infer_calls=322973 feedback_calls=322973 infer_wait=490.860s infer_obs_pack=5.264s env_step=33.072s feedback_total=25.596s feedback_obs_pack=6.909s feedback_info_pack=1.183s effective_fps=582.15 +2026-09-27 18:15:07.606 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=341515 infer_calls=341515 feedback_calls=341515 infer_wait=516.086s infer_obs_pack=5.574s env_step=35.050s feedback_total=27.149s feedback_obs_pack=7.324s feedback_info_pack=1.254s effective_fps=584.93 +2026-09-27 18:15:37.607 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=361040 infer_calls=361040 feedback_calls=361040 infer_wait=541.037s infer_obs_pack=5.907s env_step=37.154s feedback_total=28.784s feedback_obs_pack=7.768s feedback_info_pack=1.328s effective_fps=589.09 +2026-09-27 18:16:07.608 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=380281 infer_calls=380281 feedback_calls=380281 infer_wait=566.008s infer_obs_pack=6.238s env_step=39.237s feedback_total=30.419s feedback_obs_pack=8.203s feedback_info_pack=1.403s effective_fps=592.43 +2026-09-27 18:16:37.609 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=399727 infer_calls=399727 feedback_calls=399727 infer_wait=590.960s infer_obs_pack=6.567s env_step=41.324s feedback_total=32.082s feedback_obs_pack=8.639s feedback_info_pack=1.476s effective_fps=595.78 +2026-09-27 18:17:07.610 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=419257 infer_calls=419257 feedback_calls=419257 infer_wait=615.776s infer_obs_pack=6.910s env_step=43.477s feedback_total=33.766s feedback_obs_pack=9.085s feedback_info_pack=1.556s effective_fps=599.00 +2026-09-27 18:17:37.610 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=438468 infer_calls=438468 feedback_calls=438468 infer_wait=640.646s infer_obs_pack=7.250s env_step=45.602s feedback_total=35.428s feedback_obs_pack=9.528s feedback_info_pack=1.633s effective_fps=601.53 +2026-09-27 18:18:07.610 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=458218 infer_calls=458218 feedback_calls=458218 infer_wait=665.473s infer_obs_pack=7.591s env_step=47.727s feedback_total=37.128s feedback_obs_pack=9.978s feedback_info_pack=1.708s effective_fps=604.57 +2026-09-27 18:18:37.610 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=477393 infer_calls=477393 feedback_calls=477393 infer_wait=690.439s infer_obs_pack=7.922s env_step=49.823s feedback_total=38.755s feedback_obs_pack=10.423s feedback_info_pack=1.780s effective_fps=606.65 +2026-09-27 18:19:07.611 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=496818 infer_calls=496818 feedback_calls=496818 infer_wait=715.412s infer_obs_pack=8.254s env_step=51.903s feedback_total=40.391s feedback_obs_pack=10.864s feedback_info_pack=1.855s effective_fps=608.88 +2026-09-27 18:19:37.891 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=516098 infer_calls=516098 feedback_calls=516098 infer_wait=740.658s infer_obs_pack=8.588s env_step=53.990s feedback_total=42.025s feedback_obs_pack=11.303s feedback_info_pack=1.930s effective_fps=610.58 +2026-09-27 18:20:07.892 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=535363 infer_calls=535363 feedback_calls=535363 infer_wait=765.629s infer_obs_pack=8.916s env_step=56.083s feedback_total=43.650s feedback_obs_pack=11.739s feedback_info_pack=2.006s effective_fps=612.35 +2026-09-27 18:20:37.894 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=553495 infer_calls=553495 feedback_calls=553495 infer_wait=791.005s infer_obs_pack=9.226s env_step=57.981s feedback_total=45.152s feedback_obs_pack=12.139s feedback_info_pack=2.075s effective_fps=612.70 +2026-09-27 18:21:07.966 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=571394 infer_calls=571394 feedback_calls=571394 infer_wait=816.553s infer_obs_pack=9.528s env_step=59.856s feedback_total=46.605s feedback_obs_pack=12.534s feedback_info_pack=2.144s effective_fps=612.73 +2026-09-27 18:21:37.967 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=589677 infer_calls=589677 feedback_calls=589677 infer_wait=841.895s infer_obs_pack=9.843s env_step=61.764s feedback_total=48.120s feedback_obs_pack=12.939s feedback_info_pack=2.214s effective_fps=613.21 +2026-09-27 18:22:08.059 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=606210 infer_calls=606210 feedback_calls=606210 infer_wait=868.020s infer_obs_pack=10.105s env_step=63.411s feedback_total=49.392s feedback_obs_pack=13.288s feedback_info_pack=2.276s effective_fps=611.76 +2026-09-27 18:22:38.059 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=621387 infer_calls=621387 feedback_calls=621387 infer_wait=894.449s infer_obs_pack=10.343s env_step=64.901s feedback_total=50.517s feedback_obs_pack=13.596s feedback_info_pack=2.329s effective_fps=609.08 +2026-09-27 18:23:08.060 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=636598 infer_calls=636598 feedback_calls=636598 infer_wait=920.904s infer_obs_pack=10.582s env_step=66.367s feedback_total=51.649s feedback_obs_pack=13.901s feedback_info_pack=2.383s effective_fps=606.57 +2026-09-27 18:23:38.061 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=651805 infer_calls=651805 feedback_calls=651805 infer_wait=947.380s infer_obs_pack=10.820s env_step=67.828s feedback_total=52.765s feedback_obs_pack=14.205s feedback_info_pack=2.433s effective_fps=604.20 +2026-09-27 18:24:08.062 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=667254 infer_calls=667254 feedback_calls=667254 infer_wait=973.747s infer_obs_pack=11.060s env_step=69.331s feedback_total=53.926s feedback_obs_pack=14.519s feedback_info_pack=2.486s effective_fps=602.18 +2026-09-27 18:24:38.064 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=682601 infer_calls=682601 feedback_calls=682601 infer_wait=1000.151s infer_obs_pack=11.307s env_step=70.801s feedback_total=55.095s feedback_obs_pack=14.829s feedback_info_pack=2.541s effective_fps=600.17 +2026-09-27 18:25:08.065 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=697852 infer_calls=697852 feedback_calls=697852 infer_wait=1026.608s infer_obs_pack=11.544s env_step=72.266s feedback_total=56.225s feedback_obs_pack=15.138s feedback_info_pack=2.592s effective_fps=598.17 +2026-09-27 18:25:38.067 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=713143 infer_calls=713143 feedback_calls=713143 infer_wait=1053.022s infer_obs_pack=11.777s env_step=73.753s feedback_total=57.372s feedback_obs_pack=15.449s feedback_info_pack=2.643s effective_fps=596.31 +2026-09-27 18:26:08.067 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=730507 infer_calls=730507 feedback_calls=730507 infer_wait=1078.659s infer_obs_pack=12.062s env_step=75.555s feedback_total=58.782s feedback_obs_pack=15.826s feedback_info_pack=2.707s effective_fps=596.30 +2026-09-27 18:26:38.068 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=748527 infer_calls=748527 feedback_calls=748527 infer_wait=1104.054s infer_obs_pack=12.370s env_step=77.467s feedback_total=60.271s feedback_obs_pack=16.231s feedback_info_pack=2.778s effective_fps=596.83 +2026-09-27 18:27:08.069 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=766757 infer_calls=766757 feedback_calls=766757 infer_wait=1129.461s infer_obs_pack=12.678s env_step=79.363s feedback_total=61.766s feedback_obs_pack=16.631s feedback_info_pack=2.847s effective_fps=597.50 +2026-09-27 18:27:38.069 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=784943 infer_calls=784943 feedback_calls=784943 infer_wait=1154.824s infer_obs_pack=12.990s env_step=81.278s feedback_total=63.265s feedback_obs_pack=17.040s feedback_info_pack=2.916s effective_fps=598.12 +2026-09-27 18:28:08.071 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=803213 infer_calls=803213 feedback_calls=803213 infer_wait=1180.140s infer_obs_pack=13.305s env_step=83.226s feedback_total=64.769s feedback_obs_pack=17.452s feedback_info_pack=2.985s effective_fps=598.77 +2026-09-27 18:28:38.071 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=822940 infer_calls=822940 feedback_calls=822940 infer_wait=1204.984s infer_obs_pack=13.649s env_step=85.363s feedback_total=66.446s feedback_obs_pack=17.904s feedback_info_pack=3.063s effective_fps=600.49 +2026-09-27 18:29:08.072 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=842174 infer_calls=842174 feedback_calls=842174 infer_wait=1230.023s infer_obs_pack=13.982s env_step=87.410s feedback_total=68.067s feedback_obs_pack=18.341s feedback_info_pack=3.138s effective_fps=601.78 +2026-09-27 18:29:38.072 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=861060 infer_calls=861060 feedback_calls=861060 infer_wait=1254.990s infer_obs_pack=14.318s env_step=89.505s feedback_total=69.690s feedback_obs_pack=18.782s feedback_info_pack=3.214s effective_fps=602.77 +2026-09-27 18:30:08.237 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=880642 infer_calls=880642 feedback_calls=880642 infer_wait=1279.993s infer_obs_pack=14.660s env_step=91.632s feedback_total=71.385s feedback_obs_pack=19.233s feedback_info_pack=3.291s effective_fps=604.14 +2026-09-27 18:30:38.238 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=900245 infer_calls=900245 feedback_calls=900245 infer_wait=1304.785s infer_obs_pack=15.001s env_step=93.794s feedback_total=73.080s feedback_obs_pack=19.679s feedback_info_pack=3.370s effective_fps=605.55 +2026-09-27 18:31:08.239 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=919407 infer_calls=919407 feedback_calls=919407 infer_wait=1329.698s infer_obs_pack=15.342s env_step=95.900s feedback_total=74.728s feedback_obs_pack=20.127s feedback_info_pack=3.448s effective_fps=606.60 +2026-09-27 18:31:38.242 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=938237 infer_calls=938237 feedback_calls=938237 infer_wait=1354.769s infer_obs_pack=15.669s env_step=97.943s feedback_total=76.325s feedback_obs_pack=20.558s feedback_info_pack=3.520s effective_fps=607.39 +2026-09-27 18:32:08.243 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=957392 infer_calls=957392 feedback_calls=957392 infer_wait=1379.758s infer_obs_pack=15.999s env_step=100.011s feedback_total=77.961s feedback_obs_pack=20.991s feedback_info_pack=3.596s effective_fps=608.36 +2026-09-27 18:32:38.243 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=976529 infer_calls=976529 feedback_calls=976529 infer_wait=1404.732s infer_obs_pack=16.340s env_step=102.076s feedback_total=79.613s feedback_obs_pack=21.434s feedback_info_pack=3.671s effective_fps=609.28 +2026-09-27 18:33:08.243 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=994806 infer_calls=994806 feedback_calls=994806 infer_wait=1430.090s infer_obs_pack=16.651s env_step=103.981s feedback_total=81.125s feedback_obs_pack=21.840s feedback_info_pack=3.739s effective_fps=609.62 +2026-09-27 18:33:16.826 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Final rollout timing summary: env_steps=999425 infer_calls=999425 feedback_calls=999425 infer_wait=1437.261s infer_obs_pack=16.723s env_step=104.424s feedback_total=81.480s feedback_obs_pack=21.933s feedback_info_pack=3.755s effective_fps=609.45 +2026-09-27 18:33:16.827 | INFO | plugrl_env_client.runner.run:run:156 - Server signalled the end of the run; collection stopped after 999/3997696 episodes. diff --git a/experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed2.log b/experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed2.log new file mode 100644 index 0000000..a716c81 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-cheetah/client-seed2.log @@ -0,0 +1,66 @@ +2026-09-27 18:05:12.040 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.classic.classic_env: pygame is not installed. Please install it with pip install "plugrl-env-client[classic]". +2026-09-27 18:05:12.040 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.atari.atari_env: Atari is not installed. Please install it with the 'atari' extra, e.g. 'pip install plugrl-env-client[atari]' +2026-09-27 18:05:12.040 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.d4rl.d4rl_env: d4rl is not installed. Please install it with pip install "plugrl-env-client[d4rl]". +2026-09-27 18:05:12.040 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.libero.libero_env: libero is not installed. Please install it with pip install "plugrl-env-client[libero]". +2026-09-27 18:05:12.041 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.robomimic.robomimic_env: Robomimic is not installed. Please install it with the 'robomimic' extra, e.g. 'pip install plugrl-env-client[robomimic]' +2026-09-27 18:05:12.083 | INFO | __main__:main:83 - Starting env client exp_name=mujoco-v1-nenv1-20260927-180512-09ab20c3 output_dir=runs/mujoco-v1-nenv1-20260927-180512-09ab20c3 recorder={'episode_freq': 0, 'thread0_only': True, 'record_video': False, 'video_fps': 30.0, 'record_full_rollout': False, 'record_obs_stats': True, 'record_episode_metrics': True, 'metric_window': 100} +2026-09-27 18:05:12.129 | INFO | plugrl_env_client.agent.websocket_env_client_agent:_wait_for_server:67 - Waiting for server at ws://127.0.0.1:9942... +2026-09-27 18:05:12.132 | INFO | plugrl_env_client.runner.run:report_server_metadata:56 - Server metadata: {'protocol_version': 1, 'server': 'plugrl-server', 'server_version': '0.1.0', 'algorithm': 'PPOAlgorithm', 'policy': 'GaussianPolicy', 'action_dim': 6, 'action_horizon': 1} +2026-09-27 18:05:42.135 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=16768 infer_calls=16768 feedback_calls=16768 infer_wait=25.937s infer_obs_pack=0.293s env_step=1.661s feedback_total=1.310s feedback_obs_pack=0.341s feedback_info_pack=0.056s effective_fps=574.24 +2026-09-27 18:06:12.136 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=36238 infer_calls=36238 feedback_calls=36238 infer_wait=50.893s infer_obs_pack=0.630s env_step=3.759s feedback_total=2.948s feedback_obs_pack=0.782s feedback_info_pack=0.130s effective_fps=622.34 +2026-09-27 18:06:42.138 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=55430 infer_calls=55430 feedback_calls=55430 infer_wait=75.875s infer_obs_pack=0.969s env_step=5.847s feedback_total=4.568s feedback_obs_pack=1.221s feedback_info_pack=0.200s effective_fps=635.24 +2026-09-27 18:07:12.138 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=74317 infer_calls=74317 feedback_calls=74317 infer_wait=100.960s infer_obs_pack=1.301s env_step=7.917s feedback_total=6.133s feedback_obs_pack=1.647s feedback_info_pack=0.270s effective_fps=638.95 +2026-09-27 18:07:42.149 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=92162 infer_calls=92162 feedback_calls=92162 infer_wait=126.344s infer_obs_pack=1.605s env_step=9.861s feedback_total=7.618s feedback_obs_pack=2.048s feedback_info_pack=0.335s effective_fps=633.73 +2026-09-27 18:08:12.149 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=110325 infer_calls=110325 feedback_calls=110325 infer_wait=151.652s infer_obs_pack=1.923s env_step=11.831s feedback_total=9.116s feedback_obs_pack=2.458s feedback_info_pack=0.400s effective_fps=632.15 +2026-09-27 18:08:42.151 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=127936 infer_calls=127936 feedback_calls=127936 infer_wait=177.166s infer_obs_pack=2.220s env_step=13.698s feedback_total=10.549s feedback_obs_pack=2.852s feedback_info_pack=0.463s effective_fps=628.26 +2026-09-27 18:09:12.395 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=143362 infer_calls=143362 feedback_calls=143362 infer_wait=203.834s infer_obs_pack=2.460s env_step=15.183s feedback_total=11.683s feedback_obs_pack=3.156s feedback_info_pack=0.514s effective_fps=614.87 +2026-09-27 18:09:42.396 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=158786 infer_calls=158786 feedback_calls=158786 infer_wait=230.231s infer_obs_pack=2.702s env_step=16.685s feedback_total=12.820s feedback_obs_pack=3.462s feedback_info_pack=0.565s effective_fps=605.04 +2026-09-27 18:10:12.398 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=173974 infer_calls=173974 feedback_calls=173974 infer_wait=256.717s infer_obs_pack=2.936s env_step=18.146s feedback_total=13.935s feedback_obs_pack=3.766s feedback_info_pack=0.619s effective_fps=596.35 +2026-09-27 18:10:42.398 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=189097 infer_calls=189097 feedback_calls=189097 infer_wait=283.159s infer_obs_pack=3.169s env_step=19.630s feedback_total=15.058s feedback_obs_pack=4.074s feedback_info_pack=0.668s effective_fps=589.06 +2026-09-27 18:11:12.400 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=204500 infer_calls=204500 feedback_calls=204500 infer_wait=309.556s infer_obs_pack=3.416s env_step=21.111s feedback_total=16.207s feedback_obs_pack=4.383s feedback_info_pack=0.720s effective_fps=583.80 +2026-09-27 18:11:42.400 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=219624 infer_calls=219624 feedback_calls=219624 infer_wait=336.034s infer_obs_pack=3.653s env_step=22.570s feedback_total=17.322s feedback_obs_pack=4.687s feedback_info_pack=0.768s effective_fps=578.60 +2026-09-27 18:12:12.401 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=235165 infer_calls=235165 feedback_calls=235165 infer_wait=362.400s infer_obs_pack=3.902s env_step=24.072s feedback_total=18.473s feedback_obs_pack=4.998s feedback_info_pack=0.821s effective_fps=575.19 +2026-09-27 18:12:42.402 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=250634 infer_calls=250634 feedback_calls=250634 infer_wait=388.738s infer_obs_pack=4.151s env_step=25.577s feedback_total=19.648s feedback_obs_pack=5.315s feedback_info_pack=0.874s effective_fps=572.08 +2026-09-27 18:13:12.402 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=268840 infer_calls=268840 feedback_calls=268840 infer_wait=414.060s infer_obs_pack=4.470s env_step=27.525s feedback_total=21.143s feedback_obs_pack=5.716s feedback_info_pack=0.941s effective_fps=575.43 +2026-09-27 18:13:42.403 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=287135 infer_calls=287135 feedback_calls=287135 infer_wait=439.380s infer_obs_pack=4.787s env_step=29.451s feedback_total=22.669s feedback_obs_pack=6.124s feedback_info_pack=1.009s effective_fps=578.57 +2026-09-27 18:14:12.404 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=305163 infer_calls=305163 feedback_calls=305163 infer_wait=464.755s infer_obs_pack=5.101s env_step=31.372s feedback_total=24.159s feedback_obs_pack=6.522s feedback_info_pack=1.075s effective_fps=580.83 +2026-09-27 18:14:42.404 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=323385 infer_calls=323385 feedback_calls=323385 infer_wait=490.201s infer_obs_pack=5.412s env_step=33.265s feedback_total=25.610s feedback_obs_pack=6.915s feedback_info_pack=1.142s effective_fps=583.21 +2026-09-27 18:15:12.638 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=342018 infer_calls=342018 feedback_calls=342018 infer_wait=515.647s infer_obs_pack=5.735s env_step=35.264s feedback_total=27.152s feedback_obs_pack=7.333s feedback_info_pack=1.214s effective_fps=585.85 +2026-09-27 18:15:42.640 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=361527 infer_calls=361527 feedback_calls=361527 infer_wait=540.528s infer_obs_pack=6.085s env_step=37.393s feedback_total=28.800s feedback_obs_pack=7.776s feedback_info_pack=1.287s effective_fps=589.95 +2026-09-27 18:16:12.641 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=380862 infer_calls=380862 feedback_calls=380862 infer_wait=565.496s infer_obs_pack=6.428s env_step=39.487s feedback_total=30.419s feedback_obs_pack=8.216s feedback_info_pack=1.358s effective_fps=593.40 +2026-09-27 18:16:42.642 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=399722 infer_calls=399722 feedback_calls=399722 infer_wait=590.573s infer_obs_pack=6.758s env_step=41.538s feedback_total=32.008s feedback_obs_pack=8.641s feedback_info_pack=1.430s effective_fps=595.82 +2026-09-27 18:17:12.642 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=419287 infer_calls=419287 feedback_calls=419287 infer_wait=615.287s infer_obs_pack=7.114s env_step=43.741s feedback_total=33.717s feedback_obs_pack=9.092s feedback_info_pack=1.505s effective_fps=599.10 +2026-09-27 18:17:42.643 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=438887 infer_calls=438887 feedback_calls=438887 infer_wait=640.055s infer_obs_pack=7.459s env_step=45.923s feedback_total=35.417s feedback_obs_pack=9.551s feedback_info_pack=1.580s effective_fps=602.16 +2026-09-27 18:18:12.644 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=458447 infer_calls=458447 feedback_calls=458447 infer_wait=664.896s infer_obs_pack=7.802s env_step=48.059s feedback_total=37.108s feedback_obs_pack=10.002s feedback_info_pack=1.654s effective_fps=604.92 +2026-09-27 18:18:42.644 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=477665 infer_calls=477665 feedback_calls=477665 infer_wait=689.870s infer_obs_pack=8.145s env_step=50.143s feedback_total=38.734s feedback_obs_pack=10.437s feedback_info_pack=1.723s effective_fps=607.03 +2026-09-27 18:19:12.645 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=496947 infer_calls=496947 feedback_calls=496947 infer_wait=714.844s infer_obs_pack=8.488s env_step=52.236s feedback_total=40.348s feedback_obs_pack=10.873s feedback_info_pack=1.795s effective_fps=609.07 +2026-09-27 18:19:42.645 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=516426 infer_calls=516426 feedback_calls=516426 infer_wait=739.749s infer_obs_pack=8.839s env_step=54.352s feedback_total=41.995s feedback_obs_pack=11.311s feedback_info_pack=1.866s effective_fps=611.20 +2026-09-27 18:20:12.647 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=535940 infer_calls=535940 feedback_calls=535940 infer_wait=764.620s infer_obs_pack=9.185s env_step=56.469s feedback_total=43.665s feedback_obs_pack=11.757s feedback_info_pack=1.939s effective_fps=613.25 +2026-09-27 18:20:42.649 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=554132 infer_calls=554132 feedback_calls=554132 infer_wait=789.974s infer_obs_pack=9.498s env_step=58.405s feedback_total=45.157s feedback_obs_pack=12.162s feedback_info_pack=2.008s effective_fps=613.63 +2026-09-27 18:21:12.649 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=572022 infer_calls=572022 feedback_calls=572022 infer_wait=815.410s infer_obs_pack=9.805s env_step=60.296s feedback_total=46.631s feedback_obs_pack=12.563s feedback_info_pack=2.077s effective_fps=613.66 +2026-09-27 18:21:42.650 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=590133 infer_calls=590133 feedback_calls=590133 infer_wait=840.794s infer_obs_pack=10.116s env_step=62.228s feedback_total=48.108s feedback_obs_pack=12.965s feedback_info_pack=2.143s effective_fps=613.93 +2026-09-27 18:22:12.651 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=606438 infer_calls=606438 feedback_calls=606438 infer_wait=866.837s infer_obs_pack=10.384s env_step=63.882s feedback_total=49.367s feedback_obs_pack=13.305s feedback_info_pack=2.201s effective_fps=612.27 +2026-09-27 18:22:42.652 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=621694 infer_calls=621694 feedback_calls=621694 infer_wait=893.232s infer_obs_pack=10.625s env_step=65.387s feedback_total=50.507s feedback_obs_pack=13.616s feedback_info_pack=2.250s effective_fps=609.65 +2026-09-27 18:23:12.652 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=636952 infer_calls=636952 feedback_calls=636952 infer_wait=919.646s infer_obs_pack=10.867s env_step=66.878s feedback_total=51.639s feedback_obs_pack=13.924s feedback_info_pack=2.303s effective_fps=607.18 +2026-09-27 18:23:42.653 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=652399 infer_calls=652399 feedback_calls=652399 infer_wait=946.020s infer_obs_pack=11.113s env_step=68.383s feedback_total=52.790s feedback_obs_pack=14.233s feedback_info_pack=2.353s effective_fps=605.02 +2026-09-27 18:24:12.655 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=667740 infer_calls=667740 feedback_calls=667740 infer_wait=972.419s infer_obs_pack=11.360s env_step=69.885s feedback_total=53.924s feedback_obs_pack=14.542s feedback_info_pack=2.403s effective_fps=602.88 +2026-09-27 18:24:42.656 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=683213 infer_calls=683213 feedback_calls=683213 infer_wait=998.785s infer_obs_pack=11.605s env_step=71.397s feedback_total=55.069s feedback_obs_pack=14.857s feedback_info_pack=2.456s effective_fps=600.97 +2026-09-27 18:25:12.656 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=698643 infer_calls=698643 feedback_calls=698643 infer_wait=1025.156s infer_obs_pack=11.851s env_step=72.909s feedback_total=56.216s feedback_obs_pack=15.171s feedback_info_pack=2.509s effective_fps=599.11 +2026-09-27 18:25:42.657 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=714109 infer_calls=714109 feedback_calls=714109 infer_wait=1051.533s infer_obs_pack=12.092s env_step=74.421s feedback_total=57.353s feedback_obs_pack=15.484s feedback_info_pack=2.560s effective_fps=597.38 +2026-09-27 18:26:12.658 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=731811 infer_calls=731811 feedback_calls=731811 infer_wait=1077.009s infer_obs_pack=12.398s env_step=76.294s feedback_total=58.811s feedback_obs_pack=15.879s feedback_info_pack=2.627s effective_fps=597.63 +2026-09-27 18:26:42.659 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=749715 infer_calls=749715 feedback_calls=749715 infer_wait=1102.403s infer_obs_pack=12.711s env_step=78.212s feedback_total=60.284s feedback_obs_pack=16.284s feedback_info_pack=2.697s effective_fps=598.04 +2026-09-27 18:27:12.659 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=767910 infer_calls=767910 feedback_calls=767910 infer_wait=1127.783s infer_obs_pack=13.023s env_step=80.131s feedback_total=61.778s feedback_obs_pack=16.683s feedback_info_pack=2.763s effective_fps=598.66 +2026-09-27 18:27:42.660 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=785779 infer_calls=785779 feedback_calls=785779 infer_wait=1153.295s infer_obs_pack=13.324s env_step=81.987s feedback_total=63.226s feedback_obs_pack=17.069s feedback_info_pack=2.832s effective_fps=598.99 +2026-09-27 18:28:12.660 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=804760 infer_calls=804760 feedback_calls=804760 infer_wait=1178.423s infer_obs_pack=13.646s env_step=84.013s feedback_total=64.803s feedback_obs_pack=17.491s feedback_info_pack=2.902s effective_fps=600.17 +2026-09-27 18:28:42.661 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=824097 infer_calls=824097 feedback_calls=824097 infer_wait=1203.350s infer_obs_pack=13.986s env_step=86.113s feedback_total=66.469s feedback_obs_pack=17.932s feedback_info_pack=2.976s effective_fps=601.57 +2026-09-27 18:29:12.662 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=843255 infer_calls=843255 feedback_calls=843255 infer_wait=1228.332s infer_obs_pack=14.326s env_step=88.200s feedback_total=68.098s feedback_obs_pack=18.371s feedback_info_pack=3.051s effective_fps=602.77 +2026-09-27 18:29:42.662 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=862657 infer_calls=862657 feedback_calls=862657 infer_wait=1253.235s infer_obs_pack=14.662s env_step=90.329s feedback_total=69.747s feedback_obs_pack=18.811s feedback_info_pack=3.124s effective_fps=604.11 +2026-09-27 18:30:12.662 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=882611 infer_calls=882611 feedback_calls=882611 infer_wait=1277.946s infer_obs_pack=15.016s env_step=92.526s feedback_total=71.473s feedback_obs_pack=19.261s feedback_info_pack=3.201s effective_fps=605.79 +2026-09-27 18:30:42.663 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=902344 infer_calls=902344 feedback_calls=902344 infer_wait=1302.716s infer_obs_pack=15.363s env_step=94.700s feedback_total=73.178s feedback_obs_pack=19.713s feedback_info_pack=3.275s effective_fps=607.25 +2026-09-27 18:31:12.664 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=921702 infer_calls=921702 feedback_calls=921702 infer_wait=1327.656s infer_obs_pack=15.703s env_step=96.813s feedback_total=74.804s feedback_obs_pack=20.147s feedback_info_pack=3.349s effective_fps=608.39 +2026-09-27 18:31:42.664 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=940868 infer_calls=940868 feedback_calls=940868 infer_wait=1352.622s infer_obs_pack=16.042s env_step=98.909s feedback_total=76.427s feedback_obs_pack=20.588s feedback_info_pack=3.421s effective_fps=609.37 +2026-09-27 18:32:12.664 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=960246 infer_calls=960246 feedback_calls=960246 infer_wait=1377.532s infer_obs_pack=16.383s env_step=101.021s feedback_total=78.082s feedback_obs_pack=21.029s feedback_info_pack=3.496s effective_fps=610.45 +2026-09-27 18:32:42.665 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=979577 infer_calls=979577 feedback_calls=979577 infer_wait=1402.534s infer_obs_pack=16.725s env_step=103.096s feedback_total=79.698s feedback_obs_pack=21.466s feedback_info_pack=3.569s effective_fps=611.45 +2026-09-27 18:33:12.665 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=997746 infer_calls=997746 feedback_calls=997746 infer_wait=1428.047s infer_obs_pack=17.030s env_step=104.962s feedback_total=81.137s feedback_obs_pack=21.852s feedback_info_pack=3.631s effective_fps=611.67 +2026-09-27 18:33:15.703 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Final rollout timing summary: env_steps=999425 infer_calls=999425 feedback_calls=999425 infer_wait=1430.368s infer_obs_pack=17.057s env_step=105.123s feedback_total=81.264s feedback_obs_pack=21.886s feedback_info_pack=3.636s effective_fps=611.71 +2026-09-27 18:33:15.704 | INFO | plugrl_env_client.runner.run:run:156 - Server signalled the end of the run; collection stopped after 999/3997696 episodes. diff --git a/experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed0.log b/experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed0.log new file mode 100644 index 0000000..4242931 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed0.log @@ -0,0 +1,43 @@ +18:05:01|INFO|plugrl_server version: 0.1.0 +18:05:01|INFO|Algorithm: ppo, Config: PPOAlgoConfig(global_steps=999424, learning_rate=0.0003, anneal_lr=True, buffer_size=2048, batch_size=64, update_epochs=10, gamma=0.99, gae_lambda=0.95, norm_adv=True, clip_coef=0.2, clip_vloss=True, ent_coef=0.0, vf_coef=0.5, max_grad_norm=0.5, target_kl=None, normalize_rewards=True, reward_clip=10.0, train_itrs=488, save_interval=20) +18:05:01|INFO|Policy: gaussian-policy, Config: GaussianPolicyConfig(algo='ppo', device=device(type='cpu'), obs_dim=17, action_dim=6, hidden_dims=(64, 64), state_keys=('obs',), normalize_observations=True, obs_clip=10.0, freeze_obs_stats=False, action_clip=1.0) +18:05:01|INFO|Seeded python, numpy and torch with 0. One env client is then reproducible; several are not, because batch composition depends on when their requests arrive. +18:05:01|INFO|Checkpoint Manager created: + at /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0 +18:05:01|INFO|Policy created... +18:05:01|INFO|Initialized RolloutBuffer buffer_size=2048 action_shape=(2048, 6) value_shape=(2048,) +18:05:01|INFO|Algorithm created: + +18:05:01|INFO|Agent Server is listening on 0.0.0.0:9940 +18:06:11|INFO|Checkpoint saved at step 40961 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/40961 +18:07:14|INFO|Checkpoint saved at step 81921 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/81921 +18:08:22|INFO|Checkpoint saved at step 122881 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/122881 +18:09:38|INFO|Checkpoint saved at step 163841 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/163841 +18:10:57|INFO|Checkpoint saved at step 204801 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/204801 +18:12:16|INFO|Checkpoint saved at step 245761 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/245761 +18:13:27|INFO|Checkpoint saved at step 286721 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/286721 +18:14:36|INFO|Checkpoint saved at step 327681 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/327681 +18:15:40|INFO|Checkpoint saved at step 368641 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/368641 +18:16:43|INFO|Checkpoint saved at step 409601 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/409601 +18:17:45|INFO|Checkpoint saved at step 450561 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/450561 +18:18:47|INFO|Checkpoint saved at step 491521 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/491521 +18:19:51|INFO|Checkpoint saved at step 532481 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/532481 +18:20:57|INFO|Checkpoint saved at step 573441 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/573441 +18:22:07|INFO|Checkpoint saved at step 614401 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/614401 +18:23:26|INFO|Checkpoint saved at step 655361 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/655361 +18:24:45|INFO|Checkpoint saved at step 696321 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/696321 +18:26:01|INFO|Checkpoint saved at step 737281 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/737281 +18:27:09|INFO|Checkpoint saved at step 778241 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/778241 +18:28:16|INFO|Checkpoint saved at step 819201 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/819201 +18:29:18|INFO|Checkpoint saved at step 860161 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/860161 +18:30:21|INFO|Checkpoint saved at step 901121 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/901121 +18:31:24|INFO|Checkpoint saved at step 942080 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/942080 +18:32:27|INFO|Checkpoint saved at step 983041 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/983041 +18:32:53|INFO|Checkpoint saved at step 999425 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed0/999425 +18:32:53|INFO|Stopping server as the algorithm signaled to stop. +18:32:53|INFO|Shutdown started: aborting pending infer requests. +18:32:53|INFO|Shutdown cleanup finished: pending_futures=1, drained_requests=1 +18:32:53|INFO|Shutdown closing 1 websocket connection(s). +18:32:53|INFO|Shutdown interrupted an in-flight inference request from ('127.0.0.1', 48534). +18:32:53|INFO|WebSocket server closed. +18:32:53|INFO|Scheduler task cancelled and cleaned up. diff --git a/experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed1.log b/experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed1.log new file mode 100644 index 0000000..c5c7f57 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed1.log @@ -0,0 +1,43 @@ +18:05:06|INFO|plugrl_server version: 0.1.0 +18:05:06|INFO|Algorithm: ppo, Config: PPOAlgoConfig(global_steps=999424, learning_rate=0.0003, anneal_lr=True, buffer_size=2048, batch_size=64, update_epochs=10, gamma=0.99, gae_lambda=0.95, norm_adv=True, clip_coef=0.2, clip_vloss=True, ent_coef=0.0, vf_coef=0.5, max_grad_norm=0.5, target_kl=None, normalize_rewards=True, reward_clip=10.0, train_itrs=488, save_interval=20) +18:05:06|INFO|Policy: gaussian-policy, Config: GaussianPolicyConfig(algo='ppo', device=device(type='cpu'), obs_dim=17, action_dim=6, hidden_dims=(64, 64), state_keys=('obs',), normalize_observations=True, obs_clip=10.0, freeze_obs_stats=False, action_clip=1.0) +18:05:06|INFO|Seeded python, numpy and torch with 1. One env client is then reproducible; several are not, because batch composition depends on when their requests arrive. +18:05:06|INFO|Checkpoint Manager created: + at /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1 +18:05:06|INFO|Policy created... +18:05:06|INFO|Initialized RolloutBuffer buffer_size=2048 action_shape=(2048, 6) value_shape=(2048,) +18:05:06|INFO|Algorithm created: + +18:05:06|INFO|Agent Server is listening on 0.0.0.0:9941 +18:06:15|INFO|Checkpoint saved at step 40961 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/40961 +18:07:20|INFO|Checkpoint saved at step 81921 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/81921 +18:08:28|INFO|Checkpoint saved at step 122881 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/122881 +18:09:47|INFO|Checkpoint saved at step 163841 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/163841 +18:11:07|INFO|Checkpoint saved at step 204801 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/204801 +18:12:27|INFO|Checkpoint saved at step 245761 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/245761 +18:13:37|INFO|Checkpoint saved at step 286721 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/286721 +18:14:45|INFO|Checkpoint saved at step 327681 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/327681 +18:15:49|INFO|Checkpoint saved at step 368641 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/368641 +18:16:53|INFO|Checkpoint saved at step 409601 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/409601 +18:17:56|INFO|Checkpoint saved at step 450561 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/450561 +18:18:59|INFO|Checkpoint saved at step 491521 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/491521 +18:20:03|INFO|Checkpoint saved at step 532481 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/532481 +18:21:11|INFO|Checkpoint saved at step 573441 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/573441 +18:22:24|INFO|Checkpoint saved at step 614401 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/614401 +18:23:45|INFO|Checkpoint saved at step 655361 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/655361 +18:25:05|INFO|Checkpoint saved at step 696321 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/696321 +18:26:19|INFO|Checkpoint saved at step 737281 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/737281 +18:27:27|INFO|Checkpoint saved at step 778241 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/778241 +18:28:32|INFO|Checkpoint saved at step 819201 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/819201 +18:29:36|INFO|Checkpoint saved at step 860161 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/860161 +18:30:39|INFO|Checkpoint saved at step 901121 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/901121 +18:31:44|INFO|Checkpoint saved at step 942081 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/942081 +18:32:48|INFO|Checkpoint saved at step 983041 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/983041 +18:33:16|INFO|Checkpoint saved at step 999425 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed1/999425 +18:33:16|INFO|Stopping server as the algorithm signaled to stop. +18:33:16|INFO|Shutdown started: aborting pending infer requests. +18:33:16|INFO|Shutdown cleanup finished: pending_futures=1, drained_requests=1 +18:33:16|INFO|Shutdown closing 1 websocket connection(s). +18:33:16|INFO|Shutdown interrupted an in-flight inference request from ('127.0.0.1', 43314). +18:33:16|INFO|WebSocket server closed. +18:33:16|INFO|Scheduler task cancelled and cleaned up. diff --git a/experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed2.log b/experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed2.log new file mode 100644 index 0000000..860ea37 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-cheetah/server-seed2.log @@ -0,0 +1,43 @@ +18:05:11|INFO|plugrl_server version: 0.1.0 +18:05:11|INFO|Algorithm: ppo, Config: PPOAlgoConfig(global_steps=999424, learning_rate=0.0003, anneal_lr=True, buffer_size=2048, batch_size=64, update_epochs=10, gamma=0.99, gae_lambda=0.95, norm_adv=True, clip_coef=0.2, clip_vloss=True, ent_coef=0.0, vf_coef=0.5, max_grad_norm=0.5, target_kl=None, normalize_rewards=True, reward_clip=10.0, train_itrs=488, save_interval=20) +18:05:11|INFO|Policy: gaussian-policy, Config: GaussianPolicyConfig(algo='ppo', device=device(type='cpu'), obs_dim=17, action_dim=6, hidden_dims=(64, 64), state_keys=('obs',), normalize_observations=True, obs_clip=10.0, freeze_obs_stats=False, action_clip=1.0) +18:05:11|INFO|Seeded python, numpy and torch with 2. One env client is then reproducible; several are not, because batch composition depends on when their requests arrive. +18:05:11|INFO|Checkpoint Manager created: + at /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2 +18:05:11|INFO|Policy created... +18:05:11|INFO|Initialized RolloutBuffer buffer_size=2048 action_shape=(2048, 6) value_shape=(2048,) +18:05:11|INFO|Algorithm created: + +18:05:11|INFO|Agent Server is listening on 0.0.0.0:9942 +18:06:19|INFO|Checkpoint saved at step 40961 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/40961 +18:07:25|INFO|Checkpoint saved at step 81921 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/81921 +18:08:33|INFO|Checkpoint saved at step 122881 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/122881 +18:09:52|INFO|Checkpoint saved at step 163841 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/163841 +18:11:13|INFO|Checkpoint saved at step 204801 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/204801 +18:12:33|INFO|Checkpoint saved at step 245761 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/245761 +18:13:41|INFO|Checkpoint saved at step 286721 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/286721 +18:14:49|INFO|Checkpoint saved at step 327681 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/327681 +18:15:53|INFO|Checkpoint saved at step 368641 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/368641 +18:16:57|INFO|Checkpoint saved at step 409601 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/409601 +18:18:00|INFO|Checkpoint saved at step 450561 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/450561 +18:19:04|INFO|Checkpoint saved at step 491521 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/491521 +18:20:07|INFO|Checkpoint saved at step 532481 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/532481 +18:21:15|INFO|Checkpoint saved at step 573441 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/573441 +18:22:28|INFO|Checkpoint saved at step 614401 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/614401 +18:23:48|INFO|Checkpoint saved at step 655361 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/655361 +18:25:08|INFO|Checkpoint saved at step 696321 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/696321 +18:26:21|INFO|Checkpoint saved at step 737281 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/737281 +18:27:30|INFO|Checkpoint saved at step 778241 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/778241 +18:28:35|INFO|Checkpoint saved at step 819201 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/819201 +18:29:38|INFO|Checkpoint saved at step 860161 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/860161 +18:30:41|INFO|Checkpoint saved at step 901121 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/901121 +18:31:44|INFO|Checkpoint saved at step 942081 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/942081 +18:32:48|INFO|Checkpoint saved at step 983041 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/983041 +18:33:15|INFO|Checkpoint saved at step 999425 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-cheetah/ppo/gaussian-policy/ppo-cheetah-seed2/999425 +18:33:15|INFO|Stopping server as the algorithm signaled to stop. +18:33:15|INFO|Shutdown started: aborting pending infer requests. +18:33:15|INFO|Shutdown cleanup finished: pending_futures=1, drained_requests=1 +18:33:15|INFO|Shutdown closing 1 websocket connection(s). +18:33:15|INFO|Shutdown interrupted an in-flight inference request from ('127.0.0.1', 60930). +18:33:15|INFO|WebSocket server closed. +18:33:15|INFO|Scheduler task cancelled and cleaned up. diff --git a/experiments/e38-gaussian-ppo/results/ppo-hopper.out b/experiments/e38-gaussian-ppo/results/ppo-hopper.out new file mode 100644 index 0000000..6db292c --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-hopper.out @@ -0,0 +1,11 @@ +cell: ppo-hopper = gaussian-policy/default x ppo/default x Hopper-v5 +iters: 488 x 2048 batch: 64 replan: 1 seeds: 0 1 2 +extra: algo none; policy --policy.obs-dim 11 --policy.action-dim 3 +out: /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper +start 2026-09-27 18:05:18 +seed 0 finished rc=0 at 18:32:53 +seed 2 finished rc=0 at 18:32:55 +seed 1 finished rc=0 at 18:33:02 +end 2026-09-27 18:33:02 +failed seeds: 0 +CELL_DONE ppo-hopper diff --git a/experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed0.log b/experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed0.log new file mode 100644 index 0000000..fb7c399 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed0.log @@ -0,0 +1,65 @@ +2026-09-27 18:05:22.031 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.classic.classic_env: pygame is not installed. Please install it with pip install "plugrl-env-client[classic]". +2026-09-27 18:05:22.032 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.atari.atari_env: Atari is not installed. Please install it with the 'atari' extra, e.g. 'pip install plugrl-env-client[atari]' +2026-09-27 18:05:22.032 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.d4rl.d4rl_env: d4rl is not installed. Please install it with pip install "plugrl-env-client[d4rl]". +2026-09-27 18:05:22.032 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.libero.libero_env: libero is not installed. Please install it with pip install "plugrl-env-client[libero]". +2026-09-27 18:05:22.033 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.robomimic.robomimic_env: Robomimic is not installed. Please install it with the 'robomimic' extra, e.g. 'pip install plugrl-env-client[robomimic]' +2026-09-27 18:05:22.076 | INFO | __main__:main:83 - Starting env client exp_name=mujoco-v1-nenv1-20260927-180522-5422ee55 output_dir=runs/mujoco-v1-nenv1-20260927-180522-5422ee55 recorder={'episode_freq': 0, 'thread0_only': True, 'record_video': False, 'video_fps': 30.0, 'record_full_rollout': False, 'record_obs_stats': True, 'record_episode_metrics': True, 'metric_window': 100} +2026-09-27 18:05:22.112 | INFO | plugrl_env_client.agent.websocket_env_client_agent:_wait_for_server:67 - Waiting for server at ws://127.0.0.1:9950... +2026-09-27 18:05:22.115 | INFO | plugrl_env_client.runner.run:report_server_metadata:56 - Server metadata: {'protocol_version': 1, 'server': 'plugrl-server', 'server_version': '0.1.0', 'algorithm': 'PPOAlgorithm', 'policy': 'GaussianPolicy', 'action_dim': 3, 'action_horizon': 1} +2026-09-27 18:05:52.117 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=18594 infer_calls=18594 feedback_calls=18594 infer_wait=24.324s infer_obs_pack=0.327s env_step=2.937s feedback_total=1.478s feedback_obs_pack=0.398s feedback_info_pack=0.085s effective_fps=639.71 +2026-09-27 18:06:22.118 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=38382 infer_calls=38382 feedback_calls=38382 infer_wait=47.975s infer_obs_pack=0.670s env_step=6.284s feedback_total=3.125s feedback_obs_pack=0.848s feedback_info_pack=0.184s effective_fps=661.14 +2026-09-27 18:06:52.118 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=57687 infer_calls=57687 feedback_calls=57687 infer_wait=71.807s infer_obs_pack=1.014s env_step=9.498s feedback_total=4.755s feedback_obs_pack=1.288s feedback_info_pack=0.278s effective_fps=662.51 +2026-09-27 18:07:22.119 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=76693 infer_calls=76693 feedback_calls=76693 infer_wait=95.932s infer_obs_pack=1.338s env_step=12.532s feedback_total=6.336s feedback_obs_pack=1.718s feedback_info_pack=0.369s effective_fps=660.37 +2026-09-27 18:07:52.119 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=95451 infer_calls=95451 feedback_calls=95451 infer_wait=120.212s infer_obs_pack=1.657s env_step=15.485s feedback_total=7.863s feedback_obs_pack=2.133s feedback_info_pack=0.459s effective_fps=657.30 +2026-09-27 18:08:22.120 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=113985 infer_calls=113985 feedback_calls=113985 infer_wait=144.542s infer_obs_pack=1.971s env_step=18.420s feedback_total=9.376s feedback_obs_pack=2.547s feedback_info_pack=0.546s effective_fps=653.93 +2026-09-27 18:08:52.121 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=131500 infer_calls=131500 feedback_calls=131500 infer_wait=169.346s infer_obs_pack=2.260s env_step=21.095s feedback_total=10.754s feedback_obs_pack=2.922s feedback_info_pack=0.626s effective_fps=646.33 +2026-09-27 18:09:22.121 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=147256 infer_calls=147256 feedback_calls=147256 infer_wait=194.968s infer_obs_pack=2.504s env_step=23.352s feedback_total=11.917s feedback_obs_pack=3.230s feedback_info_pack=0.692s effective_fps=632.70 +2026-09-27 18:09:52.122 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=162924 infer_calls=162924 feedback_calls=162924 infer_wait=220.627s infer_obs_pack=2.747s env_step=25.578s feedback_total=13.079s feedback_obs_pack=3.544s feedback_info_pack=0.757s effective_fps=621.77 +2026-09-27 18:10:22.124 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=178723 infer_calls=178723 feedback_calls=178723 infer_wait=246.270s infer_obs_pack=2.989s env_step=27.822s feedback_total=14.234s feedback_obs_pack=3.861s feedback_info_pack=0.821s effective_fps=613.50 +2026-09-27 18:10:52.125 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=194353 infer_calls=194353 feedback_calls=194353 infer_wait=271.941s infer_obs_pack=3.234s env_step=30.049s feedback_total=15.381s feedback_obs_pack=4.166s feedback_info_pack=0.888s effective_fps=606.21 +2026-09-27 18:11:22.126 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=210069 infer_calls=210069 feedback_calls=210069 infer_wait=297.602s infer_obs_pack=3.484s env_step=32.286s feedback_total=16.528s feedback_obs_pack=4.481s feedback_info_pack=0.953s effective_fps=600.37 +2026-09-27 18:11:52.126 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=225784 infer_calls=225784 feedback_calls=225784 infer_wait=323.304s infer_obs_pack=3.721s env_step=34.491s feedback_total=17.669s feedback_obs_pack=4.792s feedback_info_pack=1.020s effective_fps=595.45 +2026-09-27 18:12:22.154 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=241666 infer_calls=241666 feedback_calls=241666 infer_wait=348.932s infer_obs_pack=3.969s env_step=36.753s feedback_total=18.833s feedback_obs_pack=5.112s feedback_info_pack=1.086s effective_fps=591.61 +2026-09-27 18:12:52.156 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=258571 infer_calls=258571 feedback_calls=258571 infer_wait=374.152s infer_obs_pack=4.238s env_step=39.212s feedback_total=20.097s feedback_obs_pack=5.465s feedback_info_pack=1.161s effective_fps=590.75 +2026-09-27 18:13:22.156 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=277147 infer_calls=277147 feedback_calls=277147 infer_wait=398.541s infer_obs_pack=4.552s env_step=42.082s feedback_total=21.620s feedback_obs_pack=5.877s feedback_info_pack=1.251s effective_fps=593.72 +2026-09-27 18:13:52.156 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=295520 infer_calls=295520 feedback_calls=295520 infer_wait=422.992s infer_obs_pack=4.865s env_step=44.927s feedback_total=23.098s feedback_obs_pack=6.280s feedback_info_pack=1.335s effective_fps=595.95 +2026-09-27 18:14:22.156 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=314056 infer_calls=314056 feedback_calls=314056 infer_wait=447.426s infer_obs_pack=5.183s env_step=47.788s feedback_total=24.581s feedback_obs_pack=6.693s feedback_info_pack=1.424s effective_fps=598.23 +2026-09-27 18:14:52.156 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=332414 infer_calls=332414 feedback_calls=332414 infer_wait=471.924s infer_obs_pack=5.491s env_step=50.622s feedback_total=26.056s feedback_obs_pack=7.100s feedback_info_pack=1.511s effective_fps=599.93 +2026-09-27 18:15:22.157 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=352256 infer_calls=352256 feedback_calls=352256 infer_wait=495.825s infer_obs_pack=5.835s env_step=53.754s feedback_total=27.697s feedback_obs_pack=7.547s feedback_info_pack=1.604s effective_fps=604.10 +2026-09-27 18:15:52.158 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=371609 infer_calls=371609 feedback_calls=371609 infer_wait=519.802s infer_obs_pack=6.166s env_step=56.847s feedback_total=29.334s feedback_obs_pack=7.986s feedback_info_pack=1.697s effective_fps=607.06 +2026-09-27 18:16:22.453 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=391170 infer_calls=391170 feedback_calls=391170 infer_wait=543.967s infer_obs_pack=6.503s env_step=59.982s feedback_total=31.001s feedback_obs_pack=8.433s feedback_info_pack=1.796s effective_fps=609.82 +2026-09-27 18:16:52.454 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=410507 infer_calls=410507 feedback_calls=410507 infer_wait=567.869s infer_obs_pack=6.841s env_step=63.099s feedback_total=32.673s feedback_obs_pack=8.891s feedback_info_pack=1.889s effective_fps=612.26 +2026-09-27 18:17:22.623 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=430082 infer_calls=430082 feedback_calls=430082 infer_wait=591.788s infer_obs_pack=7.196s env_step=66.274s feedback_total=34.386s feedback_obs_pack=9.348s feedback_info_pack=1.986s effective_fps=614.72 +2026-09-27 18:17:52.623 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=449977 infer_calls=449977 feedback_calls=449977 infer_wait=615.508s infer_obs_pack=7.535s env_step=69.500s feedback_total=36.093s feedback_obs_pack=9.812s feedback_info_pack=2.084s effective_fps=617.56 +2026-09-27 18:18:22.623 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=469320 infer_calls=469320 feedback_calls=469320 infer_wait=639.482s infer_obs_pack=7.874s env_step=72.581s feedback_total=37.729s feedback_obs_pack=10.265s feedback_info_pack=2.179s effective_fps=619.43 +2026-09-27 18:18:52.624 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=488814 infer_calls=488814 feedback_calls=488814 infer_wait=663.378s infer_obs_pack=8.211s env_step=75.708s feedback_total=39.386s feedback_obs_pack=10.718s feedback_info_pack=2.278s effective_fps=621.36 +2026-09-27 18:19:22.625 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=508279 infer_calls=508279 feedback_calls=508279 infer_wait=687.325s infer_obs_pack=8.553s env_step=78.822s feedback_total=41.014s feedback_obs_pack=11.168s feedback_info_pack=2.372s effective_fps=623.11 +2026-09-27 18:19:52.626 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=527941 infer_calls=527941 feedback_calls=527941 infer_wait=711.170s infer_obs_pack=8.901s env_step=81.990s feedback_total=42.671s feedback_obs_pack=11.627s feedback_info_pack=2.472s effective_fps=624.98 +2026-09-27 18:20:22.628 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=546932 infer_calls=546932 feedback_calls=546932 infer_wait=735.280s infer_obs_pack=9.233s env_step=85.015s feedback_total=44.258s feedback_obs_pack=12.068s feedback_info_pack=2.567s effective_fps=625.93 +2026-09-27 18:20:52.628 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=566000 infer_calls=566000 feedback_calls=566000 infer_wait=759.576s infer_obs_pack=9.552s env_step=87.941s feedback_total=45.793s feedback_obs_pack=12.490s feedback_info_pack=2.656s effective_fps=626.90 +2026-09-27 18:21:22.630 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=584390 infer_calls=584390 feedback_calls=584390 infer_wait=784.080s infer_obs_pack=9.867s env_step=90.758s feedback_total=47.263s feedback_obs_pack=12.892s feedback_info_pack=2.741s effective_fps=627.05 +2026-09-27 18:21:52.631 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=602766 infer_calls=602766 feedback_calls=602766 infer_wait=808.564s infer_obs_pack=10.170s env_step=93.591s feedback_total=48.748s feedback_obs_pack=13.306s feedback_info_pack=2.826s effective_fps=627.18 +2026-09-27 18:22:22.631 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=618580 infer_calls=618580 feedback_calls=618580 infer_wait=834.225s infer_obs_pack=10.413s env_step=95.822s feedback_total=49.898s feedback_obs_pack=13.616s feedback_info_pack=2.895s effective_fps=624.60 +2026-09-27 18:22:52.631 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=634657 infer_calls=634657 feedback_calls=634657 infer_wait=859.761s infer_obs_pack=10.666s env_step=98.116s feedback_total=51.082s feedback_obs_pack=13.936s feedback_info_pack=2.964s effective_fps=622.44 +2026-09-27 18:23:22.631 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=650462 infer_calls=650462 feedback_calls=650462 infer_wait=885.384s infer_obs_pack=10.914s env_step=100.357s feedback_total=52.233s feedback_obs_pack=14.252s feedback_info_pack=3.032s effective_fps=620.14 +2026-09-27 18:23:52.634 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=666316 infer_calls=666316 feedback_calls=666316 infer_wait=911.038s infer_obs_pack=11.156s env_step=102.576s feedback_total=53.397s feedback_obs_pack=14.568s feedback_info_pack=3.099s effective_fps=618.01 +2026-09-27 18:24:22.634 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=682158 infer_calls=682158 feedback_calls=682158 infer_wait=936.649s infer_obs_pack=11.401s env_step=104.821s feedback_total=54.566s feedback_obs_pack=14.882s feedback_info_pack=3.164s effective_fps=615.98 +2026-09-27 18:24:52.634 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=698258 infer_calls=698258 feedback_calls=698258 infer_wait=962.214s infer_obs_pack=11.652s env_step=107.083s feedback_total=55.758s feedback_obs_pack=15.205s feedback_info_pack=3.232s effective_fps=614.28 +2026-09-27 18:25:22.635 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=714004 infer_calls=714004 feedback_calls=714004 infer_wait=987.821s infer_obs_pack=11.900s env_step=109.334s feedback_total=56.928s feedback_obs_pack=15.524s feedback_info_pack=3.298s effective_fps=612.36 +2026-09-27 18:25:52.635 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=730520 infer_calls=730520 feedback_calls=730520 infer_wait=1013.101s infer_obs_pack=12.165s env_step=111.748s feedback_total=58.192s feedback_obs_pack=15.872s feedback_info_pack=3.373s effective_fps=611.21 +2026-09-27 18:26:22.637 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=748790 infer_calls=748790 feedback_calls=748790 infer_wait=1037.539s infer_obs_pack=12.471s env_step=114.593s feedback_total=59.696s feedback_obs_pack=16.282s feedback_info_pack=3.459s effective_fps=611.61 +2026-09-27 18:26:52.638 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=767171 infer_calls=767171 feedback_calls=767171 infer_wait=1061.980s infer_obs_pack=12.780s env_step=117.445s feedback_total=61.193s feedback_obs_pack=16.691s feedback_info_pack=3.546s effective_fps=612.07 +2026-09-27 18:27:22.638 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=785951 infer_calls=785951 feedback_calls=785951 infer_wait=1086.277s infer_obs_pack=13.099s env_step=120.368s feedback_total=62.724s feedback_obs_pack=17.109s feedback_info_pack=3.633s effective_fps=612.84 +2026-09-27 18:27:52.825 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=804866 infer_calls=804866 feedback_calls=804866 infer_wait=1110.833s infer_obs_pack=13.415s env_step=123.252s feedback_total=64.257s feedback_obs_pack=17.529s feedback_info_pack=3.718s effective_fps=613.58 +2026-09-27 18:28:22.826 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=824111 infer_calls=824111 feedback_calls=824111 infer_wait=1134.852s infer_obs_pack=13.751s env_step=126.330s feedback_total=65.868s feedback_obs_pack=17.968s feedback_info_pack=3.812s effective_fps=614.64 +2026-09-27 18:28:52.827 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=843909 infer_calls=843909 feedback_calls=843909 infer_wait=1158.729s infer_obs_pack=14.092s env_step=129.476s feedback_total=67.525s feedback_obs_pack=18.423s feedback_info_pack=3.909s effective_fps=616.07 +2026-09-27 18:29:22.828 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=863517 infer_calls=863517 feedback_calls=863517 infer_wait=1182.567s infer_obs_pack=14.438s env_step=132.631s feedback_total=69.206s feedback_obs_pack=18.882s feedback_info_pack=4.010s effective_fps=617.31 +2026-09-27 18:29:52.829 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=882786 infer_calls=882786 feedback_calls=882786 infer_wait=1206.467s infer_obs_pack=14.783s env_step=135.750s feedback_total=70.866s feedback_obs_pack=19.334s feedback_info_pack=4.106s effective_fps=618.26 +2026-09-27 18:30:22.830 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=901935 infer_calls=901935 feedback_calls=901935 infer_wait=1230.304s infer_obs_pack=15.119s env_step=138.919s feedback_total=72.547s feedback_obs_pack=19.791s feedback_info_pack=4.202s effective_fps=619.08 +2026-09-27 18:30:52.830 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=921491 infer_calls=921491 feedback_calls=921491 infer_wait=1254.115s infer_obs_pack=15.458s env_step=142.092s feedback_total=74.237s feedback_obs_pack=20.240s feedback_info_pack=4.299s effective_fps=620.16 +2026-09-27 18:31:22.830 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=941451 infer_calls=941451 feedback_calls=941451 infer_wait=1277.948s infer_obs_pack=15.808s env_step=145.245s feedback_total=75.926s feedback_obs_pack=20.692s feedback_info_pack=4.395s effective_fps=621.45 +2026-09-27 18:31:52.832 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=960864 infer_calls=960864 feedback_calls=960864 infer_wait=1301.841s infer_obs_pack=16.155s env_step=148.356s feedback_total=77.598s feedback_obs_pack=21.145s feedback_info_pack=4.493s effective_fps=622.34 +2026-09-27 18:32:22.832 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=980001 infer_calls=980001 feedback_calls=980001 infer_wait=1325.812s infer_obs_pack=16.499s env_step=151.434s feedback_total=79.227s feedback_obs_pack=21.583s feedback_info_pack=4.593s effective_fps=623.02 +2026-09-27 18:32:52.833 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=999100 infer_calls=999100 feedback_calls=999100 infer_wait=1349.756s infer_obs_pack=16.840s env_step=154.542s feedback_total=80.863s feedback_obs_pack=22.030s feedback_info_pack=4.692s effective_fps=623.66 +2026-09-27 18:32:53.660 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Final rollout timing summary: env_steps=999425 infer_calls=999425 feedback_calls=999425 infer_wait=1350.151s infer_obs_pack=16.845s env_step=154.597s feedback_total=80.892s feedback_obs_pack=22.037s feedback_info_pack=4.694s effective_fps=623.67 +2026-09-27 18:32:53.660 | INFO | plugrl_env_client.runner.run:run:156 - Server signalled the end of the run; collection stopped after 2750/3997696 episodes. diff --git a/experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed1.log b/experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed1.log new file mode 100644 index 0000000..c6f19c9 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed1.log @@ -0,0 +1,65 @@ +2026-09-27 18:05:27.025 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.classic.classic_env: pygame is not installed. Please install it with pip install "plugrl-env-client[classic]". +2026-09-27 18:05:27.026 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.atari.atari_env: Atari is not installed. Please install it with the 'atari' extra, e.g. 'pip install plugrl-env-client[atari]' +2026-09-27 18:05:27.026 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.d4rl.d4rl_env: d4rl is not installed. Please install it with pip install "plugrl-env-client[d4rl]". +2026-09-27 18:05:27.026 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.libero.libero_env: libero is not installed. Please install it with pip install "plugrl-env-client[libero]". +2026-09-27 18:05:27.027 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.robomimic.robomimic_env: Robomimic is not installed. Please install it with the 'robomimic' extra, e.g. 'pip install plugrl-env-client[robomimic]' +2026-09-27 18:05:27.072 | INFO | __main__:main:83 - Starting env client exp_name=mujoco-v1-nenv1-20260927-180527-82897ffb output_dir=runs/mujoco-v1-nenv1-20260927-180527-82897ffb recorder={'episode_freq': 0, 'thread0_only': True, 'record_video': False, 'video_fps': 30.0, 'record_full_rollout': False, 'record_obs_stats': True, 'record_episode_metrics': True, 'metric_window': 100} +2026-09-27 18:05:27.112 | INFO | plugrl_env_client.agent.websocket_env_client_agent:_wait_for_server:67 - Waiting for server at ws://127.0.0.1:9951... +2026-09-27 18:05:27.115 | INFO | plugrl_env_client.runner.run:report_server_metadata:56 - Server metadata: {'protocol_version': 1, 'server': 'plugrl-server', 'server_version': '0.1.0', 'algorithm': 'PPOAlgorithm', 'policy': 'GaussianPolicy', 'action_dim': 3, 'action_horizon': 1} +2026-09-27 18:05:57.117 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=18860 infer_calls=18860 feedback_calls=18860 infer_wait=24.117s infer_obs_pack=0.334s env_step=3.057s feedback_total=1.530s feedback_obs_pack=0.411s feedback_info_pack=0.085s effective_fps=649.51 +2026-09-27 18:06:27.118 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=38654 infer_calls=38654 feedback_calls=38654 infer_wait=47.729s infer_obs_pack=0.686s env_step=6.394s feedback_total=3.213s feedback_obs_pack=0.869s feedback_info_pack=0.181s effective_fps=666.19 +2026-09-27 18:06:57.119 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=57918 infer_calls=57918 feedback_calls=57918 infer_wait=71.493s infer_obs_pack=1.029s env_step=9.673s feedback_total=4.846s feedback_obs_pack=1.322s feedback_info_pack=0.270s effective_fps=665.41 +2026-09-27 18:07:27.119 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=76524 infer_calls=76524 feedback_calls=76524 infer_wait=95.704s infer_obs_pack=1.343s env_step=12.708s feedback_total=6.368s feedback_obs_pack=1.734s feedback_info_pack=0.352s effective_fps=658.99 +2026-09-27 18:07:57.120 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=95060 infer_calls=95060 feedback_calls=95060 infer_wait=119.886s infer_obs_pack=1.655s env_step=15.755s feedback_total=7.907s feedback_obs_pack=2.156s feedback_info_pack=0.434s effective_fps=654.67 +2026-09-27 18:08:27.120 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=113237 infer_calls=113237 feedback_calls=113237 infer_wait=144.202s infer_obs_pack=1.967s env_step=18.728s feedback_total=9.391s feedback_obs_pack=2.563s feedback_info_pack=0.514s effective_fps=649.71 +2026-09-27 18:08:57.120 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=130266 infer_calls=130266 feedback_calls=130266 infer_wait=169.203s infer_obs_pack=2.239s env_step=21.318s feedback_total=10.712s feedback_obs_pack=2.929s feedback_info_pack=0.587s effective_fps=640.21 +2026-09-27 18:09:27.121 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=145964 infer_calls=145964 feedback_calls=145964 infer_wait=194.827s infer_obs_pack=2.483s env_step=23.577s feedback_total=11.865s feedback_obs_pack=3.248s feedback_info_pack=0.649s effective_fps=627.12 +2026-09-27 18:09:57.122 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=161628 infer_calls=161628 feedback_calls=161628 infer_wait=220.450s infer_obs_pack=2.726s env_step=25.831s feedback_total=13.018s feedback_obs_pack=3.559s feedback_info_pack=0.713s effective_fps=616.84 +2026-09-27 18:10:27.123 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=177096 infer_calls=177096 feedback_calls=177096 infer_wait=246.137s infer_obs_pack=2.964s env_step=28.037s feedback_total=14.174s feedback_obs_pack=3.874s feedback_info_pack=0.777s effective_fps=607.93 +2026-09-27 18:10:57.123 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=192551 infer_calls=192551 feedback_calls=192551 infer_wait=271.827s infer_obs_pack=3.207s env_step=30.247s feedback_total=15.315s feedback_obs_pack=4.187s feedback_info_pack=0.838s effective_fps=600.60 +2026-09-27 18:11:27.124 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=208242 infer_calls=208242 feedback_calls=208242 infer_wait=297.469s infer_obs_pack=3.448s env_step=32.477s feedback_total=16.485s feedback_obs_pack=4.507s feedback_info_pack=0.900s effective_fps=595.18 +2026-09-27 18:11:57.126 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=224023 infer_calls=224023 feedback_calls=224023 infer_wait=323.118s infer_obs_pack=3.685s env_step=34.707s feedback_total=17.650s feedback_obs_pack=4.824s feedback_info_pack=0.966s effective_fps=590.84 +2026-09-27 18:12:27.174 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=239618 infer_calls=239618 feedback_calls=239618 infer_wait=348.841s infer_obs_pack=3.926s env_step=36.923s feedback_total=18.806s feedback_obs_pack=5.142s feedback_info_pack=1.030s effective_fps=586.59 +2026-09-27 18:12:57.174 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=256805 infer_calls=256805 feedback_calls=256805 infer_wait=373.764s infer_obs_pack=4.213s env_step=39.512s feedback_total=20.184s feedback_obs_pack=5.517s feedback_info_pack=1.104s effective_fps=586.75 +2026-09-27 18:13:27.174 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=275451 infer_calls=275451 feedback_calls=275451 infer_wait=398.176s infer_obs_pack=4.523s env_step=42.394s feedback_total=21.682s feedback_obs_pack=5.924s feedback_info_pack=1.187s effective_fps=590.12 +2026-09-27 18:13:57.175 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=293890 infer_calls=293890 feedback_calls=293890 infer_wait=422.526s infer_obs_pack=4.835s env_step=45.294s feedback_total=23.212s feedback_obs_pack=6.345s feedback_info_pack=1.270s effective_fps=592.68 +2026-09-27 18:14:27.175 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=312080 infer_calls=312080 feedback_calls=312080 infer_wait=447.008s infer_obs_pack=5.143s env_step=48.107s feedback_total=24.712s feedback_obs_pack=6.754s feedback_info_pack=1.350s effective_fps=594.47 +2026-09-27 18:14:57.177 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=330448 infer_calls=330448 feedback_calls=330448 infer_wait=471.494s infer_obs_pack=5.456s env_step=50.930s feedback_total=26.193s feedback_obs_pack=7.165s feedback_info_pack=1.430s effective_fps=596.40 +2026-09-27 18:15:27.179 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=349997 infer_calls=349997 feedback_calls=349997 infer_wait=495.445s infer_obs_pack=5.794s env_step=54.039s feedback_total=27.830s feedback_obs_pack=7.619s feedback_info_pack=1.520s effective_fps=600.23 +2026-09-27 18:15:57.181 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=369430 infer_calls=369430 feedback_calls=369430 infer_wait=519.336s infer_obs_pack=6.131s env_step=57.181s feedback_total=29.493s feedback_obs_pack=8.074s feedback_info_pack=1.612s effective_fps=603.50 +2026-09-27 18:16:27.181 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=389142 infer_calls=389142 feedback_calls=389142 infer_wait=543.265s infer_obs_pack=6.476s env_step=60.302s feedback_total=31.124s feedback_obs_pack=8.523s feedback_info_pack=1.699s effective_fps=606.93 +2026-09-27 18:16:57.182 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=408952 infer_calls=408952 feedback_calls=408952 infer_wait=566.975s infer_obs_pack=6.831s env_step=63.515s feedback_total=32.843s feedback_obs_pack=8.985s feedback_info_pack=1.792s effective_fps=610.23 +2026-09-27 18:17:27.182 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=428177 infer_calls=428177 feedback_calls=428177 infer_wait=590.768s infer_obs_pack=7.177s env_step=66.700s feedback_total=34.530s feedback_obs_pack=9.442s feedback_info_pack=1.883s effective_fps=612.40 +2026-09-27 18:17:57.183 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=447715 infer_calls=447715 feedback_calls=447715 infer_wait=614.519s infer_obs_pack=7.523s env_step=69.890s feedback_total=36.241s feedback_obs_pack=9.904s feedback_info_pack=1.976s effective_fps=614.85 +2026-09-27 18:18:27.221 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=466946 infer_calls=466946 feedback_calls=466946 infer_wait=638.566s infer_obs_pack=7.860s env_step=72.957s feedback_total=37.866s feedback_obs_pack=10.339s feedback_info_pack=2.066s effective_fps=616.63 +2026-09-27 18:18:57.221 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=486709 infer_calls=486709 feedback_calls=486709 infer_wait=662.372s infer_obs_pack=8.201s env_step=76.146s feedback_total=39.542s feedback_obs_pack=10.790s feedback_info_pack=2.158s effective_fps=619.02 +2026-09-27 18:19:27.223 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=506036 infer_calls=506036 feedback_calls=506036 infer_wait=686.342s infer_obs_pack=8.540s env_step=79.232s feedback_total=41.182s feedback_obs_pack=11.244s feedback_info_pack=2.250s effective_fps=620.68 +2026-09-27 18:19:57.223 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=526054 infer_calls=526054 feedback_calls=526054 infer_wait=710.124s infer_obs_pack=8.891s env_step=82.401s feedback_total=42.883s feedback_obs_pack=11.711s feedback_info_pack=2.347s effective_fps=623.07 +2026-09-27 18:20:27.224 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=545058 infer_calls=545058 feedback_calls=545058 infer_wait=734.260s infer_obs_pack=9.224s env_step=85.402s feedback_total=44.469s feedback_obs_pack=12.133s feedback_info_pack=2.433s effective_fps=624.10 +2026-09-27 18:20:57.225 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=563455 infer_calls=563455 feedback_calls=563455 infer_wait=758.736s infer_obs_pack=9.537s env_step=88.228s feedback_total=45.951s feedback_obs_pack=12.546s feedback_info_pack=2.513s effective_fps=624.36 +2026-09-27 18:21:27.225 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=582134 infer_calls=582134 feedback_calls=582134 infer_wait=783.105s infer_obs_pack=9.852s env_step=91.112s feedback_total=47.462s feedback_obs_pack=12.956s feedback_info_pack=2.597s effective_fps=624.92 +2026-09-27 18:21:57.227 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=600241 infer_calls=600241 feedback_calls=600241 infer_wait=807.649s infer_obs_pack=10.160s env_step=93.891s feedback_total=48.938s feedback_obs_pack=13.357s feedback_info_pack=2.674s effective_fps=624.83 +2026-09-27 18:22:27.229 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=615975 infer_calls=615975 feedback_calls=615975 infer_wait=833.304s infer_obs_pack=10.407s env_step=96.090s feedback_total=50.112s feedback_obs_pack=13.676s feedback_info_pack=2.736s effective_fps=622.25 +2026-09-27 18:22:57.231 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=631642 infer_calls=631642 feedback_calls=631642 infer_wait=859.042s infer_obs_pack=10.645s env_step=98.255s feedback_total=51.259s feedback_obs_pack=13.984s feedback_info_pack=2.796s effective_fps=619.74 +2026-09-27 18:23:27.231 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=647357 infer_calls=647357 feedback_calls=647357 infer_wait=884.705s infer_obs_pack=10.892s env_step=100.465s feedback_total=52.421s feedback_obs_pack=14.300s feedback_info_pack=2.859s effective_fps=617.42 +2026-09-27 18:23:57.232 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=663114 infer_calls=663114 feedback_calls=663114 infer_wait=910.289s infer_obs_pack=11.143s env_step=102.721s feedback_total=53.597s feedback_obs_pack=14.621s feedback_info_pack=2.926s effective_fps=615.28 +2026-09-27 18:24:27.232 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=678740 infer_calls=678740 feedback_calls=678740 infer_wait=936.000s infer_obs_pack=11.385s env_step=104.903s feedback_total=54.748s feedback_obs_pack=14.934s feedback_info_pack=2.989s effective_fps=613.11 +2026-09-27 18:24:57.233 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=694426 infer_calls=694426 feedback_calls=694426 infer_wait=961.623s infer_obs_pack=11.635s env_step=107.127s feedback_total=55.926s feedback_obs_pack=15.259s feedback_info_pack=3.051s effective_fps=611.12 +2026-09-27 18:25:27.235 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=710288 infer_calls=710288 feedback_calls=710288 infer_wait=987.226s infer_obs_pack=11.888s env_step=109.359s feedback_total=57.103s feedback_obs_pack=15.586s feedback_info_pack=3.114s effective_fps=609.39 +2026-09-27 18:25:57.431 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=727042 infer_calls=727042 feedback_calls=727042 infer_wait=1012.570s infer_obs_pack=12.163s env_step=111.839s feedback_total=58.403s feedback_obs_pack=15.941s feedback_info_pack=3.185s effective_fps=608.42 +2026-09-27 18:26:27.431 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=745607 infer_calls=745607 feedback_calls=745607 infer_wait=1036.922s infer_obs_pack=12.488s env_step=114.740s feedback_total=59.916s feedback_obs_pack=16.356s feedback_info_pack=3.271s effective_fps=609.12 +2026-09-27 18:26:57.433 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=763943 infer_calls=763943 feedback_calls=763943 infer_wait=1061.407s infer_obs_pack=12.801s env_step=117.566s feedback_total=61.410s feedback_obs_pack=16.769s feedback_info_pack=3.355s effective_fps=609.60 +2026-09-27 18:27:27.435 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=782818 infer_calls=782818 feedback_calls=782818 infer_wait=1085.778s infer_obs_pack=13.122s env_step=120.438s feedback_total=62.942s feedback_obs_pack=17.181s feedback_info_pack=3.440s effective_fps=610.49 +2026-09-27 18:27:57.436 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=801195 infer_calls=801195 feedback_calls=801195 infer_wait=1110.262s infer_obs_pack=13.427s env_step=123.270s feedback_total=64.428s feedback_obs_pack=17.583s feedback_info_pack=3.519s effective_fps=610.95 +2026-09-27 18:28:27.437 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=820132 infer_calls=820132 feedback_calls=820132 infer_wait=1134.389s infer_obs_pack=13.751s env_step=126.283s feedback_total=66.018s feedback_obs_pack=18.014s feedback_info_pack=3.607s effective_fps=611.84 +2026-09-27 18:28:57.511 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=839682 infer_calls=839682 feedback_calls=839682 infer_wait=1158.402s infer_obs_pack=14.095s env_step=129.391s feedback_total=67.657s feedback_obs_pack=18.456s feedback_info_pack=3.696s effective_fps=613.11 +2026-09-27 18:29:27.511 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=859457 infer_calls=859457 feedback_calls=859457 infer_wait=1182.247s infer_obs_pack=14.441s env_step=132.541s feedback_total=69.330s feedback_obs_pack=18.912s feedback_info_pack=3.786s effective_fps=614.53 +2026-09-27 18:29:57.512 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=878693 infer_calls=878693 feedback_calls=878693 infer_wait=1206.111s infer_obs_pack=14.780s env_step=135.690s feedback_total=71.004s feedback_obs_pack=19.371s feedback_info_pack=3.877s effective_fps=615.51 +2026-09-27 18:30:27.512 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=898649 infer_calls=898649 feedback_calls=898649 infer_wait=1229.726s infer_obs_pack=15.139s env_step=138.953s feedback_total=72.766s feedback_obs_pack=19.851s feedback_info_pack=3.968s effective_fps=616.96 +2026-09-27 18:30:57.514 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=918283 infer_calls=918283 feedback_calls=918283 infer_wait=1253.554s infer_obs_pack=15.483s env_step=142.108s feedback_total=74.452s feedback_obs_pack=20.302s feedback_info_pack=4.060s effective_fps=618.12 +2026-09-27 18:31:27.515 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=937678 infer_calls=937678 feedback_calls=937678 infer_wait=1277.448s infer_obs_pack=15.833s env_step=145.227s feedback_total=76.112s feedback_obs_pack=20.750s feedback_info_pack=4.152s effective_fps=619.09 +2026-09-27 18:31:57.515 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=957230 infer_calls=957230 feedback_calls=957230 infer_wait=1301.361s infer_obs_pack=16.173s env_step=148.334s feedback_total=77.777s feedback_obs_pack=21.203s feedback_info_pack=4.245s effective_fps=620.11 +2026-09-27 18:32:27.550 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=976898 infer_calls=976898 feedback_calls=976898 infer_wait=1325.240s infer_obs_pack=16.518s env_step=151.485s feedback_total=79.452s feedback_obs_pack=21.658s feedback_info_pack=4.337s effective_fps=621.16 +2026-09-27 18:32:57.550 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=996250 infer_calls=996250 feedback_calls=996250 infer_wait=1349.225s infer_obs_pack=16.850s env_step=154.571s feedback_total=81.093s feedback_obs_pack=22.098s feedback_info_pack=4.429s effective_fps=621.98 +2026-09-27 18:33:02.890 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Final rollout timing summary: env_steps=999425 infer_calls=999425 feedback_calls=999425 infer_wait=1353.320s infer_obs_pack=16.902s env_step=155.040s feedback_total=81.340s feedback_obs_pack=22.165s feedback_info_pack=4.443s effective_fps=622.07 +2026-09-27 18:33:02.890 | INFO | plugrl_env_client.runner.run:run:156 - Server signalled the end of the run; collection stopped after 2784/3997696 episodes. diff --git a/experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed2.log b/experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed2.log new file mode 100644 index 0000000..b7e3ba4 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-hopper/client-seed2.log @@ -0,0 +1,64 @@ +2026-09-27 18:05:32.031 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.classic.classic_env: pygame is not installed. Please install it with pip install "plugrl-env-client[classic]". +2026-09-27 18:05:32.031 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.atari.atari_env: Atari is not installed. Please install it with the 'atari' extra, e.g. 'pip install plugrl-env-client[atari]' +2026-09-27 18:05:32.031 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.d4rl.d4rl_env: d4rl is not installed. Please install it with pip install "plugrl-env-client[d4rl]". +2026-09-27 18:05:32.032 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.libero.libero_env: libero is not installed. Please install it with pip install "plugrl-env-client[libero]". +2026-09-27 18:05:32.032 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.robomimic.robomimic_env: Robomimic is not installed. Please install it with the 'robomimic' extra, e.g. 'pip install plugrl-env-client[robomimic]' +2026-09-27 18:05:32.076 | INFO | __main__:main:83 - Starting env client exp_name=mujoco-v1-nenv1-20260927-180532-7ac69910 output_dir=runs/mujoco-v1-nenv1-20260927-180532-7ac69910 recorder={'episode_freq': 0, 'thread0_only': True, 'record_video': False, 'video_fps': 30.0, 'record_full_rollout': False, 'record_obs_stats': True, 'record_episode_metrics': True, 'metric_window': 100} +2026-09-27 18:05:32.118 | INFO | plugrl_env_client.agent.websocket_env_client_agent:_wait_for_server:67 - Waiting for server at ws://127.0.0.1:9952... +2026-09-27 18:05:32.121 | INFO | plugrl_env_client.runner.run:report_server_metadata:56 - Server metadata: {'protocol_version': 1, 'server': 'plugrl-server', 'server_version': '0.1.0', 'algorithm': 'PPOAlgorithm', 'policy': 'GaussianPolicy', 'action_dim': 3, 'action_horizon': 1} +2026-09-27 18:06:02.124 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=19292 infer_calls=19292 feedback_calls=19292 infer_wait=23.905s infer_obs_pack=0.344s env_step=3.176s feedback_total=1.577s feedback_obs_pack=0.420s feedback_info_pack=0.089s effective_fps=665.18 +2026-09-27 18:06:32.253 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=38914 infer_calls=38914 feedback_calls=38914 infer_wait=47.655s infer_obs_pack=0.681s env_step=6.535s feedback_total=3.245s feedback_obs_pack=0.867s feedback_info_pack=0.183s effective_fps=669.59 +2026-09-27 18:07:02.254 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=58151 infer_calls=58151 feedback_calls=58151 infer_wait=71.412s infer_obs_pack=1.014s env_step=9.805s feedback_total=4.885s feedback_obs_pack=1.312s feedback_info_pack=0.276s effective_fps=667.51 +2026-09-27 18:07:32.255 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=76299 infer_calls=76299 feedback_calls=76299 infer_wait=95.789s infer_obs_pack=1.316s env_step=12.716s feedback_total=6.383s feedback_obs_pack=1.708s feedback_info_pack=0.360s effective_fps=656.59 +2026-09-27 18:08:02.256 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=94758 infer_calls=94758 feedback_calls=94758 infer_wait=120.106s infer_obs_pack=1.621s env_step=15.685s feedback_total=7.871s feedback_obs_pack=2.110s feedback_info_pack=0.443s effective_fps=652.23 +2026-09-27 18:08:32.257 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=113505 infer_calls=113505 feedback_calls=113505 infer_wait=144.411s infer_obs_pack=1.931s env_step=18.642s feedback_total=9.379s feedback_obs_pack=2.512s feedback_info_pack=0.528s effective_fps=650.97 +2026-09-27 18:09:02.260 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=130430 infer_calls=130430 feedback_calls=130430 infer_wait=169.533s infer_obs_pack=2.199s env_step=21.173s feedback_total=10.663s feedback_obs_pack=2.853s feedback_info_pack=0.600s effective_fps=640.72 +2026-09-27 18:09:32.260 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=146470 infer_calls=146470 feedback_calls=146470 infer_wait=195.119s infer_obs_pack=2.442s env_step=23.441s feedback_total=11.842s feedback_obs_pack=3.169s feedback_info_pack=0.663s effective_fps=629.05 +2026-09-27 18:10:02.261 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=162341 infer_calls=162341 feedback_calls=162341 infer_wait=220.707s infer_obs_pack=2.686s env_step=25.723s feedback_total=13.000s feedback_obs_pack=3.482s feedback_info_pack=0.729s effective_fps=619.35 +2026-09-27 18:10:32.263 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=178437 infer_calls=178437 feedback_calls=178437 infer_wait=246.265s infer_obs_pack=2.928s env_step=27.999s feedback_total=14.194s feedback_obs_pack=3.801s feedback_info_pack=0.794s effective_fps=612.37 +2026-09-27 18:11:02.424 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=194562 infer_calls=194562 feedback_calls=194562 infer_wait=271.960s infer_obs_pack=3.178s env_step=30.288s feedback_total=15.383s feedback_obs_pack=4.119s feedback_info_pack=0.858s effective_fps=606.47 +2026-09-27 18:11:32.425 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=210747 infer_calls=210747 feedback_calls=210747 infer_wait=297.492s infer_obs_pack=3.429s env_step=32.585s feedback_total=16.565s feedback_obs_pack=4.443s feedback_info_pack=0.923s effective_fps=602.01 +2026-09-27 18:12:02.426 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=226878 infer_calls=226878 feedback_calls=226878 infer_wait=323.066s infer_obs_pack=3.676s env_step=34.858s feedback_total=17.742s feedback_obs_pack=4.757s feedback_info_pack=0.988s effective_fps=598.08 +2026-09-27 18:12:32.426 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=243112 infer_calls=243112 feedback_calls=243112 infer_wait=348.559s infer_obs_pack=3.929s env_step=37.180s feedback_total=18.934s feedback_obs_pack=5.079s feedback_info_pack=1.053s effective_fps=594.99 +2026-09-27 18:13:02.427 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=261396 infer_calls=261396 feedback_calls=261396 infer_wait=373.166s infer_obs_pack=4.227s env_step=39.980s feedback_total=20.358s feedback_obs_pack=5.461s feedback_info_pack=1.132s effective_fps=597.16 +2026-09-27 18:13:32.428 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=280038 infer_calls=280038 feedback_calls=280038 infer_wait=397.484s infer_obs_pack=4.536s env_step=42.912s feedback_total=21.882s feedback_obs_pack=5.865s feedback_info_pack=1.217s effective_fps=599.89 +2026-09-27 18:14:02.428 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=298822 infer_calls=298822 feedback_calls=298822 infer_wait=421.804s infer_obs_pack=4.852s env_step=45.855s feedback_total=23.385s feedback_obs_pack=6.266s feedback_info_pack=1.302s effective_fps=602.59 +2026-09-27 18:14:32.648 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=317442 infer_calls=317442 feedback_calls=317442 infer_wait=446.328s infer_obs_pack=5.165s env_step=48.787s feedback_total=24.915s feedback_obs_pack=6.676s feedback_info_pack=1.384s effective_fps=604.43 +2026-09-27 18:15:02.648 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=336220 infer_calls=336220 feedback_calls=336220 infer_wait=470.624s infer_obs_pack=5.481s env_step=51.735s feedback_total=26.443s feedback_obs_pack=7.085s feedback_info_pack=1.480s effective_fps=606.59 +2026-09-27 18:15:32.648 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=355695 infer_calls=355695 feedback_calls=355695 infer_wait=494.473s infer_obs_pack=5.812s env_step=54.902s feedback_total=28.110s feedback_obs_pack=7.522s feedback_info_pack=1.590s effective_fps=609.80 +2026-09-27 18:16:02.650 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=375266 infer_calls=375266 feedback_calls=375266 infer_wait=518.294s infer_obs_pack=6.154s env_step=58.097s feedback_total=29.767s feedback_obs_pack=7.965s feedback_info_pack=1.698s effective_fps=612.87 +2026-09-27 18:16:32.651 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=394972 infer_calls=394972 feedback_calls=394972 infer_wait=542.211s infer_obs_pack=6.496s env_step=61.243s feedback_total=31.398s feedback_obs_pack=8.399s feedback_info_pack=1.804s effective_fps=615.85 +2026-09-27 18:17:02.651 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=414229 infer_calls=414229 feedback_calls=414229 infer_wait=566.062s infer_obs_pack=6.837s env_step=64.396s feedback_total=33.071s feedback_obs_pack=8.849s feedback_info_pack=1.910s effective_fps=617.92 +2026-09-27 18:17:32.652 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=433601 infer_calls=433601 feedback_calls=433601 infer_wait=589.849s infer_obs_pack=7.175s env_step=67.585s feedback_total=34.754s feedback_obs_pack=9.292s feedback_info_pack=2.020s effective_fps=619.99 +2026-09-27 18:18:02.654 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=452972 infer_calls=452972 feedback_calls=452972 infer_wait=613.691s infer_obs_pack=7.520s env_step=70.739s feedback_total=36.429s feedback_obs_pack=9.740s feedback_info_pack=2.126s effective_fps=621.89 +2026-09-27 18:18:32.654 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=472487 infer_calls=472487 feedback_calls=472487 infer_wait=637.578s infer_obs_pack=7.861s env_step=73.868s feedback_total=38.092s feedback_obs_pack=10.182s feedback_info_pack=2.232s effective_fps=623.83 +2026-09-27 18:19:02.654 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=491796 infer_calls=491796 feedback_calls=491796 infer_wait=661.585s infer_obs_pack=8.196s env_step=76.947s feedback_total=39.717s feedback_obs_pack=10.615s feedback_info_pack=2.339s effective_fps=625.34 +2026-09-27 18:19:32.656 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=511027 infer_calls=511027 feedback_calls=511027 infer_wait=685.524s infer_obs_pack=8.528s env_step=80.064s feedback_total=41.354s feedback_obs_pack=11.044s feedback_info_pack=2.446s effective_fps=626.67 +2026-09-27 18:20:02.656 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=530586 infer_calls=530586 feedback_calls=530586 infer_wait=709.417s infer_obs_pack=8.867s env_step=83.215s feedback_total=43.005s feedback_obs_pack=11.491s feedback_info_pack=2.551s effective_fps=628.28 +2026-09-27 18:20:32.658 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=549697 infer_calls=549697 feedback_calls=549697 infer_wait=733.535s infer_obs_pack=9.197s env_step=86.231s feedback_total=44.604s feedback_obs_pack=11.911s feedback_info_pack=2.651s effective_fps=629.26 +2026-09-27 18:21:02.659 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=568330 infer_calls=568330 feedback_calls=568330 infer_wait=757.953s infer_obs_pack=9.504s env_step=89.085s feedback_total=46.123s feedback_obs_pack=12.321s feedback_info_pack=2.746s effective_fps=629.61 +2026-09-27 18:21:32.660 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=586795 infer_calls=586795 feedback_calls=586795 infer_wait=782.407s infer_obs_pack=9.814s env_step=91.927s feedback_total=47.619s feedback_obs_pack=12.724s feedback_info_pack=2.840s effective_fps=629.77 +2026-09-27 18:22:02.661 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=604587 infer_calls=604587 feedback_calls=604587 infer_wait=807.186s infer_obs_pack=10.106s env_step=94.599s feedback_total=49.021s feedback_obs_pack=13.098s feedback_info_pack=2.930s effective_fps=629.18 +2026-09-27 18:22:32.663 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=620641 infer_calls=620641 feedback_calls=620641 infer_wait=832.809s infer_obs_pack=10.348s env_step=96.838s feedback_total=50.193s feedback_obs_pack=13.411s feedback_info_pack=3.002s effective_fps=626.79 +2026-09-27 18:23:02.753 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=636930 infer_calls=636930 feedback_calls=636930 infer_wait=858.453s infer_obs_pack=10.598s env_step=99.103s feedback_total=51.392s feedback_obs_pack=13.726s feedback_info_pack=3.075s effective_fps=624.72 +2026-09-27 18:23:32.924 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=653314 infer_calls=653314 feedback_calls=653314 infer_wait=884.152s infer_obs_pack=10.851s env_step=101.386s feedback_total=52.579s feedback_obs_pack=14.052s feedback_info_pack=3.148s effective_fps=622.82 +2026-09-27 18:24:03.048 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=669698 infer_calls=669698 feedback_calls=669698 infer_wait=909.806s infer_obs_pack=11.104s env_step=103.656s feedback_total=53.789s feedback_obs_pack=14.366s feedback_info_pack=3.223s effective_fps=621.04 +2026-09-27 18:24:33.049 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=685960 infer_calls=685960 feedback_calls=685960 infer_wait=935.371s infer_obs_pack=11.352s env_step=105.921s feedback_total=54.974s feedback_obs_pack=14.688s feedback_info_pack=3.294s effective_fps=619.31 +2026-09-27 18:25:03.049 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=702079 infer_calls=702079 feedback_calls=702079 infer_wait=960.958s infer_obs_pack=11.601s env_step=108.183s feedback_total=56.139s feedback_obs_pack=15.005s feedback_info_pack=3.367s effective_fps=617.55 +2026-09-27 18:25:33.051 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=718341 infer_calls=718341 feedback_calls=718341 infer_wait=986.490s infer_obs_pack=11.857s env_step=110.452s feedback_total=57.344s feedback_obs_pack=15.322s feedback_info_pack=3.442s effective_fps=616.00 +2026-09-27 18:26:03.051 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=735874 infer_calls=735874 feedback_calls=735874 infer_wait=1011.434s infer_obs_pack=12.136s env_step=113.063s feedback_total=58.689s feedback_obs_pack=15.681s feedback_info_pack=3.527s effective_fps=615.63 +2026-09-27 18:26:33.052 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=754192 infer_calls=754192 feedback_calls=754192 infer_wait=1035.873s infer_obs_pack=12.451s env_step=115.914s feedback_total=60.186s feedback_obs_pack=16.095s feedback_info_pack=3.616s effective_fps=615.96 +2026-09-27 18:27:03.053 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=772473 infer_calls=772473 feedback_calls=772473 infer_wait=1060.302s infer_obs_pack=12.764s env_step=118.762s feedback_total=61.686s feedback_obs_pack=16.493s feedback_info_pack=3.712s effective_fps=616.25 +2026-09-27 18:27:33.055 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=790764 infer_calls=790764 feedback_calls=790764 infer_wait=1084.826s infer_obs_pack=13.067s env_step=121.566s feedback_total=63.159s feedback_obs_pack=16.884s feedback_info_pack=3.804s effective_fps=616.52 +2026-09-27 18:28:03.055 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=809721 infer_calls=809721 feedback_calls=809721 infer_wait=1109.105s infer_obs_pack=13.383s env_step=124.507s feedback_total=64.703s feedback_obs_pack=17.294s feedback_info_pack=3.902s effective_fps=617.31 +2026-09-27 18:28:33.055 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=829063 infer_calls=829063 feedback_calls=829063 infer_wait=1132.996s infer_obs_pack=13.721s env_step=127.650s feedback_total=66.351s feedback_obs_pack=17.733s feedback_info_pack=4.010s effective_fps=618.37 +2026-09-27 18:29:03.056 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=848305 infer_calls=848305 feedback_calls=848305 infer_wait=1157.010s infer_obs_pack=14.050s env_step=130.717s feedback_total=67.979s feedback_obs_pack=18.174s feedback_info_pack=4.113s effective_fps=619.31 +2026-09-27 18:29:33.060 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=868093 infer_calls=868093 feedback_calls=868093 infer_wait=1180.815s infer_obs_pack=14.390s env_step=133.894s feedback_total=69.661s feedback_obs_pack=18.621s feedback_info_pack=4.219s effective_fps=620.62 +2026-09-27 18:30:03.062 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=887314 infer_calls=887314 feedback_calls=887314 infer_wait=1204.679s infer_obs_pack=14.725s env_step=137.028s feedback_total=71.340s feedback_obs_pack=19.073s feedback_info_pack=4.334s effective_fps=621.47 +2026-09-27 18:30:33.063 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=906746 infer_calls=906746 feedback_calls=906746 infer_wait=1228.532s infer_obs_pack=15.062s env_step=140.194s feedback_total=73.004s feedback_obs_pack=19.512s feedback_info_pack=4.440s effective_fps=622.43 +2026-09-27 18:31:03.063 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=926225 infer_calls=926225 feedback_calls=926225 infer_wait=1252.449s infer_obs_pack=15.391s env_step=143.334s feedback_total=74.649s feedback_obs_pack=19.953s feedback_info_pack=4.549s effective_fps=623.38 +2026-09-27 18:31:33.063 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=945983 infer_calls=945983 feedback_calls=945983 infer_wait=1276.317s infer_obs_pack=15.726s env_step=146.467s feedback_total=76.324s feedback_obs_pack=20.398s feedback_info_pack=4.662s effective_fps=624.48 +2026-09-27 18:32:03.066 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=964978 infer_calls=964978 feedback_calls=964978 infer_wait=1300.433s infer_obs_pack=16.051s env_step=149.480s feedback_total=77.933s feedback_obs_pack=20.833s feedback_info_pack=4.763s effective_fps=625.03 +2026-09-27 18:32:33.067 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=984666 infer_calls=984666 feedback_calls=984666 infer_wait=1324.329s infer_obs_pack=16.385s env_step=152.621s feedback_total=79.586s feedback_obs_pack=21.276s feedback_info_pack=4.871s effective_fps=626.01 +2026-09-27 18:32:55.416 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Final rollout timing summary: env_steps=999425 infer_calls=999425 feedback_calls=999425 infer_wait=1341.759s infer_obs_pack=16.638s env_step=154.974s feedback_total=80.841s feedback_obs_pack=21.607s feedback_info_pack=4.954s effective_fps=626.91 +2026-09-27 18:32:55.416 | INFO | plugrl_env_client.runner.run:run:156 - Server signalled the end of the run; collection stopped after 3240/3997696 episodes. diff --git a/experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed0.log b/experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed0.log new file mode 100644 index 0000000..0b4cd24 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed0.log @@ -0,0 +1,43 @@ +18:05:21|INFO|plugrl_server version: 0.1.0 +18:05:21|INFO|Algorithm: ppo, Config: PPOAlgoConfig(global_steps=999424, learning_rate=0.0003, anneal_lr=True, buffer_size=2048, batch_size=64, update_epochs=10, gamma=0.99, gae_lambda=0.95, norm_adv=True, clip_coef=0.2, clip_vloss=True, ent_coef=0.0, vf_coef=0.5, max_grad_norm=0.5, target_kl=None, normalize_rewards=True, reward_clip=10.0, train_itrs=488, save_interval=20) +18:05:21|INFO|Policy: gaussian-policy, Config: GaussianPolicyConfig(algo='ppo', device=device(type='cpu'), obs_dim=11, action_dim=3, hidden_dims=(64, 64), state_keys=('obs',), normalize_observations=True, obs_clip=10.0, freeze_obs_stats=False, action_clip=1.0) +18:05:21|INFO|Seeded python, numpy and torch with 0. One env client is then reproducible; several are not, because batch composition depends on when their requests arrive. +18:05:21|INFO|Checkpoint Manager created: + at /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0 +18:05:21|INFO|Policy created... +18:05:21|INFO|Initialized RolloutBuffer buffer_size=2048 action_shape=(2048, 3) value_shape=(2048,) +18:05:21|INFO|Algorithm created: + +18:05:21|INFO|Agent Server is listening on 0.0.0.0:9950 +18:06:26|INFO|Checkpoint saved at step 40961 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/40961 +18:07:30|INFO|Checkpoint saved at step 81921 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/81921 +18:08:36|INFO|Checkpoint saved at step 122881 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/122881 +18:09:53|INFO|Checkpoint saved at step 163841 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/163841 +18:11:12|INFO|Checkpoint saved at step 204801 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/204801 +18:12:29|INFO|Checkpoint saved at step 245761 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/245761 +18:13:37|INFO|Checkpoint saved at step 286721 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/286721 +18:14:44|INFO|Checkpoint saved at step 327681 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/327681 +18:15:47|INFO|Checkpoint saved at step 368641 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/368641 +18:16:51|INFO|Checkpoint saved at step 409601 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/409601 +18:17:53|INFO|Checkpoint saved at step 450561 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/450561 +18:18:56|INFO|Checkpoint saved at step 491521 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/491521 +18:19:59|INFO|Checkpoint saved at step 532481 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/532481 +18:21:04|INFO|Checkpoint saved at step 573441 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/573441 +18:22:14|INFO|Checkpoint saved at step 614401 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/614401 +18:23:32|INFO|Checkpoint saved at step 655361 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/655361 +18:24:49|INFO|Checkpoint saved at step 696321 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/696321 +18:26:04|INFO|Checkpoint saved at step 737281 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/737281 +18:27:10|INFO|Checkpoint saved at step 778241 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/778241 +18:28:15|INFO|Checkpoint saved at step 819201 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/819201 +18:29:17|INFO|Checkpoint saved at step 860161 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/860161 +18:30:21|INFO|Checkpoint saved at step 901121 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/901121 +18:31:24|INFO|Checkpoint saved at step 942081 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/942081 +18:32:27|INFO|Checkpoint saved at step 983041 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/983041 +18:32:53|INFO|Checkpoint saved at step 999425 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed0/999425 +18:32:53|INFO|Stopping server as the algorithm signaled to stop. +18:32:53|INFO|Shutdown started: aborting pending infer requests. +18:32:53|INFO|Shutdown cleanup finished: pending_futures=1, drained_requests=1 +18:32:53|INFO|Shutdown closing 1 websocket connection(s). +18:32:53|INFO|Shutdown interrupted an in-flight inference request from ('127.0.0.1', 41568). +18:32:53|INFO|WebSocket server closed. +18:32:53|INFO|Scheduler task cancelled and cleaned up. diff --git a/experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed1.log b/experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed1.log new file mode 100644 index 0000000..af59416 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed1.log @@ -0,0 +1,43 @@ +18:05:26|INFO|plugrl_server version: 0.1.0 +18:05:26|INFO|Algorithm: ppo, Config: PPOAlgoConfig(global_steps=999424, learning_rate=0.0003, anneal_lr=True, buffer_size=2048, batch_size=64, update_epochs=10, gamma=0.99, gae_lambda=0.95, norm_adv=True, clip_coef=0.2, clip_vloss=True, ent_coef=0.0, vf_coef=0.5, max_grad_norm=0.5, target_kl=None, normalize_rewards=True, reward_clip=10.0, train_itrs=488, save_interval=20) +18:05:26|INFO|Policy: gaussian-policy, Config: GaussianPolicyConfig(algo='ppo', device=device(type='cpu'), obs_dim=11, action_dim=3, hidden_dims=(64, 64), state_keys=('obs',), normalize_observations=True, obs_clip=10.0, freeze_obs_stats=False, action_clip=1.0) +18:05:26|INFO|Seeded python, numpy and torch with 1. One env client is then reproducible; several are not, because batch composition depends on when their requests arrive. +18:05:26|INFO|Checkpoint Manager created: + at /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1 +18:05:26|INFO|Policy created... +18:05:26|INFO|Initialized RolloutBuffer buffer_size=2048 action_shape=(2048, 3) value_shape=(2048,) +18:05:26|INFO|Algorithm created: + +18:05:26|INFO|Agent Server is listening on 0.0.0.0:9951 +18:06:30|INFO|Checkpoint saved at step 40961 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/40961 +18:07:35|INFO|Checkpoint saved at step 81921 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/81921 +18:08:43|INFO|Checkpoint saved at step 122881 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/122881 +18:10:01|INFO|Checkpoint saved at step 163841 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/163841 +18:11:20|INFO|Checkpoint saved at step 204801 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/204801 +18:12:38|INFO|Checkpoint saved at step 245761 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/245761 +18:13:45|INFO|Checkpoint saved at step 286721 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/286721 +18:14:52|INFO|Checkpoint saved at step 327681 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/327681 +18:15:56|INFO|Checkpoint saved at step 368641 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/368641 +18:16:58|INFO|Checkpoint saved at step 409601 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/409601 +18:18:01|INFO|Checkpoint saved at step 450561 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/450561 +18:19:04|INFO|Checkpoint saved at step 491521 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/491521 +18:20:07|INFO|Checkpoint saved at step 532481 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/532481 +18:21:13|INFO|Checkpoint saved at step 573441 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/573441 +18:22:24|INFO|Checkpoint saved at step 614401 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/614401 +18:23:42|INFO|Checkpoint saved at step 655361 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/655361 +18:25:00|INFO|Checkpoint saved at step 696321 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/696321 +18:26:14|INFO|Checkpoint saved at step 737281 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/737281 +18:27:20|INFO|Checkpoint saved at step 778241 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/778241 +18:28:26|INFO|Checkpoint saved at step 819201 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/819201 +18:29:28|INFO|Checkpoint saved at step 860161 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/860161 +18:30:31|INFO|Checkpoint saved at step 901121 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/901121 +18:31:34|INFO|Checkpoint saved at step 942081 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/942081 +18:32:37|INFO|Checkpoint saved at step 983041 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/983041 +18:33:02|INFO|Checkpoint saved at step 999425 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed1/999425 +18:33:02|INFO|Stopping server as the algorithm signaled to stop. +18:33:02|INFO|Shutdown started: aborting pending infer requests. +18:33:02|INFO|Shutdown cleanup finished: pending_futures=1, drained_requests=1 +18:33:02|INFO|Shutdown closing 1 websocket connection(s). +18:33:02|INFO|Shutdown interrupted an in-flight inference request from ('127.0.0.1', 36974). +18:33:02|INFO|WebSocket server closed. +18:33:02|INFO|Scheduler task cancelled and cleaned up. diff --git a/experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed2.log b/experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed2.log new file mode 100644 index 0000000..bbe462b --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-hopper/server-seed2.log @@ -0,0 +1,43 @@ +18:05:31|INFO|plugrl_server version: 0.1.0 +18:05:31|INFO|Algorithm: ppo, Config: PPOAlgoConfig(global_steps=999424, learning_rate=0.0003, anneal_lr=True, buffer_size=2048, batch_size=64, update_epochs=10, gamma=0.99, gae_lambda=0.95, norm_adv=True, clip_coef=0.2, clip_vloss=True, ent_coef=0.0, vf_coef=0.5, max_grad_norm=0.5, target_kl=None, normalize_rewards=True, reward_clip=10.0, train_itrs=488, save_interval=20) +18:05:31|INFO|Policy: gaussian-policy, Config: GaussianPolicyConfig(algo='ppo', device=device(type='cpu'), obs_dim=11, action_dim=3, hidden_dims=(64, 64), state_keys=('obs',), normalize_observations=True, obs_clip=10.0, freeze_obs_stats=False, action_clip=1.0) +18:05:31|INFO|Seeded python, numpy and torch with 2. One env client is then reproducible; several are not, because batch composition depends on when their requests arrive. +18:05:31|INFO|Checkpoint Manager created: + at /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2 +18:05:31|INFO|Policy created... +18:05:31|INFO|Initialized RolloutBuffer buffer_size=2048 action_shape=(2048, 3) value_shape=(2048,) +18:05:31|INFO|Algorithm created: + +18:05:31|INFO|Agent Server is listening on 0.0.0.0:9952 +18:06:35|INFO|Checkpoint saved at step 40961 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/40961 +18:07:41|INFO|Checkpoint saved at step 81921 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/81921 +18:08:48|INFO|Checkpoint saved at step 122881 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/122881 +18:10:05|INFO|Checkpoint saved at step 163841 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/163841 +18:11:21|INFO|Checkpoint saved at step 204801 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/204801 +18:12:37|INFO|Checkpoint saved at step 245761 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/245761 +18:13:43|INFO|Checkpoint saved at step 286721 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/286721 +18:14:49|INFO|Checkpoint saved at step 327681 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/327681 +18:15:52|INFO|Checkpoint saved at step 368641 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/368641 +18:16:55|INFO|Checkpoint saved at step 409601 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/409601 +18:17:58|INFO|Checkpoint saved at step 450561 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/450561 +18:19:02|INFO|Checkpoint saved at step 491521 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/491521 +18:20:05|INFO|Checkpoint saved at step 532481 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/532481 +18:21:11|INFO|Checkpoint saved at step 573441 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/573441 +18:22:20|INFO|Checkpoint saved at step 614401 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/614401 +18:23:36|INFO|Checkpoint saved at step 655361 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/655361 +18:24:52|INFO|Checkpoint saved at step 696321 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/696321 +18:26:05|INFO|Checkpoint saved at step 737281 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/737281 +18:27:12|INFO|Checkpoint saved at step 778241 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/778241 +18:28:18|INFO|Checkpoint saved at step 819201 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/819201 +18:29:21|INFO|Checkpoint saved at step 860161 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/860161 +18:30:24|INFO|Checkpoint saved at step 901121 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/901121 +18:31:27|INFO|Checkpoint saved at step 942081 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/942081 +18:32:30|INFO|Checkpoint saved at step 983041 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/983041 +18:32:55|INFO|Checkpoint saved at step 999425 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-hopper/ppo/gaussian-policy/ppo-hopper-seed2/999425 +18:32:55|INFO|Stopping server as the algorithm signaled to stop. +18:32:55|INFO|Shutdown started: aborting pending infer requests. +18:32:55|INFO|Shutdown cleanup finished: pending_futures=1, drained_requests=1 +18:32:55|INFO|Shutdown closing 1 websocket connection(s). +18:32:55|INFO|Shutdown interrupted an in-flight inference request from ('127.0.0.1', 49598). +18:32:55|INFO|WebSocket server closed. +18:32:55|INFO|Scheduler task cancelled and cleaned up. diff --git a/experiments/e38-gaussian-ppo/results/ppo-walker.out b/experiments/e38-gaussian-ppo/results/ppo-walker.out new file mode 100644 index 0000000..8f7ecda --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-walker.out @@ -0,0 +1,11 @@ +cell: ppo-walker = gaussian-policy/default x ppo/default x Walker2d-v5 +iters: 488 x 2048 batch: 64 replan: 1 seeds: 0 1 2 +extra: algo none; policy --policy.obs-dim 17 --policy.action-dim 6 +out: /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker +start 2026-09-27 18:05:38 +seed 1 finished rc=0 at 18:33:35 +seed 0 finished rc=0 at 18:33:55 +seed 2 finished rc=0 at 18:34:01 +end 2026-09-27 18:34:01 +failed seeds: 0 +CELL_DONE ppo-walker diff --git a/experiments/e38-gaussian-ppo/results/ppo-walker/client-seed0.log b/experiments/e38-gaussian-ppo/results/ppo-walker/client-seed0.log new file mode 100644 index 0000000..6f5844e --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-walker/client-seed0.log @@ -0,0 +1,66 @@ +2026-09-27 18:05:42.027 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.classic.classic_env: pygame is not installed. Please install it with pip install "plugrl-env-client[classic]". +2026-09-27 18:05:42.027 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.atari.atari_env: Atari is not installed. Please install it with the 'atari' extra, e.g. 'pip install plugrl-env-client[atari]' +2026-09-27 18:05:42.027 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.d4rl.d4rl_env: d4rl is not installed. Please install it with pip install "plugrl-env-client[d4rl]". +2026-09-27 18:05:42.028 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.libero.libero_env: libero is not installed. Please install it with pip install "plugrl-env-client[libero]". +2026-09-27 18:05:42.028 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.robomimic.robomimic_env: Robomimic is not installed. Please install it with the 'robomimic' extra, e.g. 'pip install plugrl-env-client[robomimic]' +2026-09-27 18:05:42.078 | INFO | __main__:main:83 - Starting env client exp_name=mujoco-v1-nenv1-20260927-180542-f96b537c output_dir=runs/mujoco-v1-nenv1-20260927-180542-f96b537c recorder={'episode_freq': 0, 'thread0_only': True, 'record_video': False, 'video_fps': 30.0, 'record_full_rollout': False, 'record_obs_stats': True, 'record_episode_metrics': True, 'metric_window': 100} +2026-09-27 18:05:42.114 | INFO | plugrl_env_client.agent.websocket_env_client_agent:_wait_for_server:67 - Waiting for server at ws://127.0.0.1:9960... +2026-09-27 18:05:42.119 | INFO | plugrl_env_client.runner.run:report_server_metadata:56 - Server metadata: {'protocol_version': 1, 'server': 'plugrl-server', 'server_version': '0.1.0', 'algorithm': 'PPOAlgorithm', 'policy': 'GaussianPolicy', 'action_dim': 6, 'action_horizon': 1} +2026-09-27 18:06:12.122 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=18795 infer_calls=18795 feedback_calls=18795 infer_wait=23.846s infer_obs_pack=0.333s env_step=3.270s feedback_total=1.553s feedback_obs_pack=0.425s feedback_info_pack=0.086s effective_fps=648.05 +2026-09-27 18:06:42.123 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=38289 infer_calls=38289 feedback_calls=38289 infer_wait=47.391s infer_obs_pack=0.678s env_step=6.743s feedback_total=3.209s feedback_obs_pack=0.870s feedback_info_pack=0.176s effective_fps=659.91 +2026-09-27 18:07:12.372 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=57346 infer_calls=57346 feedback_calls=57346 infer_wait=71.368s infer_obs_pack=1.008s env_step=10.147s feedback_total=4.789s feedback_obs_pack=1.307s feedback_info_pack=0.262s effective_fps=656.78 +2026-09-27 18:07:42.372 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=75508 infer_calls=75508 feedback_calls=75508 infer_wait=95.499s infer_obs_pack=1.317s env_step=13.328s feedback_total=6.277s feedback_obs_pack=1.712s feedback_info_pack=0.344s effective_fps=648.57 +2026-09-27 18:08:12.372 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=93387 infer_calls=93387 feedback_calls=93387 infer_wait=119.654s infer_obs_pack=1.615s env_step=16.540s feedback_total=7.714s feedback_obs_pack=2.123s feedback_info_pack=0.424s effective_fps=641.73 +2026-09-27 18:08:42.375 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=111202 infer_calls=111202 feedback_calls=111202 infer_wait=143.865s infer_obs_pack=1.912s env_step=19.685s feedback_total=9.172s feedback_obs_pack=2.520s feedback_info_pack=0.504s effective_fps=636.78 +2026-09-27 18:09:12.376 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=126706 infer_calls=126706 feedback_calls=126706 infer_wait=169.265s infer_obs_pack=2.160s env_step=22.176s feedback_total=10.312s feedback_obs_pack=2.833s feedback_info_pack=0.564s effective_fps=621.37 +2026-09-27 18:09:42.376 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=142161 infer_calls=142161 feedback_calls=142161 infer_wait=194.670s infer_obs_pack=2.401s env_step=24.678s feedback_total=11.449s feedback_obs_pack=3.147s feedback_info_pack=0.631s effective_fps=609.62 +2026-09-27 18:10:12.377 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=157648 infer_calls=157648 feedback_calls=157648 infer_wait=220.109s infer_obs_pack=2.641s env_step=27.147s feedback_total=12.582s feedback_obs_pack=3.461s feedback_info_pack=0.691s effective_fps=600.61 +2026-09-27 18:10:42.378 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=173095 infer_calls=173095 feedback_calls=173095 infer_wait=245.551s infer_obs_pack=2.877s env_step=29.613s feedback_total=13.718s feedback_obs_pack=3.767s feedback_info_pack=0.753s effective_fps=593.28 +2026-09-27 18:11:12.570 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=188418 infer_calls=188418 feedback_calls=188418 infer_wait=271.161s infer_obs_pack=3.117s env_step=32.083s feedback_total=14.867s feedback_obs_pack=4.077s feedback_info_pack=0.816s effective_fps=586.56 +2026-09-27 18:11:42.571 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=204084 infer_calls=204084 feedback_calls=204084 infer_wait=296.601s infer_obs_pack=3.357s env_step=34.555s feedback_total=15.994s feedback_obs_pack=4.390s feedback_info_pack=0.875s effective_fps=582.25 +2026-09-27 18:12:12.571 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=219532 infer_calls=219532 feedback_calls=219532 infer_wait=322.025s infer_obs_pack=3.599s env_step=37.000s feedback_total=17.157s feedback_obs_pack=4.710s feedback_info_pack=0.940s effective_fps=578.05 +2026-09-27 18:12:42.572 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=235255 infer_calls=235255 feedback_calls=235255 infer_wait=347.379s infer_obs_pack=3.843s env_step=39.502s feedback_total=18.319s feedback_obs_pack=5.028s feedback_info_pack=1.001s effective_fps=575.13 +2026-09-27 18:13:12.572 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=253250 infer_calls=253250 feedback_calls=253250 infer_wait=371.611s infer_obs_pack=4.150s env_step=42.596s feedback_total=19.780s feedback_obs_pack=5.436s feedback_info_pack=1.080s effective_fps=578.02 +2026-09-27 18:13:42.573 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=271444 infer_calls=271444 feedback_calls=271444 infer_wait=395.888s infer_obs_pack=4.457s env_step=45.641s feedback_total=21.261s feedback_obs_pack=5.842s feedback_info_pack=1.163s effective_fps=580.94 +2026-09-27 18:14:12.574 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=289282 infer_calls=289282 feedback_calls=289282 infer_wait=420.234s infer_obs_pack=4.748s env_step=48.669s feedback_total=22.716s feedback_obs_pack=6.241s feedback_info_pack=1.242s effective_fps=582.80 +2026-09-27 18:14:42.739 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=307202 infer_calls=307202 feedback_calls=307202 infer_wait=444.689s infer_obs_pack=5.050s env_step=51.725s feedback_total=24.167s feedback_obs_pack=6.644s feedback_info_pack=1.319s effective_fps=584.44 +2026-09-27 18:15:12.955 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=325634 infer_calls=325634 feedback_calls=325634 infer_wait=468.882s infer_obs_pack=5.368s env_step=54.968s feedback_total=25.700s feedback_obs_pack=7.067s feedback_info_pack=1.402s effective_fps=586.82 +2026-09-27 18:15:42.957 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=345296 infer_calls=345296 feedback_calls=345296 infer_wait=492.416s infer_obs_pack=5.715s env_step=58.421s feedback_total=27.371s feedback_obs_pack=7.517s feedback_info_pack=1.491s effective_fps=591.34 +2026-09-27 18:16:13.213 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=364546 infer_calls=364546 feedback_calls=364546 infer_wait=516.416s infer_obs_pack=6.042s env_step=61.778s feedback_total=28.979s feedback_obs_pack=7.953s feedback_info_pack=1.580s effective_fps=594.48 +2026-09-27 18:16:43.214 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=384083 infer_calls=384083 feedback_calls=384083 infer_wait=540.044s infer_obs_pack=6.382s env_step=65.185s feedback_total=30.628s feedback_obs_pack=8.400s feedback_info_pack=1.670s effective_fps=598.04 +2026-09-27 18:17:13.215 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=403220 infer_calls=403220 feedback_calls=403220 infer_wait=563.634s infer_obs_pack=6.717s env_step=68.579s feedback_total=32.313s feedback_obs_pack=8.858s feedback_info_pack=1.762s effective_fps=600.71 +2026-09-27 18:17:43.216 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=422351 infer_calls=422351 feedback_calls=422351 infer_wait=587.247s infer_obs_pack=7.049s env_step=71.979s feedback_total=33.987s feedback_obs_pack=9.309s feedback_info_pack=1.855s effective_fps=603.13 +2026-09-27 18:18:13.218 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=442002 infer_calls=442002 feedback_calls=442002 infer_wait=610.839s infer_obs_pack=7.386s env_step=75.401s feedback_total=35.650s feedback_obs_pack=9.760s feedback_info_pack=1.947s effective_fps=606.08 +2026-09-27 18:18:43.221 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=460861 infer_calls=460861 feedback_calls=460861 infer_wait=634.620s infer_obs_pack=7.717s env_step=78.697s feedback_total=37.287s feedback_obs_pack=10.209s feedback_info_pack=2.034s effective_fps=607.74 +2026-09-27 18:19:13.222 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=480070 infer_calls=480070 feedback_calls=480070 infer_wait=658.267s infer_obs_pack=8.060s env_step=82.089s feedback_total=38.935s feedback_obs_pack=10.659s feedback_info_pack=2.123s effective_fps=609.73 +2026-09-27 18:19:43.223 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=499507 infer_calls=499507 feedback_calls=499507 infer_wait=681.963s infer_obs_pack=8.404s env_step=85.456s feedback_total=40.559s feedback_obs_pack=11.105s feedback_info_pack=2.208s effective_fps=611.85 +2026-09-27 18:20:13.224 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=518261 infer_calls=518261 feedback_calls=518261 infer_wait=705.789s infer_obs_pack=8.733s env_step=88.773s feedback_total=42.143s feedback_obs_pack=11.548s feedback_info_pack=2.297s effective_fps=613.01 +2026-09-27 18:20:43.478 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=536578 infer_calls=536578 feedback_calls=536578 infer_wait=730.238s infer_obs_pack=9.040s env_step=91.881s feedback_total=43.634s feedback_obs_pack=11.948s feedback_info_pack=2.377s effective_fps=613.38 +2026-09-27 18:21:13.479 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=554700 infer_calls=554700 feedback_calls=554700 infer_wait=754.505s infer_obs_pack=9.344s env_step=94.945s feedback_total=45.110s feedback_obs_pack=12.346s feedback_info_pack=2.460s effective_fps=613.67 +2026-09-27 18:21:43.479 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=572694 infer_calls=572694 feedback_calls=572694 infer_wait=778.778s infer_obs_pack=9.648s env_step=98.001s feedback_total=46.592s feedback_obs_pack=12.755s feedback_info_pack=2.543s effective_fps=613.81 +2026-09-27 18:22:13.479 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=589102 infer_calls=589102 feedback_calls=589102 infer_wait=803.837s infer_obs_pack=9.915s env_step=100.637s feedback_total=47.851s feedback_obs_pack=13.101s feedback_info_pack=2.609s effective_fps=612.22 +2026-09-27 18:22:43.480 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=604586 infer_calls=604586 feedback_calls=604586 infer_wait=829.302s infer_obs_pack=10.156s env_step=103.052s feedback_total=49.013s feedback_obs_pack=13.421s feedback_info_pack=2.671s effective_fps=609.76 +2026-09-27 18:23:13.482 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=620218 infer_calls=620218 feedback_calls=620218 infer_wait=854.818s infer_obs_pack=10.395s env_step=105.434s feedback_total=50.163s feedback_obs_pack=13.730s feedback_info_pack=2.734s effective_fps=607.57 +2026-09-27 18:23:43.483 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=635644 infer_calls=635644 feedback_calls=635644 infer_wait=880.322s infer_obs_pack=10.637s env_step=107.828s feedback_total=51.301s feedback_obs_pack=14.047s feedback_info_pack=2.795s effective_fps=605.32 +2026-09-27 18:24:13.484 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=651183 infer_calls=651183 feedback_calls=651183 infer_wait=905.788s infer_obs_pack=10.877s env_step=110.251s feedback_total=52.442s feedback_obs_pack=14.365s feedback_info_pack=2.856s effective_fps=603.31 +2026-09-27 18:24:43.485 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=666585 infer_calls=666585 feedback_calls=666585 infer_wait=931.322s infer_obs_pack=11.116s env_step=112.631s feedback_total=53.576s feedback_obs_pack=14.682s feedback_info_pack=2.919s effective_fps=601.26 +2026-09-27 18:25:13.487 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=682069 infer_calls=682069 feedback_calls=682069 infer_wait=956.825s infer_obs_pack=11.359s env_step=115.035s feedback_total=54.713s feedback_obs_pack=14.997s feedback_info_pack=2.986s effective_fps=599.39 +2026-09-27 18:25:43.488 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=697756 infer_calls=697756 feedback_calls=697756 infer_wait=982.299s infer_obs_pack=11.596s env_step=117.451s feedback_total=55.865s feedback_obs_pack=15.312s feedback_info_pack=3.051s effective_fps=597.80 +2026-09-27 18:26:13.489 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=715661 infer_calls=715661 feedback_calls=715661 infer_wait=1006.584s infer_obs_pack=11.901s env_step=120.508s feedback_total=57.331s feedback_obs_pack=15.708s feedback_info_pack=3.134s effective_fps=598.22 +2026-09-27 18:26:43.490 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=733767 infer_calls=733767 feedback_calls=733767 infer_wait=1030.836s infer_obs_pack=12.209s env_step=123.595s feedback_total=58.802s feedback_obs_pack=16.113s feedback_info_pack=3.213s effective_fps=598.78 +2026-09-27 18:27:13.491 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=751903 infer_calls=751903 feedback_calls=751903 infer_wait=1055.106s infer_obs_pack=12.519s env_step=126.664s feedback_total=60.264s feedback_obs_pack=16.523s feedback_info_pack=3.291s effective_fps=599.34 +2026-09-27 18:27:43.491 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=770201 infer_calls=770201 feedback_calls=770201 infer_wait=1079.323s infer_obs_pack=12.822s env_step=129.749s feedback_total=61.769s feedback_obs_pack=16.928s feedback_info_pack=3.373s effective_fps=600.00 +2026-09-27 18:28:13.493 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=788680 infer_calls=788680 feedback_calls=788680 infer_wait=1103.433s infer_obs_pack=13.141s env_step=132.895s feedback_total=63.291s feedback_obs_pack=17.339s feedback_info_pack=3.455s effective_fps=600.78 +2026-09-27 18:28:43.493 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=808344 infer_calls=808344 feedback_calls=808344 infer_wait=1127.068s infer_obs_pack=13.482s env_step=136.306s feedback_total=64.937s feedback_obs_pack=17.791s feedback_info_pack=3.547s effective_fps=602.44 +2026-09-27 18:29:13.493 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=827272 infer_calls=827272 feedback_calls=827272 infer_wait=1150.812s infer_obs_pack=13.819s env_step=139.648s feedback_total=66.551s feedback_obs_pack=18.229s feedback_info_pack=3.637s effective_fps=603.48 +2026-09-27 18:29:43.494 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=846119 infer_calls=846119 feedback_calls=846119 infer_wait=1174.631s infer_obs_pack=14.142s env_step=142.959s feedback_total=68.155s feedback_obs_pack=18.664s feedback_info_pack=3.724s effective_fps=604.42 +2026-09-27 18:30:13.495 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=865272 infer_calls=865272 feedback_calls=865272 infer_wait=1198.241s infer_obs_pack=14.477s env_step=146.372s feedback_total=69.826s feedback_obs_pack=19.120s feedback_info_pack=3.816s effective_fps=605.54 +2026-09-27 18:30:43.495 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=884521 infer_calls=884521 feedback_calls=884521 infer_wait=1221.825s infer_obs_pack=14.818s env_step=149.799s feedback_total=71.493s feedback_obs_pack=19.570s feedback_info_pack=3.906s effective_fps=606.69 +2026-09-27 18:31:13.496 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=903365 infer_calls=903365 feedback_calls=903365 infer_wait=1245.632s infer_obs_pack=15.149s env_step=153.116s feedback_total=73.094s feedback_obs_pack=20.006s feedback_info_pack=3.994s effective_fps=607.51 +2026-09-27 18:31:43.496 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=922270 infer_calls=922270 feedback_calls=922270 infer_wait=1269.462s infer_obs_pack=15.479s env_step=156.432s feedback_total=74.678s feedback_obs_pack=20.441s feedback_info_pack=4.081s effective_fps=608.34 +2026-09-27 18:32:13.496 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=941534 infer_calls=941534 feedback_calls=941534 infer_wait=1293.151s infer_obs_pack=15.817s env_step=159.804s feedback_total=76.311s feedback_obs_pack=20.892s feedback_info_pack=4.171s effective_fps=609.37 +2026-09-27 18:32:43.765 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=960514 infer_calls=960514 feedback_calls=960514 infer_wait=1317.245s infer_obs_pack=16.145s env_step=163.119s feedback_total=77.898s feedback_obs_pack=21.329s feedback_info_pack=4.260s effective_fps=610.08 +2026-09-27 18:33:13.765 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=978889 infer_calls=978889 feedback_calls=978889 infer_wait=1341.486s infer_obs_pack=16.451s env_step=166.197s feedback_total=79.378s feedback_obs_pack=21.733s feedback_info_pack=4.342s effective_fps=610.47 +2026-09-27 18:33:43.767 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=993671 infer_calls=993671 feedback_calls=993671 infer_wait=1367.534s infer_obs_pack=16.658s env_step=168.285s feedback_total=80.402s feedback_obs_pack=22.002s feedback_info_pack=4.394s effective_fps=608.54 +2026-09-27 18:33:55.512 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Final rollout timing summary: env_steps=999425 infer_calls=999425 feedback_calls=999425 infer_wait=1377.485s infer_obs_pack=16.733s env_step=169.069s feedback_total=80.766s feedback_obs_pack=22.105s feedback_info_pack=4.413s effective_fps=607.90 +2026-09-27 18:33:55.513 | INFO | plugrl_env_client.runner.run:run:156 - Server signalled the end of the run; collection stopped after 3464/3997696 episodes. diff --git a/experiments/e38-gaussian-ppo/results/ppo-walker/client-seed1.log b/experiments/e38-gaussian-ppo/results/ppo-walker/client-seed1.log new file mode 100644 index 0000000..128aeb3 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-walker/client-seed1.log @@ -0,0 +1,65 @@ +2026-09-27 18:05:47.036 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.classic.classic_env: pygame is not installed. Please install it with pip install "plugrl-env-client[classic]". +2026-09-27 18:05:47.036 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.atari.atari_env: Atari is not installed. Please install it with the 'atari' extra, e.g. 'pip install plugrl-env-client[atari]' +2026-09-27 18:05:47.036 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.d4rl.d4rl_env: d4rl is not installed. Please install it with pip install "plugrl-env-client[d4rl]". +2026-09-27 18:05:47.037 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.libero.libero_env: libero is not installed. Please install it with pip install "plugrl-env-client[libero]". +2026-09-27 18:05:47.037 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.robomimic.robomimic_env: Robomimic is not installed. Please install it with the 'robomimic' extra, e.g. 'pip install plugrl-env-client[robomimic]' +2026-09-27 18:05:47.083 | INFO | __main__:main:83 - Starting env client exp_name=mujoco-v1-nenv1-20260927-180547-7f5bb064 output_dir=runs/mujoco-v1-nenv1-20260927-180547-7f5bb064 recorder={'episode_freq': 0, 'thread0_only': True, 'record_video': False, 'video_fps': 30.0, 'record_full_rollout': False, 'record_obs_stats': True, 'record_episode_metrics': True, 'metric_window': 100} +2026-09-27 18:05:47.126 | INFO | plugrl_env_client.agent.websocket_env_client_agent:_wait_for_server:67 - Waiting for server at ws://127.0.0.1:9961... +2026-09-27 18:05:47.130 | INFO | plugrl_env_client.runner.run:report_server_metadata:56 - Server metadata: {'protocol_version': 1, 'server': 'plugrl-server', 'server_version': '0.1.0', 'algorithm': 'PPOAlgorithm', 'policy': 'GaussianPolicy', 'action_dim': 6, 'action_horizon': 1} +2026-09-27 18:06:17.132 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=19061 infer_calls=19061 feedback_calls=19061 infer_wait=23.705s infer_obs_pack=0.337s env_step=3.323s feedback_total=1.607s feedback_obs_pack=0.435s feedback_info_pack=0.088s effective_fps=657.91 +2026-09-27 18:06:47.132 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=38235 infer_calls=38235 feedback_calls=38235 infer_wait=47.340s infer_obs_pack=0.673s env_step=6.724s feedback_total=3.246s feedback_obs_pack=0.876s feedback_info_pack=0.184s effective_fps=659.42 +2026-09-27 18:07:17.132 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=56880 infer_calls=56880 feedback_calls=56880 infer_wait=71.115s infer_obs_pack=0.999s env_step=10.060s feedback_total=4.854s feedback_obs_pack=1.318s feedback_info_pack=0.275s effective_fps=653.58 +2026-09-27 18:07:47.132 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=75215 infer_calls=75215 feedback_calls=75215 infer_wait=95.183s infer_obs_pack=1.311s env_step=13.275s feedback_total=6.339s feedback_obs_pack=1.730s feedback_info_pack=0.353s effective_fps=647.80 +2026-09-27 18:08:17.134 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=93418 infer_calls=93418 feedback_calls=93418 infer_wait=119.220s infer_obs_pack=1.625s env_step=16.520s feedback_total=7.830s feedback_obs_pack=2.141s feedback_info_pack=0.432s effective_fps=643.40 +2026-09-27 18:08:47.135 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=111417 infer_calls=111417 feedback_calls=111417 infer_wait=143.467s infer_obs_pack=1.926s env_step=19.653s feedback_total=9.281s feedback_obs_pack=2.548s feedback_info_pack=0.513s effective_fps=639.13 +2026-09-27 18:09:17.136 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=127089 infer_calls=127089 feedback_calls=127089 infer_wait=168.765s infer_obs_pack=2.174s env_step=22.194s feedback_total=10.455s feedback_obs_pack=2.870s feedback_info_pack=0.576s effective_fps=624.25 +2026-09-27 18:09:47.137 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=142949 infer_calls=142949 feedback_calls=142949 infer_wait=194.079s infer_obs_pack=2.426s env_step=24.718s feedback_total=11.628s feedback_obs_pack=3.191s feedback_info_pack=0.642s effective_fps=613.91 +2026-09-27 18:10:17.138 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=158646 infer_calls=158646 feedback_calls=158646 infer_wait=219.355s infer_obs_pack=2.678s env_step=27.260s feedback_total=12.821s feedback_obs_pack=3.520s feedback_info_pack=0.703s effective_fps=605.26 +2026-09-27 18:10:47.138 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=174329 infer_calls=174329 feedback_calls=174329 infer_wait=244.702s infer_obs_pack=2.929s env_step=29.741s feedback_total=14.005s feedback_obs_pack=3.841s feedback_info_pack=0.767s effective_fps=598.29 +2026-09-27 18:11:17.138 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=190046 infer_calls=190046 feedback_calls=190046 infer_wait=270.056s infer_obs_pack=3.179s env_step=32.249s feedback_total=15.168s feedback_obs_pack=4.159s feedback_info_pack=0.828s effective_fps=592.69 +2026-09-27 18:11:47.140 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=205724 infer_calls=205724 feedback_calls=205724 infer_wait=295.458s infer_obs_pack=3.426s env_step=34.720s feedback_total=16.314s feedback_obs_pack=4.477s feedback_info_pack=0.891s effective_fps=587.92 +2026-09-27 18:12:17.141 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=221468 infer_calls=221468 feedback_calls=221468 infer_wait=320.789s infer_obs_pack=3.674s env_step=37.220s feedback_total=17.499s feedback_obs_pack=4.803s feedback_info_pack=0.954s effective_fps=584.07 +2026-09-27 18:12:47.141 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=237884 infer_calls=237884 feedback_calls=237884 infer_wait=345.845s infer_obs_pack=3.937s env_step=39.863s feedback_total=18.764s feedback_obs_pack=5.152s feedback_info_pack=1.020s effective_fps=582.47 +2026-09-27 18:13:17.141 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=256299 infer_calls=256299 feedback_calls=256299 infer_wait=369.862s infer_obs_pack=4.249s env_step=43.061s feedback_total=20.307s feedback_obs_pack=5.572s feedback_info_pack=1.102s effective_fps=585.85 +2026-09-27 18:13:47.143 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=274643 infer_calls=274643 feedback_calls=274643 infer_wait=394.019s infer_obs_pack=4.570s env_step=46.193s feedback_total=21.790s feedback_obs_pack=5.987s feedback_info_pack=1.183s effective_fps=588.64 +2026-09-27 18:14:17.145 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=293027 infer_calls=293027 feedback_calls=293027 infer_wait=418.099s infer_obs_pack=4.888s env_step=49.359s feedback_total=23.310s feedback_obs_pack=6.407s feedback_info_pack=1.263s effective_fps=591.19 +2026-09-27 18:14:47.146 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=311741 infer_calls=311741 feedback_calls=311741 infer_wait=442.139s infer_obs_pack=5.207s env_step=52.547s feedback_total=24.844s feedback_obs_pack=6.823s feedback_info_pack=1.344s effective_fps=594.09 +2026-09-27 18:15:17.146 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=330461 infer_calls=330461 feedback_calls=330461 infer_wait=466.003s infer_obs_pack=5.536s env_step=55.817s feedback_total=26.426s feedback_obs_pack=7.266s feedback_info_pack=1.431s effective_fps=596.73 +2026-09-27 18:15:47.147 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=349695 infer_calls=349695 feedback_calls=349695 infer_wait=489.633s infer_obs_pack=5.879s env_step=59.213s feedback_total=28.060s feedback_obs_pack=7.716s feedback_info_pack=1.523s effective_fps=600.04 +2026-09-27 18:16:17.147 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=369230 infer_calls=369230 feedback_calls=369230 infer_wait=513.254s infer_obs_pack=6.216s env_step=62.617s feedback_total=29.714s feedback_obs_pack=8.167s feedback_info_pack=1.608s effective_fps=603.51 +2026-09-27 18:16:47.150 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=388757 infer_calls=388757 feedback_calls=388757 infer_wait=536.809s infer_obs_pack=6.560s env_step=66.043s feedback_total=31.400s feedback_obs_pack=8.623s feedback_info_pack=1.700s effective_fps=606.66 +2026-09-27 18:17:17.151 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=407701 infer_calls=407701 feedback_calls=407701 infer_wait=560.470s infer_obs_pack=6.902s env_step=69.404s feedback_total=33.054s feedback_obs_pack=9.076s feedback_info_pack=1.786s effective_fps=608.66 +2026-09-27 18:17:47.152 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=426684 infer_calls=426684 feedback_calls=426684 infer_wait=584.078s infer_obs_pack=7.241s env_step=72.803s feedback_total=34.721s feedback_obs_pack=9.534s feedback_info_pack=1.875s effective_fps=610.56 +2026-09-27 18:18:17.153 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=446105 infer_calls=446105 feedback_calls=446105 infer_wait=607.700s infer_obs_pack=7.584s env_step=76.200s feedback_total=36.378s feedback_obs_pack=9.988s feedback_info_pack=1.966s effective_fps=612.90 +2026-09-27 18:18:47.154 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=465294 infer_calls=465294 feedback_calls=465294 infer_wait=631.465s infer_obs_pack=7.918s env_step=79.502s feedback_total=38.009s feedback_obs_pack=10.427s feedback_info_pack=2.053s effective_fps=614.74 +2026-09-27 18:19:17.155 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=484559 infer_calls=484559 feedback_calls=484559 infer_wait=655.117s infer_obs_pack=8.255s env_step=82.881s feedback_total=39.670s feedback_obs_pack=10.885s feedback_info_pack=2.143s effective_fps=616.55 +2026-09-27 18:19:47.157 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=503636 infer_calls=503636 feedback_calls=503636 infer_wait=678.855s infer_obs_pack=8.588s env_step=86.207s feedback_total=41.300s feedback_obs_pack=11.337s feedback_info_pack=2.230s effective_fps=618.00 +2026-09-27 18:20:17.161 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=522476 infer_calls=522476 feedback_calls=522476 infer_wait=702.627s infer_obs_pack=8.927s env_step=89.514s feedback_total=42.931s feedback_obs_pack=11.785s feedback_info_pack=2.317s effective_fps=619.05 +2026-09-27 18:20:47.201 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=540674 infer_calls=540674 feedback_calls=540674 infer_wait=726.915s infer_obs_pack=9.245s env_step=92.573s feedback_total=44.399s feedback_obs_pack=12.190s feedback_info_pack=2.396s effective_fps=619.24 +2026-09-27 18:21:17.203 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=559166 infer_calls=559166 feedback_calls=559166 infer_wait=751.092s infer_obs_pack=9.558s env_step=95.677s feedback_total=45.905s feedback_obs_pack=12.605s feedback_info_pack=2.474s effective_fps=619.76 +2026-09-27 18:21:47.204 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=577709 infer_calls=577709 feedback_calls=577709 infer_wait=775.218s infer_obs_pack=9.875s env_step=98.807s feedback_total=47.424s feedback_obs_pack=13.014s feedback_info_pack=2.555s effective_fps=620.31 +2026-09-27 18:22:17.372 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=593922 infer_calls=593922 feedback_calls=593922 infer_wait=800.570s infer_obs_pack=10.138s env_step=101.359s feedback_total=48.662s feedback_obs_pack=13.362s feedback_info_pack=2.619s effective_fps=618.20 +2026-09-27 18:22:47.372 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=609985 infer_calls=609985 feedback_calls=609985 infer_wait=825.898s infer_obs_pack=10.397s env_step=103.824s feedback_total=49.872s feedback_obs_pack=13.689s feedback_info_pack=2.683s effective_fps=616.15 +2026-09-27 18:23:17.373 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=625904 infer_calls=625904 feedback_calls=625904 infer_wait=851.303s infer_obs_pack=10.652s env_step=106.256s feedback_total=51.047s feedback_obs_pack=14.012s feedback_info_pack=2.746s effective_fps=614.08 +2026-09-27 18:23:47.373 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=641826 infer_calls=641826 feedback_calls=641826 infer_wait=876.723s infer_obs_pack=10.905s env_step=108.678s feedback_total=52.207s feedback_obs_pack=14.329s feedback_info_pack=2.807s effective_fps=612.13 +2026-09-27 18:24:17.374 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=657678 infer_calls=657678 feedback_calls=657678 infer_wait=902.121s infer_obs_pack=11.154s env_step=111.114s feedback_total=53.396s feedback_obs_pack=14.656s feedback_info_pack=2.874s effective_fps=610.21 +2026-09-27 18:24:47.490 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=673794 infer_calls=673794 feedback_calls=673794 infer_wait=927.567s infer_obs_pack=11.409s env_step=113.593s feedback_total=54.594s feedback_obs_pack=14.982s feedback_info_pack=2.937s effective_fps=608.58 +2026-09-27 18:25:17.491 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=689878 infer_calls=689878 feedback_calls=689878 infer_wait=952.853s infer_obs_pack=11.667s env_step=116.093s feedback_total=55.804s feedback_obs_pack=15.307s feedback_info_pack=3.000s effective_fps=607.06 +2026-09-27 18:25:47.491 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=706016 infer_calls=706016 feedback_calls=706016 infer_wait=978.093s infer_obs_pack=11.923s env_step=118.617s feedback_total=57.031s feedback_obs_pack=15.644s feedback_info_pack=3.065s effective_fps=605.68 +2026-09-27 18:26:17.492 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=724194 infer_calls=724194 feedback_calls=724194 infer_wait=1002.306s infer_obs_pack=12.234s env_step=121.705s feedback_total=58.527s feedback_obs_pack=16.060s feedback_info_pack=3.145s effective_fps=606.14 +2026-09-27 18:26:47.492 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=742599 infer_calls=742599 feedback_calls=742599 infer_wait=1026.540s infer_obs_pack=12.551s env_step=124.767s feedback_total=60.018s feedback_obs_pack=16.479s feedback_info_pack=3.223s effective_fps=606.76 +2026-09-27 18:27:17.493 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=760714 infer_calls=760714 feedback_calls=760714 infer_wait=1050.779s infer_obs_pack=12.865s env_step=127.829s feedback_total=61.507s feedback_obs_pack=16.889s feedback_info_pack=3.308s effective_fps=607.12 +2026-09-27 18:27:47.495 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=779199 infer_calls=779199 feedback_calls=779199 infer_wait=1075.033s infer_obs_pack=13.178s env_step=130.888s feedback_total=62.993s feedback_obs_pack=17.297s feedback_info_pack=3.387s effective_fps=607.76 +2026-09-27 18:28:17.496 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=798575 infer_calls=798575 feedback_calls=798575 infer_wait=1098.785s infer_obs_pack=13.516s env_step=134.190s feedback_total=64.630s feedback_obs_pack=17.745s feedback_info_pack=3.476s effective_fps=609.08 +2026-09-27 18:28:47.497 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=817234 infer_calls=817234 feedback_calls=817234 infer_wait=1122.621s infer_obs_pack=13.849s env_step=137.454s feedback_total=66.234s feedback_obs_pack=18.185s feedback_info_pack=3.562s effective_fps=609.80 +2026-09-27 18:29:17.499 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=836729 infer_calls=836729 feedback_calls=836729 infer_wait=1146.329s infer_obs_pack=14.189s env_step=140.786s feedback_total=67.882s feedback_obs_pack=18.636s feedback_info_pack=3.654s effective_fps=611.11 +2026-09-27 18:29:47.499 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=856111 infer_calls=856111 feedback_calls=856111 infer_wait=1170.022s infer_obs_pack=14.535s env_step=144.119s feedback_total=69.533s feedback_obs_pack=19.085s feedback_info_pack=3.741s effective_fps=612.29 +2026-09-27 18:30:17.500 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=875699 infer_calls=875699 feedback_calls=875699 infer_wait=1193.478s infer_obs_pack=14.878s env_step=147.580s feedback_total=71.261s feedback_obs_pack=19.550s feedback_info_pack=3.834s effective_fps=613.58 +2026-09-27 18:30:47.500 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=895328 infer_calls=895328 feedback_calls=895328 infer_wait=1217.000s infer_obs_pack=15.225s env_step=150.994s feedback_total=72.984s feedback_obs_pack=20.014s feedback_info_pack=3.926s effective_fps=614.84 +2026-09-27 18:31:17.500 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=914897 infer_calls=914897 feedback_calls=914897 infer_wait=1240.614s infer_obs_pack=15.567s env_step=154.379s feedback_total=74.663s feedback_obs_pack=20.472s feedback_info_pack=4.017s effective_fps=616.00 +2026-09-27 18:31:47.816 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=933890 infer_calls=933890 feedback_calls=933890 infer_wait=1264.702s infer_obs_pack=15.902s env_step=157.673s feedback_total=76.296s feedback_obs_pack=20.921s feedback_info_pack=4.105s effective_fps=616.60 +2026-09-27 18:32:17.818 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=952983 infer_calls=952983 feedback_calls=952983 infer_wait=1288.464s infer_obs_pack=16.239s env_step=160.992s feedback_total=77.922s feedback_obs_pack=21.379s feedback_info_pack=4.193s effective_fps=617.37 +2026-09-27 18:32:47.820 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=972620 infer_calls=972620 feedback_calls=972620 infer_wait=1312.033s infer_obs_pack=16.592s env_step=164.402s feedback_total=79.611s feedback_obs_pack=21.842s feedback_info_pack=4.284s effective_fps=618.46 +2026-09-27 18:33:17.822 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=990700 infer_calls=990700 feedback_calls=990700 infer_wait=1336.502s infer_obs_pack=16.893s env_step=167.333s feedback_total=81.044s feedback_obs_pack=22.228s feedback_info_pack=4.359s effective_fps=618.50 +2026-09-27 18:33:35.526 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Final rollout timing summary: env_steps=999425 infer_calls=999425 feedback_calls=999425 infer_wait=1351.634s infer_obs_pack=17.018s env_step=168.510s feedback_total=81.624s feedback_obs_pack=22.382s feedback_info_pack=4.388s effective_fps=617.39 +2026-09-27 18:33:35.526 | INFO | plugrl_env_client.runner.run:run:156 - Server signalled the end of the run; collection stopped after 3396/3997696 episodes. diff --git a/experiments/e38-gaussian-ppo/results/ppo-walker/client-seed2.log b/experiments/e38-gaussian-ppo/results/ppo-walker/client-seed2.log new file mode 100644 index 0000000..e83c706 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-walker/client-seed2.log @@ -0,0 +1,66 @@ +2026-09-27 18:05:52.031 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.classic.classic_env: pygame is not installed. Please install it with pip install "plugrl-env-client[classic]". +2026-09-27 18:05:52.032 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.atari.atari_env: Atari is not installed. Please install it with the 'atari' extra, e.g. 'pip install plugrl-env-client[atari]' +2026-09-27 18:05:52.032 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.d4rl.d4rl_env: d4rl is not installed. Please install it with pip install "plugrl-env-client[d4rl]". +2026-09-27 18:05:52.032 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.libero.libero_env: libero is not installed. Please install it with pip install "plugrl-env-client[libero]". +2026-09-27 18:05:52.033 | WARNING | plugrl_env_client.envs:_load_env_modules:27 - Skip loading env module plugrl_env_client.envs.robomimic.robomimic_env: Robomimic is not installed. Please install it with the 'robomimic' extra, e.g. 'pip install plugrl-env-client[robomimic]' +2026-09-27 18:05:52.080 | INFO | __main__:main:83 - Starting env client exp_name=mujoco-v1-nenv1-20260927-180552-5393ca80 output_dir=runs/mujoco-v1-nenv1-20260927-180552-5393ca80 recorder={'episode_freq': 0, 'thread0_only': True, 'record_video': False, 'video_fps': 30.0, 'record_full_rollout': False, 'record_obs_stats': True, 'record_episode_metrics': True, 'metric_window': 100} +2026-09-27 18:05:52.126 | INFO | plugrl_env_client.agent.websocket_env_client_agent:_wait_for_server:67 - Waiting for server at ws://127.0.0.1:9962... +2026-09-27 18:05:52.129 | INFO | plugrl_env_client.runner.run:report_server_metadata:56 - Server metadata: {'protocol_version': 1, 'server': 'plugrl-server', 'server_version': '0.1.0', 'algorithm': 'PPOAlgorithm', 'policy': 'GaussianPolicy', 'action_dim': 6, 'action_horizon': 1} +2026-09-27 18:06:22.132 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=19152 infer_calls=19152 feedback_calls=19152 infer_wait=23.706s infer_obs_pack=0.348s env_step=3.343s feedback_total=1.584s feedback_obs_pack=0.443s feedback_info_pack=0.091s effective_fps=660.83 +2026-09-27 18:06:52.132 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=38143 infer_calls=38143 feedback_calls=38143 infer_wait=47.336s infer_obs_pack=0.680s env_step=6.800s feedback_total=3.202s feedback_obs_pack=0.898s feedback_info_pack=0.182s effective_fps=657.44 +2026-09-27 18:07:22.133 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=56735 infer_calls=56735 feedback_calls=56735 infer_wait=71.195s infer_obs_pack=0.998s env_step=10.137s feedback_total=4.748s feedback_obs_pack=1.332s feedback_info_pack=0.267s effective_fps=651.54 +2026-09-27 18:07:52.134 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=74735 infer_calls=74735 feedback_calls=74735 infer_wait=95.373s infer_obs_pack=1.298s env_step=13.317s feedback_total=6.202s feedback_obs_pack=1.735s feedback_info_pack=0.348s effective_fps=643.21 +2026-09-27 18:08:22.137 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=93090 infer_calls=93090 feedback_calls=93090 infer_wait=119.337s infer_obs_pack=1.613s env_step=16.611s feedback_total=7.710s feedback_obs_pack=2.155s feedback_info_pack=0.433s effective_fps=640.80 +2026-09-27 18:08:52.138 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=110337 infer_calls=110337 feedback_calls=110337 infer_wait=143.917s infer_obs_pack=1.894s env_step=19.568s feedback_total=9.058s feedback_obs_pack=2.530s feedback_info_pack=0.506s effective_fps=632.53 +2026-09-27 18:09:22.140 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=125712 infer_calls=125712 feedback_calls=125712 infer_wait=169.304s infer_obs_pack=2.137s env_step=22.072s feedback_total=10.206s feedback_obs_pack=2.847s feedback_info_pack=0.569s effective_fps=617.09 +2026-09-27 18:09:52.141 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=141190 infer_calls=141190 feedback_calls=141190 infer_wait=194.667s infer_obs_pack=2.383s env_step=24.584s feedback_total=11.371s feedback_obs_pack=3.174s feedback_info_pack=0.635s effective_fps=605.95 +2026-09-27 18:10:22.141 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=156312 infer_calls=156312 feedback_calls=156312 infer_wait=220.114s infer_obs_pack=2.626s env_step=27.057s feedback_total=12.495s feedback_obs_pack=3.493s feedback_info_pack=0.696s effective_fps=595.95 +2026-09-27 18:10:52.142 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=171778 infer_calls=171778 feedback_calls=171778 infer_wait=245.517s infer_obs_pack=2.868s env_step=29.537s feedback_total=13.651s feedback_obs_pack=3.821s feedback_info_pack=0.758s effective_fps=589.14 +2026-09-27 18:11:22.145 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=187010 infer_calls=187010 feedback_calls=187010 infer_wait=270.973s infer_obs_pack=3.112s env_step=31.981s feedback_total=14.803s feedback_obs_pack=4.148s feedback_info_pack=0.820s effective_fps=582.82 +2026-09-27 18:11:52.145 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=202547 infer_calls=202547 feedback_calls=202547 infer_wait=296.387s infer_obs_pack=3.355s env_step=34.445s feedback_total=15.966s feedback_obs_pack=4.473s feedback_info_pack=0.886s effective_fps=578.45 +2026-09-27 18:12:22.148 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=217873 infer_calls=217873 feedback_calls=217873 infer_wait=321.807s infer_obs_pack=3.599s env_step=36.905s feedback_total=17.131s feedback_obs_pack=4.794s feedback_info_pack=0.948s effective_fps=574.19 +2026-09-27 18:12:52.151 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=234478 infer_calls=234478 feedback_calls=234478 infer_wait=346.719s infer_obs_pack=3.866s env_step=39.646s feedback_total=18.424s feedback_obs_pack=5.157s feedback_info_pack=1.021s effective_fps=573.78 +2026-09-27 18:13:22.152 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=252425 infer_calls=252425 feedback_calls=252425 infer_wait=370.939s infer_obs_pack=4.176s env_step=42.747s feedback_total=19.898s feedback_obs_pack=5.570s feedback_info_pack=1.101s effective_fps=576.63 +2026-09-27 18:13:52.154 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=270681 infer_calls=270681 feedback_calls=270681 infer_wait=395.043s infer_obs_pack=4.490s env_step=45.917s feedback_total=21.403s feedback_obs_pack=5.993s feedback_info_pack=1.185s effective_fps=579.80 +2026-09-27 18:14:22.351 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=288770 infer_calls=288770 feedback_calls=288770 infer_wait=419.470s infer_obs_pack=4.793s env_step=49.023s feedback_total=22.876s feedback_obs_pack=6.406s feedback_info_pack=1.270s effective_fps=582.01 +2026-09-27 18:14:52.352 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=307045 infer_calls=307045 feedback_calls=307045 infer_wait=443.706s infer_obs_pack=5.098s env_step=52.127s feedback_total=24.342s feedback_obs_pack=6.812s feedback_info_pack=1.351s effective_fps=584.55 +2026-09-27 18:15:22.352 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=325639 infer_calls=325639 feedback_calls=325639 infer_wait=467.554s infer_obs_pack=5.422s env_step=55.404s feedback_total=25.956s feedback_obs_pack=7.248s feedback_info_pack=1.443s effective_fps=587.44 +2026-09-27 18:15:52.353 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=344756 infer_calls=344756 feedback_calls=344756 infer_wait=491.214s infer_obs_pack=5.758s env_step=58.807s feedback_total=27.586s feedback_obs_pack=7.711s feedback_info_pack=1.533s effective_fps=590.98 +2026-09-27 18:16:22.354 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=364350 infer_calls=364350 feedback_calls=364350 infer_wait=514.815s infer_obs_pack=6.099s env_step=62.246s feedback_total=29.235s feedback_obs_pack=8.165s feedback_info_pack=1.626s effective_fps=594.96 +2026-09-27 18:16:52.356 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=383256 infer_calls=383256 feedback_calls=383256 infer_wait=538.603s infer_obs_pack=6.429s env_step=65.582s feedback_total=30.839s feedback_obs_pack=8.605s feedback_info_pack=1.713s effective_fps=597.48 +2026-09-27 18:17:22.356 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=402420 infer_calls=402420 feedback_calls=402420 infer_wait=562.115s infer_obs_pack=6.764s env_step=69.032s feedback_total=32.542s feedback_obs_pack=9.067s feedback_info_pack=1.809s effective_fps=600.22 +2026-09-27 18:17:52.356 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=421642 infer_calls=421642 feedback_calls=421642 infer_wait=585.623s infer_obs_pack=7.107s env_step=72.498s feedback_total=34.246s feedback_obs_pack=9.531s feedback_info_pack=1.902s effective_fps=602.80 +2026-09-27 18:18:22.358 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=441129 infer_calls=441129 feedback_calls=441129 infer_wait=609.175s infer_obs_pack=7.449s env_step=75.958s feedback_total=35.916s feedback_obs_pack=9.997s feedback_info_pack=1.993s effective_fps=605.53 +2026-09-27 18:18:52.359 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=460756 infer_calls=460756 feedback_calls=460756 infer_wait=632.704s infer_obs_pack=7.796s env_step=79.446s feedback_total=37.568s feedback_obs_pack=10.465s feedback_info_pack=2.085s effective_fps=608.25 +2026-09-27 18:19:22.360 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=480180 infer_calls=480180 feedback_calls=480180 infer_wait=656.231s infer_obs_pack=8.140s env_step=82.909s feedback_total=39.258s feedback_obs_pack=10.932s feedback_info_pack=2.179s effective_fps=610.50 +2026-09-27 18:19:52.360 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=499205 infer_calls=499205 feedback_calls=499205 infer_wait=679.955s infer_obs_pack=8.470s env_step=86.279s feedback_total=40.875s feedback_obs_pack=11.391s feedback_info_pack=2.268s effective_fps=612.09 +2026-09-27 18:20:22.555 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=518146 infer_calls=518146 feedback_calls=518146 infer_wait=703.976s infer_obs_pack=8.795s env_step=89.592s feedback_total=42.462s feedback_obs_pack=11.837s feedback_info_pack=2.353s effective_fps=613.32 +2026-09-27 18:20:52.557 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=536752 infer_calls=536752 feedback_calls=536752 infer_wait=728.054s infer_obs_pack=9.111s env_step=92.775s feedback_total=43.984s feedback_obs_pack=12.259s feedback_info_pack=2.437s effective_fps=614.19 +2026-09-27 18:21:22.558 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=554879 infer_calls=554879 feedback_calls=554879 infer_wait=752.275s infer_obs_pack=9.417s env_step=95.878s feedback_total=45.466s feedback_obs_pack=12.677s feedback_info_pack=2.517s effective_fps=614.46 +2026-09-27 18:21:52.559 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=572909 infer_calls=572909 feedback_calls=572909 infer_wait=776.455s infer_obs_pack=9.732s env_step=98.999s feedback_total=46.956s feedback_obs_pack=13.093s feedback_info_pack=2.598s effective_fps=614.62 +2026-09-27 18:22:22.560 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=588332 infer_calls=588332 feedback_calls=588332 infer_wait=801.920s infer_obs_pack=9.972s env_step=101.436s feedback_total=48.100s feedback_obs_pack=13.414s feedback_info_pack=2.660s effective_fps=611.94 +2026-09-27 18:22:52.560 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=604004 infer_calls=604004 feedback_calls=604004 infer_wait=827.288s infer_obs_pack=10.219s env_step=103.930s feedback_total=49.271s feedback_obs_pack=13.737s feedback_info_pack=2.724s effective_fps=609.67 +2026-09-27 18:23:22.560 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=619419 infer_calls=619419 feedback_calls=619419 infer_wait=852.784s infer_obs_pack=10.461s env_step=106.345s feedback_total=50.408s feedback_obs_pack=14.057s feedback_info_pack=2.787s effective_fps=607.28 +2026-09-27 18:23:52.721 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=634882 infer_calls=634882 feedback_calls=634882 infer_wait=878.399s infer_obs_pack=10.698s env_step=108.775s feedback_total=51.571s feedback_obs_pack=14.377s feedback_info_pack=2.850s effective_fps=604.97 +2026-09-27 18:24:22.722 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=650537 infer_calls=650537 feedback_calls=650537 infer_wait=903.822s infer_obs_pack=10.941s env_step=111.238s feedback_total=52.733s feedback_obs_pack=14.695s feedback_info_pack=2.913s effective_fps=603.06 +2026-09-27 18:24:52.722 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=666077 infer_calls=666077 feedback_calls=666077 infer_wait=929.287s infer_obs_pack=11.186s env_step=113.650s feedback_total=53.890s feedback_obs_pack=15.017s feedback_info_pack=2.974s effective_fps=601.15 +2026-09-27 18:25:22.723 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=681845 infer_calls=681845 feedback_calls=681845 infer_wait=954.613s infer_obs_pack=11.444s env_step=116.154s feedback_total=55.078s feedback_obs_pack=15.350s feedback_info_pack=3.038s effective_fps=599.54 +2026-09-27 18:25:52.724 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=698362 infer_calls=698362 feedback_calls=698362 infer_wait=979.571s infer_obs_pack=11.711s env_step=118.847s feedback_total=56.379s feedback_obs_pack=15.711s feedback_info_pack=3.109s effective_fps=598.68 +2026-09-27 18:26:22.725 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=716531 infer_calls=716531 feedback_calls=716531 infer_wait=1003.750s infer_obs_pack=12.018s env_step=121.964s feedback_total=57.882s feedback_obs_pack=16.137s feedback_info_pack=3.191s effective_fps=599.30 +2026-09-27 18:26:52.727 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=734547 infer_calls=734547 feedback_calls=734547 infer_wait=1027.963s infer_obs_pack=12.327s env_step=125.078s feedback_total=59.367s feedback_obs_pack=16.552s feedback_info_pack=3.275s effective_fps=599.76 +2026-09-27 18:27:22.727 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=752656 infer_calls=752656 feedback_calls=752656 infer_wait=1052.185s infer_obs_pack=12.640s env_step=128.180s feedback_total=60.841s feedback_obs_pack=16.961s feedback_info_pack=3.355s effective_fps=600.28 +2026-09-27 18:27:52.728 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=771034 infer_calls=771034 feedback_calls=771034 infer_wait=1076.333s infer_obs_pack=12.956s env_step=131.329s feedback_total=62.330s feedback_obs_pack=17.386s feedback_info_pack=3.438s effective_fps=600.99 +2026-09-27 18:28:22.730 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=789666 infer_calls=789666 feedback_calls=789666 infer_wait=1100.250s infer_obs_pack=13.277s env_step=134.591s feedback_total=63.896s feedback_obs_pack=17.824s feedback_info_pack=3.527s effective_fps=601.87 +2026-09-27 18:28:52.852 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=808962 infer_calls=808962 feedback_calls=808962 infer_wait=1124.014s infer_obs_pack=13.623s env_step=137.995s feedback_total=65.540s feedback_obs_pack=18.277s feedback_info_pack=3.618s effective_fps=603.18 +2026-09-27 18:29:22.852 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=828482 infer_calls=828482 feedback_calls=828482 infer_wait=1147.615s infer_obs_pack=13.961s env_step=141.410s feedback_total=67.214s feedback_obs_pack=18.742s feedback_info_pack=3.714s effective_fps=604.64 +2026-09-27 18:29:53.150 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=847874 infer_calls=847874 feedback_calls=847874 infer_wait=1171.493s infer_obs_pack=14.303s env_step=144.852s feedback_total=68.878s feedback_obs_pack=19.204s feedback_info_pack=3.805s effective_fps=605.83 +2026-09-27 18:30:23.150 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=867282 infer_calls=867282 feedback_calls=867282 infer_wait=1194.980s infer_obs_pack=14.643s env_step=148.328s feedback_total=70.586s feedback_obs_pack=19.677s feedback_info_pack=3.899s effective_fps=607.11 +2026-09-27 18:30:53.151 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=886602 infer_calls=886602 feedback_calls=886602 infer_wait=1218.502s infer_obs_pack=14.982s env_step=151.786s feedback_total=72.280s feedback_obs_pack=20.144s feedback_info_pack=3.995s effective_fps=608.28 +2026-09-27 18:31:23.152 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=905806 infer_calls=905806 feedback_calls=905806 infer_wait=1242.149s infer_obs_pack=15.316s env_step=155.196s feedback_total=73.923s feedback_obs_pack=20.598s feedback_info_pack=4.090s effective_fps=609.32 +2026-09-27 18:31:53.153 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=925210 infer_calls=925210 feedback_calls=925210 infer_wait=1265.697s infer_obs_pack=15.655s env_step=158.657s feedback_total=75.598s feedback_obs_pack=21.060s feedback_info_pack=4.181s effective_fps=610.45 +2026-09-27 18:32:23.155 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=944156 infer_calls=944156 feedback_calls=944156 infer_wait=1289.465s infer_obs_pack=15.985s env_step=161.991s feedback_total=77.214s feedback_obs_pack=21.504s feedback_info_pack=4.271s effective_fps=611.24 +2026-09-27 18:32:53.155 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=963475 infer_calls=963475 feedback_calls=963475 infer_wait=1313.129s infer_obs_pack=16.316s env_step=165.404s feedback_total=78.838s feedback_obs_pack=21.963s feedback_info_pack=4.358s effective_fps=612.24 +2026-09-27 18:33:23.156 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=980598 infer_calls=980598 feedback_calls=980598 infer_wait=1337.966s infer_obs_pack=16.592s env_step=168.173s feedback_total=80.155s feedback_obs_pack=22.328s feedback_info_pack=4.432s effective_fps=611.77 +2026-09-27 18:33:53.157 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Intermediate rollout timing summary: env_steps=995267 infer_calls=995267 feedback_calls=995267 infer_wait=1364.172s infer_obs_pack=16.786s env_step=170.177s feedback_total=81.136s feedback_obs_pack=22.587s feedback_info_pack=4.481s effective_fps=609.74 +2026-09-27 18:34:01.878 | INFO | plugrl_env_client.runner.rollout:log_timing_summary:98 - Final rollout timing summary: env_steps=999425 infer_calls=999425 feedback_calls=999425 infer_wait=1371.466s infer_obs_pack=16.841s env_step=170.745s feedback_total=81.449s feedback_obs_pack=22.658s feedback_info_pack=4.495s effective_fps=609.22 +2026-09-27 18:34:01.879 | INFO | plugrl_env_client.runner.run:run:156 - Server signalled the end of the run; collection stopped after 2636/3997696 episodes. diff --git a/experiments/e38-gaussian-ppo/results/ppo-walker/server-seed0.log b/experiments/e38-gaussian-ppo/results/ppo-walker/server-seed0.log new file mode 100644 index 0000000..0326cd9 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-walker/server-seed0.log @@ -0,0 +1,43 @@ +18:05:41|INFO|plugrl_server version: 0.1.0 +18:05:41|INFO|Algorithm: ppo, Config: PPOAlgoConfig(global_steps=999424, learning_rate=0.0003, anneal_lr=True, buffer_size=2048, batch_size=64, update_epochs=10, gamma=0.99, gae_lambda=0.95, norm_adv=True, clip_coef=0.2, clip_vloss=True, ent_coef=0.0, vf_coef=0.5, max_grad_norm=0.5, target_kl=None, normalize_rewards=True, reward_clip=10.0, train_itrs=488, save_interval=20) +18:05:41|INFO|Policy: gaussian-policy, Config: GaussianPolicyConfig(algo='ppo', device=device(type='cpu'), obs_dim=17, action_dim=6, hidden_dims=(64, 64), state_keys=('obs',), normalize_observations=True, obs_clip=10.0, freeze_obs_stats=False, action_clip=1.0) +18:05:41|INFO|Seeded python, numpy and torch with 0. One env client is then reproducible; several are not, because batch composition depends on when their requests arrive. +18:05:41|INFO|Checkpoint Manager created: + at /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0 +18:05:41|INFO|Policy created... +18:05:41|INFO|Initialized RolloutBuffer buffer_size=2048 action_shape=(2048, 6) value_shape=(2048,) +18:05:41|INFO|Algorithm created: + +18:05:41|INFO|Agent Server is listening on 0.0.0.0:9960 +18:06:46|INFO|Checkpoint saved at step 40961 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/40961 +18:07:53|INFO|Checkpoint saved at step 81921 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/81921 +18:09:05|INFO|Checkpoint saved at step 122881 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/122881 +18:10:24|INFO|Checkpoint saved at step 163841 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/163841 +18:11:44|INFO|Checkpoint saved at step 204801 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/204801 +18:13:00|INFO|Checkpoint saved at step 245761 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/245761 +18:14:08|INFO|Checkpoint saved at step 286720 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/286720 +18:15:16|INFO|Checkpoint saved at step 327681 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/327681 +18:16:19|INFO|Checkpoint saved at step 368641 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/368641 +18:17:23|INFO|Checkpoint saved at step 409601 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/409601 +18:18:26|INFO|Checkpoint saved at step 450561 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/450561 +18:19:31|INFO|Checkpoint saved at step 491521 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/491521 +18:20:36|INFO|Checkpoint saved at step 532481 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/532481 +18:21:44|INFO|Checkpoint saved at step 573441 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/573441 +18:23:02|INFO|Checkpoint saved at step 614401 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/614401 +18:24:21|INFO|Checkpoint saved at step 655361 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/655361 +18:25:40|INFO|Checkpoint saved at step 696321 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/696321 +18:26:49|INFO|Checkpoint saved at step 737281 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/737281 +18:27:56|INFO|Checkpoint saved at step 778241 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/778241 +18:29:01|INFO|Checkpoint saved at step 819201 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/819201 +18:30:05|INFO|Checkpoint saved at step 860161 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/860161 +18:31:09|INFO|Checkpoint saved at step 901121 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/901121 +18:32:14|INFO|Checkpoint saved at step 942081 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/942081 +18:33:21|INFO|Checkpoint saved at step 983041 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/983041 +18:33:55|INFO|Checkpoint saved at step 999425 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed0/999425 +18:33:55|INFO|Stopping server as the algorithm signaled to stop. +18:33:55|INFO|Shutdown started: aborting pending infer requests. +18:33:55|INFO|Shutdown cleanup finished: pending_futures=1, drained_requests=1 +18:33:55|INFO|Shutdown closing 1 websocket connection(s). +18:33:55|INFO|Shutdown interrupted an in-flight inference request from ('127.0.0.1', 34980). +18:33:55|INFO|WebSocket server closed. +18:33:55|INFO|Scheduler task cancelled and cleaned up. diff --git a/experiments/e38-gaussian-ppo/results/ppo-walker/server-seed1.log b/experiments/e38-gaussian-ppo/results/ppo-walker/server-seed1.log new file mode 100644 index 0000000..ce9ca44 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-walker/server-seed1.log @@ -0,0 +1,43 @@ +18:05:46|INFO|plugrl_server version: 0.1.0 +18:05:46|INFO|Algorithm: ppo, Config: PPOAlgoConfig(global_steps=999424, learning_rate=0.0003, anneal_lr=True, buffer_size=2048, batch_size=64, update_epochs=10, gamma=0.99, gae_lambda=0.95, norm_adv=True, clip_coef=0.2, clip_vloss=True, ent_coef=0.0, vf_coef=0.5, max_grad_norm=0.5, target_kl=None, normalize_rewards=True, reward_clip=10.0, train_itrs=488, save_interval=20) +18:05:46|INFO|Policy: gaussian-policy, Config: GaussianPolicyConfig(algo='ppo', device=device(type='cpu'), obs_dim=17, action_dim=6, hidden_dims=(64, 64), state_keys=('obs',), normalize_observations=True, obs_clip=10.0, freeze_obs_stats=False, action_clip=1.0) +18:05:46|INFO|Seeded python, numpy and torch with 1. One env client is then reproducible; several are not, because batch composition depends on when their requests arrive. +18:05:46|INFO|Checkpoint Manager created: + at /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1 +18:05:46|INFO|Policy created... +18:05:46|INFO|Initialized RolloutBuffer buffer_size=2048 action_shape=(2048, 6) value_shape=(2048,) +18:05:46|INFO|Algorithm created: + +18:05:46|INFO|Agent Server is listening on 0.0.0.0:9961 +18:06:51|INFO|Checkpoint saved at step 40961 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/40961 +18:07:58|INFO|Checkpoint saved at step 81921 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/81921 +18:09:09|INFO|Checkpoint saved at step 122881 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/122881 +18:10:27|INFO|Checkpoint saved at step 163841 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/163841 +18:11:45|INFO|Checkpoint saved at step 204801 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/204801 +18:12:59|INFO|Checkpoint saved at step 245761 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/245761 +18:14:06|INFO|Checkpoint saved at step 286721 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/286721 +18:15:12|INFO|Checkpoint saved at step 327681 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/327681 +18:16:16|INFO|Checkpoint saved at step 368641 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/368641 +18:17:20|INFO|Checkpoint saved at step 409601 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/409601 +18:18:24|INFO|Checkpoint saved at step 450561 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/450561 +18:19:28|INFO|Checkpoint saved at step 491521 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/491521 +18:20:33|INFO|Checkpoint saved at step 532481 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/532481 +18:21:40|INFO|Checkpoint saved at step 573441 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/573441 +18:22:56|INFO|Checkpoint saved at step 614401 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/614401 +18:24:13|INFO|Checkpoint saved at step 655361 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/655361 +18:25:29|INFO|Checkpoint saved at step 696321 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/696321 +18:26:38|INFO|Checkpoint saved at step 737280 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/737280 +18:27:46|INFO|Checkpoint saved at step 778241 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/778241 +18:28:50|INFO|Checkpoint saved at step 819201 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/819201 +18:29:53|INFO|Checkpoint saved at step 860161 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/860161 +18:30:56|INFO|Checkpoint saved at step 901121 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/901121 +18:32:00|INFO|Checkpoint saved at step 942081 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/942081 +18:33:04|INFO|Checkpoint saved at step 983041 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/983041 +18:33:35|INFO|Checkpoint saved at step 999425 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed1/999425 +18:33:35|INFO|Stopping server as the algorithm signaled to stop. +18:33:35|INFO|Shutdown started: aborting pending infer requests. +18:33:35|INFO|Shutdown cleanup finished: pending_futures=1, drained_requests=1 +18:33:35|INFO|Shutdown closing 1 websocket connection(s). +18:33:35|INFO|Shutdown interrupted an in-flight inference request from ('127.0.0.1', 45962). +18:33:35|INFO|WebSocket server closed. +18:33:35|INFO|Scheduler task cancelled and cleaned up. diff --git a/experiments/e38-gaussian-ppo/results/ppo-walker/server-seed2.log b/experiments/e38-gaussian-ppo/results/ppo-walker/server-seed2.log new file mode 100644 index 0000000..e68fcae --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/ppo-walker/server-seed2.log @@ -0,0 +1,43 @@ +18:05:51|INFO|plugrl_server version: 0.1.0 +18:05:51|INFO|Algorithm: ppo, Config: PPOAlgoConfig(global_steps=999424, learning_rate=0.0003, anneal_lr=True, buffer_size=2048, batch_size=64, update_epochs=10, gamma=0.99, gae_lambda=0.95, norm_adv=True, clip_coef=0.2, clip_vloss=True, ent_coef=0.0, vf_coef=0.5, max_grad_norm=0.5, target_kl=None, normalize_rewards=True, reward_clip=10.0, train_itrs=488, save_interval=20) +18:05:51|INFO|Policy: gaussian-policy, Config: GaussianPolicyConfig(algo='ppo', device=device(type='cpu'), obs_dim=17, action_dim=6, hidden_dims=(64, 64), state_keys=('obs',), normalize_observations=True, obs_clip=10.0, freeze_obs_stats=False, action_clip=1.0) +18:05:51|INFO|Seeded python, numpy and torch with 2. One env client is then reproducible; several are not, because batch composition depends on when their requests arrive. +18:05:51|INFO|Checkpoint Manager created: + at /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2 +18:05:51|INFO|Policy created... +18:05:51|INFO|Initialized RolloutBuffer buffer_size=2048 action_shape=(2048, 6) value_shape=(2048,) +18:05:51|INFO|Algorithm created: + +18:05:51|INFO|Agent Server is listening on 0.0.0.0:9962 +18:06:56|INFO|Checkpoint saved at step 40961 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/40961 +18:08:04|INFO|Checkpoint saved at step 81921 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/81921 +18:09:16|INFO|Checkpoint saved at step 122881 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/122881 +18:10:36|INFO|Checkpoint saved at step 163841 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/163841 +18:11:56|INFO|Checkpoint saved at step 204801 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/204801 +18:13:11|INFO|Checkpoint saved at step 245761 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/245761 +18:14:19|INFO|Checkpoint saved at step 286721 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/286721 +18:15:25|INFO|Checkpoint saved at step 327681 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/327681 +18:16:29|INFO|Checkpoint saved at step 368641 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/368641 +18:17:34|INFO|Checkpoint saved at step 409601 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/409601 +18:18:37|INFO|Checkpoint saved at step 450561 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/450561 +18:19:40|INFO|Checkpoint saved at step 491521 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/491521 +18:20:45|INFO|Checkpoint saved at step 532481 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/532481 +18:21:53|INFO|Checkpoint saved at step 573441 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/573441 +18:23:12|INFO|Checkpoint saved at step 614401 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/614401 +18:24:32|INFO|Checkpoint saved at step 655361 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/655361 +18:25:49|INFO|Checkpoint saved at step 696321 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/696321 +18:26:57|INFO|Checkpoint saved at step 737281 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/737281 +18:28:04|INFO|Checkpoint saved at step 778241 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/778241 +18:29:08|INFO|Checkpoint saved at step 819201 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/819201 +18:30:12|INFO|Checkpoint saved at step 860161 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/860161 +18:31:15|INFO|Checkpoint saved at step 901121 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/901121 +18:32:19|INFO|Checkpoint saved at step 942081 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/942081 +18:33:28|INFO|Checkpoint saved at step 983041 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/983041 +18:34:01|INFO|Checkpoint saved at step 999425 to /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/results/ppo-walker/ppo/gaussian-policy/ppo-walker-seed2/999425 +18:34:01|INFO|Stopping server as the algorithm signaled to stop. +18:34:01|INFO|Shutdown started: aborting pending infer requests. +18:34:01|INFO|Shutdown cleanup finished: pending_futures=1, drained_requests=1 +18:34:01|INFO|Shutdown closing 1 websocket connection(s). +18:34:01|INFO|Shutdown interrupted an in-flight inference request from ('127.0.0.1', 53540). +18:34:01|INFO|WebSocket server closed. +18:34:01|INFO|Scheduler task cancelled and cleaned up. diff --git a/experiments/e38-gaussian-ppo/results/run.out b/experiments/e38-gaussian-ppo/results/run.out new file mode 100644 index 0000000..567905d --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/run.out @@ -0,0 +1,9 @@ +start 2026-09-27 18:04:58 +code: 2f70733 (src /home/guangzhao/zuogou/plugrl/e38/src) +client: 931ab56 +iters: 488 seeds: 0 1 2 +end 2026-09-27 18:34:01 +ppo-cheetah failed seeds: 0 +ppo-hopper failed seeds: 0 +ppo-walker failed seeds: 0 +E38_DONE diff --git a/experiments/e38-gaussian-ppo/results/verdicts.txt b/experiments/e38-gaussian-ppo/results/verdicts.txt new file mode 100644 index 0000000..70af830 --- /dev/null +++ b/experiments/e38-gaussian-ppo/results/verdicts.txt @@ -0,0 +1,50 @@ +P1 every cell runs end to end + ppo-cheetah HalfCheetah-v5 HOLDS (28 min) + ppo-hopper Hopper-v5 HOLDS (28 min) + ppo-walker Walker2d-v5 HOLDS (28 min) + +V1 every iteration learned at CleanRL's annealed rate + ppo-cheetah 0.0003 to 6.15e-07 PASS + ppo-hopper 0.0003 to 6.15e-07 PASS + ppo-walker 0.0003 to 6.15e-07 PASS + +status learns: Hopper, Walker2d 479-488 >= 500; HalfCheetah 479-488 minus the first >= +200; on 2 of 3 + ppo-cheetah +1848.2, +1827.1, +1951.7 3 of 3 -> learns + ppo-hopper 2298.7, 2207.4, 2181.3 3 of 3 -> learns + ppo-walker 2957.2, 3260.0, 3154.2 3 of 3 -> learns + +P2 gaussian-policy learns Hopper under ppo + HOLDS + +P3 gaussian-policy learns Walker2d under ppo + HOLDS + +P4 gaussian-policy learns HalfCheetah under ppo + HOLDS + +reported: mean return over iterations 91-100, 241-250 and 479-488, beside CleanRL's on the -v4 task + ppo-cheetah seed 0 744.4 1322.5 1512.9 + ppo-cheetah seed 1 999.0 1348.1 1442.5 + ppo-cheetah seed 2 925.5 1428.5 1557.8 + ppo-cheetah CleanRL 1442.64 +/- 46.03 + ppo-hopper seed 0 1623.2 2242.8 2298.7 + ppo-hopper seed 1 1297.1 2486.9 2207.4 + ppo-hopper seed 2 920.3 2440.6 2181.3 + ppo-hopper CleanRL 2382.86 +/- 271.74 + ppo-walker seed 0 514.9 2747.0 2957.2 + ppo-walker seed 1 476.6 2037.8 3260.0 + ppo-walker seed 2 652.6 3344.2 3154.2 + ppo-walker CleanRL 2287.95 +/- 571.78 + +reported: first iteration and mean of the last ten, return and episode length + ppo-cheetah seed 0 return -335.2 -> 1512.9 length 1000.0 -> 1000.0 + ppo-cheetah seed 1 return -384.6 -> 1442.5 length 1000.0 -> 1000.0 + ppo-cheetah seed 2 return -393.9 -> 1557.8 length 1000.0 -> 1000.0 + ppo-hopper seed 0 return 12.1 -> 2298.7 length 17.6 -> 636.3 + ppo-hopper seed 1 return 15.2 -> 2207.4 length 19.5 -> 585.8 + ppo-hopper seed 2 return 14.2 -> 2181.3 length 18.7 -> 600.9 + ppo-walker seed 0 return -0.7 -> 2957.2 length 19.1 -> 685.8 + ppo-walker seed 1 return -1.8 -> 3260.0 length 17.7 -> 754.6 + ppo-walker seed 2 return -0.6 -> 3154.2 length 18.9 -> 792.4 + +wrote /home/guangzhao/zuogou/plugrl/e38/experiments/e38-gaussian-ppo/summary.tsv diff --git a/experiments/e38-gaussian-ppo/summary.tsv b/experiments/e38-gaussian-ppo/summary.tsv new file mode 100644 index 0000000..b1a44fa --- /dev/null +++ b/experiments/e38-gaussian-ppo/summary.tsv @@ -0,0 +1,10 @@ +cell policy algorithm task claim seed iterations runs first_return last10_return first_length last10_length +ppo-cheetah gaussian-policy ppo HalfCheetah-v5 learns 0 488 true -335.2 1512.9 1000.0 1000.0 +ppo-cheetah gaussian-policy ppo HalfCheetah-v5 learns 1 488 true -384.6 1442.5 1000.0 1000.0 +ppo-cheetah gaussian-policy ppo HalfCheetah-v5 learns 2 488 true -393.9 1557.8 1000.0 1000.0 +ppo-hopper gaussian-policy ppo Hopper-v5 learns 0 488 true 12.1 2298.7 17.6 636.3 +ppo-hopper gaussian-policy ppo Hopper-v5 learns 1 488 true 15.2 2207.4 19.5 585.8 +ppo-hopper gaussian-policy ppo Hopper-v5 learns 2 488 true 14.2 2181.3 18.7 600.9 +ppo-walker gaussian-policy ppo Walker2d-v5 learns 0 488 true -0.7 2957.2 19.1 685.8 +ppo-walker gaussian-policy ppo Walker2d-v5 learns 1 488 true -1.8 3260.0 17.7 754.6 +ppo-walker gaussian-policy ppo Walker2d-v5 learns 2 488 true -0.6 3154.2 18.9 792.4 From 3906e45c95799120231fc2cac3059e55b96731cd Mon Sep 17 00:00:00 2001 From: tactino <18781106300@163.com> Date: Sun, 27 Sep 2026 18:48:43 -0400 Subject: [PATCH 8/8] test: value_loss_coeff changes nothing to float32 rounding, not to the bit The factors are powers of two, but Adam's eps is added unscaled and leaves a trace in the last bits of elements whose gradient is within a few orders of it. After this branch changed the toy's returns, one parameter came out a few bits apart on CI and identical on Windows and on a Linux workstation, all torch 2.7.1. --- tests/test_fpo_value_loss_coeff.py | 39 ++++++++++++++++++++---------- 1 file changed, 26 insertions(+), 13 deletions(-) diff --git a/tests/test_fpo_value_loss_coeff.py b/tests/test_fpo_value_loss_coeff.py index b717e11..4ae2565 100644 --- a/tests/test_fpo_value_loss_coeff.py +++ b/tests/test_fpo_value_loss_coeff.py @@ -17,9 +17,10 @@ trunk and the disjoint critic and nothing else Measured on the bandit: coefficients 0.25, 1.0 and 4.0 give byte-identical -results across five seeds. The apparent effect below about 0.05 is Adam's -epsilon becoming comparable to the scaled second moment, not the critic -learning differently. +results across five seeds on the machine that measured it - identical to +float32 rounding is what holds everywhere (see the test below). The apparent +effect below about 0.05 is Adam's epsilon becoming comparable to the scaled +second moment, not the critic learning differently. This is not a behaviour change. Redefining the knob would silently alter what it means for anyone reading it, and nothing can depend on it today because it @@ -70,20 +71,32 @@ def test_the_critic_receives_no_gradient_from_the_policy_loss(): def test_scaling_the_value_loss_changes_nothing(): - """0.25, 1.0 and 4.0 give the same weights, to the bit.""" + """0.25, 1.0 and 4.0 give the same weights, to float32 rounding. + + Not to the bit. The factors are powers of two, so the scaled gradients + and moments are exact, but `eps` is added to sqrt(v) unscaled and leaves + a trace in the last bits of any element whose gradient is within a few + orders of it. Which elements, and whether the trace survives rounding, + depends on the platform's kernels: after #86 changed this toy's returns, + one parameter came out a few bits apart on CI and identical on Windows + and on a Linux workstation, all on torch 2.7.1. + """ baseline = _run(0.25) for coeff in (1.0, 4.0): other = _run(coeff) assert len(baseline) == len(other) - differing = [ - i for i, (a, b) in enumerate(zip(baseline, other)) if not torch.equal(a, b) - ] - assert not differing, ( - f"value_loss_coeff={coeff} changed {len(differing)} parameters " - "against 0.25. If this starts failing, either the critic is no " - "longer disjoint or the optimizer is no longer scale-invariant, " - "and the docstring above needs rewriting rather than the test." - ) + for i, (a, b) in enumerate(zip(baseline, other)): + torch.testing.assert_close( + b, + a, + msg=lambda m, i=i: ( + f"value_loss_coeff={coeff} moved parameter {i} against 0.25 " + f"beyond float32 rounding: {m} If this fails, either the " + "critic is no longer disjoint or the optimizer is no " + "longer scale-invariant, and the docstring above needs " + "rewriting rather than the test." + ), + ) def test_a_learning_rate_is_not_cancelled():