In [None]:
import numpy as np
import numba
import umap
import pynndescent

print("NumPy version:", np.__version__)
print("Numba version:", numba.__version__)
print("UMAP version:", umap.__version__)
print("PyNNDescent version:", pynndescent.__version__)


In [1]:
import os
import json
import ale_py


import torch as T
import torch.nn as nn
from torch import optim
import numpy as np
# import pandas as pd
# from umap import UMAP


import torch_utils
from torch import distributions

import gymnasium as gym
import gymnasium_robotics
from gymnasium.vector import VectorEnv, SyncVectorEnv
# import models
from models import ValueModel, StochasticContinuousPolicy, ActorModel, CriticModel, StochasticDiscretePolicy
from rl_agents import PPO, DDPG, Reinforce, ActorCritic, TD3, HER
import rl_callbacks
from rl_callbacks import WandbCallback
# from helper import Normalizer
from buffer import ReplayBuffer, PrioritizedReplayBuffer
from noise import NormalNoise
import gym_helper
import wandb_support
import wandb
import gym_helper
import dash_utils
from env_wrapper import EnvWrapper, GymnasiumWrapper
from schedulers import ScheduleWrapper

import matplotlib.pyplot as plt

# from mpi4py import MPI

In [None]:
import mujoco

In [None]:
print(f'mujoco version: {mujoco.__version__}')

In [None]:
env = gym.make('FetchReach-v4')
env_spec = env.spec
wrap_env = GymnasiumWrapper(env_spec)

In [None]:
state, _ = env.reset()

In [None]:
env.env.env.env.initial_qpos

In [None]:
wrap_env.env = wrap_env._initialize_env(num_envs=8)

In [None]:
states, _ = wrap_env.reset()

In [None]:
states

In [None]:
mujoco.MjModel

In [None]:
gym_robo.__version__

In [None]:
def check_cuda():
    cuda_available = T.cuda.is_available()
    if cuda_available:
        print("CUDA is available.")
        num_gpus = T.cuda.device_count()
        print(f"Number of GPUs detected: {num_gpus}")
        
        for i in range(num_gpus):
            gpu_name = T.cuda.get_device_name(i)
            gpu_memory = T.cuda.get_device_properties(i).total_memory / (1024 ** 3)  # Convert bytes to GB
            print(f"GPU {i}: {gpu_name}")
            print(f"Total memory: {gpu_memory:.2f} GB")
    else:
        print("CUDA is not available.")

check_cuda()

In [None]:
def get_default_device():
    """Returns the default device for computations, GPU if available, otherwise CPU"""
    if T.cuda.is_available():
        return T.device('cuda')
    else:
        return T.device('cpu')

device = get_default_device()
print(f"Using device: {device}")

# TEST

In [None]:
gym_robo.register_robotics_envs()

In [None]:
gym.register_envs(gymnasium_robotics)

In [None]:
gym.envs.registration.registry

In [None]:
wandb.login(key='758ac5ba01e12a3df504d2db2fec8ba4f391f7e6')

In [None]:
env = gym.make('FetchPush-v2', max_episode_steps=100, render_mode='rgb_array')
env = gym.wrappers.RecordVideo(env, 'test/', episode_trigger=lambda i: i%1==0)

episodes = 10


for episode in range(episodes):
    done = False
    obs, _ = env.reset()
    while not done:
        obs, r, term, trunc, dict = env.step(env.action_space.sample())
        if term or trunc:
            done = True
env.close()

In [None]:
env = gym.make("FetchReach-v2")
env.reset()
obs, reward, terminated, truncated, info = env.step(env.action_space.sample())

# The following always has to hold:
assert reward == env.compute_reward(obs["achieved_goal"], obs["desired_goal"], info)
assert truncated == env.compute_truncated(obs["achieved_goal"], obs["desired_goal"], info)
assert terminated == env.compute_terminated(obs["achieved_goal"], obs["desired_goal"], info)

In [None]:
env.compute_reward()

In [None]:
env = gym.make('FetchPush-v2', render_mode='rgb_array')

In [None]:
if hasattr(env, "distance_threshold"):
    print('true')
else:
    print('false')

In [None]:
if env.get_wrapper_attr("distance_threshold"):
    print('true')

In [None]:
print(dir(env))


# DDPG

In [2]:
# env = gym.make('BipedalWalker-v3')
env = gym.make('Pendulum-v1')

env_spec = env.spec
env_wrap = GymnasiumWrapper(env_spec)

In [None]:
for e in env_wrap.env.envs:
    print(e.spec)

In [4]:
# build actor
device = 'cuda'
optimizer = {'type': 'Adam','params': { 'lr': 0.001 }}

layer_config = [
    {'type': 'dense', 'params': {'units': 400, 'kernel': 'variance_scaling', 'kernel params':{"scale": 1.0, "mode": "fan_in", "distribution": "uniform"}}},
    {'type': 'relu'},
    {'type': 'dense', 'params': {'units': 300, 'kernel': 'variance_scaling', 'kernel params':{"scale": 1.0, "mode": "fan_in", "distribution": "uniform"}}},
    {'type': 'relu'},
]
output_layer_config = [{'type': 'dense', 'params': {'kernel': 'default', 'kernel params':{}}}]

actor = ActorModel(env_wrap, layer_config, output_layer_config, optimizer_params=optimizer, device=device)

In [None]:
actor

In [6]:
# build critic

state_layer_config = [
    {'type': 'dense', 'params': {'units': 400, 'kernel': 'variance_scaling', 'kernel params':{"scale": 1.0, "mode": "fan_in", "distribution": "uniform"}}},
    {'type': 'relu'}
]

merged_layer_config = [
    {'type': 'dense', 'params': {'units': 300, 'kernel': 'variance_scaling', 'kernel params':{"scale": 1.0, "mode": "fan_in", "distribution": "uniform"}}},
    {'type': 'relu'},
]
# output_layer_config = {'type': 'dense', 'params': {'kernel': 'default', 'kernel params':{}}},

critic = CriticModel(env_wrap, state_layers=state_layer_config, merged_layers=merged_layer_config,
                    output_layer_kernel=output_layer_config, optimizer_params=optimizer, device=device)

In [None]:
critic

In [None]:
# replay_buffer = ReplayBuffer(env_wrap, 100000, device=device)
replay_buffer = PrioritizedReplayBuffer(env_wrap, 200000, alpha=0.6, beta_start=0.4, beta_iter=300, beta_update_freq=1, normalize=True, epsilon=0.01,device=device)
noise = NormalNoise(shape=env_wrap.action_space.shape, stddev=0.1, device=device)

In [None]:
replay_buffer.get_config()

In [None]:
noise.get_config()

In [11]:
ddpg_agent = DDPG(env=env_wrap,
                actor_model=actor,
                critic_model=critic,
                replay_buffer=replay_buffer,
                discount=0.99,
                tau=0.05,
                action_epsilon=0.2,
                batch_size=128,
                noise=noise,
                warmup=500,
                callbacks=[rl_callbacks.WandbCallback('Pendulum-v1')],
                device=device)

In [None]:
ddpg_agent.train(2000, 16, 42, 0)

In [None]:
T.unique(ddpg_agent.replay_buffer.states).size()

In [None]:
ddpg_agent.test(10, True, 1)

In [None]:
config_file_path = '/workspaces/RL_Agents/src/app/models/ddpg/config.json'
with open(config_file_path, 'r') as file:
    config = json.load(file)

In [None]:
ddpg = DDPG.load(config)

In [None]:
ddpg.get_config()

In [None]:
ddpg.test(10, 1)

# TD3

In [2]:
# env = gym.make('BipedalWalker-v3')
env = gym.make('Pendulum-v1')
env_spec = env.spec
env_wrap = GymnasiumWrapper(env_spec)

In [3]:
# build actor
device = 'cuda'
optimizer = {'type': 'Adam','params': { 'lr': 0.001 }}

layer_config = [
    {'type': 'dense', 'params': {'units': 400, 'kernel': 'variance_scaling', 'kernel params':{"scale": 1.0, "mode": "fan_in", "distribution": "uniform"}}},
    {'type': 'relu'},
    {'type': 'dense', 'params': {'units': 300, 'kernel': 'variance_scaling', 'kernel params':{"scale": 1.0, "mode": "fan_in", "distribution": "uniform"}}},
    {'type': 'relu'},
]
output_layer_config = [{'type': 'dense', 'params': {'kernel': 'default', 'kernel params':{}}}]

actor = ActorModel(env_wrap, layer_config, output_layer_config, device=device)

In [4]:
# build critic

state_layer_config = [
    {'type': 'dense', 'params': {'units': 400, 'kernel': 'variance_scaling', 'kernel params':{"scale": 1.0, "mode": "fan_in", "distribution": "uniform"}}},
    {'type': 'relu'}
]

merged_layer_config = [
    {'type': 'dense', 'params': {'units': 300, 'kernel': 'variance_scaling', 'kernel params':{"scale": 1.0, "mode": "fan_in", "distribution": "uniform"}}},
    {'type': 'relu'},
]
# output_layer_config = {'type': 'dense', 'params': {'kernel': 'default', 'kernel params':{}}},

critic = CriticModel(env_wrap, state_layers=state_layer_config, merged_layers=merged_layer_config,
                    output_layer_kernel=output_layer_config, optimizer_params=optimizer, device=device)

In [None]:
replay_buffer = ReplayBuffer(env_wrap, 100000, device=device)
noise = NormalNoise(shape=env_wrap.action_space.shape, stddev=0.1, device=device)

In [6]:
td3 = TD3(
    env=env_wrap,
    actor_model=actor,
    critic_model=critic,
    discount=0.99,
    tau=0.05,
    action_epsilon=0.2,
    replay_buffer=replay_buffer,
    noise=noise,
    target_noise=noise,
    actor_update_delay = 2,
    normalize_inputs=True,
    warmup=200,
    # callbacks=[rl_callbacks.WandbCallback('Pendulum-v1')],
    device='cuda'
)

In [None]:
td3.target_noise.device

In [None]:
td3.train(200, 8, 42, 0)

In [None]:
float(td3.env.action_space.low[-1])

In [8]:
td3.save()

In [2]:
config_file_path = '/workspaces/RL_Agents/src/app/test/td3/config.json'
with open(config_file_path, 'r') as file:
    config = json.load(file)

In [3]:
td3 = TD3.load(config)

In [None]:
td3.get_config()

In [None]:
td3.state_normalizer.device

# HER/DDPG

In [2]:
env = gym.make('FetchReach-v4')
env_spec = env.spec
env_wrap = GymnasiumWrapper(env_spec)

In [3]:
# GOAL SHAPE
goal_shape = env.observation_space['desired_goal'].shape
print(f'goal_shape: {goal_shape}')

goal_shape: (3,)


In [4]:
# build actor
device = 'cuda'
optimizer = {'type': 'Adam','params': { 'lr': 0.001 }}

layer_config = [
    {'type': 'dense', 'params': {'units': 64, 'kernel': 'xavier_uniform', 'kernel params':{"gain": 1.0}}},
    {'type': 'relu'},
    {'type': 'dense', 'params': {'units': 64, 'kernel': 'xavier_uniform', 'kernel params':{"gain": 1.0}}},
    {'type': 'relu'},
    {'type': 'dense', 'params': {'units': 64, 'kernel': 'xavier_uniform', 'kernel params':{"gain": 1.0}}},
    {'type': 'relu'},
]
output_layer_config = [{'type': 'dense', 'params': {'kernel': 'uniform', 'kernel params':{'a':-3e-3, 'b':3e-3}}}]

actor = ActorModel(env_wrap, layer_config, output_layer_config, device=device)

In [5]:
# build critic

state_layer_config = [
    {'type': 'dense', 'params': {'units': 64, 'kernel': 'xavier_uniform', 'kernel params':{"gain": 1.0}}},
    {'type': 'relu'},
    {'type': 'dense', 'params': {'units': 64, 'kernel': 'xavier_uniform', 'kernel params':{"gain": 1.0}}},
    {'type': 'relu'},
]

merged_layer_config = [
    
    {'type': 'dense', 'params': {'units': 64, 'kernel': 'xavier_uniform', 'kernel params':{"gain": 1.0}}},
    {'type': 'relu'}
]
# output_layer_config = {'type': 'dense', 'params': {'kernel': 'default', 'kernel params':{}}},

critic = CriticModel(env_wrap, state_layers=state_layer_config, merged_layers=merged_layer_config,
                    output_layer_kernel=output_layer_config, optimizer_params=optimizer, device=device)

In [9]:
# replay_buffer = ReplayBuffer(env_wrap, 100000, goal_shape=env.observation_space['desired_goal'].shape, device=device)
replay_buffer = PrioritizedReplayBuffer(env_wrap, 10000, beta_start=0.4, beta_iter=3000, beta_update_freq=1, normalize=False, goal_shape=goal_shape, epsilon=0.01, device=device)
noise = NormalNoise(shape=env_wrap.action_space.shape, mean=0.0, stddev=0.1, device=device)
# schedule_config = {'type':'Linear', 'params':{'start_factor':1.0, 'end_factor':0.1, 'total_iters':5000}}
# noise_schedule = ScheduleWrapper(schedule_config)
noise_schedule = None

shape: (10000, 10)


In [10]:
replay_buffer.get_config()

{'class_name': 'PrioritizedReplayBuffer',
 'config': {'env': '{"type": "GymnasiumWrapper", "env": "{\\"id\\": \\"FetchReach-v4\\", \\"entry_point\\": \\"gymnasium_robotics.envs.fetch.reach:MujocoFetchReachEnv\\", \\"reward_threshold\\": null, \\"nondeterministic\\": false, \\"max_episode_steps\\": 50, \\"order_enforce\\": true, \\"disable_env_checker\\": false, \\"kwargs\\": {\\"reward_type\\": \\"sparse\\"}, \\"additional_wrappers\\": [], \\"vector_entry_point\\": null}", "wrappers": null}',
  'buffer_size': 10000,
  'alpha': 0.6,
  'beta_start': 0.4,
  'beta_iter': 3000,
  'beta_update_freq': 1,
  'priority': 'proportional',
  'normalize': False,
  'goal_shape': (3,),
  'epsilon': 0.01,
  'device': 'cuda'}}

In [11]:
ddpg_agent = DDPG(env=env_wrap,
                actor_model=actor,
                critic_model=critic,
                replay_buffer=replay_buffer,
                discount=0.95,
                tau=0.05,
                action_epsilon=0.2,
                batch_size=128,
                noise=noise,
                noise_schedule=noise_schedule,
                normalize_inputs=True,
                warmup=0,
                callbacks=[rl_callbacks.WandbCallback('FetchReach-v4')],
                device=device)

In [12]:
her = HER(
    agent=ddpg_agent,
    strategy='future',
    tolerance=0.05,
    num_goals=4,
)

In [13]:
num_epochs = 100
num_cycles = 50
num_episodes = 1
num_updates = 40
render_freq = 100
num_envs = 16
seed = 42

her.train(num_epochs, num_cycles, num_episodes, num_updates, render_freq, num_envs, seed)

[34m[1mwandb[0m: Using wandb-core as the SDK backend.  Please refer to https://wandb.me/wandb-core for more information.
[34m[1mwandb[0m: Currently logged in as: [33mjasonhayes1987[0m to [32mhttps://api.wandb.ai[0m. Use [1m`wandb login --relogin`[0m to force relogin


[34m[1mwandb[0m: logging graph, to disable use `wandb.watch(log_graph=False)`
[34m[1mwandb[0m: logging graph, to disable use `wandb.watch(log_graph=False)`


Environment 0: episode 1, score -50.0, avg_score -50.0
Environment 1: episode 1, score -50.0, avg_score -50.0
Environment 2: episode 1, score -50.0, avg_score -50.0
Environment 3: episode 1, score -50.0, avg_score -50.0
Environment 4: episode 1, score -50.0, avg_score -50.0
Environment 5: episode 1, score -50.0, avg_score -50.0
Environment 6: episode 1, score -32.0, avg_score -47.42857142857143
Environment 7: episode 1, score -22.0, avg_score -44.25
Environment 8: episode 1, score -50.0, avg_score -44.888888888888886
Environment 9: episode 1, score -50.0, avg_score -45.4
Environment 10: episode 1, score -28.0, avg_score -43.81818181818182
Environment 11: episode 1, score -50.0, avg_score -44.333333333333336
Environment 12: episode 1, score -50.0, avg_score -44.76923076923077
Environment 13: episode 1, score -50.0, avg_score -45.142857142857146
Environment 14: episode 1, score -50.0, avg_score -45.46666666666667
Environment 15: episode 1, score -50.0, avg_score -45.75
Environment 0: epi

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_100.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 3: episode 7, score -50.0, avg_score -48.97
Environment 4: episode 7, score -50.0, avg_score -48.97
Environment 5: episode 7, score -50.0, avg_score -48.97
Environment 6: episode 7, score -50.0, avg_score -48.97
Environment 7: episode 7, score -50.0, avg_score -48.97
Environment 8: episode 7, score -50.0, avg_score -48.97
Environment 9: episode 7, score -50.0, avg_score -48.97
Environment 10: episode 7, score -50.0, avg_score -49.15
Environment 11: episode 7, score -50.0, avg_score -49.43
Environment 12: episode 7, score -50.0, avg_score -49.43
Environment 13: episode 7, score -50.0, avg_score -49.43
Environment 14: episode 7, score -50.0, avg_score -49.65
Environment 15: episode 7, score -50.0, avg_score -49.65
Environment 0: episode 8, score -50.0, avg_score -49.65
Environment 1: episode 8, score -50.0, avg_score -49.65
Environment 2: episode 8, score -50.0, avg_score -49.65
Environment 3: episode 8, score -50.0, avg_score -49.65
Environment 4: episode 8, score -50.0, avg

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_200.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 7: episode 13, score -50.0, avg_score -49.58
Environment 8: episode 13, score -50.0, avg_score -49.58
Environment 9: episode 13, score -50.0, avg_score -49.58
Environment 10: episode 13, score -50.0, avg_score -49.58
Environment 11: episode 13, score -50.0, avg_score -49.58
Environment 12: episode 13, score -50.0, avg_score -49.58
Environment 13: episode 13, score -50.0, avg_score -49.58
Environment 14: episode 13, score -50.0, avg_score -49.58
Environment 15: episode 13, score -50.0, avg_score -49.58
Environment 0: episode 14, score -50.0, avg_score -49.58
Environment 1: episode 14, score -50.0, avg_score -49.58
Environment 2: episode 14, score -50.0, avg_score -49.58
Environment 3: episode 14, score -50.0, avg_score -49.58
Environment 4: episode 14, score -34.0, avg_score -49.42
Environment 5: episode 14, score -50.0, avg_score -49.42
Environment 6: episode 14, score -30.0, avg_score -49.22
Environment 7: episode 14, score -50.0, avg_score -49.22
Environment 8: episode 14

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_300.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 11: episode 19, score -50.0, avg_score -47.98
Environment 12: episode 19, score -50.0, avg_score -47.98
Environment 13: episode 19, score -50.0, avg_score -47.98
Environment 14: episode 19, score -50.0, avg_score -47.98
Environment 15: episode 19, score -50.0, avg_score -47.98
Environment 0: episode 20, score -50.0, avg_score -47.98
Environment 1: episode 20, score -50.0, avg_score -47.98
Environment 2: episode 20, score -41.0, avg_score -47.89
Environment 3: episode 20, score -50.0, avg_score -47.89
Environment 4: episode 20, score -41.0, avg_score -47.8
Environment 5: episode 20, score -50.0, avg_score -47.8
Environment 6: episode 20, score -50.0, avg_score -47.8
Environment 7: episode 20, score -50.0, avg_score -47.8
Environment 8: episode 20, score -50.0, avg_score -47.96
Environment 9: episode 20, score -50.0, avg_score -47.96
Environment 10: episode 20, score -50.0, avg_score -48.16
Environment 11: episode 20, score -31.0, avg_score -47.97
Environment 12: episode 20, 

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_400.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 15: episode 25, score -50.0, avg_score -48.73
Environment 0: episode 26, score -50.0, avg_score -48.73
Environment 1: episode 26, score -50.0, avg_score -48.73
Environment 2: episode 26, score -50.0, avg_score -48.73
Environment 3: episode 26, score -50.0, avg_score -48.73
Environment 4: episode 26, score -47.0, avg_score -48.7
Environment 5: episode 26, score -50.0, avg_score -48.7
Environment 6: episode 26, score -50.0, avg_score -48.79
Environment 7: episode 26, score -50.0, avg_score -48.79
Environment 8: episode 26, score -50.0, avg_score -48.88
Environment 9: episode 26, score -50.0, avg_score -48.88
Environment 10: episode 26, score -50.0, avg_score -48.88
Environment 11: episode 26, score -50.0, avg_score -48.88
Environment 12: episode 26, score -50.0, avg_score -48.88
Environment 13: episode 26, score -50.0, avg_score -48.88
Environment 14: episode 26, score -50.0, avg_score -48.88
Environment 15: episode 26, score -50.0, avg_score -49.07
Environment 0: episode 27,

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_500.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 3: episode 32, score -50.0, avg_score -49.15
Environment 4: episode 32, score -50.0, avg_score -49.15
Environment 5: episode 32, score -50.0, avg_score -49.15
Environment 6: episode 32, score -50.0, avg_score -49.15
Environment 7: episode 32, score -50.0, avg_score -49.15
Environment 8: episode 32, score -50.0, avg_score -49.18
Environment 9: episode 32, score -50.0, avg_score -49.18
Environment 10: episode 32, score -50.0, avg_score -49.18
Environment 11: episode 32, score -50.0, avg_score -49.18
Environment 12: episode 32, score -50.0, avg_score -49.18
Environment 13: episode 32, score -50.0, avg_score -49.18
Environment 14: episode 32, score -50.0, avg_score -49.18
Environment 15: episode 32, score -50.0, avg_score -49.18
Environment 0: episode 33, score -50.0, avg_score -49.18
Environment 1: episode 33, score -50.0, avg_score -49.18
Environment 2: episode 33, score -50.0, avg_score -49.18
Environment 3: episode 33, score -43.0, avg_score -49.11
Environment 4: episode 33

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_600.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 7: episode 38, score -50.0, avg_score -48.55
Environment 8: episode 38, score -50.0, avg_score -48.55
Environment 9: episode 38, score -50.0, avg_score -48.55
Environment 10: episode 38, score -50.0, avg_score -48.55
Environment 11: episode 38, score -50.0, avg_score -48.55
Environment 12: episode 38, score -50.0, avg_score -48.55
Environment 13: episode 38, score -50.0, avg_score -48.55
Environment 14: episode 38, score -49.0, avg_score -48.54
Environment 15: episode 38, score -50.0, avg_score -48.54
Environment 0: episode 39, score -50.0, avg_score -48.54
Environment 1: episode 39, score -50.0, avg_score -48.54
Environment 2: episode 39, score -50.0, avg_score -48.54
Environment 3: episode 39, score -50.0, avg_score -48.54
Environment 4: episode 39, score -50.0, avg_score -48.54
Environment 5: episode 39, score -50.0, avg_score -48.54
Environment 6: episode 39, score -31.0, avg_score -48.35
Environment 7: episode 39, score -50.0, avg_score -48.42
Environment 8: episode 39

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_700.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 11: episode 44, score -50.0, avg_score -49.28
Environment 12: episode 44, score -50.0, avg_score -49.28
Environment 13: episode 44, score -50.0, avg_score -49.28
Environment 14: episode 44, score -50.0, avg_score -49.28
Environment 15: episode 44, score -50.0, avg_score -49.28
Environment 0: episode 45, score -50.0, avg_score -49.28
Environment 1: episode 45, score -50.0, avg_score -49.28
Environment 2: episode 45, score -50.0, avg_score -49.29
Environment 3: episode 45, score -50.0, avg_score -49.29
Environment 4: episode 45, score -50.0, avg_score -49.29
Environment 5: episode 45, score -50.0, avg_score -49.29
Environment 6: episode 45, score -50.0, avg_score -49.29
Environment 7: episode 45, score -50.0, avg_score -49.29
Environment 8: episode 45, score -50.0, avg_score -49.29
Environment 9: episode 45, score -50.0, avg_score -49.29
Environment 10: episode 45, score -50.0, avg_score -49.48
Environment 11: episode 45, score -50.0, avg_score -49.48
Environment 12: episode 

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_800.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 15: episode 50, score -50.0, avg_score -49.24




Environment 0: episode 51, score -50.0, avg_score -49.24
Environment 1: episode 51, score -50.0, avg_score -49.24
Environment 2: episode 51, score -50.0, avg_score -49.24
Environment 3: episode 51, score -50.0, avg_score -49.24
Environment 4: episode 51, score -50.0, avg_score -49.24
Environment 5: episode 51, score -50.0, avg_score -49.24
Environment 6: episode 51, score -50.0, avg_score -49.24
Environment 7: episode 51, score -50.0, avg_score -49.24
Environment 8: episode 51, score -42.0, avg_score -49.16
Environment 9: episode 51, score -50.0, avg_score -49.16
Environment 10: episode 51, score -50.0, avg_score -49.16
Environment 11: episode 51, score -50.0, avg_score -49.16
Environment 12: episode 51, score -33.0, avg_score -48.99
Environment 13: episode 51, score -50.0, avg_score -48.99
Environment 14: episode 51, score -50.0, avg_score -48.99
Environment 15: episode 51, score -50.0, avg_score -48.99
Environment 0: episode 52, score -50.0, avg_score -48.99
Environment 1: episode 52

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_900.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 3: episode 57, score -50.0, avg_score -49.54
Environment 4: episode 57, score -50.0, avg_score -49.54
Environment 5: episode 57, score -50.0, avg_score -49.54
Environment 6: episode 57, score -50.0, avg_score -49.54
Environment 7: episode 57, score -50.0, avg_score -49.54
Environment 8: episode 57, score -50.0, avg_score -49.54
Environment 9: episode 57, score -50.0, avg_score -49.54
Environment 10: episode 57, score -50.0, avg_score -49.54
Environment 11: episode 57, score -50.0, avg_score -49.54
Environment 12: episode 57, score -50.0, avg_score -49.62
Environment 13: episode 57, score -34.0, avg_score -49.46
Environment 14: episode 57, score -50.0, avg_score -49.46
Environment 15: episode 57, score -50.0, avg_score -49.46
Environment 0: episode 58, score -50.0, avg_score -49.63
Environment 1: episode 58, score -50.0, avg_score -49.63
Environment 2: episode 58, score -50.0, avg_score -49.63
Environment 3: episode 58, score -50.0, avg_score -49.63
Environment 4: episode 58

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_1000.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 7: episode 63, score -50.0, avg_score -49.21
Environment 8: episode 63, score -50.0, avg_score -49.21
Environment 9: episode 63, score -50.0, avg_score -49.21
Environment 10: episode 63, score -50.0, avg_score -49.21
Environment 11: episode 63, score -50.0, avg_score -49.21
Environment 12: episode 63, score -50.0, avg_score -49.21
Environment 13: episode 63, score -50.0, avg_score -49.21
Environment 14: episode 63, score -50.0, avg_score -49.21
Environment 15: episode 63, score -50.0, avg_score -49.21
Environment 0: episode 64, score -50.0, avg_score -49.21
Environment 1: episode 64, score -50.0, avg_score -49.37
Environment 2: episode 64, score -38.0, avg_score -49.25
Environment 3: episode 64, score -50.0, avg_score -49.25
Environment 4: episode 64, score -49.0, avg_score -49.24
Environment 5: episode 64, score -50.0, avg_score -49.24
Environment 6: episode 64, score -50.0, avg_score -49.24
Environment 7: episode 64, score -50.0, avg_score -49.24
Environment 8: episode 64

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_1100.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 11: episode 69, score -50.0, avg_score -49.16
Environment 12: episode 69, score -50.0, avg_score -49.16
Environment 13: episode 69, score -50.0, avg_score -49.16
Environment 14: episode 69, score -50.0, avg_score -49.16
Environment 15: episode 69, score -50.0, avg_score -49.16
Environment 0: episode 70, score -50.0, avg_score -49.16
Environment 1: episode 70, score -28.0, avg_score -48.94
Environment 2: episode 70, score -50.0, avg_score -48.94
Environment 3: episode 70, score -50.0, avg_score -48.94
Environment 4: episode 70, score -12.0, avg_score -48.56
Environment 5: episode 70, score -50.0, avg_score -48.56
Environment 6: episode 70, score -50.0, avg_score -48.68
Environment 7: episode 70, score -50.0, avg_score -48.68
Environment 8: episode 70, score -50.0, avg_score -48.69
Environment 9: episode 70, score -50.0, avg_score -48.69
Environment 10: episode 70, score -46.0, avg_score -48.65
Environment 11: episode 70, score -50.0, avg_score -48.65
Environment 12: episode 

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_1200.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 15: episode 75, score -50.0, avg_score -48.32
Environment 0: episode 76, score -50.0, avg_score -48.32
Environment 1: episode 76, score -50.0, avg_score -48.32
Environment 2: episode 76, score -50.0, avg_score -48.32
Environment 3: episode 76, score -50.0, avg_score -48.32
Environment 4: episode 76, score -50.0, avg_score -48.32
Environment 5: episode 76, score -50.0, avg_score -48.54
Environment 6: episode 76, score -50.0, avg_score -48.54
Environment 7: episode 76, score -50.0, avg_score -48.54
Environment 8: episode 76, score -50.0, avg_score -48.92
Environment 9: episode 76, score -50.0, avg_score -48.92
Environment 10: episode 76, score -50.0, avg_score -48.92
Environment 11: episode 76, score -50.0, avg_score -48.92
Environment 12: episode 76, score -50.0, avg_score -48.92
Environment 13: episode 76, score -50.0, avg_score -48.92
Environment 14: episode 76, score -50.0, avg_score -48.96
Environment 15: episode 76, score -50.0, avg_score -48.96
Environment 0: episode 7

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_1300.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 3: episode 82, score -40.0, avg_score -49.54
Environment 4: episode 82, score -50.0, avg_score -49.54
Environment 5: episode 82, score -50.0, avg_score -49.54
Environment 6: episode 82, score -50.0, avg_score -49.54
Environment 7: episode 82, score -50.0, avg_score -49.54
Environment 8: episode 82, score -50.0, avg_score -49.54
Environment 9: episode 82, score -50.0, avg_score -49.54
Environment 10: episode 82, score -50.0, avg_score -49.54
Environment 11: episode 82, score -50.0, avg_score -49.54
Environment 12: episode 82, score -50.0, avg_score -49.54
Environment 13: episode 82, score -50.0, avg_score -49.54
Environment 14: episode 82, score -40.0, avg_score -49.44
Environment 15: episode 82, score -50.0, avg_score -49.44
Environment 0: episode 83, score -19.0, avg_score -49.13
Environment 1: episode 83, score -50.0, avg_score -49.13
Environment 2: episode 83, score -50.0, avg_score -49.13
Environment 3: episode 83, score -50.0, avg_score -49.13
Environment 4: episode 83

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_1400.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 7: episode 88, score -50.0, avg_score -48.68
Environment 8: episode 88, score -48.0, avg_score -48.66
Environment 9: episode 88, score -50.0, avg_score -48.66
Environment 10: episode 88, score -50.0, avg_score -48.66
Environment 11: episode 88, score -50.0, avg_score -48.66
Environment 12: episode 88, score -50.0, avg_score -48.66
Environment 13: episode 88, score -50.0, avg_score -48.66
Environment 14: episode 88, score -50.0, avg_score -48.66
Environment 15: episode 88, score -50.0, avg_score -48.66
Environment 0: episode 89, score -50.0, avg_score -48.66
Environment 1: episode 89, score -50.0, avg_score -48.66
Environment 2: episode 89, score -50.0, avg_score -48.76
Environment 3: episode 89, score -50.0, avg_score -48.76
Environment 4: episode 89, score -50.0, avg_score -49.07
Environment 5: episode 89, score -50.0, avg_score -49.07
Environment 6: episode 89, score -50.0, avg_score -49.07
Environment 7: episode 89, score -50.0, avg_score -49.07
Environment 8: episode 89

                                                   

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_1500.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 11: episode 94, score -45.0, avg_score -49.33
Environment 12: episode 94, score -50.0, avg_score -49.35
Environment 13: episode 94, score -50.0, avg_score -49.35
Environment 14: episode 94, score -50.0, avg_score -49.35
Environment 15: episode 94, score -50.0, avg_score -49.35
Environment 0: episode 95, score -50.0, avg_score -49.35
Environment 1: episode 95, score -50.0, avg_score -49.35
Environment 2: episode 95, score -50.0, avg_score -49.35
Environment 3: episode 95, score -50.0, avg_score -49.35
Environment 4: episode 95, score -50.0, avg_score -49.35
Environment 5: episode 95, score -50.0, avg_score -49.35
Environment 6: episode 95, score -50.0, avg_score -49.35
Environment 7: episode 95, score -50.0, avg_score -49.35
Environment 8: episode 95, score -50.0, avg_score -49.35
Environment 9: episode 95, score -50.0, avg_score -49.35
Environment 10: episode 95, score -50.0, avg_score -49.35
Environment 11: episode 95, score -43.0, avg_score -49.28
Environment 12: episode 

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_1600.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 15: episode 100, score -50.0, avg_score -49.1
Environment 0: episode 101, score -44.0, avg_score -49.04
Environment 1: episode 101, score -50.0, avg_score -49.04
Environment 2: episode 101, score -50.0, avg_score -49.04
Environment 3: episode 101, score -50.0, avg_score -49.04
Environment 4: episode 101, score -50.0, avg_score -49.04
Environment 5: episode 101, score -32.0, avg_score -48.86
Environment 6: episode 101, score -50.0, avg_score -48.86
Environment 7: episode 101, score -50.0, avg_score -48.86
Environment 8: episode 101, score -50.0, avg_score -48.86
Environment 9: episode 101, score -45.0, avg_score -48.81
Environment 10: episode 101, score -50.0, avg_score -48.81
Environment 11: episode 101, score -50.0, avg_score -48.81
Environment 12: episode 101, score -50.0, avg_score -48.81
Environment 13: episode 101, score -50.0, avg_score -48.81
Environment 14: episode 101, score -50.0, avg_score -48.81
Environment 15: episode 101, score -50.0, avg_score -48.88




Environment 0: episode 102, score -50.0, avg_score -48.88
Environment 1: episode 102, score -50.0, avg_score -48.88
Environment 2: episode 102, score -50.0, avg_score -49.03
Environment 3: episode 102, score -50.0, avg_score -49.03
Environment 4: episode 102, score -50.0, avg_score -49.03
Environment 5: episode 102, score -50.0, avg_score -49.03
Environment 6: episode 102, score -50.0, avg_score -49.03
Environment 7: episode 102, score -50.0, avg_score -49.03
Environment 8: episode 102, score -47.0, avg_score -49.0
Environment 9: episode 102, score -50.0, avg_score -49.0
Environment 10: episode 102, score -50.0, avg_score -49.0
Environment 11: episode 102, score -50.0, avg_score -49.0
Environment 12: episode 102, score -50.0, avg_score -49.0
Environment 13: episode 102, score -50.0, avg_score -49.0
Environment 14: episode 102, score -50.0, avg_score -49.0
Environment 15: episode 102, score -50.0, avg_score -49.0
Environment 0: episode 103, score -50.0, avg_score -49.05
Environment 1: e

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_1700.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 3: episode 107, score -50.0, avg_score -49.03
Environment 4: episode 107, score -50.0, avg_score -49.09
Environment 5: episode 107, score -50.0, avg_score -49.09
Environment 6: episode 107, score -50.0, avg_score -49.09
Environment 7: episode 107, score -50.0, avg_score -49.09
Environment 8: episode 107, score -50.0, avg_score -49.09
Environment 9: episode 107, score -50.0, avg_score -49.27
Environment 10: episode 107, score -50.0, avg_score -49.27
Environment 11: episode 107, score -50.0, avg_score -49.27
Environment 12: episode 107, score -50.0, avg_score -49.27
Environment 13: episode 107, score -50.0, avg_score -49.32
Environment 14: episode 107, score -50.0, avg_score -49.32
Environment 15: episode 107, score -50.0, avg_score -49.32
Environment 0: episode 108, score -50.0, avg_score -49.32
Environment 1: episode 108, score -50.0, avg_score -49.32
Environment 2: episode 108, score -50.0, avg_score -49.32
Environment 3: episode 108, score -50.0, avg_score -49.32
Environm

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_1800.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 7: episode 113, score -50.0, avg_score -49.29
Environment 8: episode 113, score -50.0, avg_score -49.29
Environment 9: episode 113, score -50.0, avg_score -49.29
Environment 10: episode 113, score -50.0, avg_score -49.29
Environment 11: episode 113, score -50.0, avg_score -49.29
Environment 12: episode 113, score -50.0, avg_score -49.29
Environment 13: episode 113, score -50.0, avg_score -49.29
Environment 14: episode 113, score -50.0, avg_score -49.29
Environment 15: episode 113, score -50.0, avg_score -49.29
Environment 0: episode 114, score -50.0, avg_score -49.29
Environment 1: episode 114, score -50.0, avg_score -49.29
Environment 2: episode 114, score -50.0, avg_score -49.29
Environment 3: episode 114, score -50.0, avg_score -49.29
Environment 4: episode 114, score -50.0, avg_score -49.29
Environment 5: episode 114, score -50.0, avg_score -49.29
Environment 6: episode 114, score -50.0, avg_score -49.29
Environment 7: episode 114, score -50.0, avg_score -49.29
Environm

                                                   

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_1900.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 11: episode 119, score -50.0, avg_score -49.2
Environment 12: episode 119, score -50.0, avg_score -49.2
Environment 13: episode 119, score -41.0, avg_score -49.11
Environment 14: episode 119, score -48.0, avg_score -49.09
Environment 15: episode 119, score -50.0, avg_score -49.09
Environment 0: episode 120, score -50.0, avg_score -49.09
Environment 1: episode 120, score -50.0, avg_score -49.09
Environment 2: episode 120, score -50.0, avg_score -49.09
Environment 3: episode 120, score -50.0, avg_score -49.09
Environment 4: episode 120, score -50.0, avg_score -49.09
Environment 5: episode 120, score -49.0, avg_score -49.08
Environment 6: episode 120, score -50.0, avg_score -49.08
Environment 7: episode 120, score -50.0, avg_score -49.08
Environment 8: episode 120, score -50.0, avg_score -49.08
Environment 9: episode 120, score -50.0, avg_score -49.08
Environment 10: episode 120, score -50.0, avg_score -49.08
Environment 11: episode 120, score -37.0, avg_score -48.95
Environme

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_2000.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 15: episode 125, score -50.0, avg_score -48.69
Environment 0: episode 126, score -50.0, avg_score -48.69
Environment 1: episode 126, score -50.0, avg_score -48.78
Environment 2: episode 126, score -50.0, avg_score -48.8
Environment 3: episode 126, score -50.0, avg_score -48.8
Environment 4: episode 126, score -50.0, avg_score -48.8
Environment 5: episode 126, score -50.0, avg_score -48.8
Environment 6: episode 126, score -50.0, avg_score -48.8
Environment 7: episode 126, score -40.0, avg_score -48.7
Environment 8: episode 126, score -50.0, avg_score -48.7
Environment 9: episode 126, score -50.0, avg_score -48.71
Environment 10: episode 126, score -50.0, avg_score -48.71
Environment 11: episode 126, score -50.0, avg_score -48.71
Environment 12: episode 126, score -50.0, avg_score -48.71
Environment 13: episode 126, score -50.0, avg_score -48.71
Environment 14: episode 126, score -50.0, avg_score -48.71
Environment 15: episode 126, score -50.0, avg_score -48.84
Environment 0:

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_2100.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 3: episode 132, score -50.0, avg_score -49.37
Environment 4: episode 132, score -50.0, avg_score -49.37
Environment 5: episode 132, score -42.0, avg_score -49.29
Environment 6: episode 132, score -50.0, avg_score -49.29
Environment 7: episode 132, score -50.0, avg_score -49.29
Environment 8: episode 132, score -50.0, avg_score -49.29
Environment 9: episode 132, score -50.0, avg_score -49.29
Environment 10: episode 132, score -50.0, avg_score -49.29
Environment 11: episode 132, score -50.0, avg_score -49.39
Environment 12: episode 132, score -50.0, avg_score -49.39
Environment 13: episode 132, score -50.0, avg_score -49.39
Environment 14: episode 132, score -42.0, avg_score -49.31
Environment 15: episode 132, score -50.0, avg_score -49.31
Environment 0: episode 133, score -50.0, avg_score -49.31
Environment 1: episode 133, score -50.0, avg_score -49.31
Environment 2: episode 133, score -50.0, avg_score -49.31
Environment 3: episode 133, score -50.0, avg_score -49.31
Environm

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_2200.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 7: episode 138, score -50.0, avg_score -49.34
Environment 8: episode 138, score -50.0, avg_score -49.34
Environment 9: episode 138, score -50.0, avg_score -49.42
Environment 10: episode 138, score -50.0, avg_score -49.42
Environment 11: episode 138, score -50.0, avg_score -49.42
Environment 12: episode 138, score -30.0, avg_score -49.22
Environment 13: episode 138, score -50.0, avg_score -49.22
Environment 14: episode 138, score -50.0, avg_score -49.22
Environment 15: episode 138, score -50.0, avg_score -49.22
Environment 0: episode 139, score -50.0, avg_score -49.22
Environment 1: episode 139, score -50.0, avg_score -49.22
Environment 2: episode 139, score -50.0, avg_score -49.3
Environment 3: episode 139, score -50.0, avg_score -49.3
Environment 4: episode 139, score -50.0, avg_score -49.3
Environment 5: episode 139, score -50.0, avg_score -49.3
Environment 6: episode 139, score -32.0, avg_score -49.12
Environment 7: episode 139, score -50.0, avg_score -49.12
Environment 

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_2300.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 11: episode 144, score -50.0, avg_score -49.11
Environment 12: episode 144, score -50.0, avg_score -49.11
Environment 13: episode 144, score -50.0, avg_score -49.11
Environment 14: episode 144, score -50.0, avg_score -49.11
Environment 15: episode 144, score -50.0, avg_score -49.11
Environment 0: episode 145, score -50.0, avg_score -49.31
Environment 1: episode 145, score -50.0, avg_score -49.31
Environment 2: episode 145, score -50.0, avg_score -49.31
Environment 3: episode 145, score -50.0, avg_score -49.31
Environment 4: episode 145, score -50.0, avg_score -49.31
Environment 5: episode 145, score -50.0, avg_score -49.31
Environment 6: episode 145, score -50.0, avg_score -49.31
Environment 7: episode 145, score -50.0, avg_score -49.31
Environment 8: episode 145, score -50.0, avg_score -49.31
Environment 9: episode 145, score -48.0, avg_score -49.29
Environment 10: episode 145, score -50.0, avg_score -49.47
Environment 11: episode 145, score -50.0, avg_score -49.47
Environ

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_2400.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 15: episode 150, score -50.0, avg_score -48.83
Environment 0: episode 151, score -50.0, avg_score -48.83
Environment 1: episode 151, score -50.0, avg_score -48.83
Environment 2: episode 151, score -50.0, avg_score -48.83
Environment 3: episode 151, score -50.0, avg_score -48.83
Environment 4: episode 151, score -44.0, avg_score -48.77
Environment 5: episode 151, score -50.0, avg_score -48.77
Environment 6: episode 151, score -50.0, avg_score -48.77
Environment 7: episode 151, score -50.0, avg_score -48.77
Environment 8: episode 151, score -50.0, avg_score -48.77
Environment 9: episode 151, score -50.0, avg_score -48.77
Environment 10: episode 151, score -50.0, avg_score -48.77
Environment 11: episode 151, score -46.0, avg_score -48.73
Environment 12: episode 151, score -50.0, avg_score -48.73
Environment 13: episode 151, score -50.0, avg_score -48.75
Environment 14: episode 151, score -50.0, avg_score -48.75
Environment 15: episode 151, score -40.0, avg_score -48.65




Environment 0: episode 152, score -50.0, avg_score -48.65
Environment 1: episode 152, score -50.0, avg_score -48.65
Environment 2: episode 152, score -38.0, avg_score -48.53
Environment 3: episode 152, score -44.0, avg_score -48.47
Environment 4: episode 152, score -50.0, avg_score -48.61
Environment 5: episode 152, score -50.0, avg_score -48.61
Environment 6: episode 152, score -50.0, avg_score -48.61
Environment 7: episode 152, score -49.0, avg_score -48.6
Environment 8: episode 152, score -50.0, avg_score -48.6
Environment 9: episode 152, score -50.0, avg_score -48.6
Environment 10: episode 152, score -50.0, avg_score -48.6
Environment 11: episode 152, score -50.0, avg_score -48.6
Environment 12: episode 152, score -50.0, avg_score -48.6
Environment 13: episode 152, score -50.0, avg_score -48.6
Environment 14: episode 152, score -50.0, avg_score -48.6
Environment 15: episode 152, score -50.0, avg_score -48.6
Environment 0: episode 153, score -50.0, avg_score -48.6
Environment 1: epi

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_2500.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 3: episode 157, score -48.0, avg_score -48.08
Environment 4: episode 157, score -50.0, avg_score -48.08
Environment 5: episode 157, score -50.0, avg_score -48.08
Environment 6: episode 157, score -50.0, avg_score -48.08
Environment 7: episode 157, score -50.0, avg_score -48.08
Environment 8: episode 157, score -50.0, avg_score -48.14
Environment 9: episode 157, score -50.0, avg_score -48.14
Environment 10: episode 157, score -50.0, avg_score -48.14
Environment 11: episode 157, score -47.0, avg_score -48.11
Environment 12: episode 157, score -50.0, avg_score -48.11
Environment 13: episode 157, score -50.0, avg_score -48.11
Environment 14: episode 157, score -50.0, avg_score -48.11
Environment 15: episode 157, score -50.0, avg_score -48.15
Environment 0: episode 158, score -50.0, avg_score -48.15
Environment 1: episode 158, score -50.0, avg_score -48.15
Environment 2: episode 158, score -50.0, avg_score -48.15
Environment 3: episode 158, score -47.0, avg_score -48.22
Environm

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_2600.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 7: episode 163, score -50.0, avg_score -48.41
Environment 8: episode 163, score -50.0, avg_score -48.41
Environment 9: episode 163, score -50.0, avg_score -48.41
Environment 10: episode 163, score -35.0, avg_score -48.26
Environment 11: episode 163, score -50.0, avg_score -48.26
Environment 12: episode 163, score -50.0, avg_score -48.26
Environment 13: episode 163, score -50.0, avg_score -48.26
Environment 14: episode 163, score -50.0, avg_score -48.26
Environment 15: episode 163, score -35.0, avg_score -48.14
Environment 0: episode 164, score -48.0, avg_score -48.12
Environment 1: episode 164, score -50.0, avg_score -48.12
Environment 2: episode 164, score -50.0, avg_score -48.12
Environment 3: episode 164, score -50.0, avg_score -48.12
Environment 4: episode 164, score -50.0, avg_score -48.12
Environment 5: episode 164, score -36.0, avg_score -47.98
Environment 6: episode 164, score -50.0, avg_score -47.98
Environment 7: episode 164, score -50.0, avg_score -48.01
Environm

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_2700.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 11: episode 169, score -50.0, avg_score -48.35
Environment 12: episode 169, score -50.0, avg_score -48.35
Environment 13: episode 169, score -50.0, avg_score -48.35
Environment 14: episode 169, score -50.0, avg_score -48.5
Environment 15: episode 169, score -50.0, avg_score -48.5
Environment 0: episode 170, score -37.0, avg_score -48.37
Environment 1: episode 170, score -50.0, avg_score -48.37
Environment 2: episode 170, score -50.0, avg_score -48.37
Environment 3: episode 170, score -50.0, avg_score -48.52
Environment 4: episode 170, score -50.0, avg_score -48.54
Environment 5: episode 170, score -50.0, avg_score -48.54
Environment 6: episode 170, score -50.0, avg_score -48.54
Environment 7: episode 170, score -50.0, avg_score -48.54
Environment 8: episode 170, score -50.0, avg_score -48.54
Environment 9: episode 170, score -24.0, avg_score -48.42
Environment 10: episode 170, score -50.0, avg_score -48.42
Environment 11: episode 170, score -50.0, avg_score -48.42
Environme

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_2800.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 15: episode 175, score -50.0, avg_score -49.04
Environment 0: episode 176, score -50.0, avg_score -49.04
Environment 1: episode 176, score -41.0, avg_score -48.95
Environment 2: episode 176, score -50.0, avg_score -48.95
Environment 3: episode 176, score -47.0, avg_score -48.92
Environment 4: episode 176, score -50.0, avg_score -49.05
Environment 5: episode 176, score -50.0, avg_score -49.05
Environment 6: episode 176, score -49.0, avg_score -49.04
Environment 7: episode 176, score -22.0, avg_score -48.76
Environment 8: episode 176, score -50.0, avg_score -48.76
Environment 9: episode 176, score -50.0, avg_score -48.76
Environment 10: episode 176, score -50.0, avg_score -48.76
Environment 11: episode 176, score -50.0, avg_score -48.76
Environment 12: episode 176, score -50.0, avg_score -48.76
Environment 13: episode 176, score -39.0, avg_score -48.91
Environment 14: episode 176, score -40.0, avg_score -48.81
Environment 15: episode 176, score -50.0, avg_score -48.81
Environ

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_2900.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 3: episode 182, score -50.0, avg_score -47.94
Environment 4: episode 182, score -50.0, avg_score -47.94
Environment 5: episode 182, score -50.0, avg_score -48.03
Environment 6: episode 182, score -24.0, avg_score -47.77
Environment 7: episode 182, score -50.0, avg_score -47.8
Environment 8: episode 182, score -50.0, avg_score -47.8
Environment 9: episode 182, score -50.0, avg_score -47.8
Environment 10: episode 182, score -50.0, avg_score -47.81
Environment 11: episode 182, score -50.0, avg_score -48.09
Environment 12: episode 182, score -50.0, avg_score -48.09
Environment 13: episode 182, score -50.0, avg_score -48.09
Environment 14: episode 182, score -24.0, avg_score -47.83
Environment 15: episode 182, score -50.0, avg_score -47.83
Environment 0: episode 183, score -50.0, avg_score -47.83
Environment 1: episode 183, score -50.0, avg_score -47.94
Environment 2: episode 183, score -50.0, avg_score -48.04
Environment 3: episode 183, score -50.0, avg_score -48.04
Environment

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_3000.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 7: episode 188, score -37.0, avg_score -47.48
Environment 8: episode 188, score -50.0, avg_score -47.48
Environment 9: episode 188, score -50.0, avg_score -47.48
Environment 10: episode 188, score -50.0, avg_score -47.74
Environment 11: episode 188, score -45.0, avg_score -47.69
Environment 12: episode 188, score -39.0, avg_score -47.58
Environment 13: episode 188, score -46.0, avg_score -47.54
Environment 14: episode 188, score -50.0, avg_score -47.54
Environment 15: episode 188, score -50.0, avg_score -47.54
Environment 0: episode 189, score -50.0, avg_score -47.54
Environment 1: episode 189, score -50.0, avg_score -47.54
Environment 2: episode 189, score -50.0, avg_score -47.8
Environment 3: episode 189, score -50.0, avg_score -47.8
Environment 4: episode 189, score -38.0, avg_score -47.68
Environment 5: episode 189, score -50.0, avg_score -47.68
Environment 6: episode 189, score -50.0, avg_score -47.68
Environment 7: episode 189, score -50.0, avg_score -47.68
Environmen

                                                             

Moviepy - Done !
Moviepy - video ready models/her/renders/train/episode_3100.0.mp4
episode rendered
Environment 0: Episode 1/1 Score: -50.0 Avg Score: -50.0




Environment 11: episode 194, score -32.0, avg_score -47.81
Environment 12: episode 194, score -50.0, avg_score -47.81
Environment 13: episode 194, score -50.0, avg_score -47.81
Environment 14: episode 194, score -46.0, avg_score -47.77
Environment 15: episode 194, score -50.0, avg_score -47.82
Environment 0: episode 195, score -50.0, avg_score -47.93
Environment 1: episode 195, score -41.0, avg_score -47.88
Environment 2: episode 195, score -50.0, avg_score -47.88
Environment 3: episode 195, score -50.0, avg_score -47.88
Environment 4: episode 195, score -50.0, avg_score -47.88
Environment 5: episode 195, score -50.0, avg_score -47.88
Environment 6: episode 195, score -45.0, avg_score -47.83
Environment 7: episode 195, score -50.0, avg_score -47.83
Environment 8: episode 195, score -50.0, avg_score -47.95
Environment 9: episode 195, score -50.0, avg_score -47.95
Environment 10: episode 195, score -35.0, avg_score -47.8
Environment 11: episode 195, score -41.0, avg_score -47.71
Environm

KeyboardInterrupt: 

In [None]:
T.unique(her.agent.replay_buffer.states, dim=0).size()

In [None]:
her.agent.replay_buffer.states.size()

In [None]:
T.count_nonzero(her.agent.replay_buffer.states, dim=0)

In [2]:
config_file_path = '/workspaces/RL_Agents/src/app/FetchPickAndPlace_HER_DDPG_PER_b/her/config.json'
with open(config_file_path, 'r') as file:
    config = json.load(file)

In [None]:
her = HER.load(config, load_weights=False)

In [None]:
her.agent.replay_buffer.sum_tree.tree[her.agent.replay_buffer.sum_tree.capacity-1:].size()

In [None]:
her.agent.critic_model

In [None]:
num_epochs = 200
num_cycles = 50
num_episodes = 1
num_updates = 40
render_freq = 100
num_envs = 16
seed = 42

her.train(num_epochs, num_cycles, num_episodes, num_updates, render_freq, num_envs, seed)

# Actor Critic

In [None]:
env = gym.make("CartPole-v1")

In [None]:
dense_layers = [
    (128, 'relu', "kaiming normal"),
    (256, 'relu', "kaiming normal"),
    ]



In [None]:
policy_model = models.PolicyModel(env=env, dense_layers=dense_layers, optimizer='Adam', learning_rate=0.001,)

In [None]:
for param in policy_model.parameters():
    print(param)

In [None]:
value_model = models.ValueModel(env, dense_layers=dense_layers, optimizer='Adam', learning_rate=0.001)

In [None]:
value_model

In [None]:
for params in value_model.parameters():
    print(params)

In [None]:
actor_critic = rl_agents.ActorCritic(env,
                                     policy_model,
                                     value_model,
                                     discount=0.99,
                                     policy_trace_decay=0.5,
                                     value_trace_decay=0.5,
                                     callbacks=[rl_callbacks.WandbCallback('CartPole-v1-Actor-Critic')])

In [None]:
actor_critic.train(200)

In [None]:
actor_critic.test(10, True, 1)

# REINFORCE

In [None]:
env = gym.make("CartPole-v1")

In [None]:
dense_layers = [
    (128, 'relu', {
                    "kaiming normal": {
                        "a":1.0,
                        "mode":'fan_in'
                    }
                },
    ),
    # (256, 'relu', {
    #                 "kaiming_normal": {
    #                     "a":0.0,
    #                     "mode":'fan_in'
    #                 }
    #             },
    # )
    ]

In [None]:
dense_layers = [(128, 'relu', "kaiming normal")]

In [None]:
value_model = models.ValueModel(env, dense_layers, 'Adam', 0.001)

In [None]:
for param in value_model.parameters():
    print(param)

In [None]:
policy_model = models.PolicyModel(env, dense_layers, 'Adam', 0.001)

In [None]:
for param in policy_model.parameters():
    print(param)

In [None]:
reinforce = rl_agents.Reinforce(env, policy_model, value_model, 0.99, [rl_callbacks.WandbCallback('CartPole-v0_REINFORCE', chkpt_freq=100)])

In [None]:
reinforce.train(200, True, 50)

In [None]:
reinforce.test(10, True, 1)

# DDPG w/CNN

In [None]:
env = gym.make('CarRacing-v2')

In [None]:
cnn_layers = [
    # {
    #     "batchnorm":
    #     {
    #         "num_features":3
    #     }
    # },
    {
        "conv":
        {
            "out_channels": 32,
            "kernel_size": 7,
            "stride": 3,
            "padding": 'valid',
            "bias": False
        }
    },
    {
        "relu":
        {

        }
    },
    {
        "batchnorm":
        {
            "num_features":32
        }
    },
    {
        "conv":
        {
            "out_channels": 32,
            "kernel_size": 5,
            "stride": 3,
            "padding": 'valid',
            "bias": False,
        }
    },
    {
        "relu":
        {

        }
    },
    {
        "batchnorm":
        {
            "num_features":32
        }
    },
    {
        "conv":
        {
            "out_channels": 32,
            "kernel_size": 3,
            "stride": 3,
            "padding": 'valid',
            "bias": False,
        }
    },
]

In [None]:
cnn = cnn_models.CNN(cnn_layers, env)

In [None]:
cnn

In [None]:
# build actor

dense_layers = [
    (
        64,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        64,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        64,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
]

actor = models.ActorModel(env, cnn_model=cnn, dense_layers=dense_layers, optimizer="Adam", optimizer_params={'weight_decay':0.0}, learning_rate=0.0001, normalize=False)

In [None]:
actor

In [None]:
# build critic

state_layers = [
    
]

merged_layers = [
    (
        64,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        64,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        64,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    )
]


critic = models.CriticModel(env=env, cnn_model=cnn, state_layers=state_layers, merged_layers=merged_layers, optimizer="Adam", optimizer_params={'weight_decay':0.0}, learning_rate=0.0001, normalize=False)

In [None]:
critic

In [None]:
replay_buffer = helper.ReplayBuffer(env, 1000000, goal_shape=(1,))
noise = helper.OUNoise(shape=env.action_space.shape, mean=0.0, theta=0.15, sigma=0.01, dt=1.0, device='cuda')

In [None]:
ddpg_agent = rl_agents.DDPG(
    env,
    actor,
    critic,
    discount=0.98,
    tau=0.05,
    action_epsilon=0.2,
    replay_buffer=replay_buffer,
    batch_size=128,
    noise=noise,
    callbacks=[rl_callbacks.WandbCallback("CarRacing-v2")]
)

In [None]:
ddpg_agent.train(1000, True, 10)

In [None]:
wandb.finish()

In [None]:
wandb.login()

# HER

In [2]:
env = gym.make('FetchReach-v4')
env_spec = env.spec
env_wrap = GymnasiumWrapper(env_spec)

In [None]:
env_wrap.env_spec

In [None]:
desired_goal_func, achieved_goal_func, reward_func = gym_helper.get_her_goal_functions(env)

In [None]:
desired_goal_func(env).shape

In [None]:
# build actor

dense_layers = [
    (
        64,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
]

actor = models.ActorModel(env,
                          cnn_model=None,
                          dense_layers=dense_layers,
                          goal_shape=(3,),
                          optimizer="Adam",
                          optimizer_params={'weight_decay':0.0},
                          learning_rate=0.0001, normalize=False)

In [None]:
# build critic

state_layers = [
    
]

merged_layers = [
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    )
]


critic = models.CriticModel(env=env,
                            cnn_model=None,
                            state_layers=state_layers,
                            merged_layers=merged_layers,
                            goal_shape=(3,),
                            optimizer="Adam",
                            optimizer_params={'weight_decay':0.0},
                            learning_rate=0.0001,
                            normalize=False)

In [None]:
goal_shape = desired_goal_func(env).shape
replay_buffer = helper.ReplayBuffer(env, 100000, goal_shape)
# noise = helper.OUNoise(shape=env.action_space.shape,
#                        mean=0.0,
#                        theta=0.05,
#                        sigma=0.15,
#                        dt=1.0, device='cuda')

noise=helper.NormalNoise(shape=env.action_space.shape,
                         mean = 0.0,
                         stddev=0.05,
                         )

In [None]:
ddpg_agent = rl_agents.DDPG(env=env,
                            actor_model=actor,
                            critic_model=critic,
                            discount=0.98,
                            tau=0.05,
                            action_epsilon=0.2,
                            replay_buffer=replay_buffer,
                            batch_size=256,
                            noise=noise,
                            callbacks=[rl_callbacks.WandbCallback('Reacher-v4')])

In [None]:
her = rl_agents.HER(ddpg_agent,
                    strategy='future',
                    num_goals=4,
                    tolerance=0.001,
                    desired_goal=desired_goal_func,
                    achieved_goal=achieved_goal_func,
                    reward_fn=reward_func)

In [None]:
her.train(10, 50, 16, 40, True, 1000)

In [None]:
wandb.finish()

In [None]:
her.test(10, True, 1)

In [None]:
her.save()

In [None]:
her.agent.goal_normalizer.running_std

In [None]:
loaded_her = rl_agents.HER.load("/workspaces/RL_Agents/pytorch/src/app/assets/models/her")

In [None]:
loaded_her.agent.replay_buffer.sample(10)

In [None]:
loaded_her.agent.state_normalizer.running_cnt

In [None]:
loaded_her.get_config()

In [None]:
loaded_her.test(10, True, 1)

In [None]:
10e4

# HER w/CNN

In [None]:
env = gym.make('CarRacing-v2')

In [None]:
_,_ = env.reset()

In [None]:
desired_goal_func, achieved_goal_func, reward_func = gym_helper.get_her_goal_functions(env)

In [None]:
desired_goal(env).shape

In [None]:
cnn_layers = [
    # {
    #     "batchnorm":
    #     {
    #         "num_features":3
    #     }
    # },
    {
        "conv":
        {
            "out_channels": 32,
            "kernel_size": 7,
            "stride": 3,
            "padding": 'valid',
            "bias": False
        }
    },
    {
        "relu":
        {

        }
    },
    {
        "batchnorm":
        {
            "num_features":32
        }
    },
    {
        "conv":
        {
            "out_channels": 32,
            "kernel_size": 5,
            "stride": 3,
            "padding": 'valid',
            "bias": False,
        }
    },
    {
        "relu":
        {

        }
    },
    {
        "batchnorm":
        {
            "num_features":32
        }
    },
    {
        "conv":
        {
            "out_channels": 32,
            "kernel_size": 3,
            "stride": 3,
            "padding": 'valid',
            "bias": False,
        }
    },
]

cnn = cnn_models.CNN(cnn_layers, env)

In [None]:
# build actor

dense_layers = [
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
]

actor = models.ActorModel(env,
                          cnn_model=cnn,
                          dense_layers=dense_layers,
                          goal_shape=(1,),
                          optimizer="Adam",
                          optimizer_params={'weight_decay':0.0},
                          learning_rate=0.001, normalize=False)

In [None]:
actor

In [None]:
# build critic

state_layers = [
    
]

merged_layers = [
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        256,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    )
]


critic = models.CriticModel(env=env,
                            cnn_model=cnn,
                            state_layers=state_layers,
                            merged_layers=merged_layers,
                            goal_shape=(1,),
                            optimizer="Adam",
                            optimizer_params={'weight_decay':0.0},
                            learning_rate=0.001,
                            normalize=False)

In [None]:
critic

In [None]:
goal_shape = desired_goal_func(env).shape
replay_buffer = helper.ReplayBuffer(env, 100000, goal_shape)
# noise = helper.OUNoise(shape=env.action_space.shape,
#                        mean=0.0,
#                        theta=0.05,
#                        sigma=0.15,
#                        dt=1.0, device='cuda')

noise=helper.NormalNoise(shape=env.action_space.shape,
                         mean = 0.0,
                         stddev=0.05,
                         )

In [None]:
ddpg_agent = rl_agents.DDPG(env=env,
                            actor_model=actor,
                            critic_model=critic,
                            discount=0.98,
                            tau=0.05,
                            action_epsilon=0.2,
                            replay_buffer=replay_buffer,
                            batch_size=256,
                            noise=noise,
                            callbacks=[rl_callbacks.WandbCallback('CarRacing-v2')])

In [None]:
ddpg_agent.actor_model

In [None]:
her = rl_agents.HER(ddpg_agent,
                    strategy='future',
                    num_goals=4,
                    tolerance=1,
                    desired_goal=desired_goal_func,
                    achieved_goal=achieved_goal_func,
                    reward_fn=reward_func)

In [None]:
her.agent.actor_model

In [None]:
her.train(num_epochs=20,
          num_cycles=50,
          num_episodes=16,
          num_updates=40,
          render=True,
          render_freq=20
        )

In [None]:
her = rl_agents.HER.load("/workspaces/RL_Agents/pytorch/src/app/models/her")

In [None]:
wandb.finish()

In [None]:
# reset environment
state, _ = her.agent.env.reset()
# instantiate empty lists to store current episode trajectory
states, actions, next_states, dones, state_achieved_goals, \
next_state_achieved_goals, desired_goals = [], [], [], [], [], [], []
# set desired goal
desired_goal = her.desired_goal_func(her.agent.env)
# set achieved goal
state_achieved_goal = her.achieved_goal_func(her.agent.env)
# add initial state and goals to local normalizer stats
her.state_normalizer.update_local_stats(state)
her.goal_normalizer.update_local_stats(desired_goal)
her.goal_normalizer.update_local_stats(state_achieved_goal)
# set done flag
done = False
# reset episode reward to 0
episode_reward = 0
# reset steps counter for the episode
episode_steps = 0

while not done:
    # get normalized values for state and desired goal
    state_norm = her.state_normalizer.normalize(state)
    desired_goal_norm = her.goal_normalizer.normalize(desired_goal)
    # get action
    action = her.agent.get_action(state_norm, desired_goal_norm, grad=False)
    # take action
    next_state, reward, term, trunc, _ = her.agent.env.step(action)
    # get next state achieved goal
    next_state_achieved_goal = her.achieved_goal_func(her.agent.env)
    # add next state and next state achieved goal to normalizers
    her.state_normalizer.update_local_stats(next_state)
    her.goal_normalizer.update_local_stats(next_state_achieved_goal)
    # store trajectory in replay buffer (non normalized!)
    her.agent.replay_buffer.add(state, action, reward, next_state, done,\
                                    state_achieved_goal, next_state_achieved_goal, desired_goal)
    
    # append step state, action, next state, and goals to respective lists
    states.append(state)
    actions.append(action)
    next_states.append(next_state)
    dones.append(done)
    state_achieved_goals.append(state_achieved_goal)
    next_state_achieved_goals.append(next_state_achieved_goal)
    desired_goals.append(desired_goal)

    # add to episode reward and increment steps counter
    episode_reward += reward
    episode_steps += 1
    # update state and state achieved goal
    state = next_state
    state_achieved_goal = next_state_achieved_goal
    # update done flag
    if term or trunc:
        done = True

In [None]:
# package episode states, actions, next states, and goals into trajectory tuple
trajectory = (states, actions, next_states, dones, state_achieved_goals, next_state_achieved_goals, desired_goals)

In [None]:
states, actions, next_states, dones, state_achieved_goals, next_state_achieved_goals, desired_goals = trajectory

In [None]:
for idx, (s, a, ns, d, sag, nsag, dg) in enumerate(zip(states, actions, next_states, dones, state_achieved_goals, next_state_achieved_goals, desired_goals)):
    print(f'a={a}, d={d}, sag={sag}, nsag={nsag}, dg={dg}')

In [None]:
strategy = "future"
num_goals = 4

# loop over each step in the trajectory to set new achieved goals, calculate new reward, and save to replay buffer
for idx, (state, action, next_state, done, state_achieved_goal, next_state_achieved_goal, desired_goal) in enumerate(zip(states, actions, next_states, dones, state_achieved_goals, next_state_achieved_goals, desired_goals)):

    if strategy == "final":
        new_desired_goal = next_state_achieved_goals[-1]
        new_reward = her.reward_fn(state_achieved_goal, next_state_achieved_goal, new_desired_goal)
        print(f'transition: action={action}, reward={new_reward}, done={done}, state_achieved_goal={state_achieved_goal}, next_state_achieved_goal={next_state_achieved_goal}, desired_goal={new_desired_goal}')
        her.agent.replay_buffer.add(state, action, new_reward, next_state, done, state_achieved_goal, next_state_achieved_goal, new_desired_goal)

    if strategy == 'future':
        for i in range(num_goals):
            if idx + i + 1 >= len(states):
                break
            goal_idx = np.random.randint(idx + 1, len(states))
            new_desired_goal = next_state_achieved_goals[goal_idx]
            new_reward = her.reward_fn(state_achieved_goal, next_state_achieved_goal, new_desired_goal)
            print(f'transition: action={action}, reward={new_reward}, done={done}, state_achieved_goal={state_achieved_goal}, next_state_achieved_goal={next_state_achieved_goal}, desired_goal={new_desired_goal}')
            her.agent.replay_buffer.add(state, action, new_reward, next_state, done, state_achieved_goal, next_state_achieved_goal, new_desired_goal)
    

    


In [None]:
s, a, r, ns, d, sag, nsag, dg = her.agent.replay_buffer.sample(100)

In [None]:
for i in range(100):
    print(f'{i}: a={a[i]}, r={r[i]}, d={d[i]}, sag={sag[i]}, nsag={nsag[i]}, dg={dg[i]} ')

# HER Pendulum

In [None]:
env = gym.make('Pendulum-v1')

In [None]:
# build actor

dense_layers = [
    (
        400,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    ),
    (
        300,
        "relu",
        {
            "variance scaling": {
                "scale": 1.0,
                "mode": "fan_in",
                "distribution": "uniform",
            }
        },
    )
]

actor = models.ActorModel(env, cnn_model=None, dense_layers=dense_layers, optimizer='Adam',
                          optimizer_params={'weight_decay':0.01}, learning_rate=0.001, normalize=False)

In [None]:
# build critic

state_layers = [
    
]

merged_layers = [
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    )
]


critic = models.CriticModel(env=env, cnn_model=None, state_layers=state_layers, merged_layers=merged_layers, optimizer="Adam", optimizer_params={'weight_decay':0.0}, learning_rate=0.001, normalize=False)

In [None]:
replay_buffer = helper.ReplayBuffer(env, 100000, (3,))
noise = helper.OUNoise(shape=env.action_space.shape, dt=1.0, device='cuda')

In [None]:
ddpg_agent = rl_agents.DDPG(env=env,
                            actor_model=actor,
                            critic_model=critic,
                            discount=0.99,
                            tau=0.005,
                            replay_buffer=replay_buffer,
                            noise=noise,
                            callbacks=[rl_callbacks.WandbCallback('Pendulum-v1')])

In [None]:
def desired_goal_func(env):
    return np.array([0.0, 0.0, 0.0])

def achieved_goal_func(env):
    return env.get_wrapper_attr('_get_obs')()

def reward_func(env):
    pass

In [None]:
her = rl_agents.HER(
    agent=ddpg_agent,
    strategy='none',
    desired_goal=desired_goal_func,
    achieved_goal=achieved_goal_func,
    reward_fn=reward_func,
    normalizer_clip=10.0
)

In [None]:
her.agent.critic_model

In [None]:
her.agent.target_critic_model

In [None]:
her.train(1,1,100,1)

In [None]:
wandb.finish()

In [None]:
state = env.observation_space.sample()
state

In [None]:
her.agent.state_normalizer.normalize(state)

In [None]:
goal = her.desired_goal_func(her.agent.env)
goal

In [None]:
her.agent.goal_normalizer.normalize(goal)

In [None]:
def remove_renders(folder_path):
    # Iterate over the files in the folder
    for filename in os.listdir(folder_path):
        # Check if the file has a .mp4 or .meta.json extension
        if filename.endswith(".mp4") or filename.endswith(".meta.json"):
            # Construct the full file path
            file_path = os.path.join(folder_path, filename)
            # Remove the file
            os.remove(file_path)

In [None]:
remove_renders("/workspaces/RL_Agents/pytorch/src/app/assets/models/ddpg/renders/training")

# HER Fetch-Reach (Robotics)

In [None]:
env = gym.make("FetchReach-v3", max_episode_steps=50)

In [None]:
desired_goal_func, achieved_goal_func, reward_func = gym_helper.get_her_goal_functions(env)

In [None]:
achieved_goal_func(env)

In [None]:
env.get_wrapper_attr("_get_obs")()

In [None]:
# reset env state
env.reset()

In [None]:
goal_shape = desired_goal_func(env).shape

In [None]:
goal_shape

In [None]:
# build actor

dense_layers = [
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    )
]

actor = models.ActorModel(env, cnn_model=None, dense_layers=dense_layers, goal_shape=goal_shape, optimizer='Adam',
                          optimizer_params={'weight_decay':0.0}, learning_rate=0.00001, normalize_layers=False)

In [None]:
actor

In [None]:
# build critic

state_layers = [
    
]

merged_layers = [
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
               
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
]


critic = models.CriticModel(env=env, cnn_model=None, state_layers=state_layers, merged_layers=merged_layers, goal_shape=goal_shape, optimizer="Adam", optimizer_params={'weight_decay':0.0}, learning_rate=0.00001, normalize_layers=False)

In [None]:
critic

In [None]:
replay_buffer = helper.ReplayBuffer(env, 1000000, goal_shape)
# noise = helper.OUNoise(shape=env.action_space.shape, dt=1.0, device='cuda')
noise = helper.NormalNoise(shape=env.action_space.shape, mean=0.0, stddev=0.05)

In [None]:
ddpg_agent = rl_agents.DDPG(env=env,
                            actor_model=actor,
                            critic_model=critic,
                            discount=0.98,
                            tau=0.05,
                            action_epsilon=0.2,
                            replay_buffer=replay_buffer,
                            batch_size=256,
                            noise=noise,
                            callbacks=[rl_callbacks.WandbCallback("FetchReach-v2")])

In [None]:
ddpg_agent.critic_model

In [None]:
her = rl_agents.HER(
    agent=ddpg_agent,
    strategy='future',
    tolerance=0.05,
    num_goals=4,
    desired_goal=desired_goal_func,
    achieved_goal=achieved_goal_func,
    reward_fn=reward_func,
    normalizer_clip=5.0
)

In [None]:
her.train(num_epochs=50,
          num_cycles=50,
          num_episodes=16,
          num_updates=40,
          render=True,
          render_freq=1000)

In [None]:
states, action, rewards, next_states, dones, achieved_goals, next_achieved_goals, desired_goals = her.agent.replay_buffer.sample(2)

In [None]:
desired_goals

In [None]:
her.agent.env.get_wrapper_attr("distance_threshold")

In [None]:
# get success
her.agent.env.get_wrapper_attr("_is_success")(achieved_goal_func(her.agent.env), desired_goal_func(her.agent.env))

In [None]:
her.agent.env.get_wrapper_attr("goal_distance")(next_state_achieved_goal, desired_goal, None)

In [None]:
pusher_her = rl_agents.HER.load("/workspaces/RL_Agents/pytorch/src/app/assets/models/her")

In [None]:
pusher_her.agent.env.reset()

In [None]:
pusher_her.get_config()

In [None]:
wandb.finish()

In [None]:
np.linalg.norm(pusher_her.agent.env.get_wrapper_attr("get_body_com")("goal") - pusher_her.agent.env.get_wrapper_attr("get_body_com")("object"))

In [None]:
pusher_her.agent.replay_buffer.get_config()

In [None]:

pusher_her.agent.replay_buffer.desired_goals

In [None]:
## TEST ENV
env = gym.make("Pusher-v5", render_mode="rgb_array")

In [None]:
env = gym.wrappers.RecordVideo(
                    env,
                    "/renders/training",
                    episode_trigger=lambda x: True,
                )


In [None]:
state, _ = env.reset()

for i in range(1000):
# take action
    next_state, reward, term, trunc, _ = env.step(env.action_space.sample())
env.close()

# HER Fetch Push (Robitics)

In [None]:
env = gym.make('FetchPush-v2')

In [None]:
desired_goal_func, achieved_goal_func, reward_func = gym_helper.get_her_goal_functions(env)

In [None]:
# reset env state
env.reset()

In [None]:
goal_shape = desired_goal_func(env).shape

In [None]:
# build actor

dense_layers = [
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    )
]

actor = models.ActorModel(env, cnn_model=None, dense_layers=dense_layers, goal_shape=goal_shape, optimizer='Adam',
                          optimizer_params={'weight_decay':0.0}, learning_rate=0.00001, normalize_layers=False)

In [None]:
# build critic

state_layers = [
    
]

merged_layers = [
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
               
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
]


critic = models.CriticModel(env=env, cnn_model=None, state_layers=state_layers, merged_layers=merged_layers, goal_shape=goal_shape, optimizer="Adam", optimizer_params={'weight_decay':0.0}, learning_rate=0.00001, normalize_layers=False)

In [None]:
replay_buffer = helper.ReplayBuffer(env, 1000000, goal_shape)
# noise = helper.OUNoise(shape=env.action_space.shape, dt=1.0, device='cuda')
noise = helper.NormalNoise(shape=env.action_space.shape, mean=0.0, stddev=0.05)

In [None]:
ddpg_agent = rl_agents.DDPG(env=env,
                            actor_model=actor,
                            critic_model=critic,
                            discount=0.98,
                            tau=0.05,
                            action_epsilon=0.3,
                            replay_buffer=replay_buffer,
                            batch_size=128,
                            noise=noise,
                            callbacks=[rl_callbacks.WandbCallback("FetchPush-v2")],
                            save_dir="fetch_push/models/ddpg/"
                            )

In [None]:
her = rl_agents.HER(
    agent=ddpg_agent,
    strategy='final',
    tolerance=0.05,
    num_goals=4,
    desired_goal=desired_goal_func,
    achieved_goal=achieved_goal_func,
    reward_fn=reward_func,
    normalizer_clip=5.0,
    save_dir="fetch_push/models/her/"
)

In [None]:
her.train(num_epochs=50,
          num_cycles=50,
          num_episodes=16,
          num_updates=40,
          render=True,
          render_freq=1000)

# TESTING MULTITHREADING

In [None]:
env = gym.make('FetchPush-v2')

In [None]:
desired_goal_func, achieved_goal_func, reward_func = gym_helper.get_her_goal_functions(env)

In [None]:
# reset env state
env.reset()

In [None]:
goal_shape = desired_goal_func(env).shape

In [None]:
# build actor

dense_layers = [
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    )
]

actor = models.ActorModel(env, cnn_model=None, dense_layers=dense_layers, goal_shape=goal_shape, optimizer='Adam',
                          optimizer_params={'weight_decay':0.0}, learning_rate=0.00001, normalize_layers=False)

In [None]:
# build critic

state_layers = [
    
]

merged_layers = [
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
               
            }
        },
    ),
    (
        64,
        "relu",
        {
            "kaiming uniform": {
                
            }
        },
    ),
]


critic = models.CriticModel(env=env, cnn_model=None, state_layers=state_layers, merged_layers=merged_layers, goal_shape=goal_shape, optimizer="Adam", optimizer_params={'weight_decay':0.0}, learning_rate=0.00001, normalize_layers=False)

In [None]:
replay_buffer = helper.ReplayBuffer(env, 1000000, goal_shape)
# noise = helper.OUNoise(shape=env.action_space.shape, dt=1.0, device='cuda')
noise = helper.NormalNoise(shape=env.action_space.shape, mean=0.0, stddev=0.05)

In [None]:
ddpg_agent = rl_agents.DDPG(env=env,
                            actor_model=actor,
                            critic_model=critic,
                            discount=0.98,
                            tau=0.05,
                            action_epsilon=0.3,
                            replay_buffer=replay_buffer,
                            batch_size=128,
                            noise=noise,
                            callbacks=[rl_callbacks.WandbCallback("FetchPush-v2")],
                            save_dir="fetch_push/models/ddpg/"
                            )

In [None]:
her = rl_agents.HER(
    agent=ddpg_agent,
    strategy='final',
    num_workers=4,
    tolerance=0.05,
    num_goals=4,
    desired_goal=desired_goal_func,
    achieved_goal=achieved_goal_func,
    reward_fn=reward_func,
    normalizer_clip=5.0,
    save_dir="fetch_push/models/her/"
)

In [None]:
her.train()

# TESTING

In [None]:
# load config
config_path = "/workspaces/RL_Agents/pytorch/src/app/HER_Test/her/config.json"
with open(config_path, 'r') as file:
    config = json.load(file)

In [None]:
config

In [None]:
agent = rl_agents.HER.load(config)

In [None]:
for callback in agent.agent.callbacks:
    print(callback._sweep)

# Co Occurence

In [None]:
import subprocess

In [None]:
# Define the path to your JSON configuration file
config_file_path = 'assets/wandb_config.json'

# Read the JSON configuration file
with open(config_file_path, 'r') as file:
    wandb_config = json.load(file)

# Print the configuration to verify it has been loaded correctly
print(wandb_config)

In [None]:
# Define the path to your JSON configuration file
config_file_path = 'assets/sweep_config.json'

# Read the JSON configuration file
with open(config_file_path, 'r') as file:
    sweep_config = json.load(file)

# Print the configuration to verify it has been loaded correctly
print(sweep_config)

In [None]:
# Save the updated configuration to a train config file
os.makedirs('sweep', exist_ok=True)
train_config_path = os.path.join(os.getcwd(), 'sweep/train_config.json')
with open(train_config_path, 'w') as f:
    json.dump(sweep_config, f)

# Save and Set the sweep config path
sweep_config_path = os.path.join(os.getcwd(), 'sweep/sweep_config.json')
with open(sweep_config_path, 'w') as f:
    json.dump(wandb_config, f)

In [None]:
command = ['python', 'sweep.py']

# Set the environment variable
os.environ['WANDB_DISABLE_SERVICE'] = 'true'

subprocess.Popen(command)

In [None]:
# Set the environment variable
os.environ['WANDB_DISABLE_SERVICE'] = 'true'

In [None]:
# Define the path to your JSON configuration file
config_file_path = 'sweep/sweep_config.json'

# Read the JSON configuration file
with open(config_file_path, 'r') as file:
    sweep_config = json.load(file)

# Print the configuration to verify it has been loaded correctly
print(sweep_config)

In [None]:
# Define the path to your JSON configuration file
config_file_path = 'sweep/train_config.json'

# Read the JSON configuration file
with open(config_file_path, 'r') as file:
    train_config = json.load(file)

# Print the configuration to verify it has been loaded correctly
print(train_config)

In [None]:
sweep_id = wandb.sweep(sweep=sweep_config, project=sweep_config["project"])
# loop over num wandb agents
num_agents = 1
# for agent in range(num_agents):
wandb.agent(
    sweep_id,
    function=lambda: wandb_support._run_sweep(sweep_config, train_config,),
    count=train_config['num_sweeps'],
    project=sweep_config["project"],
)

In [None]:
sweep_config

# PPO

In [None]:
from pathlib import Path
from typing import List, Tuple
import torch.nn.functional as F
from torch.distributions import Categorical, Beta, Normal, kl_divergence
import time
import cv2

In [None]:
# PARAMS
# env_id = 'Pendulum-v1'
# env_id = 'LunarLanderContinuous-v3'
env_id = 'BipedalWalker-v3'
policy_lr = 3e-4
value_lr = 2e-5
entropy_coeff = 0.1
kl_coeff = 0.1
loss = 'kl'
timesteps = 100_000
num_envs = 10
device = 'cuda'

seed = 42
env = gym.make_vec(env_id, num_envs)
# env = gym.make('BipedalWalker-v3')
# _,_ = env.reset()
# sample = env.action_space.sample()
# if isinstance(sample, np.int64) or isinstance(sample, np.int32):
#     print(f'discrete action space of size {env.action_space.n}')
# elif isinstance(sample, np.ndarray):
#     print(f'continuous action space of size {env.action_space.shape}')

T.manual_seed(seed)
T.cuda.manual_seed(seed)
np.random.seed(seed)
gym.utils.seeding.np_random.seed = seed
# Build policy model
dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
policy = StochasticContinuousPolicy(env, num_envs, dense_layers, learning_rate=policy_lr, distribution='Beta', device=device)
dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
value_function = ValueModel(env, dense_layers, learning_rate=value_lr, device=device)
ppo_agent_hybrid1 = PPO(env, policy, value_function, distribution='Beta', discount=0.99, gae_coefficient=0.95, policy_clip=0.2, entropy_coefficient=entropy_coeff, kl_coefficient=kl_coeff, loss=loss)
hybrid_train_info_1 = ppo_agent_hybrid1.train(timesteps=timesteps, trajectory_length=2048, batch_size=640, learning_epochs=10, num_envs=num_envs)

# seed = 43
# env = gym.make(env_id)
# T.manual_seed(seed)
# T.cuda.manual_seed(seed)
# np.random.seed(seed)
# gym.utils.seeding.np_random.seed = seed
# # Build policy model
# dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
# policy = StochasticContinuousPolicy(env, dense_layers, learning_rate=3e-4)
# dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
# value_function = ValueModel(env, dense_layers, learning_rate=3e-4)
# ppo_agent_hybrid2 = PPO(env, policy, value_function, distribution='Beta', discount=0.99, gae_coefficient=0.95, policy_clip=0.2, entropy_coefficient=entropy_coeff, kl_coefficient=kl_coeff, loss=loss)
# hybrid_train_info_2 = ppo_agent_hybrid2.train(timesteps=timesteps, trajectory_length=2048, batch_size=64, learning_epochs=10)

# seed = 44
# env = gym.make(env_id)
# T.manual_seed(seed)
# T.cuda.manual_seed(seed)
# np.random.seed(seed)
# gym.utils.seeding.np_random.seed = seed
# # Build policy model
# dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
# policy = StochasticContinuousPolicy(env, dense_layers, learning_rate=3e-4)
# dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
# value_function = ValueModel(env, dense_layers, learning_rate=3e-4)
# ppo_agent_hybrid3 = PPO(env, policy, value_function, distribution='Beta', discount=0.99, gae_coefficient=0.95, policy_clip=0.2, entropy_coefficient=entropy_coeff, kl_coefficient=kl_coeff, loss=loss)
# hybrid_train_info_3 = ppo_agent_hybrid3.train(timesteps=timesteps, trajectory_length=2048, batch_size=64, learning_epochs=10)
# hybrid_test_info = ppo_agent_hybrid.test(1000, 'PPO_hybrid', 100)

In [None]:
# PARAMS
# env_id = 'Pendulum-v1'
# env_id = 'LunarLanderContinuous-v3'
env_id = 'BipedalWalker-v3'
policy_lr = 3e-4
value_lr = 2e-5
entropy_coeff = 0.1
kl_coeff = 0.01
loss = 'kl'
timesteps = 100_000
num_envs = 10
device = 'cuda'

seed = 42
env = gym.make_vec(env_id, num_envs)
# env = gym.make('BipedalWalker-v3')
# _,_ = env.reset()
# sample = env.action_space.sample()
# if isinstance(sample, np.int64) or isinstance(sample, np.int32):
#     print(f'discrete action space of size {env.action_space.n}')
# elif isinstance(sample, np.ndarray):
#     print(f'continuous action space of size {env.action_space.shape}')

T.manual_seed(seed)
T.cuda.manual_seed(seed)
np.random.seed(seed)
gym.utils.seeding.np_random.seed = seed
# Build policy model
dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
policy = StochasticContinuousPolicy(env, num_envs, dense_layers, learning_rate=policy_lr, distribution='Beta', device=device)
dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
value_function = ValueModel(env, dense_layers, learning_rate=value_lr, device=device)
ppo_agent_hybrid2 = PPO(env, policy, value_function, distribution='Beta', discount=0.99, gae_coefficient=0.95, policy_clip=0.2, entropy_coefficient=entropy_coeff, kl_coefficient=kl_coeff, loss=loss)
hybrid_train_info_2 = ppo_agent_hybrid2.train(timesteps=timesteps, trajectory_length=2048, batch_size=640, learning_epochs=10, num_envs=num_envs)

# seed = 43
# env = gym.make(env_id)
# T.manual_seed(seed)
# T.cuda.manual_seed(seed)
# np.random.seed(seed)
# gym.utils.seeding.np_random.seed = seed
# # Build policy model
# dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
# policy = StochasticContinuousPolicy(env, dense_layers, learning_rate=3e-4)
# dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
# value_function = ValueModel(env, dense_layers, learning_rate=3e-4)
# ppo_agent_hybrid2 = PPO(env, policy, value_function, distribution='Beta', discount=0.99, gae_coefficient=0.95, policy_clip=0.2, entropy_coefficient=entropy_coeff, kl_coefficient=kl_coeff, loss=loss)
# hybrid_train_info_2 = ppo_agent_hybrid2.train(timesteps=timesteps, trajectory_length=2048, batch_size=64, learning_epochs=10)

# seed = 44
# env = gym.make(env_id)
# T.manual_seed(seed)
# T.cuda.manual_seed(seed)
# np.random.seed(seed)
# gym.utils.seeding.np_random.seed = seed
# # Build policy model
# dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
# policy = StochasticContinuousPolicy(env, dense_layers, learning_rate=3e-4)
# dense_layers = [(128,"tanh",{"default":{}}),(128,"tanh",{"default":{}})]
# value_function = ValueModel(env, dense_layers, learning_rate=3e-4)
# ppo_agent_hybrid3 = PPO(env, policy, value_function, distribution='Beta', discount=0.99, gae_coefficient=0.95, policy_clip=0.2, entropy_coefficient=entropy_coeff, kl_coefficient=kl_coeff, loss=loss)
# hybrid_train_info_3 = ppo_agent_hybrid3.train(timesteps=timesteps, trajectory_length=2048, batch_size=64, learning_epochs=10)
# hybrid_test_info = ppo_agent_hybrid.test(1000, 'PPO_hybrid', 100)

In [None]:
## PARAMS ##
# env_id = 'Pendulum-v1'
# env_id = 'LunarLanderContinuous-v3'
# env_id = 'BipedalWalker-v3'
env_id = 'Humanoid-v5'
# env_id = "Reacher-v5"
# env_id = "Walker2d-v5"
# env_id = 'ALE/SpaceInvaders-ram-v5'
# env_id = "CarRacing-v2"
# env_id = "BipedalWalkerHardcore-v3"

timesteps = 1_000_000
trajectory_length = 2000
batch_size = 64
learning_epochs = 10
num_envs = 16
policy_lr = 3e-4
value_lr = 2e-5
policy_clip = 0.2
entropy_coeff = 0.001
loss = 'hybrid'
kl_coeff = 0.0
normalize_advantages = True
normalize_values = False
norm_clip = np.inf
grad_clip = 40.0
reward_clip = 1.0
lambda_ = 0.0
distribution = 'beta'
device = 'cuda'

# Render Settings
render_freq = 100

## WANDB ##
project_name = 'Humanoid-v5'
run_name = None
callbacks = [WandbCallback(project_name, run_name)]
# callbacks = []

seed = 42
env = gym.make(env_id)

save_dir = 'Humanoid'
# env = gym.make('BipedalWalker-v3')
# _,_ = env.reset()
# sample = env.action_space.sample()
# if isinstance(sample, np.int64) or isinstance(sample, np.int32):
#     print(f'discrete action space of size {env.action_space.n}')
# elif isinstance(sample, np.ndarray):
#     print(f'continuous action space of size {env.action_space.shape}')

# T.manual_seed(seed)
# T.cuda.manual_seed(seed)
# np.random.seed(seed)
# gym.utils.seeding.np_random.seed = seed

# Build policy model
# dense_layers = [(64,"tanh",{"default":{}}),(64,"tanh",{"default":{}})]
layer_config = [
    # {'type': 'cnn', 'params': {'out_channels': 32, 'kernel_size': (8, 8), 'stride': 4, 'padding': 0}},
    # {'type': 'cnn', 'params': {'out_channels': 64, 'kernel_size': (4, 4), 'stride': 2, 'padding': 0}},
    # {'type': 'cnn', 'params': {'out_channels': 64, 'kernel_size': (3, 3), 'stride': 1, 'padding': 0}},
    # {'type': 'flatten'},
    {'type': 'dense', 'params': {'units': 128, 'kernel': 'default', 'kernel params':{}}},
    {'type': 'tanh'},
    {'type': 'dense', 'params': {'units': 64, 'kernel': 'default', 'kernel params':{}}},
    {'type': 'tanh'},
]
output_layer_kernel = {'type': 'dense', 'params': {'kernel': 'default', 'kernel params':{}}},
policy = StochasticContinuousPolicy(env, layer_config, output_layer_kernel, learning_rate=policy_lr, distribution=distribution, device=device)
# dense_layers = [(64,"tanh",{"default":{}}),(64,"tanh",{"default":{}})]
value_function = ValueModel(env, layer_config, output_layer_kernel, learning_rate=value_lr, device=device)
ppo = PPO(env, policy, value_function, distribution=distribution, discount=0.99, gae_coefficient=0.95, policy_clip=policy_clip, entropy_coefficient=entropy_coeff,
          loss=loss, kl_coefficient=kl_coeff, normalize_advantages=normalize_advantages, normalize_values=normalize_values, value_normalizer_clip=norm_clip, policy_grad_clip=grad_clip,
          reward_clip=reward_clip, lambda_=lambda_, callbacks=callbacks, save_dir=save_dir,device=device)
hybrid_train_info_2 = ppo.train(timesteps=timesteps, trajectory_length=trajectory_length, batch_size=batch_size, learning_epochs=learning_epochs, num_envs=num_envs, seed=seed, render_freq=render_freq)
# ppo.test(10,"ppo_test", 1)


In [None]:
config_file_path = '/workspaces/RL_Agents/src/app/pong_v5_3/ppo/config.json'
with open(config_file_path, 'r') as file:
    config = json.load(file)

In [None]:
config['wrappers']

In [None]:
pong = PPO.load(config, False)

In [None]:
pong.env.env = pong.env._initialize_env(num_envs=2)

In [None]:
pong.env.action_space

In [None]:
num_envs = 2
action_shape = (3,1)
obs_shape = (3,)

observation_space = gym.spaces.Box(low=0, high=1, shape=(num_envs, *obs_shape))
action_space = gym.spaces.Box(low=0, high=1, shape=(num_envs, *action_shape)) if len(action_shape) > 1 else gym.spaces.MultiDiscrete([action_shape[0] for n in range(num_envs)])
single_observation_space = gym.spaces.Box(low=0, high=1, shape=obs_shape)
single_action_space = gym.spaces.Box(low=0, high=1, shape=action_shape) if len(action_shape) > 1 else gym.spaces.Discrete(action_shape[0])

In [None]:
action_space

In [None]:
single_obs = T.tensor(single_observation_space.sample())
state, info = (T.stack([single_obs for _ in range(observation_space.shape[0])]), {})

In [None]:
state

In [None]:
observation = T.stack([single_obs for _ in range(observation_space.shape[0])])
reward = T.zeros(observation_space.shape[0])
terminated = T.zeros(observation_space.shape[0], dtype=T.bool)
truncated = T.zeros(observation_space.shape[0], dtype=T.bool)
info = {}

In [None]:
vec_env = gym.make_vec("LunarLanderContinuous-v3", 2)

In [None]:
T.ones(vec_env.single_action_space.shape).dim()

In [None]:
from torch.distributions import Normal

num_envs = 2
expected_mu = T.stack([T.tensor([1.65, 1.65, 1.65]) for t in range(num_envs)])
expected_sigma = T.stack([T.tensor([3.9, 3.9, 3.9]) for t in range(num_envs)])
expected_dist = Normal(expected_mu, expected_sigma)

In [None]:
expected_dist.sample().shape

In [None]:
pong.train(2000000, 128, 32, 3, 12, 42)

In [None]:
scores = np.zeros(4)

In [None]:
scores[1] = 1
scores

In [None]:
import gymnasium.wrappers as base_wrappers

WRAPPER_REGISTRY = {
    "AtariPreprocessing": {
        "cls": base_wrappers.AtariPreprocessing,
        "default_params": {
            "frame_skip": 1,
            "grayscale_obs": True,
            "scale_obs": True
        }
    },
    "TimeLimit": {
        "cls": base_wrappers.TimeLimit,
        "default_params": {
            "max_episode_steps": 1000
        }
    },
    "TimeAwareObservation": {
        "cls": base_wrappers.TimeAwareObservation,
        "default_params": {
            "flatten": False,
            "normalize_time": False
        }
    },
    "FrameStackObservation": {
        "cls": base_wrappers.FrameStackObservation,
        "default_params": {
            "stack_size": 4
        }
    },
    "ResizeObservation": {
        "cls": base_wrappers.ResizeObservation,
        "default_params": {
            "shape": 84
        }
    }
}

In [None]:
wrappers = [
    {'type': "AtariPreprocessing", 'params': {'frame_skip':1, 'grayscale_obs':True, 'scale_obs':True}},
    {'type': "FrameStackObservation", 'params': {'stack_size':4}},
]

In [None]:
def wrap_env(vec_env, wrappers):
    wrapper_list = []
    for wrapper in wrappers:
        if wrapper['type'] in WRAPPER_REGISTRY:
            print(f'wrapper type:{wrapper["type"]}')
            # Use a copy of default_params to avoid modifying the registry
            default_params = WRAPPER_REGISTRY[wrapper['type']]["default_params"].copy()
            
            if wrapper['type'] == "ResizeObservation":
                # Ensure shape is a tuple for ResizeObservation
                default_params['shape'] = (default_params['shape'], default_params['shape']) if isinstance(default_params['shape'], int) else default_params['shape']
            
            print(f'default params:{default_params}')
            override_params = wrapper.get("params", {})
            
            if wrapper['type'] == "ResizeObservation":
                # Ensure override_params shape is a tuple
                if 'shape' in override_params:
                    override_params['shape'] = (override_params['shape'], override_params['shape']) if isinstance(override_params['shape'], int) else override_params['shape']
            
            print(f'override params:{override_params}')
            final_params = {**default_params, **override_params}
            print(f'final params:{final_params}')
            
            def wrapper_factory(env, cls=WRAPPER_REGISTRY[wrapper['type']]["cls"], params=final_params):
                return cls(env, **params)
            
            wrapper_list.append(wrapper_factory)
    
    # Define apply_wrappers outside the loop
    def apply_wrappers(env):
        for wrapper in wrapper_list:
            env = wrapper(env)
            print(f'length of obs space:{len(env.observation_space.shape)}')
            print(f'env obs space shape:{env.observation_space.shape}')
        return env
    
    print(f'wrapper list:{wrapper_list}')
    envs = [lambda: apply_wrappers(gym.make(vec_env.spec.id, render_mode="rgb_array")) for _ in range(vec_env.num_envs)]    
    return SyncVectorEnv(envs)

In [None]:
vec_env = gym.make_vec("ALE/Pong-v5", render_mode="rgb_array", num_envs=8)
wrapped_vec = wrap_env(vec_env, wrappers)

In [None]:
wrapped_vec.single_observation_space

In [None]:
for env in wrapped_vec.envs:
    print(env.spec)

In [None]:
def format_wrappers(wrapper_store):
    wrappers_dict = {}
    for key, value in wrapper_store.items():
        # Split the key into wrapper type and parameter name
        parts = key.split('_param:')
        print(f'parts:{parts}')
        wrapper_type = parts[0].split('wrapper:')[1]
        print(f'wrapper_type:{wrapper_type}')
        param_name = parts[1]
        print(f'param name:{param_name}')
        
        # If the wrapper type already exists in the dictionary, append to its params
        if wrapper_type not in wrappers_dict:
            wrappers_dict[wrapper_type] = {'type': wrapper_type, 'params': {}}
        
        wrappers_dict[wrapper_type]['params'][param_name] = value
    
    # Convert the dictionary to a list of dictionaries
    formatted_wrappers = list(wrappers_dict.values())
    
    return formatted_wrappers

In [None]:
wrapper_params = {'wrapper:AtariPreprocessing_param:frame_skip': 1, 'wrapper:AtariPreprocessing_param:grayscale_obs': True, 'wrapper:AtariPreprocessing_param:scale_obs': True, 'wrapper:FrameStackObservation_param:stack_size': 4}

In [None]:
formatted_wrappers = format_wrappers(wrapper_params)

In [None]:
formatted_wrappers

In [None]:
wrapper_params = {'wrapper:AtariPreprocessing_param:frame_skip': 1, 'wrapper:AtariPreprocessing_param:grayscale_obs': True, 'wrapper:AtariPreprocessing_param:scale_obs': True, 'wrapper:FrameStackObservation_param:stack_size': 4}
formatted_wrappers = dash_utils.format_wrappers(wrapper_params)
#DEBUG
print(f'formatted wrappers:{formatted_wrappers}')
env = dash_utils.instantiate_envwrapper_obj("gymnasium", "ALE/Pong-v5", formatted_wrappers)

In [None]:
config_file_path = '/workspaces/RL_Agents/src/app/humanoid_v5_2/ppo/config.json'
with open(config_file_path, 'r') as file:
    config = json.load(file)
ppo = PPO.load(config, False)

In [None]:
ppo.get_config()

In [None]:
ppo.env.env = ppo.env._initialize_env(0, 8, 42)

In [None]:
for env in ppo.env.env.envs:
    print(env.spec.pprint)

In [None]:
ppo.get_config()

In [None]:
ppo.callbacks = []

In [None]:
ppo.train(2_000_000, 128, 64, 10, 8, 42, render_freq=100)

In [None]:
# states, _ = ppo.env.reset()
steps = 10
all_states = []
all_next_states = []
for step in range(steps):
    actions, log_probs = ppo.get_action(states)
    next_states, rewards, terms, truncs, infos = ppo.env.step(actions)
    all_states.append(states)
    all_next_states.append(next_states)
    states = next_states

In [None]:
for step, step_states in enumerate(all_states):
    print(f'step states shape:{step_states.shape}')
    for i in range(len(step_states)):
        for j in range(i + 1, len(step_states)):  # Compare each environment with others
            print(f'step state {i} shape:{step_states[i].shape}')
            print(f'step state {j} shape:{step_states[j].shape}')
            assert np.allclose(step_states[i], step_states[j]), f"Environments {i} and {j} differ at step {step}"

In [None]:
for i in range(len(all_states)):
    for j in range(i + 1, len(all_states)):  # Note the change here
        print(np.allclose(all_states[i], all_states[j]))

In [None]:
all_obs = []
obs = np.ones((8,1,84,84))
for _ in range(10):
    all_obs.append(obs)
# all_obs = np.array(all_obs)
all_obs = T.stack([T.tensor(s, dtype=T.float32) for s in all_obs])

In [None]:
all_obs.shape

In [None]:
action_space = gym.spaces.Box(low=0, high=1, shape=(2, 3))

In [None]:
np.all

In [None]:
all_advantages = []
all_returns = []
all_values = []
advantage = T.ones(128)
return_ = T.ones(128)
value = T.ones(128)
num_envs = 2

for _ in range(num_envs):
    all_advantages.append(advantage)
    all_returns.append(return_)
    all_values.append(value)

advantages = T.stack(all_advantages, dim=1)
returns = T.stack(all_returns, dim=1)
values = T.stack(all_values, dim=1)

In [None]:
advantages.shape

In [None]:
states, _ = pong.env.reset()
states.shape

In [None]:
ns, r, term, trunc, _ = pong.env.step(pong.env.action_space.sample())

In [None]:
r.shape

In [None]:
pong.env.single_observation_space.shape

In [None]:
pong.env.observation_space.shape

In [None]:
pong.env.env.envs[0].spec

In [None]:
states, _ = pong.env.reset()
states = T.tensor(states)
dist, _ = pong.policy_model(states)
sample = dist.sample()
sample.shape

In [None]:
pong.policy_model

In [None]:
pong.env.reset()

In [None]:
def clip_reward(reward):
    """
    Clip rewards to the specified range.

    Args:
        reward (float): Reward to clip.

    Returns:
        float: Clipped reward.
    """
    if reward > 1:
        return 1
    elif reward < -1:
        return -1
    else:
        return reward

In [None]:
env = gym.make_vec("ALE/Pong-v5", 1)

In [None]:
states, _ = env.reset()

In [None]:
all_rewards = []
all_dones = []
for _ in range(10):
    next_states, rewards, terms, truncs, infos = env.step(env.action_space.sample())
    all_rewards.append(rewards)
    all_dones.append(np.logical_or(terms, truncs))
rewards = T.stack([T.tensor(r, dtype=T.float32) for r in all_rewards])
dones = T.stack([T.tensor(d, dtype=T.float32) for d in all_dones])

In [None]:
dones.shape

In [None]:
rewards[:,0].shape

In [None]:
[clip_reward(reward) for reward in rewards]

In [None]:
T_max = 6000  # Total steps
eta_max = 1.0  # Initial noise stddev
eta_min = 0.1  # Minimum noise stddev

t = np.linspace(0, T_max, 1000)  # Sample points
value = eta_min + 0.5 * (eta_max - eta_min) * (1 + np.cos(t * np.pi / T_max))

plt.figure(figsize=(10, 6))
plt.plot(t, value, 'b-', label='Cosine Annealing (stddev)')
plt.axhline(y=eta_max, color='r', linestyle='--', label='Initial (1.0)')
plt.axhline(y=eta_min, color='g', linestyle='--', label='Minimum (0.1)')
plt.xlabel('Steps')
plt.ylabel('Noise StdDev')
plt.title('Cosine Annealing Curve for Noise (stddev)')
plt.legend()
plt.grid(True)
plt.show()

In [None]:
import sys
print(sys.path)  # Shows all directories Python checks for imports

# Try to find mcp specifically
try:
    import mcp
    print(f"MCP found at: {mcp.__file__}")
except ImportError:
    print("MCP not found")

In [None]:
import mcp
print(dir(mcp))  # This will show all available attributes/modules in mcp