GPUとCPUの速度比較

episods=11 step=500

GPUで80秒

CPUで48秒

episods=5 step=500

GPUで40秒

CPUで25秒

結論 CPUのほうが速い!!うそーん！

ショックすぎて、今日はもうやめます。CPU買ったほうがいいじゃん！

スクリプト

import gymnasium as gym
import time
import torch as T
import torch.nn as nn
import torch.nn.functional as F
import torch.optim as optim
import numpy as np
import matplotlib.pyplot as plt
import os

""" nvidia CUDA Toolkit 12.1
https://developer.nvidia.com/cuda-downloads?target_os=Windows&target_arch=x86_64&target_version=10&target_type=exe_local
Download cuda_12.1.1_531.14_windows.exe
"""
""" pip install
pip3 install --pre torch torchvision torchaudio --index-url https://download.pytorch.org/whl/nightly/cu121
pip install gymnasium
pip install gymnasium[mujoco]
pip install matplotlib
pip install mujoco
"""
# 44.OUActionNOoiseクラスを作成する
class OUActionNoise(object):
    def __init__(self, mu, sigma=0.15, theta=0.2, dt=1e-2, x0=None):
        self.mu = mu
        self.sigma = sigma
        self.theta = theta
        self.dt = dt
        self.x0 = x0
        self.reset()

    def __call__(self):
        x = self.x_prev + self.theta * (self.mu - self.x_prev) * self.dt + \
        self.sigma * np.sqrt(self.dt) * np.random.normal(size=self.mu.shape)
        self.x_prev = x
        return x

    def reset(self):
        self.x_prev = self.x0 if self.x0 is not None else np.zeros_like(self.mu)


# 10. ReplayBufferクラスを新規作成する
class ReplayBuffer:
    def __init__(self, max_memory_size, n_obs_space, n_action_space):
        self.max_memory_size = max_memory_size
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.memory_count = 0

        self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))
        self.action_memory =  np.zeros((self.max_memory_size, self.n_action_space))
        self.reward_memory =  np.zeros(self.max_memory_size)
        self.next_state_memory =  np.zeros((self.max_memory_size, self.n_obs_space))
        self.terminal_memory =  np.zeros(self.max_memory_size)
        #self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

    # 11.トランジション保存のためstore_transitionメソドを作成する
    def store_transition(self, obs, action, reward, next_state, done):
        #print('store_transition is working.')
        index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック
        #print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())
        self.state_memory[index] = obs.detach().numpy().flatten()
        self.action_memory[index] = action.flatten()
        self.reward_memory[index] = reward.flatten()
        self.next_state_memory[index] = next_state.flatten()
        self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように
        #print('state_memory :', self.state_memory)
        #print('action_memory :', self.action_memory)
        #print('reward_memory :', self.reward_memory)
        #print('next_state_memory :', self.next_state_memory)
        #print('memory.state_memory :', self.terminal_memory)

        #print('type of state_memory :', type(self.state_memory[0][0]))
        #print('type of action_memory :', type(self.action_memory[0][0]))
        #print('type of reward_memory :', type(self.reward_memory[0]))
        #print('type of next_state_memory :', type(self.next_state_memory[0][0]))
        #print('type of memory.state_memory :', type(self.terminal_memory[0]))

        self.memory_count += 1
        #print('memory_count :', agent.memory.memory_count)

    # 16 バッファメモリーからランダムに抽出する
    def sample_buffer(self, batch_size):
        # indexが最大メモリに到達していない場合を想定する。
        max_index = min(self.max_memory_size, self.memory_count)
        choosed_index = np.random.choice(max_index, batch_size)
        
        observations = self.state_memory[choosed_index]
        actions = self.action_memory[choosed_index]
        rewards = self.reward_memory[choosed_index]
        next_states = self.next_state_memory[choosed_index]
        terminals = self.terminal_memory[choosed_index]

        return observations, actions, rewards, next_states, terminals


# 6.ActorNNクラスを新規作成する
class ActorNN(nn.Module):
    def __init__(self, device, alpha=0.001, n_obs_space=17, n_action_space=6,
                                    layer1_size=256, layer2_size=256, batch_size=64):
        #print('ActorNN.__init__ is working.')
        super(ActorNN, self).__init__()
        self.fc1 = nn.Linear(n_obs_space, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, n_action_space)

        #26.最適化処理としてアダムを設定する
        self.optimizer = optim.Adam(self.parameters(), lr=alpha)

        # 48. actorパラメータの読み出し
        # もし、パラメータのデータが存在していたらそのパラメータで初期化する。
        # パラメータファイルの存在チェック
        """
        # if T.cuda.is_available():
            map_location = 'cuda'
        else:
            map_location = 'cpu'
        """
        map_location = device
        if os.path.isfile('actor_params.pt'):
            # パラメータファイルが存在する場合はロード
            self.load_state_dict(T.load('actor_params.pt', map_location=map_location))
            print("パラメータファイルをロードしました:", 'actor_params.pt')
        else:
            print("パラメータファイルが見つかりません:", 'actor_params.pt')

    def forward(self, obs):
        #print('AgetDDPG.ActorNN.forward is working')
        #print('====ここまではOK1====')
        x = self.fc1(obs)
        x = F.relu(x)
        x = self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        action = F.tanh(x)

        return action

# 22.CriticNNクラスを新規作成する
class CriticNN(nn.Module):

    def __init__(self, device, beta=0.001, n_obs_space=17, n_action_space=6,
                 layer1_size=256, layer2_size=256, batch_size=64):
        #print('CriticNN.__init__ is working.')
        super(CriticNN, self).__init__()

        # クリティックNNは観察空間+行動空間の２つを入力とする構造
        input_dim = n_obs_space + n_action_space
        self.fc1 = nn.Linear(input_dim, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

        #27.最適化処理としてアダムを設定する
        self.optimizer = optim.Adam(self.parameters(), lr=beta)

        # 49. criticパラメータの読み出し
        # もし、パラメータのデータが存在していたらそのパラメータで初期化する。
        # パラメータファイルの存在チェック
        """
        # if T.cuda.is_available():
            map_location = 'cuda'
        else:
            map_location = 'cpu'
        """
        map_location = device
        if os.path.isfile('critic_params.pt'):
            # パラメータファイルが存在する場合はロード
            self.load_state_dict(T.load('critic_params.pt', map_location=map_location))
            print("パラメータファイルをロードしました:", 'critic_params.pt')
        else:
            print("パラメータファイルが見つかりません:", 'critic_params.pt')

    def forward(self, obs, action):
        input_data = T.cat([obs, action], dim=1)
        x = self.fc1(input_data)
        x = F.relu(x)
        x =self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)

        return x #一つの状態価値を出力する。

# 3.エージェントクラスを定義する
class AgentDDPG:

    def __init__(self, device, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64, mode='train_mode'):
        #print('AgentDDPG.__init__ is working.')
        # 5.ActorNNクラスのインスタンスを生成する
        self.device = device

        self.alpha = alpha
        self.beta = beta
        self.gamma = gamma
        self.tau = tau
        
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.n_state_action_value = n_state_action_value

        self.layer1_size = layer1_size
        self.layer2_size = layer2_size

        # 13.バッチサイズを決めておく
        self.batch_size = batch_size 

        self.actor = ActorNN(device=self.device, alpha=0.000025, n_obs_space=17, n_action_space=6,
                            layer1_size=64, layer2_size=64, batch_size=64) 
        
        # actorネットワークをGPUへ転送
        

        self.actor.to(device)
        
        # 9.memoryインスタンスを追加
        self.MAX_MEMORY_SIZE = 10000
        self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,
                                   n_obs_space=self.n_obs_space,
                                   n_action_space=self.n_action_space)
        
        # 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する
        # actorとtarget_actorのネットワークは同じActorNNで良い
        self.target_actor = ActorNN(device=self.device, alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)

        # target_actorネットワークをGPUへ転送
        self.target_actor.to(device)

        # 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する
        self.target_critic = CriticNN(device=self.device, beta=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)

         # target_criticネットワークをGPUへ転送
        self.target_critic.to(device)

        # 24.クリティックネットワークインスタンスcriticを作成する。
        self.critic = CriticNN(device=self.device, beta=0.000025, n_obs_space=17, n_action_space=6,
                               layer1_size=64, layer2_size=64, batch_size=64)

        # criticネットワークをGPUへ転送
        self.critic.to(device)
        
        # アクターロスとクリティックロス
        self.actor_loss = 0
        self.critic_loss = 0
        
        # 45.行動ノイズのインスタンス化
        self.mode = mode
        if self.mode == 'train_mode':
            self.noise = OUActionNoise(mu=np.zeros(n_action_space))
        elif self.mode == 'eval_mode':
            self.noise = OUActionNoise(mu=np.zeros(n_action_space), sigma=0)
        else:
            print('mode error')
        

    def choose_action(self, obs): # GPU対応済み
        #print('AgentDDPG.choose_action is working.')
        # 4.方策（アクター）はニューラルネットワークで表現する。
        #   ActorNNクラスを新規作成し、インスタンスactorとして使用する。
        obs = obs.to(device)
        action = self.actor.forward(obs)
        action = action.cpu()

        # 46.行動ノイズを入れて探索性を向上させる。  
        action += T.tensor(self.noise(), dtype=T.float32)
        action = action.detach().numpy()

        return action
    
    # 8.remenberメソドを追加
    def remember(self, obs, action, reward, next_state, done):
        self.memory.store_transition(obs, action, reward, next_state, done)

    # 13.learnメソドを追加
    def learn(self):
        # 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。
        if self.memory.memory_count < self.batch_size:
            return
        
        # 15.メモリバッファからデータを抜き出す sample_buffer()
        # バッチ化されているので変数名を複数形にする
        observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)
        #print('s:', observations)
        #print(observations.shape)
        #print('a :', actions)
        #print('r :', rewards)
        #print('s_ :', next_states)
        #print('terminal :', terminals)

        # 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する 
        observations = T.tensor(observations, dtype=T.float32).to(device)
        actions = T.tensor(actions, dtype=T.float32).to(device)
        rewards = T.tensor(rewards, dtype=T.float32).to(device)
        next_states = T.tensor(next_states, dtype=T.float32).to(device)
        terminals = T.tensor(terminals, dtype=T.float32).to(device)


        # 18.ターゲットアクターネットワークインスタンスtarget_actorに
        # 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。
        # このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。
        #print('next_states :', next_states)
        target_actions = self.target_actor.forward(next_states)


        # 20.ターゲットクリティックネットワークインスタンスtarget_criticに
        # 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して
        # 価値関数の推定値ターゲットバリューを出力する。
        # TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。
        # ターゲットクリティックバリューはターゲットアクターネットワークを使う
        target_critic_values = self.target_critic.forward(next_states, target_actions)

        
        # 23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に
        # 現在の状態observationsと行動actionsを入力して
        # クリティックバリューを算出する
        critic_values = self.critic.forward(observations, actions)
        

        # 25.TDターゲットを算出する：r + γ*V(w)[s_t+1]
        td_targets = []
        for i in range(self.batch_size):
            td_target = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]
            td_targets.append(td_target)

        # TDターゲットの形をバッチに整える
        td_targets = T.tensor(td_targets, dtype=T.float32).to(device)
        td_targets = td_targets.view(self.batch_size, 1) #viewはreshapeと同じ。64x1に見え方を変更した、という意味
        #print('td_targets :', td_targets)


        # ==== （１）クリティックの学習 ====

        # 28.クリティックの勾配をゼロに初期化する
        self.critic.optimizer.zero_grad()

        # 29. TDターゲットと状態価値の二乗誤差を算出して、クリティックの損失関数とする。バッチサイズは６４個
        critic_loss = F.mse_loss(td_targets, critic_values)
        self.critic_loss = critic_loss
        #print('critic_loss : ', critic_loss) # tensor(0.0485, grad_fn=<MseLossBackward0>)

        # 30. クリティックの損失関数を微分して、勾配を算出する
        critic_loss.backward()

        # 31. 勾配からオプティマイザーによってクリティックのパラメータ（重みとバイアス）を更新する
        self.critic.optimizer.step()


        # ==== （２）アクターの学習 ====

        # 32. アクターの勾配をゼロに初期化する
        self.actor.optimizer.zero_grad()

        # 33. アクターに観測情報を入力して行動を算出する。バッチサイズは６４個
        predicted_actions = self.actor.forward(observations)

        # 34.アクターの損失関数を算出する
        #    Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択すること。
        #    なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardし、
        #    actorだけをパラメータ更新することによりactorの学習をすることができる。

        actor_loss = -self.critic.forward(observations, predicted_actions)
        actor_loss = T.mean(actor_loss)
        self.actor_loss = actor_loss

        #print(f'actor_loss: {actor_loss}, critic_loss: {critic_loss}')

        # 35. DDPG構造全体の損失関数actor_lossを微分し、勾配を算出する
        actor_loss.backward()

        # 36. 勾配からオプティマイザーによってアクターのパラメータだけを（重みとバイアス）を更新する
        self.actor.optimizer.step()
    
        # 37. 全ニューラルネットワークのパラメータを更新する。
        self.update_network_parameters()

    # 37. パラメータ更新メソド。
    def update_network_parameters(self, tau=None):
        if tau is None:
            tau = self.tau
    
        # 38. actor, critic, target_actor, target_criticのネットワーク内の全てのパラメータ（重みとバイアス）とその名前を取得する
        # actorとcriticは先ほど更新されたばかりのパラメーター
        actor_params = self.actor.named_parameters()
        critic_params = self.critic.named_parameters()
        target_actor_params = self.target_actor.named_parameters()
        target_critic_params = self.target_critic.named_parameters()
        #print('actor_params : ', actor_params) # actor_params :  <generator object Module.named_parameters at 0x000001661B2D9D48>

        # 39. パラメータをディクショナリとして取り出す。
        actor_params_dict = dict(actor_params)
        critic_params_dict = dict(critic_params)
        target_actor_params_dict = dict(target_actor_params)
        target_critic_params_dict = dict(target_critic_params)
        #print('actor_params_dict : ', actor_params_dict)
        #print(actor_params_dict.keys())
        """
        actor_params_dict :  {'fc1.weight': Parameter containing:
                                   tensor([[-0.1895, -0.0343,  0.1138,  ...,  0.2157,  0.0527, -0.1173],/

        dict_keys(['fc1.weight', 'fc1.bias', 'fc2.weight', 'fc2.bias', 'fc3.weight', 'fc3.bias'])
        """

        # 40. クリティックの各パラメーター毎に 更新重みtau=0.0001の分だけほんの少しcriticパラメータをtarget_criticパラメータに近づける。
        for name in critic_params_dict:
            critic_params_dict[name] = tau * critic_params_dict[name].clone() + \
                                       (1-tau) * target_critic_params_dict[name].clone()
            
        # 41. 更新したcriticパラメータをtarget_criticのパラメータとしてロードする。
        self.target_critic.load_state_dict(critic_params_dict)

        # 42.アクターの各パラメーター毎に 更新重みtau=0.0001の分だけほんの少しactorパラメータをtarget_actorパラメータに近づける。
        for name in actor_params_dict:
            actor_params_dict[name] = tau * actor_params_dict[name].clone() + \
                                      (1 - tau) * target_actor_params_dict[name].clone()
            
        # 43. 更新したactorパラメータをtarget_actorのパラメータとしてロードする。
        self.target_actor.load_state_dict(actor_params_dict)


#### =================== メインスクリプト ======================= ####

device = T.device('cuda' if T.cuda.is_available() else 'cpu') # cuda追加
#device = 'cpu'
EVAL_TRAIN_MODE = 'train_mode' # 評価モードか訓練モードかを選択
EPISODES = 5 # episodes
STEPS = 500    # steps
DELAY_TIME = 0.00 # sec
print('Selected Mode : ', EVAL_TRAIN_MODE)

# 2.エージェントクラスのインスタンスを生成する
agent = AgentDDPG(device=device, alpha=0.01, beta=0.01, gamma=0.99, tau=0.01,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=256, layer2_size=256, batch_size=64, mode=EVAL_TRAIN_MODE) # cuda追加

if EVAL_TRAIN_MODE == 'train_mode':
    env = gym.make("HalfCheetah-v4", render_mode='depth_array')

elif EVAL_TRAIN_MODE == 'eval_mode':
    env = gym.make("HalfCheetah-v4", render_mode= 'human')


total_rewards = []
actor_losses = []
critic_losses = []
for episode in range(EPISODES):
    obs = env.reset()
    obs = T.tensor(obs[0], dtype=T.float)
    # tensor([ 0.0040,  0.0199, -0.0622,  0.0594, -0.0605,  0.0577, -0.0056,  0.0333,        -0.0072,  0.0532, -0.0512,  0.0173, -0.0529, -0.1104,  0.0946, -0.0559,         0.0824])
    #print(type(obs))
    # observation_space :  Box(-inf, inf, (17,), float64)
    #print('observation_space : ', env.observation_space)
    #print('obs :', obs)

    reward: float = 0
    total_reward: float = 0
    done: bool = False
    for j in range(STEPS):
        env.render()
        
        # ここをDDPGに置き換えていく
        action = agent.choose_action(obs) # 1.Agentクラスを定義していく # GPU対応済み
        #action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]
        #print('====ここまではOK4====')
        #print('action_space : ', env.action_space)
        #print('action : ', action)

        next_state, reward, done, _, info = env.step(action)
        #print('next_state, reward, done, _, info :', next_state, reward, done, _, info)
        """
        action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]

        next_state, reward, done, _, info : 
        [-0.00265179  0.0229547   0.00463243 -0.04729936 -0.00959038  0.04734605
        0.03672746  0.02857842  0.09980254 -0.32065693  0.04221647  1.58668951
        -2.31089174  1.30338924 -0.25465526  1.08250465 -0.14134398]
        0.07553858359316026
        False
        False
        {'x_position': -0.09233384215910741, 'x_velocity': 0.07920445513883267, 'reward_run': 0.07920445513883267, 'reward_ctrl': -0.0036658715456724168}
        """

        #7. トラジェクトを保存する。経験再生(ReplayBuffer)
        agent.remember(obs, action, reward, next_state, int(done))
       

        #12. ニューラルネットワークを学習する
        agent.learn()

        # 26.エピソード内での報酬を累積していく
        total_reward += reward
        
        # 27. next_stateをobsとして再出発する
        #print('next_state:', next_state)
        obs = next_state
        obs = T.tensor(obs, dtype=T.float)
        # 28. チーターの動きを見たいのでスリープを入れる
        time.sleep(DELAY_TIME)

    #print('total_reward : ', total_reward)
    total_rewards.append(total_reward)

    actor_losses.append(float(agent.actor_loss)) 
    critic_losses.append(float(agent.critic_loss)) 

    # print('epsisode', i, 'score %.2f' % score, '100 game sverage %.2f' % np.mean(score_history[-100:]))
 
    # 47. 各ニューラルネットワークのパラメータを１０エピソード毎に保存する
    print('episode, total_reward : ', episode , total_reward)
    if episode % 10 == 0:
        T.save(agent.actor.state_dict(), 'actor_params.pt')
        T.save(agent.critic.state_dict(), 'critic_params.pt')
        T.save(agent.target_actor.state_dict(), 'target_actor_params.pt')
        T.save(agent.target_critic.state_dict(), 'target_critic_params.pt')
        print('==== params were saved. ====')

#print('total_rewards : ', total_rewards)
#plt.plot(total_rewards)

#plt.plot(actor_losses, label='actor_losses')
#plt.plot(critic_losses, label='critic_losses')
plt.plot(total_rewards, label='total_rewards')
plt.legend()
plt.grid(True)
plt.ioff()
plt.show()

env.close() # 空なんですけど・・・

print('script is done.')

100

101

102

103

104

105

106

107

108

109

110

111

112

113

114

115

116

117

118

119

120

121

122

123

124

125

126

127

128

129

130

131

132

133

134

135

136

137

138

139

140

141

142

143

144

145

146

147

148

149

150

151

152

153

154

155

156

157

158

159

160

161

162

163

164

165

166

167

168

169

170

171

172

173

174

175

176

177

178

179

180

181

182

183

184

185

186

187

188

189

190

191

192

193

194

195

196

197

198

199

200

201

202

203

204

205

206

207

208

209

210

211

212

213

214

215

216

217

218

219

220

221

222

223

224

225

226

227

228

229

230

231

232

233

234

235

236

237

238

239

240

241

242

243

244

245

246

247

248

249

250

251

252

253

254

255

256

257

258

259

260

261

262

263

264

265

266

267

268

269

270

271

272

273

274

275

276

277

278

279

280

281

282

283

284

285

286

287

288

289

290

291

292

293

294

295

296

297

298

299

300

301

302

303

304

305

306

307

308

309

310

311

312

313

314

315

316

317

318

319

320

321

322

323

324

325

326

327

328

329

330

331

332

333

334

335

336

337

338

339

340

341

342

343

344

345

346

347

348

349

350

351

352

353

354

355

356

357

358

359

360

361

362

363

364

365

366

367

368

369

370

371

372

373

374

375

376

377

378

379

380

381

382

383

384

385

386

387

388

389

390

391

392

393

394

395

396

397

398

399

400

401

402

403

404

405

406

407

408

409

410

411

412

413

414

415

416

417

418

419

420

421

422

423

424

425

426

427

428

429

430

431

432

433

434

435

436

437

438

439

440

441

442

443

444

445

446

447

448

449

450

451

452

453

454

455

456

457

458

459

460

461

462

463

464

465

466

467

468

469

470

471

472

473

474

475

476

477

478

479

480

481

482

483

484

485

486

487

488

489

490

491

492

493

494

495

496

497

498

499

500

501

502

503

504

505

506

507

508

509

510

511

512

513

514

515

516

517

518

519

520

521

522

523

524

525

526

527

528

529

530

531

532

533

import gymnasium as gym

import time

import torch as T

import torch.nn as nn

import torch.nn.functional as F

import torch.optim as optim

import numpy as np

import matplotlib.pyplot as plt

import os

""" nvidia CUDA Toolkit 12.1

https://developer.nvidia.com/cuda-downloads?target_os=Windows&target_arch=x86_64&target_version=10&target_type=exe_local

Download cuda_12.1.1_531.14_windows.exe

"""

""" pip install

pip3 install --pre torch torchvision torchaudio --index-url https://download.pytorch.org/whl/nightly/cu121

pip install gymnasium

pip install gymnasium[mujoco]

pip install matplotlib

pip install mujoco

"""

# 44.OUActionNOoiseクラスを作成する

class OUActionNoise(object):

def __init__(self, mu, sigma=0.15, theta=0.2, dt=1e-2, x0=None):

self.mu = mu

self.sigma = sigma

self.theta = theta

self.dt = dt

self.x0 = x0

self.reset()

def __call__(self):

x = self.x_prev + self.theta * (self.mu - self.x_prev) * self.dt + \

self.sigma * np.sqrt(self.dt) * np.random.normal(size=self.mu.shape)

self.x_prev = x

return x

def reset(self):

self.x_prev = self.x0 if self.x0 is not None else np.zeros_like(self.mu)

# 10. ReplayBufferクラスを新規作成する

class ReplayBuffer:

def __init__(self, max_memory_size, n_obs_space, n_action_space):

self.max_memory_size = max_memory_size

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.memory_count = 0

self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.action_memory = np.zeros((self.max_memory_size, self.n_action_space))

self.reward_memory = np.zeros(self.max_memory_size)

self.next_state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.terminal_memory = np.zeros(self.max_memory_size)

#self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

# 11.トランジション保存のためstore_transitionメソドを作成する

def store_transition(self, obs, action, reward, next_state, done):

#print('store_transition is working.')

index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック

#print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())

self.state_memory[index] = obs.detach().numpy().flatten()

self.action_memory[index] = action.flatten()

self.reward_memory[index] = reward.flatten()

self.next_state_memory[index] = next_state.flatten()

self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように

#print('state_memory :', self.state_memory)

#print('action_memory :', self.action_memory)

#print('reward_memory :', self.reward_memory)

#print('next_state_memory :', self.next_state_memory)

#print('memory.state_memory :', self.terminal_memory)

#print('type of state_memory :', type(self.state_memory[0][0]))

#print('type of action_memory :', type(self.action_memory[0][0]))

#print('type of reward_memory :', type(self.reward_memory[0]))

#print('type of next_state_memory :', type(self.next_state_memory[0][0]))

#print('type of memory.state_memory :', type(self.terminal_memory[0]))

self.memory_count += 1

#print('memory_count :', agent.memory.memory_count)

# 16 バッファメモリーからランダムに抽出する

def sample_buffer(self, batch_size):

# indexが最大メモリに到達していない場合を想定する。

max_index = min(self.max_memory_size, self.memory_count)

choosed_index = np.random.choice(max_index, batch_size)

observations = self.state_memory[choosed_index]

actions = self.action_memory[choosed_index]

rewards = self.reward_memory[choosed_index]

next_states = self.next_state_memory[choosed_index]

terminals = self.terminal_memory[choosed_index]

return observations, actions, rewards, next_states, terminals

# 6.ActorNNクラスを新規作成する

class ActorNN(nn.Module):

def __init__(self, device, alpha=0.001, n_obs_space=17, n_action_space=6,

layer1_size=256, layer2_size=256, batch_size=64):

#print('ActorNN.__init__ is working.')

super(ActorNN, self).__init__()

self.fc1 = nn.Linear(n_obs_space, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, n_action_space)

#26.最適化処理としてアダムを設定する

self.optimizer = optim.Adam(self.parameters(), lr=alpha)

# 48. actorパラメータの読み出し

# もし、パラメータのデータが存在していたらそのパラメータで初期化する。

# パラメータファイルの存在チェック

"""

# if T.cuda.is_available():

map_location = 'cuda'

else:

map_location = 'cpu'

"""

map_location = device

if os.path.isfile('actor_params.pt'):

# パラメータファイルが存在する場合はロード

self.load_state_dict(T.load('actor_params.pt', map_location=map_location))

print("パラメータファイルをロードしました:", 'actor_params.pt')

else:

print("パラメータファイルが見つかりません:", 'actor_params.pt')

def forward(self, obs):

#print('AgetDDPG.ActorNN.forward is working')

#print('====ここまではOK1====')

x = self.fc1(obs)

x = F.relu(x)

x = self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

action = F.tanh(x)

return action

# 22.CriticNNクラスを新規作成する

class CriticNN(nn.Module):

def __init__(self, device, beta=0.001, n_obs_space=17, n_action_space=6,

layer1_size=256, layer2_size=256, batch_size=64):

#print('CriticNN.__init__ is working.')

super(CriticNN, self).__init__()

# クリティックNNは観察空間+行動空間の２つを入力とする構造

input_dim = n_obs_space + n_action_space

self.fc1 = nn.Linear(input_dim, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

#27.最適化処理としてアダムを設定する

self.optimizer = optim.Adam(self.parameters(), lr=beta)

# 49. criticパラメータの読み出し

# もし、パラメータのデータが存在していたらそのパラメータで初期化する。

# パラメータファイルの存在チェック

"""

# if T.cuda.is_available():

map_location = 'cuda'

else:

map_location = 'cpu'

"""

map_location = device

if os.path.isfile('critic_params.pt'):

# パラメータファイルが存在する場合はロード

self.load_state_dict(T.load('critic_params.pt', map_location=map_location))

print("パラメータファイルをロードしました:", 'critic_params.pt')

else:

print("パラメータファイルが見つかりません:", 'critic_params.pt')

def forward(self, obs, action):

input_data = T.cat([obs, action], dim=1)

x = self.fc1(input_data)

x = F.relu(x)

x =self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

return x #一つの状態価値を出力する。

# 3.エージェントクラスを定義する

class AgentDDPG:

def __init__(self, device, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64, mode='train_mode'):

#print('AgentDDPG.__init__ is working.')

# 5.ActorNNクラスのインスタンスを生成する

self.device = device

self.alpha = alpha

self.beta = beta

self.gamma = gamma

self.tau = tau

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.n_state_action_value = n_state_action_value

self.layer1_size = layer1_size

self.layer2_size = layer2_size

# 13.バッチサイズを決めておく

self.batch_size = batch_size

self.actor = ActorNN(device=self.device, alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# actorネットワークをGPUへ転送

self.actor.to(device)

# 9.memoryインスタンスを追加

self.MAX_MEMORY_SIZE = 10000

self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,

n_obs_space=self.n_obs_space,

n_action_space=self.n_action_space)

# 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する

# actorとtarget_actorのネットワークは同じActorNNで良い

self.target_actor = ActorNN(device=self.device, alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# target_actorネットワークをGPUへ転送

self.target_actor.to(device)

# 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する

self.target_critic = CriticNN(device=self.device, beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# target_criticネットワークをGPUへ転送

self.target_critic.to(device)

# 24.クリティックネットワークインスタンスcriticを作成する。

self.critic = CriticNN(device=self.device, beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# criticネットワークをGPUへ転送

self.critic.to(device)

# アクターロスとクリティックロス

self.actor_loss = 0

self.critic_loss = 0

# 45.行動ノイズのインスタンス化

self.mode = mode

if self.mode == 'train_mode':

self.noise = OUActionNoise(mu=np.zeros(n_action_space))

elif self.mode == 'eval_mode':

self.noise = OUActionNoise(mu=np.zeros(n_action_space), sigma=0)

else:

print('mode error')

def choose_action(self, obs): # GPU対応済み

#print('AgentDDPG.choose_action is working.')

# 4.方策（アクター）はニューラルネットワークで表現する。

# ActorNNクラスを新規作成し、インスタンスactorとして使用する。

obs = obs.to(device)

action = self.actor.forward(obs)

action = action.cpu()

# 46.行動ノイズを入れて探索性を向上させる。

action += T.tensor(self.noise(), dtype=T.float32)

action = action.detach().numpy()

return action

# 8.remenberメソドを追加

def remember(self, obs, action, reward, next_state, done):

self.memory.store_transition(obs, action, reward, next_state, done)

# 13.learnメソドを追加

def learn(self):

# 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。

if self.memory.memory_count < self.batch_size:

return

# 15.メモリバッファからデータを抜き出す sample_buffer()

# バッチ化されているので変数名を複数形にする

observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)

#print('s:', observations)

#print(observations.shape)

#print('a :', actions)

#print('r :', rewards)

#print('s_ :', next_states)

#print('terminal :', terminals)

# 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する

observations = T.tensor(observations, dtype=T.float32).to(device)

actions = T.tensor(actions, dtype=T.float32).to(device)

rewards = T.tensor(rewards, dtype=T.float32).to(device)

next_states = T.tensor(next_states, dtype=T.float32).to(device)

terminals = T.tensor(terminals, dtype=T.float32).to(device)

# 18.ターゲットアクターネットワークインスタンスtarget_actorに

# 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。

# このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。

#print('next_states :', next_states)

target_actions = self.target_actor.forward(next_states)

# 20.ターゲットクリティックネットワークインスタンスtarget_criticに

# 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して

# 価値関数の推定値ターゲットバリューを出力する。

# TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。

# ターゲットクリティックバリューはターゲットアクターネットワークを使う

target_critic_values = self.target_critic.forward(next_states, target_actions)

# 23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に

# 現在の状態observationsと行動actionsを入力して

# クリティックバリューを算出する

critic_values = self.critic.forward(observations, actions)

# 25.TDターゲットを算出する：r + γ*V(w)[s_t+1]

td_targets = []

for i in range(self.batch_size):

td_target = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]

td_targets.append(td_target)

# TDターゲットの形をバッチに整える

td_targets = T.tensor(td_targets, dtype=T.float32).to(device)

td_targets = td_targets.view(self.batch_size, 1) #viewはreshapeと同じ。64x1に見え方を変更した、という意味

#print('td_targets :', td_targets)

# ==== （１）クリティックの学習 ====

# 28.クリティックの勾配をゼロに初期化する

self.critic.optimizer.zero_grad()

# 29. TDターゲットと状態価値の二乗誤差を算出して、クリティックの損失関数とする。バッチサイズは６４個

critic_loss = F.mse_loss(td_targets, critic_values)

self.critic_loss = critic_loss

#print('critic_loss : ', critic_loss) # tensor(0.0485, grad_fn=<MseLossBackward0>)

# 30. クリティックの損失関数を微分して、勾配を算出する

critic_loss.backward()

# 31. 勾配からオプティマイザーによってクリティックのパラメータ（重みとバイアス）を更新する

self.critic.optimizer.step()

# ==== （２）アクターの学習 ====

# 32. アクターの勾配をゼロに初期化する

self.actor.optimizer.zero_grad()

# 33. アクターに観測情報を入力して行動を算出する。バッチサイズは６４個

predicted_actions = self.actor.forward(observations)

# 34.アクターの損失関数を算出する

# Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択すること。

# なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardし、

# actorだけをパラメータ更新することによりactorの学習をすることができる。

actor_loss = -self.critic.forward(observations, predicted_actions)

actor_loss = T.mean(actor_loss)

self.actor_loss = actor_loss

#print(f'actor_loss: {actor_loss}, critic_loss: {critic_loss}')

# 35. DDPG構造全体の損失関数actor_lossを微分し、勾配を算出する

actor_loss.backward()

# 36. 勾配からオプティマイザーによってアクターのパラメータだけを（重みとバイアス）を更新する

self.actor.optimizer.step()

# 37. 全ニューラルネットワークのパラメータを更新する。

self.update_network_parameters()

# 37. パラメータ更新メソド。

def update_network_parameters(self, tau=None):

if tau is None:

tau = self.tau

# 38. actor, critic, target_actor, target_criticのネットワーク内の全てのパラメータ（重みとバイアス）とその名前を取得する

# actorとcriticは先ほど更新されたばかりのパラメーター

actor_params = self.actor.named_parameters()

critic_params = self.critic.named_parameters()

target_actor_params = self.target_actor.named_parameters()

target_critic_params = self.target_critic.named_parameters()

#print('actor_params : ', actor_params) # actor_params : <generator object Module.named_parameters at 0x000001661B2D9D48>

# 39. パラメータをディクショナリとして取り出す。

actor_params_dict = dict(actor_params)

critic_params_dict = dict(critic_params)

target_actor_params_dict = dict(target_actor_params)

target_critic_params_dict = dict(target_critic_params)

#print('actor_params_dict : ', actor_params_dict)

#print(actor_params_dict.keys())

"""

actor_params_dict : {'fc1.weight': Parameter containing:

tensor([[-0.1895, -0.0343, 0.1138, ..., 0.2157, 0.0527, -0.1173],/

dict_keys(['fc1.weight', 'fc1.bias', 'fc2.weight', 'fc2.bias', 'fc3.weight', 'fc3.bias'])

"""

# 40. クリティックの各パラメーター毎に更新重みtau=0.0001の分だけほんの少しcriticパラメータをtarget_criticパラメータに近づける。

for name in critic_params_dict:

critic_params_dict[name] = tau * critic_params_dict[name].clone() + \

(1-tau) * target_critic_params_dict[name].clone()

# 41. 更新したcriticパラメータをtarget_criticのパラメータとしてロードする。

self.target_critic.load_state_dict(critic_params_dict)

# 42.アクターの各パラメーター毎に更新重みtau=0.0001の分だけほんの少しactorパラメータをtarget_actorパラメータに近づける。

for name in actor_params_dict:

actor_params_dict[name] = tau * actor_params_dict[name].clone() + \

(1 - tau) * target_actor_params_dict[name].clone()

# 43. 更新したactorパラメータをtarget_actorのパラメータとしてロードする。

self.target_actor.load_state_dict(actor_params_dict)

#### =================== メインスクリプト ======================= ####

device = T.device('cuda' if T.cuda.is_available() else 'cpu') # cuda追加

#device = 'cpu'

EVAL_TRAIN_MODE = 'train_mode' # 評価モードか訓練モードかを選択

EPISODES = 5 # episodes

STEPS = 500 # steps

DELAY_TIME = 0.00 # sec

print('Selected Mode : ', EVAL_TRAIN_MODE)

# 2.エージェントクラスのインスタンスを生成する

agent = AgentDDPG(device=device, alpha=0.01, beta=0.01, gamma=0.99, tau=0.01,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=256, layer2_size=256, batch_size=64, mode=EVAL_TRAIN_MODE) # cuda追加

if EVAL_TRAIN_MODE == 'train_mode':

env = gym.make("HalfCheetah-v4", render_mode='depth_array')

elif EVAL_TRAIN_MODE == 'eval_mode':

env = gym.make("HalfCheetah-v4", render_mode= 'human')

total_rewards = []

actor_losses = []

critic_losses = []

for episode in range(EPISODES):

obs = env.reset()

obs = T.tensor(obs[0], dtype=T.float)

# tensor([ 0.0040, 0.0199, -0.0622, 0.0594, -0.0605, 0.0577, -0.0056, 0.0333, -0.0072, 0.0532, -0.0512, 0.0173, -0.0529, -0.1104, 0.0946, -0.0559, 0.0824])

#print(type(obs))

# observation_space : Box(-inf, inf, (17,), float64)

#print('observation_space : ', env.observation_space)

#print('obs :', obs)

reward: float = 0

total_reward: float = 0

done: bool = False

for j in range(STEPS):

env.render()

# ここをDDPGに置き換えていく

action = agent.choose_action(obs) # 1.Agentクラスを定義していく # GPU対応済み

#action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

#print('====ここまではOK4====')

#print('action_space : ', env.action_space)

#print('action : ', action)

next_state, reward, done, _, info = env.step(action)

#print('next_state, reward, done, _, info :', next_state, reward, done, _, info)

"""

action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

next_state, reward, done, _, info :

[-0.00265179 0.0229547 0.00463243 -0.04729936 -0.00959038 0.04734605

0.03672746 0.02857842 0.09980254 -0.32065693 0.04221647 1.58668951

-2.31089174 1.30338924 -0.25465526 1.08250465 -0.14134398]

0.07553858359316026

False

{'x_position': -0.09233384215910741, 'x_velocity': 0.07920445513883267, 'reward_run': 0.07920445513883267, 'reward_ctrl': -0.0036658715456724168}

"""

#7. トラジェクトを保存する。経験再生(ReplayBuffer)

agent.remember(obs, action, reward, next_state, int(done))

#12. ニューラルネットワークを学習する

agent.learn()

# 26.エピソード内での報酬を累積していく

total_reward += reward

# 27. next_stateをobsとして再出発する

#print('next_state:', next_state)

obs = next_state

obs = T.tensor(obs, dtype=T.float)

# 28. チーターの動きを見たいのでスリープを入れる

time.sleep(DELAY_TIME)

#print('total_reward : ', total_reward)

total_rewards.append(total_reward)

actor_losses.append(float(agent.actor_loss))

critic_losses.append(float(agent.critic_loss))

# print('epsisode', i, 'score %.2f' % score, '100 game sverage %.2f' % np.mean(score_history[-100:]))

# 47. 各ニューラルネットワークのパラメータを１０エピソード毎に保存する

print('episode, total_reward : ', episode , total_reward)

if episode % 10 == 0:

T.save(agent.actor.state_dict(), 'actor_params.pt')

T.save(agent.critic.state_dict(), 'critic_params.pt')

T.save(agent.target_actor.state_dict(), 'target_actor_params.pt')

T.save(agent.target_critic.state_dict(), 'target_critic_params.pt')

print('==== params were saved. ====')

#print('total_rewards : ', total_rewards)

#plt.plot(total_rewards)

#plt.plot(actor_losses, label='actor_losses')

#plt.plot(critic_losses, label='critic_losses')

plt.plot(total_rewards, label='total_rewards')

plt.legend()

plt.grid(True)

plt.ioff()

plt.show()

env.close() # 空なんですけど・・・

print('script is done.')

GPUへの飛ばし方

#変数deviceを’cuda’にする

device = torch.device(“cuda” if torch.cuda.is_available() else “cpu”)

#ネットワークのインスタンスを.to(‘cuda’)する

net_gpu.to(device)

# ネットワークへの入力xを.to(‘cuda’)する

x = x.to(device)

# ネットワークへの正解ラベルyをy.to(‘cuda’)する

y = y.to(device)

これでGPU上のnetへxとyを入れることができるので演算可能になります。

    loss = criterion(outputs, y)
    loss.backward()
    optimizer.step()

loss = criterion(outputs, y)

loss.backward()

optimizer.step()

速度比較

class SimpleNet(nn.Module):クラスとしてネットワークを作成します。

n_inputs:2

バッチ数:4

n_output:1

n_hidden:1024

hidden layer 4層をもつ全５層のネットワークです。

ネットワークのインスタンスを２つ作って

net_cpu = SimpleNet()

net_gpu = SimpleNet()

エポック数：1000でそれぞれ回してみましょう。

結果は

CPU training time: 9.972002267837524 seconds
GPU training time: 2.5578291416168213 seconds

ということで。GPUのほうが高速です。

しかし、n_hiddenを64にすると、

CPU training time: 0.5419738292694092 seconds
GPU training time: 2.506857395172119 seconds

となり、node数が少ない場合はcpuのほうが高速になります。

GPUならいつでも高速というわけではないことに注意しましょう。

スクリプト

import torch
import torch.nn as nn
import torch.optim as optim
import time

class SimpleNet(nn.Module):
    def __init__(self):
        super(SimpleNet, self).__init__()
        hidden = 1024
        self.fc1 = nn.Linear(2, hidden)
        self.fc2 = nn.Linear(hidden, hidden)
        self.fc3 = nn.Linear(hidden, hidden)
        self.fc4 = nn.Linear(hidden, hidden)
        self.fc5 = nn.Linear(hidden, 1)

    def forward(self, x):
        x = torch.relu(self.fc1(x))
        x = torch.relu(self.fc2(x))
        x = torch.relu(self.fc3(x))
        x = torch.relu(self.fc4(x))
        x = self.fc5(x)
        return x

# データの生成
x = torch.tensor([[0, 0], [0, 1], [1, 0], [1, 1]], dtype=torch.float32)
y = torch.tensor([[0], [1], [1], [0]], dtype=torch.float32)

# ネットワークの初期化
net_cpu = SimpleNet()
net_gpu = SimpleNet()

# 損失関数とオプティマイザの定義
criterion = nn.MSELoss()
optimizer = optim.SGD(net_cpu.parameters(), lr=0.1)

# ==== CPUでのトレーニング時間の測定
start_time = time.time()

for epoch in range(1000):
    optimizer.zero_grad()
    outputs = net_cpu(x)
    loss = criterion(outputs, y)
    loss.backward()
    optimizer.step()

end_time = time.time()
cpu_training_time = end_time - start_time
print("CPU training time:", cpu_training_time, "seconds")


# GPUでのトレーニング時間の測定
device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
net_gpu.to(device)
x = x.to(device)
y = y.to(device)

start_time = time.time()

for epoch in range(1000):
    optimizer.zero_grad()
    outputs = net_gpu(x)
    loss = criterion(outputs, y)
    loss.backward()
    optimizer.step()

end_time = time.time()
gpu_training_time = end_time - start_time
print("GPU training time:", gpu_training_time, "seconds")

import torch

import torch.nn as nn

import torch.optim as optim

import time

class SimpleNet(nn.Module):

def __init__(self):

super(SimpleNet, self).__init__()

hidden = 1024

self.fc1 = nn.Linear(2, hidden)

self.fc2 = nn.Linear(hidden, hidden)

self.fc3 = nn.Linear(hidden, hidden)

self.fc4 = nn.Linear(hidden, hidden)

self.fc5 = nn.Linear(hidden, 1)

def forward(self, x):

x = torch.relu(self.fc1(x))

x = torch.relu(self.fc2(x))

x = torch.relu(self.fc3(x))

x = torch.relu(self.fc4(x))

x = self.fc5(x)

return x

# データの生成

x = torch.tensor([[0, 0], [0, 1], [1, 0], [1, 1]], dtype=torch.float32)

y = torch.tensor([[0], [1], [1], [0]], dtype=torch.float32)

# ネットワークの初期化

net_cpu = SimpleNet()

net_gpu = SimpleNet()

# 損失関数とオプティマイザの定義

criterion = nn.MSELoss()

optimizer = optim.SGD(net_cpu.parameters(), lr=0.1)

# ==== CPUでのトレーニング時間の測定

start_time = time.time()

for epoch in range(1000):

optimizer.zero_grad()

outputs = net_cpu(x)

loss = criterion(outputs, y)

loss.backward()

optimizer.step()

end_time = time.time()

cpu_training_time = end_time - start_time

print("CPU training time:", cpu_training_time, "seconds")

# GPUでのトレーニング時間の測定

device = torch.device("cuda" if torch.cuda.is_available() else "cpu")

net_gpu.to(device)

x = x.to(device)

y = y.to(device)

start_time = time.time()

for epoch in range(1000):

optimizer.zero_grad()

outputs = net_gpu(x)

loss = criterion(outputs, y)

loss.backward()

optimizer.step()

end_time = time.time()

gpu_training_time = end_time - start_time

print("GPU training time:", gpu_training_time, "seconds")

次回はDDPGコードをGPU対応していきます。

DDPG by gymnasium １１日目

計算の高速化（GPUの利用）
適切なエピソード数
適切なメモリバッファ数
ネットワークの入力値？パラメータ？の正規化。
保存したパラメータを読み出すのはactorとtarget_actorまたcriticとtarget_criticで共通で良いのだろうか。

計算の高速化：GPUを使ってみる。

今日は下準備をやっていきます。

GPUの準備ができているＰＣなら

device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
print(device)

1 2	device = torch.device('cuda' if torch.cuda.is_available() else 'cpu') print(device)

で’cuda’が出力されます。

‘cpu’が出力されたならGPUの準備から始める必要があります。

GPUの準備

PCIスロットに入っているGPUを調べる

$ nvidia-smi –query-gpu=name –format=csv

出力：NVIDIA GeForce RTX 3070 Ti

cudaバージョンを調べる

$ nvidia-smi

出力：

NVIDIA-SMI 528.49　Driver Version: 528.49　CUDA Version: 12.0

CUDA Toolkit のバージョンを調べる

$ nvcc -V

出力：

nvcc: NVIDIA (R) Cuda compiler driver

Built on Fri_Jan__6_19:04:39_Pacific_Standard_Time_2023

Cuda compilation tools, release 12.0, V12.0.140

Build cuda_12.0.r12.0/compiler.32267302_0

NVIDIAのGPUドライバを最新にする

https://www.nvidia.co.jp/Download/index.aspx?lang=jp

でNVIDIA GeForce RTX 3070 Tiのドライバをインストールします。Driver Version: 531.14 にアップデートしました。

再度 $ nvidia-smiで確認すると

CUDA Version: 12.1 にアップデートしていました。

CUDA ToolkitをGPUドライバに合わせてインストールする

https://developer.nvidia.com/cuda-toolkit-archive

GPUドライバをアップデートした結果CUDAバージョンは12.1になったので、それに合わせてCUDA Toolkit 12.1.0 (February 2023), Versioned Online Documentationを選択。

次にwindows10, exeファイルを選択して、ダウンロードしてインストール。

PytorchのGPU使用バージョンをインストールする

https://pytorch.org/get-started/locally/

Pytorchがインストールされているようであれば、アンインストールしておくのが良いです。

$ pip uninstall torch

下記のように自分に合ったOS, CUDAバージョンを指定すると、インストール用のコマンドが生成されるので、実行します。

pip3 install --pre torch torchvision torchaudio --index-url https://download.pytorch.org/whl/nightly/cu121

1	pip3 install --pre torch torchvision torchaudio --index-url https://download.pytorch.org/whl/nightly/cu121

確認する

今一度下記でdeviceが’cuda’と出力されれば完了です。

device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
print(device)

1 2	device = torch.device('cuda' if torch.cuda.is_available() else 'cpu') print(device)

Pythonからもいろいろ情報を取得できます。

print(torch.__version__)
print(torch.cuda.is_available(), torch.cuda.device_count())
if torch.cuda.is_available():
    print(torch.cuda.current_device())
    print(torch.cuda.get_device_name())
    print(torch.cuda.get_device_capability())

""" 出力
2.1.0.dev20230519+cu121
True 1
0
NVIDIA GeForce RTX 3070 Ti
(8, 6)
"""

print(torch.__version__)

print(torch.cuda.is_available(), torch.cuda.device_count())

if torch.cuda.is_available():

print(torch.cuda.current_device())

print(torch.cuda.get_device_name())

print(torch.cuda.get_device_capability())

""" 出力

2.1.0.dev20230519+cu121

True 1

NVIDIA GeForce RTX 3070 Ti

(8, 6)

"""

GPUでPytorchのテンソルを計算してみよう。

cpu_tensor = torch.rand(10)
gpu_tensor = cpu_tensor.to(device=device)
print('cpu_tensor : ', cpu_tensor)
print('gpu_tensor : ', gpu_tensor)
print('gpu culc : ', gpu_tensor * gpu_tensor)

cpu_tensor = torch.rand(10)

gpu_tensor = cpu_tensor.to(device=device)

print('cpu_tensor : ', cpu_tensor)

print('gpu_tensor : ', gpu_tensor)

print('gpu culc : ', gpu_tensor * gpu_tensor)

結果

GPU同士でないと計算できないので注意です。

cpu_tensor :  tensor([0.6008, 0.6893, 0.2151, 0.6096, 0.3254, 0.5945, 0.1834, 0.3007, 0.3145,
        0.7312])
gpu_tensor :  tensor([0.6008, 0.6893, 0.2151, 0.6096, 0.3254, 0.5945, 0.1834, 0.3007, 0.3145,
        0.7312], device='cuda:0')
gpu culc :  tensor([0.3609, 0.4752, 0.0463, 0.3716, 0.1059, 0.3535, 0.0337, 0.0904, 0.0989,
        0.5346], device='cuda:0')

cpu_tensor : tensor([0.6008, 0.6893, 0.2151, 0.6096, 0.3254, 0.5945, 0.1834, 0.3007, 0.3145,

0.7312])

gpu_tensor : tensor([0.6008, 0.6893, 0.2151, 0.6096, 0.3254, 0.5945, 0.1834, 0.3007, 0.3145,

0.7312], device='cuda:0')

gpu culc : tensor([0.3609, 0.4752, 0.0463, 0.3716, 0.1059, 0.3535, 0.0337, 0.0904, 0.0989,

0.5346], device='cuda:0')

演算後にCPU上の数値またはNumpy.arrayと演算するためにはGPU上からCPU上へ戻す必要があります。

# gpu上のテンソルをCPU上にコピーし、テンソルを計算グラフから切り離し、numpyのarrayに変換する一連の操作。
numpy_array = gpu_tensor.cpu().detach().numpy()

1 2	# gpu上のテンソルをCPU上にコピーし、テンソルを計算グラフから切り離し、numpyのarrayに変換する一連の操作。 numpy_array = gpu_tensor.cpu().detach().numpy()

また、GPU上にあるとmatplotlibでグラフが書けないので、GPU→CPUまたは、Numpy.arrayにしてからmatplotlibで描画します。

gpu上にある数値でグラフ描画を試みたときの警告

plt.plot(gpu_tensor)

#TypeError: can’t convert cuda:0 device type tensor to numpy. Use Tensor.cpu() to copy the tensor to host memory first.

次回

次回はニューラルネットワークに

device = torch.device(‘cuda’ if torch.cuda.is_available() else ‘cpu’)

と

tensor.to(device)

と

numpy_array = gpu_tensor.cpu().detach().numpy()

を入れ込んでみます。

未解決の課題・疑問点

model.train()とmodel.eval()の使い方が分からない。
計算の高速化（GPUの利用）
適切なエピソード数
適切なメモリバッファ数
ネットワークの入力値？パラメータ？の正規化。
保存したパラメータを読み出すのはactorとtarget_actorまたcriticとtarget_criticで共通で良いのだろうか。

model.train()とmodel.eval()の使い方

ニューラルネットワークの訓練モードと評価モードを切り替えるメソドのようです。

例えばNNモデルがactorの場合

インスタンス生成：actor = ActorNN(引数)　してから

actor.train()で訓練モードに設定すると、バッチ正規化やドロップアウトなどの要素が有効になります。あくまで自分でバッチ正規化、ドロップアウトを設定していた場合です。

逆に、actor.eval()にするとバッチ正規化、ドロップアウトを設定していたとしても無効化されます。

ここで重要なのはactor.eval()であっても勾配は計算するし、パラメータ更新も行われるということです。評価モードということなので、推論だけするのかと勘違いしてしまいますが違います。

証拠としてスクリプトを置いておきますので実行してみてください。ちゃんと勾配計算して損失関数の値も減少していきます。

サンプルスクリプト

import torch as T
import torch.nn as nn
import torch.nn.functional as F
import torch.optim as optim
import matplotlib.pyplot as plt

class ActorNN(nn.Module):
    def __init__(self, n_inputs, n_hidden, n_outputs):
        super(ActorNN, self).__init__()

        self.fc1 = nn.Linear(n_inputs, n_hidden)
        self.fc2 = nn.Linear(n_hidden, n_hidden)
        self.fc3 = nn.Linear(n_hidden, n_outputs)

        #self.optimizer = optim.SGD(self.parameters(), lr=0.01)
        self.optimizer = optim.SGD(self.parameters(), lr=0.01)

    def forward(self, x):
        x = F.relu(self.fc1(x))
        x = F.relu(self.fc2(x))
        y_pred = F.tanh(self.fc3(x))
        return y_pred

actor = ActorNN(n_inputs=4, n_hidden=256, n_outputs=4)
#actor.train()
actor.eval()

inputs = T.tensor([1,2,3,4], dtype=T.float32)
y_label = T.tensor([0,1,0,0], dtype=T.float32)

threshold = 1e-4
losses = []
N_EPOCHS = 10000
for epoch in range(N_EPOCHS):
    y_pred = actor.forward(inputs)
    loss = F.mse_loss(y_pred, y_label) 
    actor.optimizer.zero_grad()    
    loss.backward()
    actor.optimizer.step()

    print(epoch, loss.item(), y_pred.detach().numpy())
    losses.append(loss.detach().numpy())

    if loss.detach().numpy() < threshold:
        break

plt.plot(losses)
plt.show()そ

import torch as T

import torch.nn as nn

import torch.nn.functional as F

import torch.optim as optim

import matplotlib.pyplot as plt

class ActorNN(nn.Module):

def __init__(self, n_inputs, n_hidden, n_outputs):

super(ActorNN, self).__init__()

self.fc1 = nn.Linear(n_inputs, n_hidden)

self.fc2 = nn.Linear(n_hidden, n_hidden)

self.fc3 = nn.Linear(n_hidden, n_outputs)

#self.optimizer = optim.SGD(self.parameters(), lr=0.01)

self.optimizer = optim.SGD(self.parameters(), lr=0.01)

def forward(self, x):

x = F.relu(self.fc1(x))

x = F.relu(self.fc2(x))

y_pred = F.tanh(self.fc3(x))

return y_pred

actor = ActorNN(n_inputs=4, n_hidden=256, n_outputs=4)

#actor.train()

actor.eval()

inputs = T.tensor([1,2,3,4], dtype=T.float32)

y_label = T.tensor([0,1,0,0], dtype=T.float32)

threshold = 1e-4

losses = []

N_EPOCHS = 10000

for epoch in range(N_EPOCHS):

y_pred = actor.forward(inputs)

loss = F.mse_loss(y_pred, y_label)

actor.optimizer.zero_grad()

loss.backward()

actor.optimizer.step()

print(epoch, loss.item(), y_pred.detach().numpy())

losses.append(loss.detach().numpy())

if loss.detach().numpy() < threshold:

break

plt.plot(losses)

plt.show()そ

パラメータ更新による損失関数の減少グラフ

推論するときはOUActionNoise()を止めよう

バッチ正規化やドロップアウトで訓練した場合、actor.eval()で無効化が必要なのはわかりました。しかし行動ノイズは依然として有効なので、こちらも止めましょう。

ou_noise = OUActionNoise(mu=np.zeros(1), sigma=0)

のように sigmaを0に設定することで更新を停止するギミックが必要になります。

メインスクリプト

EVAL_TRAIN_MODE = 'eval_mode' # 評価モードか訓練モードかを選択

1	EVAL_TRAIN_MODE = 'eval_mode' # 評価モードか訓練モードかを選択

# 2.エージェントクラスのインスタンスを生成する
agent = AgentDDPG(alpha=0.01, beta=0.01, gamma=0.99, tau=0.01,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=256, layer2_size=256, batch_size=64, mode=EVAL_TRAIN_MODE)

# 2.エージェントクラスのインスタンスを生成する

agent = AgentDDPG(alpha=0.01, beta=0.01, gamma=0.99, tau=0.01,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=256, layer2_size=256, batch_size=64, mode=EVAL_TRAIN_MODE)

AgentDDPGクラス

class AgentDDPG:
    def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64, mode='train_mode'):

        # 45.行動ノイズのインスタンス化
        self.mode = mode
        if self.mode == 'train_mode':
            self.noise = OUActionNoise(mu=np.zeros(n_action_space))
        elif self.mode == 'eval_mode':
            self.noise = OUActionNoise(mu=np.zeros(n_action_space), sigma=0)
        else:
            print('mode error')

class AgentDDPG:

def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64, mode='train_mode'):

# 45.行動ノイズのインスタンス化

self.mode = mode

if self.mode == 'train_mode':

self.noise = OUActionNoise(mu=np.zeros(n_action_space))

elif self.mode == 'eval_mode':

self.noise = OUActionNoise(mu=np.zeros(n_action_space), sigma=0)

else:

print('mode error')

これでOUActionNoise()は無効化できました。

ハーフチーターがプルプルしなくなりました。

しかし、各ニューラルネットワークのパラメータ更新は止まっているわけではありません。

パラメータ更新を止める

書きかけです。おそらく、EVAL_TRAIN_MODE = ‘eval_mode’ でないときだけパラメータ更新メソドである、optim.step()を行うようにすればよいと思います。

if EVAL_TRAIN_MODE != ‘eval_mode’:

self.optimizer.step()

現在のスクリプト全体

import gymnasium as gym
import time
import torch as T
import torch.nn as nn
import torch.nn.functional as F
import torch.optim as optim
import numpy as np
import matplotlib.pyplot as plt
import os

""" nvidia CUDA Toolkit 12.1
https://developer.nvidia.com/cuda-downloads?target_os=Windows&target_arch=x86_64&target_version=10&target_type=exe_local
Download cuda_12.1.1_531.14_windows.exe
"""
""" pip install
pip3 install --pre torch torchvision torchaudio --index-url https://download.pytorch.org/whl/nightly/cu121
pip install gymnasium
pip install gymnasium[mujoco]
pip install matplotlib
pip install mujoco
"""
# 44.OUActionNOoiseクラスを作成する
class OUActionNoise(object):
    def __init__(self, mu, sigma=0.15, theta=0.2, dt=1e-2, x0=None):
        self.mu = mu
        self.sigma = sigma
        self.theta = theta
        self.dt = dt
        self.x0 = x0
        self.reset()

    def __call__(self):
        x = self.x_prev + self.theta * (self.mu - self.x_prev) * self.dt + \
        self.sigma * np.sqrt(self.dt) * np.random.normal(size=self.mu.shape)
        self.x_prev = x
        return x

    def reset(self):
        self.x_prev = self.x0 if self.x0 is not None else np.zeros_like(self.mu)


# 10. ReplayBufferクラスを新規作成する
class ReplayBuffer:
    def __init__(self, max_memory_size, n_obs_space, n_action_space):
        self.max_memory_size = max_memory_size
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.memory_count = 0

        self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))
        self.action_memory =  np.zeros((self.max_memory_size, self.n_action_space))
        self.reward_memory =  np.zeros(self.max_memory_size)
        self.next_state_memory =  np.zeros((self.max_memory_size, self.n_obs_space))
        self.terminal_memory =  np.zeros(self.max_memory_size)
        #self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

    # 11.トランジション保存のためstore_transitionメソドを作成する
    def store_transition(self, obs, action, reward, next_state, done):
        #print('store_transition is working.')
        index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック
        #print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())
        self.state_memory[index] = obs.detach().numpy().flatten()
        self.action_memory[index] = action.flatten()
        self.reward_memory[index] = reward.flatten()
        self.next_state_memory[index] = next_state.flatten()
        self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように
        #print('state_memory :', self.state_memory)
        #print('action_memory :', self.action_memory)
        #print('reward_memory :', self.reward_memory)
        #print('next_state_memory :', self.next_state_memory)
        #print('memory.state_memory :', self.terminal_memory)

        #print('type of state_memory :', type(self.state_memory[0][0]))
        #print('type of action_memory :', type(self.action_memory[0][0]))
        #print('type of reward_memory :', type(self.reward_memory[0]))
        #print('type of next_state_memory :', type(self.next_state_memory[0][0]))
        #print('type of memory.state_memory :', type(self.terminal_memory[0]))

        self.memory_count += 1
        #print('memory_count :', agent.memory.memory_count)

    # 16 バッファメモリーからランダムに抽出する
    def sample_buffer(self, batch_size):
        # indexが最大メモリに到達していない場合を想定する。
        max_index = min(self.max_memory_size, self.memory_count)
        choosed_index = np.random.choice(max_index, batch_size)
        
        observations = self.state_memory[choosed_index]
        actions = self.action_memory[choosed_index]
        rewards = self.reward_memory[choosed_index]
        next_states = self.next_state_memory[choosed_index]
        terminals = self.terminal_memory[choosed_index]

        return observations, actions, rewards, next_states, terminals


# 6.ActorNNクラスを新規作成する
class ActorNN(nn.Module):
    def __init__(self, alpha=0.001, n_obs_space=17, n_action_space=6,
                                    layer1_size=256, layer2_size=256, batch_size=64):
        #print('ActorNN.__init__ is working.')
        super(ActorNN, self).__init__()
        self.fc1 = nn.Linear(n_obs_space, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, n_action_space)

        #26.最適化処理としてアダムを設定する
        self.optimizer = optim.Adam(self.parameters(), lr=alpha)

        # 48. actorパラメータの読み出し
        # もし、パラメータのデータが存在していたらそのパラメータで初期化する。
        # パラメータファイルの存在チェック
        if T.cuda.is_available():
            map_location = 'cuda'
        else:
            map_location = 'cpu'

        if os.path.isfile('actor_params.pt'):
            # パラメータファイルが存在する場合はロード
            self.load_state_dict(T.load('actor_params.pt', map_location=map_location))
            print("パラメータファイルをロードしました:", 'actor_params.pt')
        else:
            print("パラメータファイルが見つかりません:", 'actor_params.pt')

    def forward(self, obs):
        #print('AgetDDPG.ActorNN.forward is working')
        #print('====ここまではOK1====')
        x = self.fc1(obs)
        x = F.relu(x)
        x = self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        action = F.tanh(x)

        return action

# 22.CriticNNクラスを新規作成する
class CriticNN(nn.Module):

    def __init__(self, beta=0.001, n_obs_space=17, n_action_space=6,
                 layer1_size=256, layer2_size=256, batch_size=64):
        #print('CriticNN.__init__ is working.')
        super(CriticNN, self).__init__()

        # クリティックNNは観察空間+行動空間の２つを入力とする構造
        input_dim = n_obs_space + n_action_space
        self.fc1 = nn.Linear(input_dim, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

        #27.最適化処理としてアダムを設定する
        self.optimizer = optim.Adam(self.parameters(), lr=beta)

        # 49. criticパラメータの読み出し
        # もし、パラメータのデータが存在していたらそのパラメータで初期化する。
        # パラメータファイルの存在チェック
        if T.cuda.is_available():
            map_location = 'cuda'
        else:
            map_location = 'cpu'

        if os.path.isfile('critic_params.pt'):
            # パラメータファイルが存在する場合はロード
            self.load_state_dict(T.load('critic_params.pt', map_location=map_location))
            print("パラメータファイルをロードしました:", 'critic_params.pt')
        else:
            print("パラメータファイルが見つかりません:", 'critic_params.pt')

    def forward(self, obs, action):
        input_data = T.cat([obs, action], dim=1)
        x = self.fc1(input_data)
        x = F.relu(x)
        x =self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        return x #一つの状態価値を出力する。

# 3.エージェントクラスを定義する
class AgentDDPG:

    def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64, mode='train_mode'):
        #print('AgentDDPG.__init__ is working.')
        # 5.ActorNNクラスのインスタンスを生成する
        self.alpha = alpha
        self.beta = beta
        self.gamma = gamma
        self.tau = tau
        
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.n_state_action_value = n_state_action_value

        self.layer1_size = layer1_size
        self.layer2_size = layer2_size

        # 13.バッチサイズを決めておく
        self.batch_size = batch_size 

        self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                            layer1_size=64, layer2_size=64, batch_size=64)        
        
        # 9.memoryインスタンスを追加
        self.MAX_MEMORY_SIZE = 10000
        self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,
                                   n_obs_space=self.n_obs_space,
                                   n_action_space=self.n_action_space)
        
        # 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する
        # actorとtarget_actorのネットワークは同じActorNNで良い
        self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)

        
        # 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する
        self.target_critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)
        
        # 24.クリティックネットワークインスタンスcriticを作成する。
        self.critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,
                               layer1_size=64, layer2_size=64, batch_size=64)
        
        # アクターロスとクリティックロス
        self.actor_loss = 0
        self.critic_loss = 0
        
        # 45.行動ノイズのインスタンス化
        self.mode = mode
        if self.mode == 'train_mode':
            self.noise = OUActionNoise(mu=np.zeros(n_action_space))
        elif self.mode == 'eval_mode':
            self.noise = OUActionNoise(mu=np.zeros(n_action_space), sigma=0)
        else:
            print('mode error')
        

    def choose_action(self, obs):
        #print('AgentDDPG.choose_action is working.')
        # 4.方策（アクター）はニューラルネットワークで表現する。
        #   ActorNNクラスを新規作成し、インスタンスactorとして使用する。
        action = self.actor.forward(obs)
        

        # 46.行動ノイズを入れて探索性を向上させる。
        action += T.tensor(self.noise(), dtype=T.float32)
        action = action.detach().numpy()

        return action
    
    # 8.remenberメソドを追加
    def remember(self, obs, action, reward, next_state, done):
        self.memory.store_transition(obs, action, reward, next_state, done)

    # 13.learnメソドを追加
    def learn(self):
        # 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。
        if self.memory.memory_count < self.batch_size:
            return
        
        # 15.メモリバッファからデータを抜き出す sample_buffer()
        # バッチ化されているので変数名を複数形にする
        observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)
        #print('s:', observations)
        #print(observations.shape)
        #print('a :', actions)
        #print('r :', rewards)
        #print('s_ :', next_states)
        #print('terminal :', terminals)

        # 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する
        observations = T.tensor(observations, dtype=T.float32)
        actions = T.tensor(actions, dtype=T.float32)
        rewards = T.tensor(rewards, dtype=T.float32)
        next_states = T.tensor(next_states, dtype=T.float32)
        terminals = T.tensor(terminals, dtype=T.float32)
       
        # 18.ターゲットアクターネットワークインスタンスtarget_actorに
        # 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。
        # このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。
        #print('next_states :', next_states)
        target_actions = self.target_actor.forward(next_states)

        # 20.ターゲットクリティックネットワークインスタンスtarget_criticに
        # 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して
        # 価値関数の推定値ターゲットバリューを出力する。
        # TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。
        # ターゲットクリティックバリューはターゲットアクターネットワークを使う
        target_critic_values = self.target_critic.forward(next_states, target_actions)

        
        # 23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に
        # 現在の状態observationsと行動actionsを入力して
        # クリティックバリューを算出する
        critic_values = self.critic.forward(observations, actions)

        # 25.TDターゲットを算出する：r + γ*V(w)[s_t+1]
        td_targets = []
        for i in range(self.batch_size):
            td_target = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]
            td_targets.append(td_target)
        
        # TDターゲットの形をバッチに整える
        td_targets = T.tensor(td_targets, dtype=T.float32)
        td_targets = td_targets.view(self.batch_size, 1) #viewはreshapeと同じ。64x1に見え方を変更した、という意味
        #print('td_targets :', td_targets)


        # ==== （１）クリティックの学習 ====

        # 28.クリティックの勾配をゼロに初期化する
        self.critic.optimizer.zero_grad()

        # 29. TDターゲットと状態価値の二乗誤差を算出して、クリティックの損失関数とする。バッチサイズは６４個
        critic_loss = F.mse_loss(td_targets, critic_values)
        self.critic_loss = critic_loss
        #print('critic_loss : ', critic_loss) # tensor(0.0485, grad_fn=<MseLossBackward0>)

        # 30. クリティックの損失関数を微分して、勾配を算出する
        critic_loss.backward()

        # 31. 勾配からオプティマイザーによってクリティックのパラメータ（重みとバイアス）を更新する
        self.critic.optimizer.step()


        # ==== （２）アクターの学習 ====

        # 32. アクターの勾配をゼロに初期化する
        self.actor.optimizer.zero_grad()

        # 33. アクターに観測情報を入力して行動を算出する。バッチサイズは６４個
        predicted_actions = self.actor.forward(observations)

        # 34.アクターの損失関数を算出する
        #    Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択すること。
        #    なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardし、
        #    actorだけをパラメータ更新することによりactorの学習をすることができる。

        actor_loss = -self.critic.forward(observations, predicted_actions)
        actor_loss = T.mean(actor_loss)
        self.actor_loss = actor_loss

        #print(f'actor_loss: {actor_loss}, critic_loss: {critic_loss}')

        # 35. DDPG構造全体の損失関数actor_lossを微分し、勾配を算出する
        actor_loss.backward()

        # 36. 勾配からオプティマイザーによってアクターのパラメータだけを（重みとバイアス）を更新する
        self.actor.optimizer.step()
    
        # 37. 全ニューラルネットワークのパラメータを更新する。
        self.update_network_parameters()

    # 37. パラメータ更新メソド。
    def update_network_parameters(self, tau=None):
        if tau is None:
            tau = self.tau
    
        # 38. actor, critic, target_actor, target_criticのネットワーク内の全てのパラメータ（重みとバイアス）とその名前を取得する
        # actorとcriticは先ほど更新されたばかりのパラメーター
        actor_params = self.actor.named_parameters()
        critic_params = self.critic.named_parameters()
        target_actor_params = self.target_actor.named_parameters()
        target_critic_params = self.target_critic.named_parameters()
        #print('actor_params : ', actor_params) # actor_params :  <generator object Module.named_parameters at 0x000001661B2D9D48>

        # 39. パラメータをディクショナリとして取り出す。
        actor_params_dict = dict(actor_params)
        critic_params_dict = dict(critic_params)
        target_actor_params_dict = dict(target_actor_params)
        target_critic_params_dict = dict(target_critic_params)
        #print('actor_params_dict : ', actor_params_dict)
        #print(actor_params_dict.keys())
        """
        actor_params_dict :  {'fc1.weight': Parameter containing:
                                   tensor([[-0.1895, -0.0343,  0.1138,  ...,  0.2157,  0.0527, -0.1173],/

        dict_keys(['fc1.weight', 'fc1.bias', 'fc2.weight', 'fc2.bias', 'fc3.weight', 'fc3.bias'])
        """

        # 40. クリティックの各パラメーター毎に 更新重みtau=0.0001の分だけほんの少しcriticパラメータをtarget_criticパラメータに近づける。
        for name in critic_params_dict:
            critic_params_dict[name] = tau * critic_params_dict[name].clone() + \
                                       (1-tau) * target_critic_params_dict[name].clone()
            
        # 41. 更新したcriticパラメータをtarget_criticのパラメータとしてロードする。
        self.target_critic.load_state_dict(critic_params_dict)

        # 42.アクターの各パラメーター毎に 更新重みtau=0.0001の分だけほんの少しactorパラメータをtarget_actorパラメータに近づける。
        for name in actor_params_dict:
            actor_params_dict[name] = tau * actor_params_dict[name].clone() + \
                                      (1 - tau) * target_actor_params_dict[name].clone()
            
        # 43. 更新したactorパラメータをtarget_actorのパラメータとしてロードする。
        self.target_actor.load_state_dict(actor_params_dict)


#### =================== メインスクリプト ======================= ####



EVAL_TRAIN_MODE = 'train_mode' # 評価モードか訓練モードかを選択
EPISODES = 1001# episodes
STEPS = 500    # steps
DELAY_TIME = 0.00 # sec

# 2.エージェントクラスのインスタンスを生成する
agent = AgentDDPG(alpha=0.01, beta=0.01, gamma=0.99, tau=0.01,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=256, layer2_size=256, batch_size=64, mode=EVAL_TRAIN_MODE)

if EVAL_TRAIN_MODE == 'train_mode':
    env = gym.make("HalfCheetah-v4", render_mode='depth_array')

elif EVAL_TRAIN_MODE == 'eval_mode':
    env = gym.make("HalfCheetah-v4", render_mode= 'human')


total_rewards = []
actor_losses = []
critic_losses = []
for episode in range(EPISODES):
    obs = env.reset()
    obs = T.tensor(obs[0], dtype=T.float)
    # tensor([ 0.0040,  0.0199, -0.0622,  0.0594, -0.0605,  0.0577, -0.0056,  0.0333,        -0.0072,  0.0532, -0.0512,  0.0173, -0.0529, -0.1104,  0.0946, -0.0559,         0.0824])
    #print(type(obs))
    # observation_space :  Box(-inf, inf, (17,), float64)
    #print('observation_space : ', env.observation_space)
    #print('obs :', obs)

    reward: float = 0
    total_reward: float = 0
    done: bool = False
    for j in range(STEPS):
        env.render()
        
        # ここをDDPGに置き換えていく
        action = agent.choose_action(obs) # 1.Agentクラスを定義していく
        #action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]
        #print('====ここまではOK4====')
        #print('action_space : ', env.action_space)
        #print('action : ', action)

        next_state, reward, done, _, info = env.step(action)
        #print('next_state, reward, done, _, info :', next_state, reward, done, _, info)
        """
        action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]

        next_state, reward, done, _, info : 
        [-0.00265179  0.0229547   0.00463243 -0.04729936 -0.00959038  0.04734605
        0.03672746  0.02857842  0.09980254 -0.32065693  0.04221647  1.58668951
        -2.31089174  1.30338924 -0.25465526  1.08250465 -0.14134398]
        0.07553858359316026
        False
        False
        {'x_position': -0.09233384215910741, 'x_velocity': 0.07920445513883267, 'reward_run': 0.07920445513883267, 'reward_ctrl': -0.0036658715456724168}
        """

        #7. トラジェクトを保存する。経験再生(ReplayBuffer)
        agent.remember(obs, action, reward, next_state, int(done))

        #12. ニューラルネットワークを学習する
        agent.learn()
        
        # 26.エピソード内での報酬を累積していく
        total_reward += reward
        
        # 27. next_stateをobsとして再出発する
        #print('next_state:', next_state)
        obs = next_state
        obs = T.tensor(obs, dtype=T.float)
        # 28. チーターの動きを見たいのでスリープを入れる
        time.sleep(DELAY_TIME)

    #print('total_reward : ', total_reward)
    total_rewards.append(total_reward)

    actor_losses.append(float(agent.actor_loss)) 
    critic_losses.append(float(agent.critic_loss)) 

    # print('epsisode', i, 'score %.2f' % score, '100 game sverage %.2f' % np.mean(score_history[-100:]))
 
    # 47. 各ニューラルネットワークのパラメータを１０エピソード毎に保存する
    print('episode, total_reward : ', episode , total_reward)
    if episode % 10 == 0:
        T.save(agent.actor.state_dict(), 'actor_params.pt')
        T.save(agent.critic.state_dict(), 'critic_params.pt')
        T.save(agent.target_actor.state_dict(), 'target_actor_params.pt')
        T.save(agent.target_critic.state_dict(), 'target_critic_params.pt')
        print('==== params were saved. ====')

#print('total_rewards : ', total_rewards)
#plt.plot(total_rewards)

#plt.plot(actor_losses, label='actor_losses')
#plt.plot(critic_losses, label='critic_losses')
plt.plot(total_rewards, label='total_rewards')
plt.legend()
plt.grid(True)
plt.ioff()
plt.show()

env.close() # 空なんですけど・・・

print('script is done.')

# https://gymnasium.farama.org/

100

101

102

103

104

105

106

107

108

109

110

111

112

113

114

115

116

117

118

119

120

121

122

123

124

125

126

127

128

129

130

131

132

133

134

135

136

137

138

139

140

141

142

143

144

145

146

147

148

149

150

151

152

153

154

155

156

157

158

159

160

161

162

163

164

165

166

167

168

169

170

171

172

173

174

175

176

177

178

179

180

181

182

183

184

185

186

187

188

189

190

191

192

193

194

195

196

197

198

199

200

201

202

203

204

205

206

207

208

209

210

211

212

213

214

215

216

217

218

219

220

221

222

223

224

225

226

227

228

229

230

231

232

233

234

235

236

237

238

239

240

241

242

243

244

245

246

247

248

249

250

251

252

253

254

255

256

257

258

259

260

261

262

263

264

265

266

267

268

269

270

271

272

273

274

275

276

277

278

279

280

281

282

283

284

285

286

287

288

289

290

291

292

293

294

295

296

297

298

299

300

301

302

303

304

305

306

307

308

309

310

311

312

313

314

315

316

317

318

319

320

321

322

323

324

325

326

327

328

329

330

331

332

333

334

335

336

337

338

339

340

341

342

343

344

345

346

347

348

349

350

351

352

353

354

355

356

357

358

359

360

361

362

363

364

365

366

367

368

369

370

371

372

373

374

375

376

377

378

379

380

381

382

383

384

385

386

387

388

389

390

391

392

393

394

395

396

397

398

399

400

401

402

403

404

405

406

407

408

409

410

411

412

413

414

415

416

417

418

419

420

421

422

423

424

425

426

427

428

429

430

431

432

433

434

435

436

437

438

439

440

441

442

443

444

445

446

447

448

449

450

451

452

453

454

455

456

457

458

459

460

461

462

463

464

465

466

467

468

469

470

471

472

473

474

475

476

477

478

479

480

481

482

483

484

485

486

487

488

489

490

491

492

493

494

495

496

497

498

499

500

501

502

503

504

505

506

507

508

509

import gymnasium as gym

import time

import torch as T

import torch.nn as nn

import torch.nn.functional as F

import torch.optim as optim

import numpy as np

import matplotlib.pyplot as plt

import os

""" nvidia CUDA Toolkit 12.1

https://developer.nvidia.com/cuda-downloads?target_os=Windows&target_arch=x86_64&target_version=10&target_type=exe_local

Download cuda_12.1.1_531.14_windows.exe

"""

""" pip install

pip3 install --pre torch torchvision torchaudio --index-url https://download.pytorch.org/whl/nightly/cu121

pip install gymnasium

pip install gymnasium[mujoco]

pip install matplotlib

pip install mujoco

"""

# 44.OUActionNOoiseクラスを作成する

class OUActionNoise(object):

def __init__(self, mu, sigma=0.15, theta=0.2, dt=1e-2, x0=None):

self.mu = mu

self.sigma = sigma

self.theta = theta

self.dt = dt

self.x0 = x0

self.reset()

def __call__(self):

x = self.x_prev + self.theta * (self.mu - self.x_prev) * self.dt + \

self.sigma * np.sqrt(self.dt) * np.random.normal(size=self.mu.shape)

self.x_prev = x

return x

def reset(self):

self.x_prev = self.x0 if self.x0 is not None else np.zeros_like(self.mu)

# 10. ReplayBufferクラスを新規作成する

class ReplayBuffer:

def __init__(self, max_memory_size, n_obs_space, n_action_space):

self.max_memory_size = max_memory_size

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.memory_count = 0

self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.action_memory = np.zeros((self.max_memory_size, self.n_action_space))

self.reward_memory = np.zeros(self.max_memory_size)

self.next_state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.terminal_memory = np.zeros(self.max_memory_size)

#self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

# 11.トランジション保存のためstore_transitionメソドを作成する

def store_transition(self, obs, action, reward, next_state, done):

#print('store_transition is working.')

index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック

#print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())

self.state_memory[index] = obs.detach().numpy().flatten()

self.action_memory[index] = action.flatten()

self.reward_memory[index] = reward.flatten()

self.next_state_memory[index] = next_state.flatten()

self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように

#print('state_memory :', self.state_memory)

#print('action_memory :', self.action_memory)

#print('reward_memory :', self.reward_memory)

#print('next_state_memory :', self.next_state_memory)

#print('memory.state_memory :', self.terminal_memory)

#print('type of state_memory :', type(self.state_memory[0][0]))

#print('type of action_memory :', type(self.action_memory[0][0]))

#print('type of reward_memory :', type(self.reward_memory[0]))

#print('type of next_state_memory :', type(self.next_state_memory[0][0]))

#print('type of memory.state_memory :', type(self.terminal_memory[0]))

self.memory_count += 1

#print('memory_count :', agent.memory.memory_count)

# 16 バッファメモリーからランダムに抽出する

def sample_buffer(self, batch_size):

# indexが最大メモリに到達していない場合を想定する。

max_index = min(self.max_memory_size, self.memory_count)

choosed_index = np.random.choice(max_index, batch_size)

observations = self.state_memory[choosed_index]

actions = self.action_memory[choosed_index]

rewards = self.reward_memory[choosed_index]

next_states = self.next_state_memory[choosed_index]

terminals = self.terminal_memory[choosed_index]

return observations, actions, rewards, next_states, terminals

# 6.ActorNNクラスを新規作成する

class ActorNN(nn.Module):

def __init__(self, alpha=0.001, n_obs_space=17, n_action_space=6,

layer1_size=256, layer2_size=256, batch_size=64):

#print('ActorNN.__init__ is working.')

super(ActorNN, self).__init__()

self.fc1 = nn.Linear(n_obs_space, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, n_action_space)

#26.最適化処理としてアダムを設定する

self.optimizer = optim.Adam(self.parameters(), lr=alpha)

# 48. actorパラメータの読み出し

# もし、パラメータのデータが存在していたらそのパラメータで初期化する。

# パラメータファイルの存在チェック

if T.cuda.is_available():

map_location = 'cuda'

else:

map_location = 'cpu'

if os.path.isfile('actor_params.pt'):

# パラメータファイルが存在する場合はロード

self.load_state_dict(T.load('actor_params.pt', map_location=map_location))

print("パラメータファイルをロードしました:", 'actor_params.pt')

else:

print("パラメータファイルが見つかりません:", 'actor_params.pt')

def forward(self, obs):

#print('AgetDDPG.ActorNN.forward is working')

#print('====ここまではOK1====')

x = self.fc1(obs)

x = F.relu(x)

x = self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

action = F.tanh(x)

return action

# 22.CriticNNクラスを新規作成する

class CriticNN(nn.Module):

def __init__(self, beta=0.001, n_obs_space=17, n_action_space=6,

layer1_size=256, layer2_size=256, batch_size=64):

#print('CriticNN.__init__ is working.')

super(CriticNN, self).__init__()

# クリティックNNは観察空間+行動空間の２つを入力とする構造

input_dim = n_obs_space + n_action_space

self.fc1 = nn.Linear(input_dim, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

#27.最適化処理としてアダムを設定する

self.optimizer = optim.Adam(self.parameters(), lr=beta)

# 49. criticパラメータの読み出し

# もし、パラメータのデータが存在していたらそのパラメータで初期化する。

# パラメータファイルの存在チェック

if T.cuda.is_available():

map_location = 'cuda'

else:

map_location = 'cpu'

if os.path.isfile('critic_params.pt'):

# パラメータファイルが存在する場合はロード

self.load_state_dict(T.load('critic_params.pt', map_location=map_location))

print("パラメータファイルをロードしました:", 'critic_params.pt')

else:

print("パラメータファイルが見つかりません:", 'critic_params.pt')

def forward(self, obs, action):

input_data = T.cat([obs, action], dim=1)

x = self.fc1(input_data)

x = F.relu(x)

x =self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

return x #一つの状態価値を出力する。

# 3.エージェントクラスを定義する

class AgentDDPG:

def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64, mode='train_mode'):

#print('AgentDDPG.__init__ is working.')

# 5.ActorNNクラスのインスタンスを生成する

self.alpha = alpha

self.beta = beta

self.gamma = gamma

self.tau = tau

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.n_state_action_value = n_state_action_value

self.layer1_size = layer1_size

self.layer2_size = layer2_size

# 13.バッチサイズを決めておく

self.batch_size = batch_size

self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 9.memoryインスタンスを追加

self.MAX_MEMORY_SIZE = 10000

self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,

n_obs_space=self.n_obs_space,

n_action_space=self.n_action_space)

# 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する

# actorとtarget_actorのネットワークは同じActorNNで良い

self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する

self.target_critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 24.クリティックネットワークインスタンスcriticを作成する。

self.critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# アクターロスとクリティックロス

self.actor_loss = 0

self.critic_loss = 0

# 45.行動ノイズのインスタンス化

self.mode = mode

if self.mode == 'train_mode':

self.noise = OUActionNoise(mu=np.zeros(n_action_space))

elif self.mode == 'eval_mode':

self.noise = OUActionNoise(mu=np.zeros(n_action_space), sigma=0)

else:

print('mode error')

def choose_action(self, obs):

#print('AgentDDPG.choose_action is working.')

# 4.方策（アクター）はニューラルネットワークで表現する。

# ActorNNクラスを新規作成し、インスタンスactorとして使用する。

action = self.actor.forward(obs)

# 46.行動ノイズを入れて探索性を向上させる。

action += T.tensor(self.noise(), dtype=T.float32)

action = action.detach().numpy()

return action

# 8.remenberメソドを追加

def remember(self, obs, action, reward, next_state, done):

self.memory.store_transition(obs, action, reward, next_state, done)

# 13.learnメソドを追加

def learn(self):

# 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。

if self.memory.memory_count < self.batch_size:

return

# 15.メモリバッファからデータを抜き出す sample_buffer()

# バッチ化されているので変数名を複数形にする

observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)

#print('s:', observations)

#print(observations.shape)

#print('a :', actions)

#print('r :', rewards)

#print('s_ :', next_states)

#print('terminal :', terminals)

# 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する

observations = T.tensor(observations, dtype=T.float32)

actions = T.tensor(actions, dtype=T.float32)

rewards = T.tensor(rewards, dtype=T.float32)

next_states = T.tensor(next_states, dtype=T.float32)

terminals = T.tensor(terminals, dtype=T.float32)

# 18.ターゲットアクターネットワークインスタンスtarget_actorに

# 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。

# このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。

#print('next_states :', next_states)

target_actions = self.target_actor.forward(next_states)

# 20.ターゲットクリティックネットワークインスタンスtarget_criticに

# 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して

# 価値関数の推定値ターゲットバリューを出力する。

# TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。

# ターゲットクリティックバリューはターゲットアクターネットワークを使う

target_critic_values = self.target_critic.forward(next_states, target_actions)

# 23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に

# 現在の状態observationsと行動actionsを入力して

# クリティックバリューを算出する

critic_values = self.critic.forward(observations, actions)

# 25.TDターゲットを算出する：r + γ*V(w)[s_t+1]

td_targets = []

for i in range(self.batch_size):

td_target = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]

td_targets.append(td_target)

# TDターゲットの形をバッチに整える

td_targets = T.tensor(td_targets, dtype=T.float32)

td_targets = td_targets.view(self.batch_size, 1) #viewはreshapeと同じ。64x1に見え方を変更した、という意味

#print('td_targets :', td_targets)

# ==== （１）クリティックの学習 ====

# 28.クリティックの勾配をゼロに初期化する

self.critic.optimizer.zero_grad()

# 29. TDターゲットと状態価値の二乗誤差を算出して、クリティックの損失関数とする。バッチサイズは６４個

critic_loss = F.mse_loss(td_targets, critic_values)

self.critic_loss = critic_loss

#print('critic_loss : ', critic_loss) # tensor(0.0485, grad_fn=<MseLossBackward0>)

# 30. クリティックの損失関数を微分して、勾配を算出する

critic_loss.backward()

# 31. 勾配からオプティマイザーによってクリティックのパラメータ（重みとバイアス）を更新する

self.critic.optimizer.step()

# ==== （２）アクターの学習 ====

# 32. アクターの勾配をゼロに初期化する

self.actor.optimizer.zero_grad()

# 33. アクターに観測情報を入力して行動を算出する。バッチサイズは６４個

predicted_actions = self.actor.forward(observations)

# 34.アクターの損失関数を算出する

# Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択すること。

# なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardし、

# actorだけをパラメータ更新することによりactorの学習をすることができる。

actor_loss = -self.critic.forward(observations, predicted_actions)

actor_loss = T.mean(actor_loss)

self.actor_loss = actor_loss

#print(f'actor_loss: {actor_loss}, critic_loss: {critic_loss}')

# 35. DDPG構造全体の損失関数actor_lossを微分し、勾配を算出する

actor_loss.backward()

# 36. 勾配からオプティマイザーによってアクターのパラメータだけを（重みとバイアス）を更新する

self.actor.optimizer.step()

# 37. 全ニューラルネットワークのパラメータを更新する。

self.update_network_parameters()

# 37. パラメータ更新メソド。

def update_network_parameters(self, tau=None):

if tau is None:

tau = self.tau

# 38. actor, critic, target_actor, target_criticのネットワーク内の全てのパラメータ（重みとバイアス）とその名前を取得する

# actorとcriticは先ほど更新されたばかりのパラメーター

actor_params = self.actor.named_parameters()

critic_params = self.critic.named_parameters()

target_actor_params = self.target_actor.named_parameters()

target_critic_params = self.target_critic.named_parameters()

#print('actor_params : ', actor_params) # actor_params : <generator object Module.named_parameters at 0x000001661B2D9D48>

# 39. パラメータをディクショナリとして取り出す。

actor_params_dict = dict(actor_params)

critic_params_dict = dict(critic_params)

target_actor_params_dict = dict(target_actor_params)

target_critic_params_dict = dict(target_critic_params)

#print('actor_params_dict : ', actor_params_dict)

#print(actor_params_dict.keys())

"""

actor_params_dict : {'fc1.weight': Parameter containing:

tensor([[-0.1895, -0.0343, 0.1138, ..., 0.2157, 0.0527, -0.1173],/

dict_keys(['fc1.weight', 'fc1.bias', 'fc2.weight', 'fc2.bias', 'fc3.weight', 'fc3.bias'])

"""

# 40. クリティックの各パラメーター毎に更新重みtau=0.0001の分だけほんの少しcriticパラメータをtarget_criticパラメータに近づける。

for name in critic_params_dict:

critic_params_dict[name] = tau * critic_params_dict[name].clone() + \

(1-tau) * target_critic_params_dict[name].clone()

# 41. 更新したcriticパラメータをtarget_criticのパラメータとしてロードする。

self.target_critic.load_state_dict(critic_params_dict)

# 42.アクターの各パラメーター毎に更新重みtau=0.0001の分だけほんの少しactorパラメータをtarget_actorパラメータに近づける。

for name in actor_params_dict:

actor_params_dict[name] = tau * actor_params_dict[name].clone() + \

(1 - tau) * target_actor_params_dict[name].clone()

# 43. 更新したactorパラメータをtarget_actorのパラメータとしてロードする。

self.target_actor.load_state_dict(actor_params_dict)

#### =================== メインスクリプト ======================= ####

EVAL_TRAIN_MODE = 'train_mode' # 評価モードか訓練モードかを選択

EPISODES = 1001# episodes

STEPS = 500 # steps

DELAY_TIME = 0.00 # sec

# 2.エージェントクラスのインスタンスを生成する

agent = AgentDDPG(alpha=0.01, beta=0.01, gamma=0.99, tau=0.01,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=256, layer2_size=256, batch_size=64, mode=EVAL_TRAIN_MODE)

if EVAL_TRAIN_MODE == 'train_mode':

env = gym.make("HalfCheetah-v4", render_mode='depth_array')

elif EVAL_TRAIN_MODE == 'eval_mode':

env = gym.make("HalfCheetah-v4", render_mode= 'human')

total_rewards = []

actor_losses = []

critic_losses = []

for episode in range(EPISODES):

obs = env.reset()

obs = T.tensor(obs[0], dtype=T.float)

# tensor([ 0.0040, 0.0199, -0.0622, 0.0594, -0.0605, 0.0577, -0.0056, 0.0333, -0.0072, 0.0532, -0.0512, 0.0173, -0.0529, -0.1104, 0.0946, -0.0559, 0.0824])

#print(type(obs))

# observation_space : Box(-inf, inf, (17,), float64)

#print('observation_space : ', env.observation_space)

#print('obs :', obs)

reward: float = 0

total_reward: float = 0

done: bool = False

for j in range(STEPS):

env.render()

# ここをDDPGに置き換えていく

action = agent.choose_action(obs) # 1.Agentクラスを定義していく

#action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

#print('====ここまではOK4====')

#print('action_space : ', env.action_space)

#print('action : ', action)

next_state, reward, done, _, info = env.step(action)

#print('next_state, reward, done, _, info :', next_state, reward, done, _, info)

"""

action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

next_state, reward, done, _, info :

[-0.00265179 0.0229547 0.00463243 -0.04729936 -0.00959038 0.04734605

0.03672746 0.02857842 0.09980254 -0.32065693 0.04221647 1.58668951

-2.31089174 1.30338924 -0.25465526 1.08250465 -0.14134398]

0.07553858359316026

False

{'x_position': -0.09233384215910741, 'x_velocity': 0.07920445513883267, 'reward_run': 0.07920445513883267, 'reward_ctrl': -0.0036658715456724168}

"""

#7. トラジェクトを保存する。経験再生(ReplayBuffer)

agent.remember(obs, action, reward, next_state, int(done))

#12. ニューラルネットワークを学習する

agent.learn()

# 26.エピソード内での報酬を累積していく

total_reward += reward

# 27. next_stateをobsとして再出発する

#print('next_state:', next_state)

obs = next_state

obs = T.tensor(obs, dtype=T.float)

# 28. チーターの動きを見たいのでスリープを入れる

time.sleep(DELAY_TIME)

#print('total_reward : ', total_reward)

total_rewards.append(total_reward)

actor_losses.append(float(agent.actor_loss))

critic_losses.append(float(agent.critic_loss))

# print('epsisode', i, 'score %.2f' % score, '100 game sverage %.2f' % np.mean(score_history[-100:]))

# 47. 各ニューラルネットワークのパラメータを１０エピソード毎に保存する

print('episode, total_reward : ', episode , total_reward)

if episode % 10 == 0:

T.save(agent.actor.state_dict(), 'actor_params.pt')

T.save(agent.critic.state_dict(), 'critic_params.pt')

T.save(agent.target_actor.state_dict(), 'target_actor_params.pt')

T.save(agent.target_critic.state_dict(), 'target_critic_params.pt')

print('==== params were saved. ====')

#print('total_rewards : ', total_rewards)

#plt.plot(total_rewards)

#plt.plot(actor_losses, label='actor_losses')

#plt.plot(critic_losses, label='critic_losses')

plt.plot(total_rewards, label='total_rewards')

plt.legend()

plt.grid(True)

plt.ioff()

plt.show()

env.close() # 空なんですけど・・・

print('script is done.')

# https://gymnasium.farama.org/

DDPG by gymnasium ９日目

前回まででうまいこと学習が進むようになりましたので、今回はパラメータの保存と読出をやってみましょう。

例えば、エピソードを１００回繰り返しある程度ハーフチーターが前に進む方策を得たらパラメータをいったん保存します。

プログラムを止めて次回動かすときは保存したパラメータを読み込んで、学習済みの状態から動かすことができます。

これで突然プログラムが途中で止まってしまっても被害を最小限にできますね。

パラメータ保存

メインスクリプトでエピソードの終わり、次のエピソードが始まる直前に下記コードを入れます。

    # 47.各ニューラルネットワークのパラメータを１０エピソード毎に保存する
    print('episode, total_reward : ', episode , total_reward)
    if episode % 10 == 0:
        T.save(agent.actor.state_dict(), 'actor_params.pt')
        T.save(agent.critic.state_dict(), 'critic_params.pt')
        T.save(agent.target_actor.state_dict(), 'target_actor_params.pt')
        T.save(agent.target_critic.state_dict(), 'target_critic_params.pt')
        print('==== params were saved. ====')

# 47.各ニューラルネットワークのパラメータを１０エピソード毎に保存する

print('episode, total_reward : ', episode , total_reward)

if episode % 10 == 0:

T.save(agent.actor.state_dict(), 'actor_params.pt')

T.save(agent.critic.state_dict(), 'critic_params.pt')

T.save(agent.target_actor.state_dict(), 'target_actor_params.pt')

T.save(agent.target_critic.state_dict(), 'target_critic_params.pt')

print('==== params were saved. ====')

パラメータ読込

ActorNN(n.Module)クラスの__init__()内に入れて、actor, target_actorのインスタンス生成と同時にパラメータを引き継いでもらうようにします。

        # 48. actorパラメータの読み出し
        # もし、パラメータのデータが存在していたらそのパラメータで初期化する。
        # パラメータファイルの存在チェック
        if T.cuda.is_available():
            map_location = 'cuda'
        else:
            map_location = 'cpu'

        if os.path.isfile('actor_params.pt'):
            # パラメータファイルが存在する場合はロード
            self.load_state_dict(T.load('actor_params.pt', map_location=map_location))
            print("パラメータファイルをロードしました:", 'actor_params.pt')
        else:
            print("パラメータファイルが見つかりません:", 'actor_params.pt')

# 48. actorパラメータの読み出し

# もし、パラメータのデータが存在していたらそのパラメータで初期化する。

# パラメータファイルの存在チェック

if T.cuda.is_available():

map_location = 'cuda'

else:

map_location = 'cpu'

if os.path.isfile('actor_params.pt'):

# パラメータファイルが存在する場合はロード

self.load_state_dict(T.load('actor_params.pt', map_location=map_location))

print("パラメータファイルをロードしました:", 'actor_params.pt')

else:

print("パラメータファイルが見つかりません:", 'actor_params.pt')

CriticNN(nn.Module)クラスも同様に。

        # 49. criticパラメータの読み出し
        # もし、パラメータのデータが存在していたらそのパラメータで初期化する。
        # パラメータファイルの存在チェック
        if T.cuda.is_available():
            map_location = 'cuda'
        else:
            map_location = 'cpu'

        if os.path.isfile('critic_params.pt'):
            # パラメータファイルが存在する場合はロード
            self.load_state_dict(T.load('critic_params.pt', map_location=map_location))
            print("パラメータファイルをロードしました:", 'critic_params.pt')
        else:
            print("パラメータファイルが見つかりません:", 'critic_params.pt')

# 49. criticパラメータの読み出し

# もし、パラメータのデータが存在していたらそのパラメータで初期化する。

# パラメータファイルの存在チェック

if T.cuda.is_available():

map_location = 'cuda'

else:

map_location = 'cpu'

if os.path.isfile('critic_params.pt'):

# パラメータファイルが存在する場合はロード

self.load_state_dict(T.load('critic_params.pt', map_location=map_location))

print("パラメータファイルをロードしました:", 'critic_params.pt')

else:

print("パラメータファイルが見つかりません:", 'critic_params.pt')

これで、動きます。

学習のノウハウ

学習のコツを編み出しました。

学習初期はエージェントの動きが小さくなかなか前進しません。

最初はステップ数を10～100程度に小さくして、スタートダッシュだけを覚えさせました。

そこでいったん止めて、ステップ数を200、400と増やしていくと安定して走り続けるハーフチーターが得られます。

計算の高速化（3Dモデルの表示をオフにする）

env = gym.make(“HalfCheetah-v4”, render_mode= ‘human’)

の中のrender_modeを’depth_array’に変更すればOKです。

学習の進行状況のリアルタイム可視化

エピソード数とそのリワードだけを表示しています。

print(‘episode, total_reward : ‘, episode , total_reward)

パラメータの保存を10エピソード毎にやっています。

if episode % 10 == 0:

print(‘==== params were saved. ====’)

↓↓↓出力

episode, total_reward :  498 259.49914064186675
episode, total_reward :  499 360.5913236729034
episode, total_reward :  500 493.48175790551875
==== params were saved. ====
episode, total_reward :  501 -38.671055063944046
episode, total_reward :  502 260.85835046795177
episode, total_reward :  503 327.2605501737727
episode, total_reward :  504 355.8699644344001
episode, total_reward :  505 381.4410905904467
episode, total_reward :  506 342.0158764598478
episode, total_reward :  507 314.5073192976202
episode, total_reward :  508 82.38588849764184
episode, total_reward :  509 355.04443705923273
episode, total_reward :  510 337.79242797960603
==== params were saved. ====
episode, total_reward :  511 380.88193759326816
episode, total_reward :  512 437.873030625086
episode, total_reward :  513 277.1307694476558

episode, total_reward : 498 259.49914064186675

episode, total_reward : 499 360.5913236729034

episode, total_reward : 500 493.48175790551875

==== params were saved. ====

episode, total_reward : 501 -38.671055063944046

episode, total_reward : 502 260.85835046795177

episode, total_reward : 503 327.2605501737727

episode, total_reward : 504 355.8699644344001

episode, total_reward : 505 381.4410905904467

episode, total_reward : 506 342.0158764598478

episode, total_reward : 507 314.5073192976202

episode, total_reward : 508 82.38588849764184

episode, total_reward : 509 355.04443705923273

episode, total_reward : 510 337.79242797960603

==== params were saved. ====

episode, total_reward : 511 380.88193759326816

episode, total_reward : 512 437.873030625086

episode, total_reward : 513 277.1307694476558

解決できた課題

パラメータのセーブとロード
途中で止まった時に続行可能にしたい
計算の高速化（print文の無効化）
計算の高速化（3Dモデルの表示をオフにする）
学習の進行状況のリアルタイム可視化
適切なステップ数
適切なニューラルネットワーク構造

未解決の課題・疑問点

計算の高速化（GPUの利用）
適切なエピソード数
適切なメモリバッファ数
model.train()とmodel.eval()の使い方が分からない。
ネットワークの入力値？パラメータ？の正規化。
保存したパラメータを読み出すのはactorとtarget_actorまたcriticとtarget_criticで共通で良いのだろうか。

これまでのスクリプト

import gymnasium as gym
import time
import torch as T
import torch.nn as nn
import torch.nn.functional as F
import torch.optim as optim
import numpy as np
import matplotlib.pyplot as plt
import os

""" pip install
pip3 install --pre torch torchvision torchaudio --index-url https://download.pytorch.org/whl/nightly/cu121
pip install gymnasium
pip install matplotlib
pip install mujoco
pip install gymnasium[mujoco]
"""
# 44.OUActionNOoiseクラスを作成する
class OUActionNoise(object):
    def __init__(self, mu, sigma=0.15, theta=0.2, dt=1e-2, x0=None):
        self.mu = mu
        self.sigma = sigma
        self.theta = theta
        self.dt = dt
        self.x0 = x0
        self.reset()

    def __call__(self):
        x = self.x_prev + self.theta * (self.mu - self.x_prev) * self.dt + \
        self.sigma * np.sqrt(self.dt) * np.random.normal(size=self.mu.shape)
        self.x_prev = x
        return x

    def reset(self):
        self.x_prev = self.x0 if self.x0 is not None else np.zeros_like(self.mu)


# 10. ReplayBufferクラスを新規作成する
class ReplayBuffer:
    def __init__(self, max_memory_size, n_obs_space, n_action_space):
        self.max_memory_size = max_memory_size
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.memory_count = 0

        self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))
        self.action_memory =  np.zeros((self.max_memory_size, self.n_action_space))
        self.reward_memory =  np.zeros(self.max_memory_size)
        self.next_state_memory =  np.zeros((self.max_memory_size, self.n_obs_space))
        self.terminal_memory =  np.zeros(self.max_memory_size)
        #self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

    # 11.トランジション保存のためstore_transitionメソドを作成する
    def store_transition(self, obs, action, reward, next_state, done):
        #print('store_transition is working.')
        index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック
        #print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())
        self.state_memory[index] = obs.detach().numpy().flatten()
        self.action_memory[index] = action.flatten()
        self.reward_memory[index] = reward.flatten()
        self.next_state_memory[index] = next_state.flatten()
        self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように
        #print('state_memory :', self.state_memory)
        #print('action_memory :', self.action_memory)
        #print('reward_memory :', self.reward_memory)
        #print('next_state_memory :', self.next_state_memory)
        #print('memory.state_memory :', self.terminal_memory)

        #print('type of state_memory :', type(self.state_memory[0][0]))
        #print('type of action_memory :', type(self.action_memory[0][0]))
        #print('type of reward_memory :', type(self.reward_memory[0]))
        #print('type of next_state_memory :', type(self.next_state_memory[0][0]))
        #print('type of memory.state_memory :', type(self.terminal_memory[0]))

        self.memory_count += 1
        #print('memory_count :', agent.memory.memory_count)

    # 16 バッファメモリーからランダムに抽出する
    def sample_buffer(self, batch_size):
        # indexが最大メモリに到達していない場合を想定する。
        max_index = min(self.max_memory_size, self.memory_count)
        choosed_index = np.random.choice(max_index, batch_size)
        
        observations = self.state_memory[choosed_index]
        actions = self.action_memory[choosed_index]
        rewards = self.reward_memory[choosed_index]
        next_states = self.next_state_memory[choosed_index]
        terminals = self.terminal_memory[choosed_index]

        return observations, actions, rewards, next_states, terminals


# 6.ActorNNクラスを新規作成する
class ActorNN(nn.Module):
    def __init__(self, alpha=0.001, n_obs_space=17, n_action_space=6,
                                    layer1_size=256, layer2_size=256, batch_size=64):
        #print('ActorNN.__init__ is working.')
        super(ActorNN, self).__init__()
        self.fc1 = nn.Linear(n_obs_space, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, n_action_space)

        #26.最適化処理としてアダムを設定する
        self.optimizer = optim.Adam(self.parameters(), lr=alpha)

        # 48. actorパラメータの読み出し
        # もし、パラメータのデータが存在していたらそのパラメータで初期化する。
        # パラメータファイルの存在チェック
        if T.cuda.is_available():
            map_location = 'cuda'
        else:
            map_location = 'cpu'

        if os.path.isfile('actor_params.pt'):
            # パラメータファイルが存在する場合はロード
            self.load_state_dict(T.load('actor_params.pt', map_location=map_location))
            print("パラメータファイルをロードしました:", 'actor_params.pt')
        else:
            print("パラメータファイルが見つかりません:", 'actor_params.pt')

    def forward(self, obs):
        #print('AgetDDPG.ActorNN.forward is working')
        #print('====ここまではOK1====')
        x = self.fc1(obs)
        x = F.relu(x)
        x = self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        action = F.tanh(x)

        return action

# 22.CriticNNクラスを新規作成する
class CriticNN(nn.Module):

    def __init__(self, beta=0.001, n_obs_space=17, n_action_space=6,
                 layer1_size=256, layer2_size=256, batch_size=64):
        #print('CriticNN.__init__ is working.')
        super(CriticNN, self).__init__()

        # クリティックNNは観察空間+行動空間の２つを入力とする構造
        input_dim = n_obs_space + n_action_space
        self.fc1 = nn.Linear(input_dim, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

        #27.最適化処理としてアダムを設定する
        self.optimizer = optim.Adam(self.parameters(), lr=beta)

        # 49. criticパラメータの読み出し
        # もし、パラメータのデータが存在していたらそのパラメータで初期化する。
        # パラメータファイルの存在チェック
        if T.cuda.is_available():
            map_location = 'cuda'
        else:
            map_location = 'cpu'

        if os.path.isfile('critic_params.pt'):
            # パラメータファイルが存在する場合はロード
            self.load_state_dict(T.load('critic_params.pt', map_location=map_location))
            print("パラメータファイルをロードしました:", 'critic_params.pt')
        else:
            print("パラメータファイルが見つかりません:", 'critic_params.pt')

    def forward(self, obs, action):
        input_data = T.cat([obs, action], dim=1)
        x = self.fc1(input_data)
        x = F.relu(x)
        x =self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        return x #一つの状態価値を出力する。

# 3.エージェントクラスを定義する
class AgentDDPG:

    def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64):
        #print('AgentDDPG.__init__ is working.')
        # 5.ActorNNクラスのインスタンスを生成する
        self.alpha = alpha
        self.beta = beta
        self.gamma = gamma
        self.tau = tau
        
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.n_state_action_value = n_state_action_value

        self.layer1_size = layer1_size
        self.layer2_size = layer2_size

        # 13.バッチサイズを決めておく
        self.batch_size = batch_size 

        self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                            layer1_size=64, layer2_size=64, batch_size=64)        
        
        # 9.memoryインスタンスを追加
        self.MAX_MEMORY_SIZE = 10000
        self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,
                                   n_obs_space=self.n_obs_space,
                                   n_action_space=self.n_action_space)
        
        # 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する
        # actorとtarget_actorのネットワークは同じActorNNで良い
        self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)

        
        # 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する
        self.target_critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)
        
        # 24.クリティックネットワークインスタンスcriticを作成する。
        self.critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,
                               layer1_size=64, layer2_size=64, batch_size=64)
        
        # アクターロスとクリティックロス
        self.actor_loss = 0
        self.critic_loss = 0
        
        # 45.行動ノイズのインスタンス化
        self.noise = OUActionNoise(mu=np.zeros(n_action_space))

    def choose_action(self, obs):
        #print('AgentDDPG.choose_action is working.')
        # 4.方策（アクター）はニューラルネットワークで表現する。
        #   ActorNNクラスを新規作成し、インスタンスactorとして使用する。
        action = self.actor.forward(obs)
        

        # 46.行動ノイズを入れて探索性を向上させる。
        action += T.tensor(self.noise(), dtype=T.float32)
        action = action.detach().numpy()

        return action
    
    # 8.remenberメソドを追加
    def remember(self, obs, action, reward, next_state, done):
        self.memory.store_transition(obs, action, reward, next_state, done)

    # 13.learnメソドを追加
    def learn(self):
        # 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。
        if self.memory.memory_count < self.batch_size:
            return
        
        # 15.メモリバッファからデータを抜き出す sample_buffer()
        # バッチ化されているので変数名を複数形にする
        observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)
        #print('s:', observations)
        #print(observations.shape)
        #print('a :', actions)
        #print('r :', rewards)
        #print('s_ :', next_states)
        #print('terminal :', terminals)

        # 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する
        observations = T.tensor(observations, dtype=T.float32)
        actions = T.tensor(actions, dtype=T.float32)
        rewards = T.tensor(rewards, dtype=T.float32)
        next_states = T.tensor(next_states, dtype=T.float32)
        terminals = T.tensor(terminals, dtype=T.float32)
       
        # 18.ターゲットアクターネットワークインスタンスtarget_actorに
        # 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。
        # このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。
        #print('next_states :', next_states)
        target_actions = self.target_actor.forward(next_states)

        # 20.ターゲットクリティックネットワークインスタンスtarget_criticに
        # 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して
        # 価値関数の推定値ターゲットバリューを出力する。
        # TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。
        # ターゲットクリティックバリューはターゲットアクターネットワークを使う
        target_critic_values = self.target_critic.forward(next_states, target_actions)

        
        # 23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に
        # 現在の状態observationsと行動actionsを入力して
        # クリティックバリューを算出する
        critic_values = self.critic.forward(observations, actions)

        # 25.TDターゲットを算出する：r + γ*V(w)[s_t+1]
        td_targets = []
        for i in range(self.batch_size):
            td_target = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]
            td_targets.append(td_target)
        
        # TDターゲットの形をバッチに整える
        td_targets = T.tensor(td_targets, dtype=T.float32)
        td_targets = td_targets.view(self.batch_size, 1) #viewはreshapeと同じ。64x1に見え方を変更した、という意味
        #print('td_targets :', td_targets)

        
        #ここから次回はやっていこう 2023/5/16
        # ==== （１）クリティックの学習 ====

        #self.critic.train()

        # 28.クリティックの勾配をゼロに初期化する
        self.critic.optimizer.zero_grad()

        # 29. TDターゲットと状態価値の二乗誤差を算出して、クリティックの損失関数とする。バッチサイズは６４個
        critic_loss = F.mse_loss(td_targets, critic_values)
        self.critic_loss = critic_loss
        #print('critic_loss : ', critic_loss) # tensor(0.0485, grad_fn=<MseLossBackward0>)

        # 30. クリティックの損失関数を微分して、勾配を算出する
        critic_loss.backward()

        # 31. 勾配からオプティマイザーによってクリティックのパラメータ（重みとバイアス）を更新する
        self.critic.optimizer.step()

        # self.critic.eval()


        # ==== （２）アクターの学習 ====

        # 32. アクターの勾配をゼロに初期化する
        self.actor.optimizer.zero_grad()

        # 33. アクターに観測情報を入力して行動を算出する。バッチサイズは６４個
        predicted_actions = self.actor.forward(observations)

        #self.actor.train()

        # 34.アクターの損失関数を算出する
        #    Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択すること。
        #    なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardし、
        #    actorだけをパラメータ更新することによりactorの学習をすることができる。

        actor_loss = -self.critic.forward(observations, predicted_actions)
        actor_loss = T.mean(actor_loss)
        self.actor_loss = actor_loss

        #print(f'actor_loss: {actor_loss}, critic_loss: {critic_loss}')

        # 35. DDPG構造全体の損失関数actor_lossを微分し、勾配を算出する
        actor_loss.backward()

        # 36. 勾配からオプティマイザーによってアクターのパラメータだけを（重みとバイアス）を更新する
        self.actor.optimizer.step()
    
        # 37. 全ニューラルネットワークのパラメータを更新する。
        self.update_network_parameters()

    # 37. パラメータ更新メソド。
    def update_network_parameters(self, tau=None):
        if tau is None:
            tau = self.tau
    
        # 38. actor, critic, target_actor, target_criticのネットワーク内の全てのパラメータ（重みとバイアス）とその名前を取得する
        # actorとcriticは先ほど更新されたばかりのパラメーター
        actor_params = self.actor.named_parameters()
        critic_params = self.critic.named_parameters()
        target_actor_params = self.target_actor.named_parameters()
        target_critic_params = self.target_critic.named_parameters()
        #print('actor_params : ', actor_params) # actor_params :  <generator object Module.named_parameters at 0x000001661B2D9D48>

        # 39. パラメータをディクショナリとして取り出す。
        actor_params_dict = dict(actor_params)
        critic_params_dict = dict(critic_params)
        target_actor_params_dict = dict(target_actor_params)
        target_critic_params_dict = dict(target_critic_params)
        #print('actor_params_dict : ', actor_params_dict)
        #print(actor_params_dict.keys())
        """
        actor_params_dict :  {'fc1.weight': Parameter containing:
                                   tensor([[-0.1895, -0.0343,  0.1138,  ...,  0.2157,  0.0527, -0.1173],/

        dict_keys(['fc1.weight', 'fc1.bias', 'fc2.weight', 'fc2.bias', 'fc3.weight', 'fc3.bias'])
        """

        # 40. クリティックの各パラメーター毎に 更新重みtau=0.0001の分だけほんの少しcriticパラメータをtarget_criticパラメータに近づける。
        for name in critic_params_dict:
            critic_params_dict[name] = tau * critic_params_dict[name].clone() + \
                                       (1-tau) * target_critic_params_dict[name].clone()
            
        # 41. 更新したcriticパラメータをtarget_criticのパラメータとしてロードする。
        self.target_critic.load_state_dict(critic_params_dict)

        # 42.アクターの各パラメーター毎に 更新重みtau=0.0001の分だけほんの少しactorパラメータをtarget_actorパラメータに近づける。
        for name in actor_params_dict:
            actor_params_dict[name] = tau * actor_params_dict[name].clone() + \
                                      (1 - tau) * target_actor_params_dict[name].clone()
            
        # 43. 更新したactorパラメータをtarget_actorのパラメータとしてロードする。
        self.target_actor.load_state_dict(actor_params_dict)

    """
    # =====================次はここから============================
    def save_models(self):
        self.actor.save_checkpoint()
        self.critic.save_checkpoint()        
        self.target_actor.save_checkpoint()
        self.target_critic.save_checkpoint()

    def load_models(self):
        self.actor.load_checkpoint()
        self.critic.load_checkpoint()        
        self.target_actor.load_checkpoint()
        self.target_critic.load_checkpoint() 
    """
# 2.エージェントクラスのインスタンスを生成する
agent = AgentDDPG(alpha=0.01, beta=0.01, gamma=0.99, tau=0.01,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=256, layer2_size=256, batch_size=64)

#env = gym.make("HalfCheetah-v4", render_mode= 'human')
env = gym.make("HalfCheetah-v4", render_mode='depth_array')

EPISODES = 1001# episodes
STEPS = 400    # steps
DELAY_TIME = 0.00 # sec

total_rewards = []
actor_losses = []
critic_losses = []
for episode in range(EPISODES):
    obs = env.reset()
    obs = T.tensor(obs[0], dtype=T.float)
    # tensor([ 0.0040,  0.0199, -0.0622,  0.0594, -0.0605,  0.0577, -0.0056,  0.0333,        -0.0072,  0.0532, -0.0512,  0.0173, -0.0529, -0.1104,  0.0946, -0.0559,         0.0824])
    #print(type(obs))
    # observation_space :  Box(-inf, inf, (17,), float64)
    #print('observation_space : ', env.observation_space)
    #print('obs :', obs)

    reward: float = 0
    total_reward: float = 0
    done: bool = False
    for j in range(STEPS):
        env.render()
        
        # ここをDDPGに置き換えていく
        action = agent.choose_action(obs) # 1.Agentクラスを定義していく
        #action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]
        #print('====ここまではOK4====')
        #print('action_space : ', env.action_space)
        #print('action : ', action)

        next_state, reward, done, _, info = env.step(action)
        #print('next_state, reward, done, _, info :', next_state, reward, done, _, info)
        """
        action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]

        next_state, reward, done, _, info : 
        [-0.00265179  0.0229547   0.00463243 -0.04729936 -0.00959038  0.04734605
        0.03672746  0.02857842  0.09980254 -0.32065693  0.04221647  1.58668951
        -2.31089174  1.30338924 -0.25465526  1.08250465 -0.14134398]
        0.07553858359316026
        False
        False
        {'x_position': -0.09233384215910741, 'x_velocity': 0.07920445513883267, 'reward_run': 0.07920445513883267, 'reward_ctrl': -0.0036658715456724168}
        """
        #print('====ここまではOK5====')

        #7. トラジェクトを保存する。経験再生(ReplayBuffer)
        agent.remember(obs, action, reward, next_state, int(done))

        #12. ニューラルネットワークを学習する
        agent.learn()
        
        # 26.エピソード内での報酬を累積していく
        total_reward += reward
        
        # 27. next_stateをobsとして再出発する
        #print('next_state:', next_state)
        obs = next_state
        obs = T.tensor(obs, dtype=T.float)
        # 28. チーターの動きを見たいのでスリープを入れる
        time.sleep(DELAY_TIME)

    #print('total_reward : ', total_reward)
    total_rewards.append(total_reward)

    actor_losses.append(float(agent.actor_loss)) 
    critic_losses.append(float(agent.critic_loss)) 

    # print('epsisode', i, 'score %.2f' % score, '100 game sverage %.2f' % np.mean(score_history[-100:]))
 
    # 47. 各ニューラルネットワークのパラメータを１０エピソード毎に保存する
    print('episode, total_reward : ', episode , total_reward)
    if episode % 10 == 0:
        T.save(agent.actor.state_dict(), 'actor_params.pt')
        T.save(agent.critic.state_dict(), 'critic_params.pt')
        T.save(agent.target_actor.state_dict(), 'target_actor_params.pt')
        T.save(agent.target_critic.state_dict(), 'target_critic_params.pt')
        print('==== params were saved. ====')

#print('total_rewards : ', total_rewards)
#plt.plot(total_rewards)

#plt.plot(actor_losses, label='actor_losses')
#plt.plot(critic_losses, label='critic_losses')
plt.plot(total_rewards, label='total_rewards')
plt.legend()
plt.grid(True)
plt.ioff()
plt.show()

env.close() # 空なんですけど・・・

print('script is done.')

# https://gymnasium.farama.org/

100

101

102

103

104

105

106

107

108

109

110

111

112

113

114

115

116

117

118

119

120

121

122

123

124

125

126

127

128

129

130

131

132

133

134

135

136

137

138

139

140

141

142

143

144

145

146

147

148

149

150

151

152

153

154

155

156

157

158

159

160

161

162

163

164

165

166

167

168

169

170

171

172

173

174

175

176

177

178

179

180

181

182

183

184

185

186

187

188

189

190

191

192

193

194

195

196

197

198

199

200

201

202

203

204

205

206

207

208

209

210

211

212

213

214

215

216

217

218

219

220

221

222

223

224

225

226

227

228

229

230

231

232

233

234

235

236

237

238

239

240

241

242

243

244

245

246

247

248

249

250

251

252

253

254

255

256

257

258

259

260

261

262

263

264

265

266

267

268

269

270

271

272

273

274

275

276

277

278

279

280

281

282

283

284

285

286

287

288

289

290

291

292

293

294

295

296

297

298

299

300

301

302

303

304

305

306

307

308

309

310

311

312

313

314

315

316

317

318

319

320

321

322

323

324

325

326

327

328

329

330

331

332

333

334

335

336

337

338

339

340

341

342

343

344

345

346

347

348

349

350

351

352

353

354

355

356

357

358

359

360

361

362

363

364

365

366

367

368

369

370

371

372

373

374

375

376

377

378

379

380

381

382

383

384

385

386

387

388

389

390

391

392

393

394

395

396

397

398

399

400

401

402

403

404

405

406

407

408

409

410

411

412

413

414

415

416

417

418

419

420

421

422

423

424

425

426

427

428

429

430

431

432

433

434

435

436

437

438

439

440

441

442

443

444

445

446

447

448

449

450

451

452

453

454

455

456

457

458

459

460

461

462

463

464

465

466

467

468

469

470

471

472

473

474

475

476

477

478

479

480

481

482

483

484

485

486

487

488

489

490

491

492

493

494

495

496

497

498

499

500

501

502

503

504

505

506

507

508

509

510

import gymnasium as gym

import time

import torch as T

import torch.nn as nn

import torch.nn.functional as F

import torch.optim as optim

import numpy as np

import matplotlib.pyplot as plt

import os

""" pip install

pip3 install --pre torch torchvision torchaudio --index-url https://download.pytorch.org/whl/nightly/cu121

pip install gymnasium

pip install matplotlib

pip install mujoco

pip install gymnasium[mujoco]

"""

# 44.OUActionNOoiseクラスを作成する

class OUActionNoise(object):

def __init__(self, mu, sigma=0.15, theta=0.2, dt=1e-2, x0=None):

self.mu = mu

self.sigma = sigma

self.theta = theta

self.dt = dt

self.x0 = x0

self.reset()

def __call__(self):

x = self.x_prev + self.theta * (self.mu - self.x_prev) * self.dt + \

self.sigma * np.sqrt(self.dt) * np.random.normal(size=self.mu.shape)

self.x_prev = x

return x

def reset(self):

self.x_prev = self.x0 if self.x0 is not None else np.zeros_like(self.mu)

# 10. ReplayBufferクラスを新規作成する

class ReplayBuffer:

def __init__(self, max_memory_size, n_obs_space, n_action_space):

self.max_memory_size = max_memory_size

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.memory_count = 0

self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.action_memory = np.zeros((self.max_memory_size, self.n_action_space))

self.reward_memory = np.zeros(self.max_memory_size)

self.next_state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.terminal_memory = np.zeros(self.max_memory_size)

#self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

# 11.トランジション保存のためstore_transitionメソドを作成する

def store_transition(self, obs, action, reward, next_state, done):

#print('store_transition is working.')

index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック

#print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())

self.state_memory[index] = obs.detach().numpy().flatten()

self.action_memory[index] = action.flatten()

self.reward_memory[index] = reward.flatten()

self.next_state_memory[index] = next_state.flatten()

self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように

#print('state_memory :', self.state_memory)

#print('action_memory :', self.action_memory)

#print('reward_memory :', self.reward_memory)

#print('next_state_memory :', self.next_state_memory)

#print('memory.state_memory :', self.terminal_memory)

#print('type of state_memory :', type(self.state_memory[0][0]))

#print('type of action_memory :', type(self.action_memory[0][0]))

#print('type of reward_memory :', type(self.reward_memory[0]))

#print('type of next_state_memory :', type(self.next_state_memory[0][0]))

#print('type of memory.state_memory :', type(self.terminal_memory[0]))

self.memory_count += 1

#print('memory_count :', agent.memory.memory_count)

# 16 バッファメモリーからランダムに抽出する

def sample_buffer(self, batch_size):

# indexが最大メモリに到達していない場合を想定する。

max_index = min(self.max_memory_size, self.memory_count)

choosed_index = np.random.choice(max_index, batch_size)

observations = self.state_memory[choosed_index]

actions = self.action_memory[choosed_index]

rewards = self.reward_memory[choosed_index]

next_states = self.next_state_memory[choosed_index]

terminals = self.terminal_memory[choosed_index]

return observations, actions, rewards, next_states, terminals

# 6.ActorNNクラスを新規作成する

class ActorNN(nn.Module):

def __init__(self, alpha=0.001, n_obs_space=17, n_action_space=6,

layer1_size=256, layer2_size=256, batch_size=64):

#print('ActorNN.__init__ is working.')

super(ActorNN, self).__init__()

self.fc1 = nn.Linear(n_obs_space, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, n_action_space)

#26.最適化処理としてアダムを設定する

self.optimizer = optim.Adam(self.parameters(), lr=alpha)

# 48. actorパラメータの読み出し

# もし、パラメータのデータが存在していたらそのパラメータで初期化する。

# パラメータファイルの存在チェック

if T.cuda.is_available():

map_location = 'cuda'

else:

map_location = 'cpu'

if os.path.isfile('actor_params.pt'):

# パラメータファイルが存在する場合はロード

self.load_state_dict(T.load('actor_params.pt', map_location=map_location))

print("パラメータファイルをロードしました:", 'actor_params.pt')

else:

print("パラメータファイルが見つかりません:", 'actor_params.pt')

def forward(self, obs):

#print('AgetDDPG.ActorNN.forward is working')

#print('====ここまではOK1====')

x = self.fc1(obs)

x = F.relu(x)

x = self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

action = F.tanh(x)

return action

# 22.CriticNNクラスを新規作成する

class CriticNN(nn.Module):

def __init__(self, beta=0.001, n_obs_space=17, n_action_space=6,

layer1_size=256, layer2_size=256, batch_size=64):

#print('CriticNN.__init__ is working.')

super(CriticNN, self).__init__()

# クリティックNNは観察空間+行動空間の２つを入力とする構造

input_dim = n_obs_space + n_action_space

self.fc1 = nn.Linear(input_dim, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

#27.最適化処理としてアダムを設定する

self.optimizer = optim.Adam(self.parameters(), lr=beta)

# 49. criticパラメータの読み出し

# もし、パラメータのデータが存在していたらそのパラメータで初期化する。

# パラメータファイルの存在チェック

if T.cuda.is_available():

map_location = 'cuda'

else:

map_location = 'cpu'

if os.path.isfile('critic_params.pt'):

# パラメータファイルが存在する場合はロード

self.load_state_dict(T.load('critic_params.pt', map_location=map_location))

print("パラメータファイルをロードしました:", 'critic_params.pt')

else:

print("パラメータファイルが見つかりません:", 'critic_params.pt')

def forward(self, obs, action):

input_data = T.cat([obs, action], dim=1)

x = self.fc1(input_data)

x = F.relu(x)

x =self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

return x #一つの状態価値を出力する。

# 3.エージェントクラスを定義する

class AgentDDPG:

def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64):

#print('AgentDDPG.__init__ is working.')

# 5.ActorNNクラスのインスタンスを生成する

self.alpha = alpha

self.beta = beta

self.gamma = gamma

self.tau = tau

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.n_state_action_value = n_state_action_value

self.layer1_size = layer1_size

self.layer2_size = layer2_size

# 13.バッチサイズを決めておく

self.batch_size = batch_size

self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 9.memoryインスタンスを追加

self.MAX_MEMORY_SIZE = 10000

self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,

n_obs_space=self.n_obs_space,

n_action_space=self.n_action_space)

# 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する

# actorとtarget_actorのネットワークは同じActorNNで良い

self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する

self.target_critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 24.クリティックネットワークインスタンスcriticを作成する。

self.critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# アクターロスとクリティックロス

self.actor_loss = 0

self.critic_loss = 0

# 45.行動ノイズのインスタンス化

self.noise = OUActionNoise(mu=np.zeros(n_action_space))

def choose_action(self, obs):

#print('AgentDDPG.choose_action is working.')

# 4.方策（アクター）はニューラルネットワークで表現する。

# ActorNNクラスを新規作成し、インスタンスactorとして使用する。

action = self.actor.forward(obs)

# 46.行動ノイズを入れて探索性を向上させる。

action += T.tensor(self.noise(), dtype=T.float32)

action = action.detach().numpy()

return action

# 8.remenberメソドを追加

def remember(self, obs, action, reward, next_state, done):

self.memory.store_transition(obs, action, reward, next_state, done)

# 13.learnメソドを追加

def learn(self):

# 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。

if self.memory.memory_count < self.batch_size:

return

# 15.メモリバッファからデータを抜き出す sample_buffer()

# バッチ化されているので変数名を複数形にする

observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)

#print('s:', observations)

#print(observations.shape)

#print('a :', actions)

#print('r :', rewards)

#print('s_ :', next_states)

#print('terminal :', terminals)

# 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する

observations = T.tensor(observations, dtype=T.float32)

actions = T.tensor(actions, dtype=T.float32)

rewards = T.tensor(rewards, dtype=T.float32)

next_states = T.tensor(next_states, dtype=T.float32)

terminals = T.tensor(terminals, dtype=T.float32)

# 18.ターゲットアクターネットワークインスタンスtarget_actorに

# 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。

# このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。

#print('next_states :', next_states)

target_actions = self.target_actor.forward(next_states)

# 20.ターゲットクリティックネットワークインスタンスtarget_criticに

# 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して

# 価値関数の推定値ターゲットバリューを出力する。

# TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。

# ターゲットクリティックバリューはターゲットアクターネットワークを使う

target_critic_values = self.target_critic.forward(next_states, target_actions)

# 23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に

# 現在の状態observationsと行動actionsを入力して

# クリティックバリューを算出する

critic_values = self.critic.forward(observations, actions)

# 25.TDターゲットを算出する：r + γ*V(w)[s_t+1]

td_targets = []

for i in range(self.batch_size):

td_target = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]

td_targets.append(td_target)

# TDターゲットの形をバッチに整える

td_targets = T.tensor(td_targets, dtype=T.float32)

td_targets = td_targets.view(self.batch_size, 1) #viewはreshapeと同じ。64x1に見え方を変更した、という意味

#print('td_targets :', td_targets)

#ここから次回はやっていこう 2023/5/16

# ==== （１）クリティックの学習 ====

#self.critic.train()

# 28.クリティックの勾配をゼロに初期化する

self.critic.optimizer.zero_grad()

# 29. TDターゲットと状態価値の二乗誤差を算出して、クリティックの損失関数とする。バッチサイズは６４個

critic_loss = F.mse_loss(td_targets, critic_values)

self.critic_loss = critic_loss

#print('critic_loss : ', critic_loss) # tensor(0.0485, grad_fn=<MseLossBackward0>)

# 30. クリティックの損失関数を微分して、勾配を算出する

critic_loss.backward()

# 31. 勾配からオプティマイザーによってクリティックのパラメータ（重みとバイアス）を更新する

self.critic.optimizer.step()

# self.critic.eval()

# ==== （２）アクターの学習 ====

# 32. アクターの勾配をゼロに初期化する

self.actor.optimizer.zero_grad()

# 33. アクターに観測情報を入力して行動を算出する。バッチサイズは６４個

predicted_actions = self.actor.forward(observations)

#self.actor.train()

# 34.アクターの損失関数を算出する

# Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択すること。

# なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardし、

# actorだけをパラメータ更新することによりactorの学習をすることができる。

actor_loss = -self.critic.forward(observations, predicted_actions)

actor_loss = T.mean(actor_loss)

self.actor_loss = actor_loss

#print(f'actor_loss: {actor_loss}, critic_loss: {critic_loss}')

# 35. DDPG構造全体の損失関数actor_lossを微分し、勾配を算出する

actor_loss.backward()

# 36. 勾配からオプティマイザーによってアクターのパラメータだけを（重みとバイアス）を更新する

self.actor.optimizer.step()

# 37. 全ニューラルネットワークのパラメータを更新する。

self.update_network_parameters()

# 37. パラメータ更新メソド。

def update_network_parameters(self, tau=None):

if tau is None:

tau = self.tau

# 38. actor, critic, target_actor, target_criticのネットワーク内の全てのパラメータ（重みとバイアス）とその名前を取得する

# actorとcriticは先ほど更新されたばかりのパラメーター

actor_params = self.actor.named_parameters()

critic_params = self.critic.named_parameters()

target_actor_params = self.target_actor.named_parameters()

target_critic_params = self.target_critic.named_parameters()

#print('actor_params : ', actor_params) # actor_params : <generator object Module.named_parameters at 0x000001661B2D9D48>

# 39. パラメータをディクショナリとして取り出す。

actor_params_dict = dict(actor_params)

critic_params_dict = dict(critic_params)

target_actor_params_dict = dict(target_actor_params)

target_critic_params_dict = dict(target_critic_params)

#print('actor_params_dict : ', actor_params_dict)

#print(actor_params_dict.keys())

"""

actor_params_dict : {'fc1.weight': Parameter containing:

tensor([[-0.1895, -0.0343, 0.1138, ..., 0.2157, 0.0527, -0.1173],/

dict_keys(['fc1.weight', 'fc1.bias', 'fc2.weight', 'fc2.bias', 'fc3.weight', 'fc3.bias'])

"""

# 40. クリティックの各パラメーター毎に更新重みtau=0.0001の分だけほんの少しcriticパラメータをtarget_criticパラメータに近づける。

for name in critic_params_dict:

critic_params_dict[name] = tau * critic_params_dict[name].clone() + \

(1-tau) * target_critic_params_dict[name].clone()

# 41. 更新したcriticパラメータをtarget_criticのパラメータとしてロードする。

self.target_critic.load_state_dict(critic_params_dict)

# 42.アクターの各パラメーター毎に更新重みtau=0.0001の分だけほんの少しactorパラメータをtarget_actorパラメータに近づける。

for name in actor_params_dict:

actor_params_dict[name] = tau * actor_params_dict[name].clone() + \

(1 - tau) * target_actor_params_dict[name].clone()

# 43. 更新したactorパラメータをtarget_actorのパラメータとしてロードする。

self.target_actor.load_state_dict(actor_params_dict)

"""

# =====================次はここから============================

def save_models(self):

self.actor.save_checkpoint()

self.critic.save_checkpoint()

self.target_actor.save_checkpoint()

self.target_critic.save_checkpoint()

def load_models(self):

self.actor.load_checkpoint()

self.critic.load_checkpoint()

self.target_actor.load_checkpoint()

self.target_critic.load_checkpoint()

"""

# 2.エージェントクラスのインスタンスを生成する

agent = AgentDDPG(alpha=0.01, beta=0.01, gamma=0.99, tau=0.01,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=256, layer2_size=256, batch_size=64)

#env = gym.make("HalfCheetah-v4", render_mode= 'human')

env = gym.make("HalfCheetah-v4", render_mode='depth_array')

EPISODES = 1001# episodes

STEPS = 400 # steps

DELAY_TIME = 0.00 # sec

total_rewards = []

actor_losses = []

critic_losses = []

for episode in range(EPISODES):

obs = env.reset()

obs = T.tensor(obs[0], dtype=T.float)

# tensor([ 0.0040, 0.0199, -0.0622, 0.0594, -0.0605, 0.0577, -0.0056, 0.0333, -0.0072, 0.0532, -0.0512, 0.0173, -0.0529, -0.1104, 0.0946, -0.0559, 0.0824])

#print(type(obs))

# observation_space : Box(-inf, inf, (17,), float64)

#print('observation_space : ', env.observation_space)

#print('obs :', obs)

reward: float = 0

total_reward: float = 0

done: bool = False

for j in range(STEPS):

env.render()

# ここをDDPGに置き換えていく

action = agent.choose_action(obs) # 1.Agentクラスを定義していく

#action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

#print('====ここまではOK4====')

#print('action_space : ', env.action_space)

#print('action : ', action)

next_state, reward, done, _, info = env.step(action)

#print('next_state, reward, done, _, info :', next_state, reward, done, _, info)

"""

action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

next_state, reward, done, _, info :

[-0.00265179 0.0229547 0.00463243 -0.04729936 -0.00959038 0.04734605

0.03672746 0.02857842 0.09980254 -0.32065693 0.04221647 1.58668951

-2.31089174 1.30338924 -0.25465526 1.08250465 -0.14134398]

0.07553858359316026

False

{'x_position': -0.09233384215910741, 'x_velocity': 0.07920445513883267, 'reward_run': 0.07920445513883267, 'reward_ctrl': -0.0036658715456724168}

"""

#print('====ここまではOK5====')

#7. トラジェクトを保存する。経験再生(ReplayBuffer)

agent.remember(obs, action, reward, next_state, int(done))

#12. ニューラルネットワークを学習する

agent.learn()

# 26.エピソード内での報酬を累積していく

total_reward += reward

# 27. next_stateをobsとして再出発する

#print('next_state:', next_state)

obs = next_state

obs = T.tensor(obs, dtype=T.float)

# 28. チーターの動きを見たいのでスリープを入れる

time.sleep(DELAY_TIME)

#print('total_reward : ', total_reward)

total_rewards.append(total_reward)

actor_losses.append(float(agent.actor_loss))

critic_losses.append(float(agent.critic_loss))

# print('epsisode', i, 'score %.2f' % score, '100 game sverage %.2f' % np.mean(score_history[-100:]))

# 47. 各ニューラルネットワークのパラメータを１０エピソード毎に保存する

print('episode, total_reward : ', episode , total_reward)

if episode % 10 == 0:

T.save(agent.actor.state_dict(), 'actor_params.pt')

T.save(agent.critic.state_dict(), 'critic_params.pt')

T.save(agent.target_actor.state_dict(), 'target_actor_params.pt')

T.save(agent.target_critic.state_dict(), 'target_critic_params.pt')

print('==== params were saved. ====')

#print('total_rewards : ', total_rewards)

#plt.plot(total_rewards)

#plt.plot(actor_losses, label='actor_losses')

#plt.plot(critic_losses, label='critic_losses')

plt.plot(total_rewards, label='total_rewards')

plt.legend()

plt.grid(True)

plt.ioff()

plt.show()

env.close() # 空なんですけど・・・

print('script is done.')

# https://gymnasium.farama.org/

DDPG by gymnasium ８日目

chatGPTより提案されたニューラルネットワーク構造を導入してみます。

actorNNは隠れ層ノードを64から256に増やしました。

criticNNも隠れ層ノードを64から256に増やしました。

actorNNの活性化関数はrelu, relu,tanhで出力のまま変わらず。

criticNnの活性化関数はrelu,reluで最終層は活性化関数なしで出力。こちらも変更なしです。

基本構造は悪くなかったようです。

# 6.ActorNNクラスを新規作成する
class ActorNN(nn.Module):
    def __init__(self, alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=256, layer2_size=256, batch_size=64):
        #print('ActorNN.__init__ is working.')
        super(ActorNN, self).__init__()
        self.fc1 = nn.Linear(n_obs_space, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, n_action_space)

        #26.最適化処理としてアダムを設定する
        self.optimizer = optim.Adam(self.parameters(), lr=alpha)

    def forward(self, obs):
        #print('AgetDDPG.ActorNN.forward is working')
        #print('====ここまではOK1====')
        x = self.fc1(obs)
        x = F.relu(x)
        x = self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        mu = F.tanh(x)
        #print('action μ:', mu)
        #print('====ここまではOK2====')

        # 必要であればあとでノイズを入れる：action = mu + noize
        action = mu
        return action
        

# 22.CriticNNクラスを新規作成する
class CriticNN(nn.Module):

    def __init__(self, beta=0.000025, n_obs_space=17, n_action_space=6,
                 layer1_size=256, layer2_size=256, batch_size=64):
        #print('CriticNN.__init__ is working.')
        super(CriticNN, self).__init__()

        # クリティックNNは観察空間+行動空間の２つを入力とする構造
        input_dim = n_obs_space + n_action_space
        self.fc1 = nn.Linear(input_dim, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

        #27.最適化処理としてアダムを設定する
        self.optimizer = optim.Adam(self.parameters(), lr=beta)

    def forward(self, obs, action):
        input_data = T.cat([obs, action], dim=1)
        x = self.fc1(input_data)
        x = F.relu(x)
        x =self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        return x #一つの状態価値を出力する。

# 6.ActorNNクラスを新規作成する

class ActorNN(nn.Module):

def __init__(self, alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=256, layer2_size=256, batch_size=64):

#print('ActorNN.__init__ is working.')

super(ActorNN, self).__init__()

self.fc1 = nn.Linear(n_obs_space, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, n_action_space)

#26.最適化処理としてアダムを設定する

self.optimizer = optim.Adam(self.parameters(), lr=alpha)

def forward(self, obs):

#print('AgetDDPG.ActorNN.forward is working')

#print('====ここまではOK1====')

x = self.fc1(obs)

x = F.relu(x)

x = self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

mu = F.tanh(x)

#print('action μ:', mu)

#print('====ここまではOK2====')

# 必要であればあとでノイズを入れる：action = mu + noize

action = mu

return action

# 22.CriticNNクラスを新規作成する

class CriticNN(nn.Module):

def __init__(self, beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=256, layer2_size=256, batch_size=64):

#print('CriticNN.__init__ is working.')

super(CriticNN, self).__init__()

# クリティックNNは観察空間+行動空間の２つを入力とする構造

input_dim = n_obs_space + n_action_space

self.fc1 = nn.Linear(input_dim, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

#27.最適化処理としてアダムを設定する

self.optimizer = optim.Adam(self.parameters(), lr=beta)

def forward(self, obs, action):

input_data = T.cat([obs, action], dim=1)

x = self.fc1(input_data)

x = F.relu(x)

x =self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

return x #一つの状態価値を出力する。

変わらず actor_lossesが上昇傾向にあります。

次は、ステップ数を１０から５０に増やしてみます。

前のめりを覚えたようで、たまにひっくり返ります。

しかし、actor_losses, critic_lossesは上昇傾向で変わらす。しかし、なんか前に行こうと頑張っているようには見えます。符号が逆になってないだろうか？

ここで行動にノイズを入れて環境の探索性を上げることで学習が良い方向に進むかやってみます。

DDPGにおけるOUActionNoiseクラスは、行動に対してオルナシュウ-ウーレンベック（Ornstein-Uhlenbeck）過程に基づくノイズを生成するために使用されるクラスです。このノイズは、環境の探索性を増加させるためにアクションに追加されます。

このクラスのインスタンス化時に、平均値（mu）、標準偏差（sigma）、タイムステップの幅（dt）、回帰係数（theta）、初期値（x0）を指定します。__call__メソッドは、ノイズを生成して返します。

DDPGの学習時には、Actorネットワークから生成されたアクションにOUActionNoiseクラスを適用してノイズを追加し、環境への探索性を高めます。これにより、探索と収束のトレードオフを実現し、より良いポリシーの探索を促進することができます。

44.OUActionNOoiseクラスを作成する

# 44.OUActionNOoiseクラスを作成する
class OUActionNoise(object):
    def __init__(self, mu, sigma=0.15, theta=0.2, dt=1e-2, x0=None):
        self.mu = mu
        self.sigma = sigma
        self.theta = theta
        self.dt = dt
        self.x0 = x0
        self.reset()

    def __call__(self):
        x = self.x_prev + self.theta * (self.mu - self.x_prev) * self.dt + \
        self.sigma * np.sqrt(self.dt) * np.random.normal(size=self.mu.shape)
        self.x_prev = x
        return x

    def reset(self):
        self.x_prev = self.x0 if self.x0 is not None else np.zeros_like(self.mu)

# 44.OUActionNOoiseクラスを作成する

class OUActionNoise(object):

def __init__(self, mu, sigma=0.15, theta=0.2, dt=1e-2, x0=None):

self.mu = mu

self.sigma = sigma

self.theta = theta

self.dt = dt

self.x0 = x0

self.reset()

def __call__(self):

x = self.x_prev + self.theta * (self.mu - self.x_prev) * self.dt + \

self.sigma * np.sqrt(self.dt) * np.random.normal(size=self.mu.shape)

self.x_prev = x

return x

def reset(self):

self.x_prev = self.x0 if self.x0 is not None else np.zeros_like(self.mu)

AgentDDPGクラスに追加

        # 45.行動ノイズのインスタンス化
        self.noise = OUActionNoise(mu=np.zeros(n_action_space))

1 2	# 45.行動ノイズのインスタンス化 self.noise = OUActionNoise(mu=np.zeros(n_action_space))

choose_actionメソド内でactionにノイズを入れる。

    def choose_action(self, obs):
        #print('AgentDDPG.choose_action is working.')
        # 4.方策（アクター）はニューラルネットワークで表現する。
        #   ActorNNクラスを新規作成し、インスタンスactorとして使用する。
        action = self.actor.forward(obs)

        # 46.行動ノイズを入れて探索性を向上させる。
        action += T.tensor(self.noise(), dtype=T.float32)

        action = action.detach().numpy()

        return action

def choose_action(self, obs):

#print('AgentDDPG.choose_action is working.')

# 4.方策（アクター）はニューラルネットワークで表現する。

# ActorNNクラスを新規作成し、インスタンスactorとして使用する。

action = self.actor.forward(obs)

# 46.行動ノイズを入れて探索性を向上させる。

action += T.tensor(self.noise(), dtype=T.float32)

action = action.detach().numpy()

return action

actorにしろcriticにしろ、常にtargetが動いているのでlossが小さくなるわけではないのかなと思い始めました。

前にぴょんぴょん跳ねるような動作が生まれてきました。ノイズのおかげでしょうか。

EPISODES = 1000 # episodes

STEPS = 100 # steps

ではどうでしょうか。

30000ステップを超えたあたりから、ハーフチーターは開始１秒で前進側にすっ飛んでいく挙動が得られました。

しかし、安定してすっ飛んでいくわけではなく、ちょっともたついてから前傾姿勢で進む場合と混ざり合っています。それでも、開始直後に後退する動作はなくなりましたので確実に成長しています。

ニューラルネットワークのパラメータ更新はうまくいっているようです。

リプレイバッファのサイズはまだ1000だけにしていますがもっと多いほうがいいのでしょうか。多すぎると古い情報がなかなか更新されないので学習が遅くなってしまう気がします。

バッチサイズ64に対してベストなバッファサイズはどのように考えればよいでしょうか。課題です。

次回は、ニューラルネットワークモデルのパラメータ保存と読み出しについて考えていきましょう。

前回

actorが行動して集めたデータから、経験再生を使って「次の状態」からtarget_actorが「次の行動」を出力し、「次の行動」と「次の状態」からtarget_criticが「次の状態価値」出力し、TDターゲットを算出しました。

一方で、criticは経験再生を使って「現在の状態」と「そのとき取った行動」から「現在の状態価値」別名ベースラインを算出しました。

今回

ここからは、本当の意味で学習・訓練、つまりパラメータ更新をやっていきます。

オプティマイザーを定義する

オプティマイザーをActorNNクラスとCriticNNクラスの__init__()に定義しておきます。

【ActorNN】#26

self.optimizer = optim.Adam(self.parameters(), lr=alpha)

【CriticNN】#27

self.optimizer = optim.Adam(self.parameters(), lr=beta)

引数のself.parameters()は、モデル自身が持っている重みやバイアスのパラメータです。それを学習率lrでAdamによって最適化（損失関数の最小化）するインスタンスself.optimizerを定義します。

criticの学習

lean()メソド内でのTDターゲット算出後からやっていきます。

# ==== （１）クリティックの学習 ====

# 28.クリティックの勾配をゼロに初期化する
self.critic.optimizer.zero_grad()

# 29. TDターゲットと状態価値の二乗誤差を算出して、クリティックの損失関数とする。バッチサイズは６４個
critic_loss = F.mse_loss(td_targets, critic_values)

# 30. クリティックの損失関数を微分して、勾配を算出する
critic_loss.backward()

# 31. 勾配からオプティマイザーによってクリティックのパラメータ（重みとバイアス）を更新する
self.critic.optimizer.step()

# ==== （１）クリティックの学習 ====

# 28.クリティックの勾配をゼロに初期化する

self.critic.optimizer.zero_grad()

# 29. TDターゲットと状態価値の二乗誤差を算出して、クリティックの損失関数とする。バッチサイズは６４個

critic_loss = F.mse_loss(td_targets, critic_values)

# 30. クリティックの損失関数を微分して、勾配を算出する

critic_loss.backward()

# 31. 勾配からオプティマイザーによってクリティックのパラメータ（重みとバイアス）を更新する

self.critic.optimizer.step()

クリティックの損失関数はtensor(0.0485, grad_fn=<MseLossBackward0>)の形で出力されます。

actorの学習

続けてactorを学習します。

Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択することです。

なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardすることによってactorにも勾配情報が届きパラメータの更新をすることができます。

# ==== （２）アクターの学習 ====

# 32. アクターの勾配をゼロに初期化する
self.actor.optimizer.zero_grad()

# 33. アクターに観測情報を入力して行動を算出する。バッチサイズは６４個
predicted_actions = self.actor.forward(observations)

#self.actor.train()

# 34.アクターの損失関数を算出する
#    Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択すること。
#    なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardし、
#    actorだけをパラメータ更新することによりactorの学習をすることができる。

actor_loss = -self.critic.forward(observations, predicted_actions)
actor_loss = T.mean(actor_loss)

# 35. DDPG構造全体の損失関数actor_lossを微分し、勾配を算出する
actor_loss.backward()

# 36. 勾配からオプティマイザーによってアクターのパラメータだけを（重みとバイアス）を更新する
self.actor.optimizer.step()

# ==== （２）アクターの学習 ====

# 32. アクターの勾配をゼロに初期化する

self.actor.optimizer.zero_grad()

# 33. アクターに観測情報を入力して行動を算出する。バッチサイズは６４個

predicted_actions = self.actor.forward(observations)

#self.actor.train()

# 34.アクターの損失関数を算出する

# Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択すること。

# なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardし、

# actorだけをパラメータ更新することによりactorの学習をすることができる。

actor_loss = -self.critic.forward(observations, predicted_actions)

actor_loss = T.mean(actor_loss)

# 35. DDPG構造全体の損失関数actor_lossを微分し、勾配を算出する

actor_loss.backward()

# 36. 勾配からオプティマイザーによってアクターのパラメータだけを（重みとバイアス）を更新する

self.actor.optimizer.step()

全ニューラルネットワークのパラメータを更新する

learn()メソドの締めくくりとして、３６の直後にself.update_network_parameters()を入れ、メソドとして定義します。

 # 37. パラメータ更新メソド。
    def update_network_parameters(self, tau=None):
        if tau is None:
            tau = self.tau
    
        # 38. actor, critic, target_actor, target_criticのネットワーク内の全てのパラメータ（重みとバイアス）とその名前を取得する
        # actorとcriticは先ほど更新されたばかりのパラメーター
        actor_params = self.actor.named_parameters()
        critic_params = self.critic.named_parameters()
        target_actor_params = self.target_actor.named_parameters()
        target_critic_params = self.target_critic.named_parameters()
        print('actor_params : ', actor_params) # actor_params :  <generator object Module.named_parameters at 0x000001661B2D9D48>

        # 39. パラメータをディクショナリとして取り出す。
        actor_params_dict = dict(actor_params)
        critic_params_dict = dict(critic_params)
        target_actor_params_dict = dict(target_actor_params)
        target_critic_params_dict = dict(target_critic_params)
        print('actor_params_dict : ', actor_params_dict)
        print(actor_params_dict.keys())
        """
        actor_params_dict :  {'fc1.weight': Parameter containing:
                                   tensor([[-0.1895, -0.0343,  0.1138,  ...,  0.2157,  0.0527, -0.1173],/

        dict_keys(['fc1.weight', 'fc1.bias', 'fc2.weight', 'fc2.bias', 'fc3.weight', 'fc3.bias'])
        """

        # 40. クリティックの各パラメーター毎に 更新重みtau=0.0001の分だけほんの少しcriticパラメータをtarget_criticパラメータに近づける。
        for name in critic_params_dict:
            critic_params_dict[name] = tau * critic_params_dict[name].clone() + \
                                       (1-tau) * target_critic_params_dict[name].clone()
            
        # 41. 更新したcriticパラメータをtarget_criticのパラメータとしてロードする。
        self.target_critic.load_state_dict(critic_params_dict)

        # 42.アクターの各パラメーター毎に 更新重みtau=0.0001の分だけほんの少しactorパラメータをtarget_actorパラメータに近づける。
        for name in actor_params_dict:
            actor_params_dict[name] = tau * actor_params_dict[name].clone() + \
                                      (1 - tau) * target_actor_params_dict[name].clone()
            
        # 43. 更新したactorパラメータをtarget_actorのパラメータとしてロードする。
        self.target_actor.load_state_dict(actor_params_dict)

# 37. パラメータ更新メソド。

def update_network_parameters(self, tau=None):

if tau is None:

tau = self.tau

# 38. actor, critic, target_actor, target_criticのネットワーク内の全てのパラメータ（重みとバイアス）とその名前を取得する

# actorとcriticは先ほど更新されたばかりのパラメーター

actor_params = self.actor.named_parameters()

critic_params = self.critic.named_parameters()

target_actor_params = self.target_actor.named_parameters()

target_critic_params = self.target_critic.named_parameters()

print('actor_params : ', actor_params) # actor_params : <generator object Module.named_parameters at 0x000001661B2D9D48>

# 39. パラメータをディクショナリとして取り出す。

actor_params_dict = dict(actor_params)

critic_params_dict = dict(critic_params)

target_actor_params_dict = dict(target_actor_params)

target_critic_params_dict = dict(target_critic_params)

print('actor_params_dict : ', actor_params_dict)

print(actor_params_dict.keys())

"""

actor_params_dict : {'fc1.weight': Parameter containing:

tensor([[-0.1895, -0.0343, 0.1138, ..., 0.2157, 0.0527, -0.1173],/

dict_keys(['fc1.weight', 'fc1.bias', 'fc2.weight', 'fc2.bias', 'fc3.weight', 'fc3.bias'])

"""

# 40. クリティックの各パラメーター毎に更新重みtau=0.0001の分だけほんの少しcriticパラメータをtarget_criticパラメータに近づける。

for name in critic_params_dict:

critic_params_dict[name] = tau * critic_params_dict[name].clone() + \

(1-tau) * target_critic_params_dict[name].clone()

# 41. 更新したcriticパラメータをtarget_criticのパラメータとしてロードする。

self.target_critic.load_state_dict(critic_params_dict)

# 42.アクターの各パラメーター毎に更新重みtau=0.0001の分だけほんの少しactorパラメータをtarget_actorパラメータに近づける。

for name in actor_params_dict:

actor_params_dict[name] = tau * actor_params_dict[name].clone() + \

(1 - tau) * target_actor_params_dict[name].clone()

# 43. 更新したactorパラメータをtarget_actorのパラメータとしてロードする。

self.target_actor.load_state_dict(actor_params_dict)

学習結果の確認

クリティックとアクタークリティックのパラメータ更新まで行ってアクターとターゲットアクターのパラメーター更新をしない状態で試してみました。

動作確認のつもりでやりましたが、学習は進んでいるようです。

10step x 100epsode　で学習したところ、ハーフチーターはエピソード開始直後に前へ倒れこむような挙動を獲得し、リワードを稼ぐようになりました。10stepでは走り続ける動作を獲得するのは無理なようです。

続いて、アクターとターゲットアクターのパラメーター更新も追加して同じことを行いました。

こちらは、足を折りたたんで低い姿勢なることでリワードを稼ぎに行っているようです。しかし80エピソードから成績が悪化していっています。うまく学習が進んでいないようです。

まあ、ニューラルネットワーク構造もまだ適当に作っているので、改善の余地があります。

また下記のようにactor_lossesとcritic_lossesの変化も可視化してみると下記のように悪化していく方向にあります。

課題

計算の高速化（print文の無効化）
適切なニューラルネットワーク構造
適切なステップ数
適切なエピソード数
パラメータのセーブとロード
途中で止まった時に続行可能にしたい
計算の高速化（GPUの利用）
学習の進行状況のリアルタイム可視化

現在までのスクリプト

import gymnasium as gym
import time
import torch as T
import torch.nn as nn
import torch.nn.functional as F
import torch.optim as optim
import numpy as np
import matplotlib.pyplot as plt

# 10. ReplayBufferクラスを新規作成する
class ReplayBuffer:
    def __init__(self, max_memory_size, n_obs_space, n_action_space):
        self.max_memory_size = max_memory_size
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.memory_count = 0

        self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))
        self.action_memory =  np.zeros((self.max_memory_size, self.n_action_space))
        self.reward_memory =  np.zeros(self.max_memory_size)
        self.next_state_memory =  np.zeros((self.max_memory_size, self.n_obs_space))
        self.terminal_memory =  np.zeros(self.max_memory_size)
        #self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

    # 11.トランジション保存のためstore_transitionメソドを作成する
    def store_transition(self, obs, action, reward, next_state, done):
        #print('store_transition is working.')
        index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック
        #print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())
        self.state_memory[index] = obs.detach().numpy().flatten()
        self.action_memory[index] = action.flatten()
        self.reward_memory[index] = reward.flatten()
        self.next_state_memory[index] = next_state.flatten()
        self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように
        #print('state_memory :', self.state_memory)
        #print('action_memory :', self.action_memory)
        #print('reward_memory :', self.reward_memory)
        #print('next_state_memory :', self.next_state_memory)
        #print('memory.state_memory :', self.terminal_memory)

        #print('type of state_memory :', type(self.state_memory[0][0]))
        #print('type of action_memory :', type(self.action_memory[0][0]))
        #print('type of reward_memory :', type(self.reward_memory[0]))
        #print('type of next_state_memory :', type(self.next_state_memory[0][0]))
        #print('type of memory.state_memory :', type(self.terminal_memory[0]))

        self.memory_count += 1
        print('memory_count :', agent.memory.memory_count)

    # 16 バッファメモリーからランダムに抽出する
    def sample_buffer(self, batch_size):
        # indexが最大メモリに到達していない場合を想定する。
        max_index = min(self.max_memory_size, self.memory_count)
        choosed_index = np.random.choice(max_index, batch_size)
        
        observations = self.state_memory[choosed_index]
        actions = self.action_memory[choosed_index]
        rewards = self.reward_memory[choosed_index]
        next_states = self.next_state_memory[choosed_index]
        terminals = self.terminal_memory[choosed_index]

        return observations, actions, rewards, next_states, terminals


# 6.ActorNNクラスを新規作成する
class ActorNN(nn.Module):
    def __init__(self, alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64):
        #print('ActorNN.__init__ is working.')
        super(ActorNN, self).__init__()
        self.fc1 = nn.Linear(n_obs_space, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, n_action_space)

        #26.最適化処理としてアダムを設定する
        self.optimizer = optim.Adam(self.parameters(), lr=alpha)

    def forward(self, obs):
        #print('AgetDDPG.ActorNN.forward is working')
        #print('====ここまではOK1====')
        x = self.fc1(obs)
        x = F.relu(x)
        x = self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        mu = F.tanh(x)
        #print('action μ:', mu)
        #print('====ここまではOK2====')

        # 必要であればあとでノイズを入れる：action = mu + noize
        action = mu
        return action
        

# 22.CriticNNクラスを新規作成する
class CriticNN(nn.Module):

    def __init__(self, beta=0.000025, n_obs_space=17, n_action_space=6,
                 layer1_size=64, layer2_size=64, batch_size=64):
        #print('CriticNN.__init__ is working.')
        super(CriticNN, self).__init__()

        # クリティックNNは観察空間+行動空間の２つを入力とする構造
        input_dim = n_obs_space + n_action_space
        self.fc1 = nn.Linear(input_dim, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

        #27.最適化処理としてアダムを設定する
        self.optimizer = optim.Adam(self.parameters(), lr=beta)

    def forward(self, obs, action):
        input_data = T.cat([obs, action], dim=1)
        x = self.fc1(input_data)
        x = F.relu(x)
        x =self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        return x #一つの状態価値を出力する。

# 3.エージェントクラスを定義する
class AgentDDPG:

    def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64):
        #print('AgentDDPG.__init__ is working.')
        # 5.ActorNNクラスのインスタンスを生成する
        self.alpha = alpha
        self.beta = beta
        self.gamma = gamma
        self.tau = tau
        
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.n_state_action_value = n_state_action_value

        self.layer1_size = layer1_size
        self.layer2_size = layer2_size

        # 13.バッチサイズを決めておく
        self.batch_size = batch_size 

        self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                            layer1_size=64, layer2_size=64, batch_size=64)        
        
        # 9.memoryインスタンスを追加
        self.MAX_MEMORY_SIZE = 1000
        self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,
                                   n_obs_space=self.n_obs_space,
                                   n_action_space=self.n_action_space)
        
        # 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する
        # actorとtarget_actorのネットワークは同じActorNNで良い
        self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)

        
        # 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する
        self.target_critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)
        
        # 24.クリティックネットワークインスタンスcriticを作成する。
        self.critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,
                               layer1_size=64, layer2_size=64, batch_size=64)
        
        # アクターロスとクリティックロス
        self.actor_loss = 0
        self.critic_loss = 0
        

    def choose_action(self, obs):
        #print('AgentDDPG.choose_action is working.')
        # 4.方策（アクター）はニューラルネットワークで表現する。
        #   ActorNNクラスを新規作成し、インスタンスactorとして使用する。
        action = self.actor.forward(obs)
        action = action.detach().numpy()
        #print('====ここまではOK3====')
        return action
    
    # 8.remenberメソドを追加
    def remember(self, obs, action, reward, next_state, done):
        self.memory.store_transition(obs, action, reward, next_state, done)

    # 13.learnメソドを追加
    def learn(self):
        # 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。
        if self.memory.memory_count < self.batch_size:
            return
        
        # 15.メモリバッファからデータを抜き出す sample_buffer()
        # バッチ化されているので変数名を複数形にする
        observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)
        #print('s:', observations)
        #print(observations.shape)
        #print('a :', actions)
        #print('r :', rewards)
        #print('s_ :', next_states)
        #print('terminal :', terminals)

        # 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する
        observations = T.tensor(observations, dtype=T.float32)
        actions = T.tensor(actions, dtype=T.float32)
        rewards = T.tensor(rewards, dtype=T.float32)
        next_states = T.tensor(next_states, dtype=T.float32)
        terminals = T.tensor(terminals, dtype=T.float32)
       
        # 18.ターゲットアクターネットワークインスタンスtarget_actorに
        # 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。
        # このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。
        #print('next_states :', next_states)
        target_actions = self.target_actor.forward(next_states)

        # 20.ターゲットクリティックネットワークインスタンスtarget_criticに
        # 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して
        # 価値関数の推定値ターゲットバリューを出力する。
        # TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。
        # ターゲットクリティックバリューはターゲットアクターネットワークを使う
        target_critic_values = self.target_critic.forward(next_states, target_actions)

        
        # 23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に
        # 現在の状態observationsと行動actionsを入力して
        # クリティックバリューを算出する
        critic_values = self.critic.forward(observations, actions)

        # 25.TDターゲットを算出する：r + γ*V(w)[s_t+1]
        td_targets = []
        for i in range(self.batch_size):
            td_target = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]
            td_targets.append(td_target)
        
        # TDターゲットの形をバッチに整える
        td_targets = T.tensor(td_targets, dtype=T.float32)
        td_targets = td_targets.view(self.batch_size, 1) #viewはreshapeと同じ。64x1に見え方を変更した、という意味
        #print('td_targets :', td_targets)

        
        #ここから次回はやっていこう 2023/5/16
        # ==== （１）クリティックの学習 ====

        #self.critic.train()

        # 28.クリティックの勾配をゼロに初期化する
        self.critic.optimizer.zero_grad()

        # 29. TDターゲットと状態価値の二乗誤差を算出して、クリティックの損失関数とする。バッチサイズは６４個
        critic_loss = F.mse_loss(td_targets, critic_values)
        self.critic_loss = critic_loss
        #print('critic_loss : ', critic_loss) # tensor(0.0485, grad_fn=<MseLossBackward0>)

        # 30. クリティックの損失関数を微分して、勾配を算出する
        critic_loss.backward()

        # 31. 勾配からオプティマイザーによってクリティックのパラメータ（重みとバイアス）を更新する
        self.critic.optimizer.step()

        # self.critic.eval()


        # ==== （２）アクターの学習 ====

        # 32. アクターの勾配をゼロに初期化する
        self.actor.optimizer.zero_grad()

        # 33. アクターに観測情報を入力して行動を算出する。バッチサイズは６４個
        predicted_actions = self.actor.forward(observations)

        #self.actor.train()

        # 34.アクターの損失関数を算出する
        #    Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択すること。
        #    なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardし、
        #    actorだけをパラメータ更新することによりactorの学習をすることができる。

        actor_loss = -self.critic.forward(observations, predicted_actions)
        actor_loss = T.mean(actor_loss)
        self.actor_loss = actor_loss

        print(f'actor_loss: {actor_loss}, critic_loss: {critic_loss}')

        # 35. DDPG構造全体の損失関数actor_lossを微分し、勾配を算出する
        actor_loss.backward()

        # 36. 勾配からオプティマイザーによってアクターのパラメータだけを（重みとバイアス）を更新する
        self.actor.optimizer.step()
    
        # 37. 全ニューラルネットワークのパラメータを更新する。
        self.update_network_parameters()

    # 37. パラメータ更新メソド。
    def update_network_parameters(self, tau=None):
        if tau is None:
            tau = self.tau
    
        # 38. actor, critic, target_actor, target_criticのネットワーク内の全てのパラメータ（重みとバイアス）とその名前を取得する
        # actorとcriticは先ほど更新されたばかりのパラメーター
        actor_params = self.actor.named_parameters()
        critic_params = self.critic.named_parameters()
        target_actor_params = self.target_actor.named_parameters()
        target_critic_params = self.target_critic.named_parameters()
        #print('actor_params : ', actor_params) # actor_params :  <generator object Module.named_parameters at 0x000001661B2D9D48>

        # 39. パラメータをディクショナリとして取り出す。
        actor_params_dict = dict(actor_params)
        critic_params_dict = dict(critic_params)
        target_actor_params_dict = dict(target_actor_params)
        target_critic_params_dict = dict(target_critic_params)
        #print('actor_params_dict : ', actor_params_dict)
        #print(actor_params_dict.keys())
        """
        actor_params_dict :  {'fc1.weight': Parameter containing:
                                   tensor([[-0.1895, -0.0343,  0.1138,  ...,  0.2157,  0.0527, -0.1173],/

        dict_keys(['fc1.weight', 'fc1.bias', 'fc2.weight', 'fc2.bias', 'fc3.weight', 'fc3.bias'])
        """

        # 40. クリティックの各パラメーター毎に 更新重みtau=0.0001の分だけほんの少しcriticパラメータをtarget_criticパラメータに近づける。
        for name in critic_params_dict:
            critic_params_dict[name] = tau * critic_params_dict[name].clone() + \
                                       (1-tau) * target_critic_params_dict[name].clone()
            
        # 41. 更新したcriticパラメータをtarget_criticのパラメータとしてロードする。
        self.target_critic.load_state_dict(critic_params_dict)

        # 42.アクターの各パラメーター毎に 更新重みtau=0.0001の分だけほんの少しactorパラメータをtarget_actorパラメータに近づける。
        for name in actor_params_dict:
            actor_params_dict[name] = tau * actor_params_dict[name].clone() + \
                                      (1 - tau) * target_actor_params_dict[name].clone()
            
        # 43. 更新したactorパラメータをtarget_actorのパラメータとしてロードする。
        self.target_actor.load_state_dict(actor_params_dict)
    

# 2.エージェントクラスのインスタンスを生成する
agent = AgentDDPG(alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64)

env = gym.make("HalfCheetah-v4", render_mode= 'human')

EPISODES = 100 # episodes
STEPS = 10    # steps
DELAY_TIME = 0.00 # sec


total_rewards = []
actor_losses = []
critic_losses = []
for eposode in range(EPISODES):
    obs = env.reset()
    obs = T.tensor(obs[0], dtype=T.float)
    # tensor([ 0.0040,  0.0199, -0.0622,  0.0594, -0.0605,  0.0577, -0.0056,  0.0333,        -0.0072,  0.0532, -0.0512,  0.0173, -0.0529, -0.1104,  0.0946, -0.0559,         0.0824])
    #print(type(obs))
    # observation_space :  Box(-inf, inf, (17,), float64)
    #print('observation_space : ', env.observation_space)
    #print('obs :', obs)

    reward: float = 0
    total_reward: float = 0
    done: bool = False
    for j in range(STEPS):
        env.render()
        
        # ここをDDPGに置き換えていく
        action = agent.choose_action(obs) # 1.Agentクラスを定義していく
        #action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]
        #print('====ここまではOK4====')
        #print('action_space : ', env.action_space)
        #print('action : ', action)

        next_state, reward, done, _, info = env.step(action)
        #print('next_state, reward, done, _, info :', next_state, reward, done, _, info)
        """
        action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]

        next_state, reward, done, _, info : 
        [-0.00265179  0.0229547   0.00463243 -0.04729936 -0.00959038  0.04734605
        0.03672746  0.02857842  0.09980254 -0.32065693  0.04221647  1.58668951
        -2.31089174  1.30338924 -0.25465526  1.08250465 -0.14134398]
        0.07553858359316026
        False
        False
        {'x_position': -0.09233384215910741, 'x_velocity': 0.07920445513883267, 'reward_run': 0.07920445513883267, 'reward_ctrl': -0.0036658715456724168}
        """
        #print('====ここまではOK5====')

        #7. トラジェクトを保存する。経験再生(ReplayBuffer)
        agent.remember(obs, action, reward, next_state, int(done))

        #12. ニューラルネットワークを学習する
        agent.learn()
        
        # 26.エピソード内での報酬を累積していく
        total_reward += reward
        
        # 27. next_stateをobsとして再出発する
        #print('next_state:', next_state)
        obs = next_state
        obs = T.tensor(obs, dtype=T.float)
        # 28. チーターの動きを見たいのでスリープを入れる
        time.sleep(DELAY_TIME)

    #print('total_reward : ', total_reward)
    total_rewards.append(total_reward)

    actor_losses.append(float(agent.actor_loss)) 
    critic_losses.append(float(agent.critic_loss)) 

#print('total_rewards : ', total_rewards)
# plt.plot(total_rewards)

plt.plot(actor_losses)
plt.plot(critic_losses)
#plt.plot(critic_losses.detach().numpy())
plt.show()

env.close() # 空なんですけど・・・

print('script is done.')

# https://gymnasium.farama.org/

100

101

102

103

104

105

106

107

108

109

110

111

112

113

114

115

116

117

118

119

120

121

122

123

124

125

126

127

128

129

130

131

132

133

134

135

136

137

138

139

140

141

142

143

144

145

146

147

148

149

150

151

152

153

154

155

156

157

158

159

160

161

162

163

164

165

166

167

168

169

170

171

172

173

174

175

176

177

178

179

180

181

182

183

184

185

186

187

188

189

190

191

192

193

194

195

196

197

198

199

200

201

202

203

204

205

206

207

208

209

210

211

212

213

214

215

216

217

218

219

220

221

222

223

224

225

226

227

228

229

230

231

232

233

234

235

236

237

238

239

240

241

242

243

244

245

246

247

248

249

250

251

252

253

254

255

256

257

258

259

260

261

262

263

264

265

266

267

268

269

270

271

272

273

274

275

276

277

278

279

280

281

282

283

284

285

286

287

288

289

290

291

292

293

294

295

296

297

298

299

300

301

302

303

304

305

306

307

308

309

310

311

312

313

314

315

316

317

318

319

320

321

322

323

324

325

326

327

328

329

330

331

332

333

334

335

336

337

338

339

340

341

342

343

344

345

346

347

348

349

350

351

352

353

354

355

356

357

358

359

360

361

362

363

364

365

366

367

368

369

370

371

372

373

374

375

376

377

378

379

380

381

382

383

384

385

386

387

388

389

390

391

392

393

394

395

396

397

398

399

400

401

402

403

404

405

406

407

408

409

410

411

412

413

414

415

416

417

418

419

420

421

422

423

424

import gymnasium as gym

import time

import torch as T

import torch.nn as nn

import torch.nn.functional as F

import torch.optim as optim

import numpy as np

import matplotlib.pyplot as plt

# 10. ReplayBufferクラスを新規作成する

class ReplayBuffer:

def __init__(self, max_memory_size, n_obs_space, n_action_space):

self.max_memory_size = max_memory_size

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.memory_count = 0

self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.action_memory = np.zeros((self.max_memory_size, self.n_action_space))

self.reward_memory = np.zeros(self.max_memory_size)

self.next_state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.terminal_memory = np.zeros(self.max_memory_size)

#self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

# 11.トランジション保存のためstore_transitionメソドを作成する

def store_transition(self, obs, action, reward, next_state, done):

#print('store_transition is working.')

index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック

#print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())

self.state_memory[index] = obs.detach().numpy().flatten()

self.action_memory[index] = action.flatten()

self.reward_memory[index] = reward.flatten()

self.next_state_memory[index] = next_state.flatten()

self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように

#print('state_memory :', self.state_memory)

#print('action_memory :', self.action_memory)

#print('reward_memory :', self.reward_memory)

#print('next_state_memory :', self.next_state_memory)

#print('memory.state_memory :', self.terminal_memory)

#print('type of state_memory :', type(self.state_memory[0][0]))

#print('type of action_memory :', type(self.action_memory[0][0]))

#print('type of reward_memory :', type(self.reward_memory[0]))

#print('type of next_state_memory :', type(self.next_state_memory[0][0]))

#print('type of memory.state_memory :', type(self.terminal_memory[0]))

self.memory_count += 1

print('memory_count :', agent.memory.memory_count)

# 16 バッファメモリーからランダムに抽出する

def sample_buffer(self, batch_size):

# indexが最大メモリに到達していない場合を想定する。

max_index = min(self.max_memory_size, self.memory_count)

choosed_index = np.random.choice(max_index, batch_size)

observations = self.state_memory[choosed_index]

actions = self.action_memory[choosed_index]

rewards = self.reward_memory[choosed_index]

next_states = self.next_state_memory[choosed_index]

terminals = self.terminal_memory[choosed_index]

return observations, actions, rewards, next_states, terminals

# 6.ActorNNクラスを新規作成する

class ActorNN(nn.Module):

def __init__(self, alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64):

#print('ActorNN.__init__ is working.')

super(ActorNN, self).__init__()

self.fc1 = nn.Linear(n_obs_space, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, n_action_space)

#26.最適化処理としてアダムを設定する

self.optimizer = optim.Adam(self.parameters(), lr=alpha)

def forward(self, obs):

#print('AgetDDPG.ActorNN.forward is working')

#print('====ここまではOK1====')

x = self.fc1(obs)

x = F.relu(x)

x = self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

mu = F.tanh(x)

#print('action μ:', mu)

#print('====ここまではOK2====')

# 必要であればあとでノイズを入れる：action = mu + noize

action = mu

return action

# 22.CriticNNクラスを新規作成する

class CriticNN(nn.Module):

def __init__(self, beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64):

#print('CriticNN.__init__ is working.')

super(CriticNN, self).__init__()

# クリティックNNは観察空間+行動空間の２つを入力とする構造

input_dim = n_obs_space + n_action_space

self.fc1 = nn.Linear(input_dim, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

#27.最適化処理としてアダムを設定する

self.optimizer = optim.Adam(self.parameters(), lr=beta)

def forward(self, obs, action):

input_data = T.cat([obs, action], dim=1)

x = self.fc1(input_data)

x = F.relu(x)

x =self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

return x #一つの状態価値を出力する。

# 3.エージェントクラスを定義する

class AgentDDPG:

def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64):

#print('AgentDDPG.__init__ is working.')

# 5.ActorNNクラスのインスタンスを生成する

self.alpha = alpha

self.beta = beta

self.gamma = gamma

self.tau = tau

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.n_state_action_value = n_state_action_value

self.layer1_size = layer1_size

self.layer2_size = layer2_size

# 13.バッチサイズを決めておく

self.batch_size = batch_size

self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 9.memoryインスタンスを追加

self.MAX_MEMORY_SIZE = 1000

self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,

n_obs_space=self.n_obs_space,

n_action_space=self.n_action_space)

# 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する

# actorとtarget_actorのネットワークは同じActorNNで良い

self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する

self.target_critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 24.クリティックネットワークインスタンスcriticを作成する。

self.critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# アクターロスとクリティックロス

self.actor_loss = 0

self.critic_loss = 0

def choose_action(self, obs):

#print('AgentDDPG.choose_action is working.')

# 4.方策（アクター）はニューラルネットワークで表現する。

# ActorNNクラスを新規作成し、インスタンスactorとして使用する。

action = self.actor.forward(obs)

action = action.detach().numpy()

#print('====ここまではOK3====')

return action

# 8.remenberメソドを追加

def remember(self, obs, action, reward, next_state, done):

self.memory.store_transition(obs, action, reward, next_state, done)

# 13.learnメソドを追加

def learn(self):

# 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。

if self.memory.memory_count < self.batch_size:

return

# 15.メモリバッファからデータを抜き出す sample_buffer()

# バッチ化されているので変数名を複数形にする

observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)

#print('s:', observations)

#print(observations.shape)

#print('a :', actions)

#print('r :', rewards)

#print('s_ :', next_states)

#print('terminal :', terminals)

# 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する

observations = T.tensor(observations, dtype=T.float32)

actions = T.tensor(actions, dtype=T.float32)

rewards = T.tensor(rewards, dtype=T.float32)

next_states = T.tensor(next_states, dtype=T.float32)

terminals = T.tensor(terminals, dtype=T.float32)

# 18.ターゲットアクターネットワークインスタンスtarget_actorに

# 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。

# このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。

#print('next_states :', next_states)

target_actions = self.target_actor.forward(next_states)

# 20.ターゲットクリティックネットワークインスタンスtarget_criticに

# 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して

# 価値関数の推定値ターゲットバリューを出力する。

# TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。

# ターゲットクリティックバリューはターゲットアクターネットワークを使う

target_critic_values = self.target_critic.forward(next_states, target_actions)

# 23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に

# 現在の状態observationsと行動actionsを入力して

# クリティックバリューを算出する

critic_values = self.critic.forward(observations, actions)

# 25.TDターゲットを算出する：r + γ*V(w)[s_t+1]

td_targets = []

for i in range(self.batch_size):

td_target = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]

td_targets.append(td_target)

# TDターゲットの形をバッチに整える

td_targets = T.tensor(td_targets, dtype=T.float32)

td_targets = td_targets.view(self.batch_size, 1) #viewはreshapeと同じ。64x1に見え方を変更した、という意味

#print('td_targets :', td_targets)

#ここから次回はやっていこう 2023/5/16

# ==== （１）クリティックの学習 ====

#self.critic.train()

# 28.クリティックの勾配をゼロに初期化する

self.critic.optimizer.zero_grad()

# 29. TDターゲットと状態価値の二乗誤差を算出して、クリティックの損失関数とする。バッチサイズは６４個

critic_loss = F.mse_loss(td_targets, critic_values)

self.critic_loss = critic_loss

#print('critic_loss : ', critic_loss) # tensor(0.0485, grad_fn=<MseLossBackward0>)

# 30. クリティックの損失関数を微分して、勾配を算出する

critic_loss.backward()

# 31. 勾配からオプティマイザーによってクリティックのパラメータ（重みとバイアス）を更新する

self.critic.optimizer.step()

# self.critic.eval()

# ==== （２）アクターの学習 ====

# 32. アクターの勾配をゼロに初期化する

self.actor.optimizer.zero_grad()

# 33. アクターに観測情報を入力して行動を算出する。バッチサイズは６４個

predicted_actions = self.actor.forward(observations)

#self.actor.train()

# 34.アクターの損失関数を算出する

# Actorの目的は、Criticネットワークの出力（行動価値）を最大化するような行動を選択すること。

# なので、actorNN→criticNNのDDPG構造全体の出力結果をactor_lossとして、actorNNとcriticNNの両方をbackwardし、

# actorだけをパラメータ更新することによりactorの学習をすることができる。

actor_loss = -self.critic.forward(observations, predicted_actions)

actor_loss = T.mean(actor_loss)

self.actor_loss = actor_loss

print(f'actor_loss: {actor_loss}, critic_loss: {critic_loss}')

# 35. DDPG構造全体の損失関数actor_lossを微分し、勾配を算出する

actor_loss.backward()

# 36. 勾配からオプティマイザーによってアクターのパラメータだけを（重みとバイアス）を更新する

self.actor.optimizer.step()

# 37. 全ニューラルネットワークのパラメータを更新する。

self.update_network_parameters()

# 37. パラメータ更新メソド。

def update_network_parameters(self, tau=None):

if tau is None:

tau = self.tau

# 38. actor, critic, target_actor, target_criticのネットワーク内の全てのパラメータ（重みとバイアス）とその名前を取得する

# actorとcriticは先ほど更新されたばかりのパラメーター

actor_params = self.actor.named_parameters()

critic_params = self.critic.named_parameters()

target_actor_params = self.target_actor.named_parameters()

target_critic_params = self.target_critic.named_parameters()

#print('actor_params : ', actor_params) # actor_params : <generator object Module.named_parameters at 0x000001661B2D9D48>

# 39. パラメータをディクショナリとして取り出す。

actor_params_dict = dict(actor_params)

critic_params_dict = dict(critic_params)

target_actor_params_dict = dict(target_actor_params)

target_critic_params_dict = dict(target_critic_params)

#print('actor_params_dict : ', actor_params_dict)

#print(actor_params_dict.keys())

"""

actor_params_dict : {'fc1.weight': Parameter containing:

tensor([[-0.1895, -0.0343, 0.1138, ..., 0.2157, 0.0527, -0.1173],/

dict_keys(['fc1.weight', 'fc1.bias', 'fc2.weight', 'fc2.bias', 'fc3.weight', 'fc3.bias'])

"""

# 40. クリティックの各パラメーター毎に更新重みtau=0.0001の分だけほんの少しcriticパラメータをtarget_criticパラメータに近づける。

for name in critic_params_dict:

critic_params_dict[name] = tau * critic_params_dict[name].clone() + \

(1-tau) * target_critic_params_dict[name].clone()

# 41. 更新したcriticパラメータをtarget_criticのパラメータとしてロードする。

self.target_critic.load_state_dict(critic_params_dict)

# 42.アクターの各パラメーター毎に更新重みtau=0.0001の分だけほんの少しactorパラメータをtarget_actorパラメータに近づける。

for name in actor_params_dict:

actor_params_dict[name] = tau * actor_params_dict[name].clone() + \

(1 - tau) * target_actor_params_dict[name].clone()

# 43. 更新したactorパラメータをtarget_actorのパラメータとしてロードする。

self.target_actor.load_state_dict(actor_params_dict)

# 2.エージェントクラスのインスタンスを生成する

agent = AgentDDPG(alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64)

env = gym.make("HalfCheetah-v4", render_mode= 'human')

EPISODES = 100 # episodes

STEPS = 10 # steps

DELAY_TIME = 0.00 # sec

total_rewards = []

actor_losses = []

critic_losses = []

for eposode in range(EPISODES):

obs = env.reset()

obs = T.tensor(obs[0], dtype=T.float)

# tensor([ 0.0040, 0.0199, -0.0622, 0.0594, -0.0605, 0.0577, -0.0056, 0.0333, -0.0072, 0.0532, -0.0512, 0.0173, -0.0529, -0.1104, 0.0946, -0.0559, 0.0824])

#print(type(obs))

# observation_space : Box(-inf, inf, (17,), float64)

#print('observation_space : ', env.observation_space)

#print('obs :', obs)

reward: float = 0

total_reward: float = 0

done: bool = False

for j in range(STEPS):

env.render()

# ここをDDPGに置き換えていく

action = agent.choose_action(obs) # 1.Agentクラスを定義していく

#action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

#print('====ここまではOK4====')

#print('action_space : ', env.action_space)

#print('action : ', action)

next_state, reward, done, _, info = env.step(action)

#print('next_state, reward, done, _, info :', next_state, reward, done, _, info)

"""

action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

next_state, reward, done, _, info :

[-0.00265179 0.0229547 0.00463243 -0.04729936 -0.00959038 0.04734605

0.03672746 0.02857842 0.09980254 -0.32065693 0.04221647 1.58668951

-2.31089174 1.30338924 -0.25465526 1.08250465 -0.14134398]

0.07553858359316026

False

{'x_position': -0.09233384215910741, 'x_velocity': 0.07920445513883267, 'reward_run': 0.07920445513883267, 'reward_ctrl': -0.0036658715456724168}

"""

#print('====ここまではOK5====')

#7. トラジェクトを保存する。経験再生(ReplayBuffer)

agent.remember(obs, action, reward, next_state, int(done))

#12. ニューラルネットワークを学習する

agent.learn()

# 26.エピソード内での報酬を累積していく

total_reward += reward

# 27. next_stateをobsとして再出発する

#print('next_state:', next_state)

obs = next_state

obs = T.tensor(obs, dtype=T.float)

# 28. チーターの動きを見たいのでスリープを入れる

time.sleep(DELAY_TIME)

#print('total_reward : ', total_reward)

total_rewards.append(total_reward)

actor_losses.append(float(agent.actor_loss))

critic_losses.append(float(agent.critic_loss))

#print('total_rewards : ', total_rewards)

# plt.plot(total_rewards)

plt.plot(actor_losses)

plt.plot(critic_losses)

#plt.plot(critic_losses.detach().numpy())

plt.show()

env.close() # 空なんですけど・・・

print('script is done.')

# https://gymnasium.farama.org/

以上。次回はニューラルネットワーク構造を見直しましょう。

前回までの動き：

agentがobsを受けてchoose_actionでactor(NN)をforwardしactionを出力する。
actionを受けてenv.stepし結果としてnext_state,reward,doneを得る。
agent.rememberで結果obs,action,rewerd,next_state,int(done)を保存する。
rememberで64データ集まったらagent.learnで学習が始まる。

学習：

sample_bufferで64データをランダムに取り出す。
dtype=T.float32に変換する。
target_actorへnext_states 64データを入力してtarget_actions 64データを得る。

ここまで作成しました。

今回

このtarget_actionsをnext_statesと共にtarget_criticへ入力するところからやっていきます。

この部分こそ連続値対応できるDDPGの核心部分なので十分に理解する必要があります。

引き続き学習learn()メソド内での処理です。

やっていこう

AgentDDPG.learn()メソド内の

target_actions = self.target_actor.forward(next_states)

の直下に

target_critic_values

　= self.target_critic.forward(next_states, target_actions)

を入れます。

算出したてのtarget_actionsとnext_statesの２つを入力として、target_critic_valuesを出力します。

ちょっと説明をいれると、DDPGはTD法なのでTDターゲットとしてr + γ*V(w)[s_t+1]を考えます。target_critic_valuesはこれのことです。

この価値関数Vの部分をtarget_criticNNで表現します。

# 21.ターゲットクリティックネットワークインスタンスtarget_criticを作成します。AgentDDPG.__init__()内に定義します。

target_actorの引数　学習率alphaをcritic用にbetaへ変更しています

self.target_critic

= CriticNN(beta=0.000025,
           n_obs_space=17, n_action_space=6,
           layer1_size=64, layer2_size=64,
           batch_size=64)

self.target_critic

= CriticNN(beta=0.000025,

n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64,

batch_size=64)

# 22.CriticNNクラスを作成します。

# 22.CriticNNクラスを新規作成する
class CriticNN(nn.Module):

    def __init__(self, beta=0.000025, n_obs_space=17, n_action_space=6,
                 layer1_size=64, layer2_size=64, batch_size=64):
        print('CriticNN.__init__ is working.')
        super(CriticNN, self).__init__()

        # クリティックNNは観察空間+行動空間の２つを入力とする構造
        input_dim = n_obs_space + n_action_space
        self.fc1 = nn.Linear(input_dim, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

    def forward(self, obs, action):
        input_data = T.cat([obs, action], dim=1)
        x = self.fc1(input_data)
        x = F.relu(x)
        x =self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        return x #一つの状態価値を出力する。

# 22.CriticNNクラスを新規作成する

class CriticNN(nn.Module):

def __init__(self, beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64):

print('CriticNN.__init__ is working.')

super(CriticNN, self).__init__()

# クリティックNNは観察空間+行動空間の２つを入力とする構造

input_dim = n_obs_space + n_action_space

self.fc1 = nn.Linear(input_dim, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

def forward(self, obs, action):

input_data = T.cat([obs, action], dim=1)

x = self.fc1(input_data)

x = F.relu(x)

x =self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

return x #一つの状態価値を出力する。

これでtarget_critic_values が返ってくる

#23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に現在の状態observationsと行動actionsを入力してcritic_valueを算出する。

AgentDDPG.learn()メソドに戻ってさっきほどの

target_critic_values

　= self.target_critic.forward(next_states, target_actions)

の直下に

critic_values

= self.critic.forward(observations, actions)

を入れる。criticインスタンスはまだ作成していないので、AgentDDPG.__init__()に追加する

# 24. AgentDDPG.__init__()にクリティックインスタンス生成を追加する

self.critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,
                       layer1_size=64, layer2_size=64, batch_size=64)

1 2	self.critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6, layer1_size=64, layer2_size=64, batch_size=64)

これで４つのNNを導入することができた。

# 25.target_criticからTDターゲット(= r + γ*V(w)[s_t+1])を算出する。

AgentDDPG.learn()メソドに戻って、

td_targets = []
for i in range(self.batch_size):
    td_target 
        = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]
    td_targets.append(td_target)

# TDターゲットの形をバッチに整える
td_targets = T.tensor(td_targets, dtype=T.float32)
td_targets = td_targets.view(self.batch_size, 1)

td_targets = []

for i in range(self.batch_size):

td_target

= rewards[i] + self.gamma * target_critic_values[i] * terminals[i]

td_targets.append(td_target)

# TDターゲットの形をバッチに整える

td_targets = T.tensor(td_targets, dtype=T.float32)

td_targets = td_targets.view(self.batch_size, 1)

まとめとこれまでのスクリプト

actorが行動して集めたデータから、経験再生を使って「次の状態」からtarget_actorが「次の行動」を出力し、「次の行動」と「次の状態」からtarget_criticが「次の状態価値」出力し、TDターゲットを算出しました。注意すべきはここで言う「次の行動」とはあくまでtarget_actorが生み出した「架空の行動」です。

一方で、criticは経験再生を使って「現在の状態」と「そのとき取った行動」から「現在の状態価値」を算出ししました。注意すべきは、こちらの「そのとき取った行動」とは実際にactorが行動して経験再生バッファに保存されたデータです。

また、この「現在の状態価値」をベースラインと呼びます。次回以降。「TDターゲット-ベースライン」の演算が出てくるので注目です。

import gymnasium as gym
import time
import torch as T
import torch.nn as nn
import torch.nn.functional as F
import torch.optim as opitm
import numpy as np

# 10. ReplayBufferクラスを新規作成する
class ReplayBuffer:
    def __init__(self, max_memory_size, n_obs_space, n_action_space):
        self.max_memory_size = max_memory_size
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.memory_count = 0

        self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))
        self.action_memory =  np.zeros((self.max_memory_size, self.n_action_space))
        self.reward_memory =  np.zeros(self.max_memory_size)
        self.next_state_memory =  np.zeros((self.max_memory_size, self.n_obs_space))
        self.terminal_memory =  np.zeros(self.max_memory_size)
        #self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

    # 11.トランジション保存のためstore_transitionメソドを作成する
    def store_transition(self, obs, action, reward, next_state, done):
        print('store_transition is working.')
        index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック
        print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())
        self.state_memory[index] = obs.detach().numpy().flatten()
        self.action_memory[index] = action.flatten()
        self.reward_memory[index] = reward.flatten()
        self.next_state_memory[index] = next_state.flatten()
        self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように
        print('state_memory :', self.state_memory)
        print('action_memory :', self.action_memory)
        print('reward_memory :', self.reward_memory)
        print('next_state_memory :', self.next_state_memory)
        print('memory.state_memory :', self.terminal_memory)

        print('type of state_memory :', type(self.state_memory[0][0]))
        print('type of action_memory :', type(self.action_memory[0][0]))
        print('type of reward_memory :', type(self.reward_memory[0]))
        print('type of next_state_memory :', type(self.next_state_memory[0][0]))
        print('type of memory.state_memory :', type(self.terminal_memory[0]))

        self.memory_count += 1
        print('memory_count :', agent.memory.memory_count)

    # 16 バッファメモリーからランダムに抽出する
    def sample_buffer(self, batch_size):
        # indexが最大メモリに到達していない場合を想定する。
        max_index = min(self.max_memory_size, self.memory_count)
        choosed_index = np.random.choice(max_index, batch_size)
        
        observations = self.state_memory[choosed_index]
        actions = self.action_memory[choosed_index]
        rewards = self.reward_memory[choosed_index]
        next_states = self.next_state_memory[choosed_index]
        terminals = self.terminal_memory[choosed_index]

        return observations, actions, rewards, next_states, terminals


# 6.ActorNNクラスを新規作成する
class ActorNN(nn.Module):
    def __init__(self, alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64):
        print('ActorNN.__init__ is working.')
        super(ActorNN, self).__init__()
        self.fc1 = nn.Linear(n_obs_space, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, n_action_space)

    def forward(self, obs):
        print('AgetDDPG.ActorNN.forward is working')
        print('====ここまではOK1====')
        x = self.fc1(obs)
        x = F.relu(x)
        x = self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        mu = F.tanh(x)
        print('action μ:', mu)
        print('====ここまではOK2====')

        # 必要であればあとでノイズを入れる：action = mu + noize
        action = mu
        return action
        

# 22.CriticNNクラスを新規作成する
class CriticNN(nn.Module):

    def __init__(self, beta=0.000025, n_obs_space=17, n_action_space=6,
                 layer1_size=64, layer2_size=64, batch_size=64):
        print('CriticNN.__init__ is working.')
        super(CriticNN, self).__init__()

        # クリティックNNは観察空間+行動空間の２つを入力とする構造
        input_dim = n_obs_space + n_action_space
        self.fc1 = nn.Linear(input_dim, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

    def forward(self, obs, action):
        input_data = T.cat([obs, action], dim=1)
        x = self.fc1(input_data)
        x = F.relu(x)
        x =self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        return x #一つの状態価値を出力する。

# 3.エージェントクラスを定義する
class AgentDDPG:

    def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64):
        print('AgentDDPG.__init__ is working.')
        # 5.ActorNNクラスのインスタンスを生成する
        self.alpha = alpha
        self.beta = beta
        self.gamma = gamma
        self.tau = tau
        
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.n_state_action_value = n_state_action_value

        self.layer1_size = layer1_size
        self.layer2_size = layer2_size

        # 13.バッチサイズを決めておく
        self.batch_size = batch_size 

        self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                            layer1_size=64, layer2_size=64, batch_size=64)        
        
        # 9.memoryインスタンスを追加
        self.MAX_MEMORY_SIZE = 1000
        self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,
                                   n_obs_space=self.n_obs_space,
                                   n_action_space=self.n_action_space)
        
        # 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する
        # actorとtarget_actorのネットワークは同じActorNNで良い
        self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)

        
        # 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する
        self.target_critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)
        
        # 24.クリティックネットワークインスタンスcriticを作成する。
        self.critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,
                               layer1_size=64, layer2_size=64, batch_size=64)        

    def choose_action(self, obs):
        print('AgentDDPG.choose_action is working.')
        # 4.方策（アクター）はニューラルネットワークで表現する。
        #   ActorNNクラスを新規作成し、インスタンスactorとして使用する。
        action = self.actor.forward(obs)
        action = action.detach().numpy()
        print('====ここまではOK3====')
        return action
    
    # 8.remenberメソドを追加
    def remember(self, obs, action, reward, next_state, done):
        self.memory.store_transition(obs, action, reward, next_state, done)

    # 13.learnメソドを追加
    def learn(self):
        # 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。
        if self.memory.memory_count < self.batch_size:
            return
        
        # 15.メモリバッファからデータを抜き出す sample_buffer()
        # バッチ化されているので変数名を複数形にする
        observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)
        print('s:', observations)
        print(observations.shape)
        print('a :', actions)
        print('r :', rewards)
        print('s_ :', next_states)
        print('terminal :', terminals)

        # 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する
        observations = T.tensor(observations, dtype=T.float32)
        actions = T.tensor(actions, dtype=T.float32)
        rewards = T.tensor(rewards, dtype=T.float32)
        next_states = T.tensor(next_states, dtype=T.float32)
        terminals = T.tensor(terminals, dtype=T.float32)
       
        # 18.ターゲットアクターネットワークインスタンスtarget_actorに
        # 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。
        # このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。
        print('next_states :', next_states)
        target_actions = self.target_actor.forward(next_states)

        # 20.ターゲットクリティックネットワークインスタンスtarget_criticに
        # 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して
        # 価値関数の推定値ターゲットバリューを出力する。
        # TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。
        # ターゲットクリティックバリューはターゲットアクターネットワークを使う
        target_critic_values = self.target_critic.forward(next_states, target_actions)

        
        # 23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に
        # 現在の状態observationsと行動actionsを入力して
        # クリティックバリューを算出する
        critic_values = self.critic.forward(observations, actions)

        # 25.TDターゲットを算出する：r + γ*V(w)[s_t+1]
        td_targets = []
        for i in range(self.batch_size):
            td_target = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]
            td_targets.append(td_target)
        
        # TDターゲットの形をバッチに整える
        td_targets = T.tensor(td_targets, dtype=T.float32)
        td_targets = td_targets.view(self.batch_size, 1)
        print('td_targets :', td_targets)
        print('td_targets :', td_targets)

 
# 2.エージェントクラスのインスタンスを生成する
agent = AgentDDPG(alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64)

env = gym.make("HalfCheetah-v4", render_mode= 'human')

EPISODES = 10 # episodes
STEPS = 10    # steps
DELAY_TIME = 0.00 # sec


total_rewards = []
for eposode in range(EPISODES):
    obs = env.reset()
    obs = T.tensor(obs[0], dtype=T.float)
    # tensor([ 0.0040,  0.0199, -0.0622,  0.0594, -0.0605,  0.0577, -0.0056,  0.0333,        -0.0072,  0.0532, -0.0512,  0.0173, -0.0529, -0.1104,  0.0946, -0.0559,         0.0824])
    print(type(obs))
    # observation_space :  Box(-inf, inf, (17,), float64)
    print('observation_space : ', env.observation_space)
    print('obs :', obs)

    reward: float = 0
    total_reward: float = 0
    done: bool = False
    for j in range(STEPS):
        env.render()
        
        # ここをDDPGに置き換えていく
        action = agent.choose_action(obs) # 1.Agentクラスを定義していく
        #action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]
        print('====ここまではOK4====')
        print('action_space : ', env.action_space)
        print('action : ', action)

        next_state, reward, done, _, info = env.step(action)
        print('next_state, reward, done, _, info :', next_state, reward, done, _, info)

        print('====ここまではOK5====')

        #7. トラジェクトを保存する。経験再生(ReplayBuffer)
        agent.remember(obs, action, reward, next_state, int(done))

        #12. ニューラルネットワークを学習する
        agent.learn()

        # 26.エピソード内での報酬を累積していく
        total_reward += reward
        
        # 27. next_stateをobsとして再出発する
        print('next_state:', next_state)
        obs = next_state
        obs = T.tensor(obs, dtype=T.float)
        # 28. チーターの動きを見たいのでスリープを入れる
        time.sleep(DELAY_TIME)

    print('total_reward : ', total_reward)
    total_rewards.append(total_reward)

print('total_rewards : ', total_rewards)

env.close() # 空なんですけど・・・
print('script is done.')
# https://gymnasium.farama.org/

100

101

102

103

104

105

106

107

108

109

110

111

112

113

114

115

116

117

118

119

120

121

122

123

124

125

126

127

128

129

130

131

132

133

134

135

136

137

138

139

140

141

142

143

144

145

146

147

148

149

150

151

152

153

154

155

156

157

158

159

160

161

162

163

164

165

166

167

168

169

170

171

172

173

174

175

176

177

178

179

180

181

182

183

184

185

186

187

188

189

190

191

192

193

194

195

196

197

198

199

200

201

202

203

204

205

206

207

208

209

210

211

212

213

214

215

216

217

218

219

220

221

222

223

224

225

226

227

228

229

230

231

232

233

234

235

236

237

238

239

240

241

242

243

244

245

246

247

248

249

250

251

252

253

254

255

256

257

258

259

260

261

262

263

264

265

266

267

268

269

270

271

272

273

274

275

276

277

278

279

280

281

282

283

284

285

286

287

288

289

290

291

292

293

import gymnasium as gym

import time

import torch as T

import torch.nn as nn

import torch.nn.functional as F

import torch.optim as opitm

import numpy as np

# 10. ReplayBufferクラスを新規作成する

class ReplayBuffer:

def __init__(self, max_memory_size, n_obs_space, n_action_space):

self.max_memory_size = max_memory_size

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.memory_count = 0

self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.action_memory = np.zeros((self.max_memory_size, self.n_action_space))

self.reward_memory = np.zeros(self.max_memory_size)

self.next_state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.terminal_memory = np.zeros(self.max_memory_size)

#self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

# 11.トランジション保存のためstore_transitionメソドを作成する

def store_transition(self, obs, action, reward, next_state, done):

print('store_transition is working.')

index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック

print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())

self.state_memory[index] = obs.detach().numpy().flatten()

self.action_memory[index] = action.flatten()

self.reward_memory[index] = reward.flatten()

self.next_state_memory[index] = next_state.flatten()

self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように

print('state_memory :', self.state_memory)

print('action_memory :', self.action_memory)

print('reward_memory :', self.reward_memory)

print('next_state_memory :', self.next_state_memory)

print('memory.state_memory :', self.terminal_memory)

print('type of state_memory :', type(self.state_memory[0][0]))

print('type of action_memory :', type(self.action_memory[0][0]))

print('type of reward_memory :', type(self.reward_memory[0]))

print('type of next_state_memory :', type(self.next_state_memory[0][0]))

print('type of memory.state_memory :', type(self.terminal_memory[0]))

self.memory_count += 1

print('memory_count :', agent.memory.memory_count)

# 16 バッファメモリーからランダムに抽出する

def sample_buffer(self, batch_size):

# indexが最大メモリに到達していない場合を想定する。

max_index = min(self.max_memory_size, self.memory_count)

choosed_index = np.random.choice(max_index, batch_size)

observations = self.state_memory[choosed_index]

actions = self.action_memory[choosed_index]

rewards = self.reward_memory[choosed_index]

next_states = self.next_state_memory[choosed_index]

terminals = self.terminal_memory[choosed_index]

return observations, actions, rewards, next_states, terminals

# 6.ActorNNクラスを新規作成する

class ActorNN(nn.Module):

def __init__(self, alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64):

print('ActorNN.__init__ is working.')

super(ActorNN, self).__init__()

self.fc1 = nn.Linear(n_obs_space, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, n_action_space)

def forward(self, obs):

print('AgetDDPG.ActorNN.forward is working')

print('====ここまではOK1====')

x = self.fc1(obs)

x = F.relu(x)

x = self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

mu = F.tanh(x)

print('action μ:', mu)

print('====ここまではOK2====')

# 必要であればあとでノイズを入れる：action = mu + noize

action = mu

return action

# 22.CriticNNクラスを新規作成する

class CriticNN(nn.Module):

def __init__(self, beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64):

print('CriticNN.__init__ is working.')

super(CriticNN, self).__init__()

# クリティックNNは観察空間+行動空間の２つを入力とする構造

input_dim = n_obs_space + n_action_space

self.fc1 = nn.Linear(input_dim, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, 1) # 最後は1個で良い

def forward(self, obs, action):

input_data = T.cat([obs, action], dim=1)

x = self.fc1(input_data)

x = F.relu(x)

x =self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

return x #一つの状態価値を出力する。

# 3.エージェントクラスを定義する

class AgentDDPG:

def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64):

print('AgentDDPG.__init__ is working.')

# 5.ActorNNクラスのインスタンスを生成する

self.alpha = alpha

self.beta = beta

self.gamma = gamma

self.tau = tau

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.n_state_action_value = n_state_action_value

self.layer1_size = layer1_size

self.layer2_size = layer2_size

# 13.バッチサイズを決めておく

self.batch_size = batch_size

self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 9.memoryインスタンスを追加

self.MAX_MEMORY_SIZE = 1000

self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,

n_obs_space=self.n_obs_space,

n_action_space=self.n_action_space)

# 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する

# actorとtarget_actorのネットワークは同じActorNNで良い

self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する

self.target_critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 24.クリティックネットワークインスタンスcriticを作成する。

self.critic = CriticNN(beta=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

def choose_action(self, obs):

print('AgentDDPG.choose_action is working.')

# 4.方策（アクター）はニューラルネットワークで表現する。

# ActorNNクラスを新規作成し、インスタンスactorとして使用する。

action = self.actor.forward(obs)

action = action.detach().numpy()

print('====ここまではOK3====')

return action

# 8.remenberメソドを追加

def remember(self, obs, action, reward, next_state, done):

self.memory.store_transition(obs, action, reward, next_state, done)

# 13.learnメソドを追加

def learn(self):

# 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。

if self.memory.memory_count < self.batch_size:

return

# 15.メモリバッファからデータを抜き出す sample_buffer()

# バッチ化されているので変数名を複数形にする

observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)

print('s:', observations)

print(observations.shape)

print('a :', actions)

print('r :', rewards)

print('s_ :', next_states)

print('terminal :', terminals)

# 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する

observations = T.tensor(observations, dtype=T.float32)

actions = T.tensor(actions, dtype=T.float32)

rewards = T.tensor(rewards, dtype=T.float32)

next_states = T.tensor(next_states, dtype=T.float32)

terminals = T.tensor(terminals, dtype=T.float32)

# 18.ターゲットアクターネットワークインスタンスtarget_actorに

# 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。

# このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。

print('next_states :', next_states)

target_actions = self.target_actor.forward(next_states)

# 20.ターゲットクリティックネットワークインスタンスtarget_criticに

# 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して

# 価値関数の推定値ターゲットバリューを出力する。

# TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。

# ターゲットクリティックバリューはターゲットアクターネットワークを使う

target_critic_values = self.target_critic.forward(next_states, target_actions)

# 23.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に

# 現在の状態observationsと行動actionsを入力して

# クリティックバリューを算出する

critic_values = self.critic.forward(observations, actions)

# 25.TDターゲットを算出する：r + γ*V(w)[s_t+1]

td_targets = []

for i in range(self.batch_size):

td_target = rewards[i] + self.gamma * target_critic_values[i] * terminals[i]

td_targets.append(td_target)

# TDターゲットの形をバッチに整える

td_targets = T.tensor(td_targets, dtype=T.float32)

td_targets = td_targets.view(self.batch_size, 1)

print('td_targets :', td_targets)

# 2.エージェントクラスのインスタンスを生成する

agent = AgentDDPG(alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64)

env = gym.make("HalfCheetah-v4", render_mode= 'human')

EPISODES = 10 # episodes

STEPS = 10 # steps

DELAY_TIME = 0.00 # sec

total_rewards = []

for eposode in range(EPISODES):

obs = env.reset()

obs = T.tensor(obs[0], dtype=T.float)

# tensor([ 0.0040, 0.0199, -0.0622, 0.0594, -0.0605, 0.0577, -0.0056, 0.0333, -0.0072, 0.0532, -0.0512, 0.0173, -0.0529, -0.1104, 0.0946, -0.0559, 0.0824])

print(type(obs))

# observation_space : Box(-inf, inf, (17,), float64)

print('observation_space : ', env.observation_space)

print('obs :', obs)

reward: float = 0

total_reward: float = 0

done: bool = False

for j in range(STEPS):

env.render()

# ここをDDPGに置き換えていく

action = agent.choose_action(obs) # 1.Agentクラスを定義していく

#action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

print('====ここまではOK4====')

print('action_space : ', env.action_space)

print('action : ', action)

next_state, reward, done, _, info = env.step(action)

print('next_state, reward, done, _, info :', next_state, reward, done, _, info)

print('====ここまではOK5====')

#7. トラジェクトを保存する。経験再生(ReplayBuffer)

agent.remember(obs, action, reward, next_state, int(done))

#12. ニューラルネットワークを学習する

agent.learn()

# 26.エピソード内での報酬を累積していく

total_reward += reward

# 27. next_stateをobsとして再出発する

print('next_state:', next_state)

obs = next_state

obs = T.tensor(obs, dtype=T.float)

# 28. チーターの動きを見たいのでスリープを入れる

time.sleep(DELAY_TIME)

print('total_reward : ', total_reward)

total_rewards.append(total_reward)

print('total_rewards : ', total_rewards)

env.close() # 空なんですけど・・・

print('script is done.')

# https://gymnasium.farama.org/

ではまた次回

DDPG by gymnasium ５日目

次回は、actorNNへの引数修正、それに伴うagentDDPG引数修正、そしてリプレイバッファへの保存内容を修正しました。

今回は64データ集まった後に学習learn()メソドが走るとエラーが出るので修正していきます。

chatGPTより

このエラーメッセージは、F.linear関数を使用して乗算されている2つのテンソルのデータ型が一致しないことを示しています。この場合、inputテンソルとweightテンソルのデータ型が異なるようです。

この問題を解決するには、次の手順を試すことができます。

inputテンソルとweightテンソルのデータ型を確認します。それらのdtype属性を出力して、同じかどうかを確認できます。
データ型が異なる場合は、to()メソッドを使用してinputテンソルをweightテンソルと同じデータ型に変換します。たとえば、weightテンソルがfloat32テンソルの場合、input.to(torch.float32)を呼び出してinputテンソルをfloat32テンソルに変換できます。
代わりに、inputテンソルと同じデータ型にweightテンソルを変換することもできます。たとえば、inputテンソルがfloat64テンソルの場合、weight.to(torch.float64)を呼び出してweightテンソルをfloat64テンソルに変換できます。
テンソル上で行われる他の操作も同じデータ型を維持するようにしてください。

テンソルのデータ型が一貫していることを確認することで、遭遇したRuntimeErrorを解決できるはずです。

とのこと。なるほど、入力データをpytorchの型に合わせる必要があるようです。

現状確認

データを保存するときにstore_transitionメソドで

self.state_memory[index] = obs.detach().numpy().flatten()

self.action_memory[index] = action.flatten()

self.reward_memory[index] = reward.flatten()

self.next_state_memory[index] = next_state.flatten()

self.terminal_memory[index] = 1 – int(done)

としているので、type()で型を見てみます。

print(‘type of state_memory :’, type(self.state_memory[0][0]))

print(‘type of action_memory :’, type(self.action_memory[0][0]))

print(‘type of reward_memory :’, type(self.reward_memory[0]))

print(‘type of next_state_memory :’, type(self.next_state_memory[0][0]))

print(‘type of memory.state_memory :’, type(self.terminal_memory[0]))

結果、値は全てnumpy.float64になっています。

type of state_memory : <class ‘numpy.float64’>
type of action_memory : <class ‘numpy.float64’>
type of reward_memory : <class ‘numpy.float64’>
type of next_state_memory : <class ‘numpy.float64’>
type of memory.state_memory : <class ‘numpy.float64’>

取り出す際もsample_buffer(self, batch_size)メソドで

observations = self.state_memory[choosed_index]

として戻り値を得ているので変わりません。

戻り値はpytorchのテンソルに変換しています。troch.float64になっている。

observations = T.tensor(observations, dtype=float)

actions = T.tensor(actions, dtype=float)

rewards = T.tensor(rewards, dtype=float)

next_states = T.tensor(next_states, dtype=float)

terminals = T.tensor(terminals, dtype=float)

それを

target_actions = self.target_actor.forward(next_states)

に入れたときに起こっているのか？

class ActorNN(nn.Module):

__init__: self.fc1 = nn.Linear(n_obs_space, layer1_size)

forward : x = self.fc1(obs) ここでエラーが発生している

obsはバッチサイズ64x観察空間17、を入力ノード17x次層ノード64で待ち受けている。数としては問題ない。

型が合わないということなので、重みパラメータの型を調べてみる

agent.actor.fc1.weight.dtype → torch.float32

agent.target_actor.fc1.weight.dtype → torch.float32

なるほど、torch.float32で入力しなければならないようなので、変更します。

修正前：observations = T.tensor(observations, dtype=float)

修正後：observations = T.tensor(observations, dtype=T.float32)

これで回るようになりました。

ここまでのスクリプト

target_actorへnext_states 　バッチサイズ64データを入力し、target_actions 64データを得ることができました。

以降、actorが１ステップ行動するごとに、target_actorへnext_states 64データを入力してtarget_actions 64データを繰り返し出力する状態になりました。

次回は、このtarget_actionsをnext_statesと共にtarget_criticへ入力するところからやっていきます。

import gymnasium as gym
import time
import torch as T
import torch.nn as nn
import torch.nn.functional as F
import torch.optim as opitm
import numpy as np

# 10. ReplayBufferクラスを新規作成する
class ReplayBuffer:
    def __init__(self, max_memory_size, n_obs_space, n_action_space):
        self.max_memory_size = max_memory_size
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.memory_count = 0

        self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))
        self.action_memory =  np.zeros((self.max_memory_size, self.n_action_space))
        self.reward_memory =  np.zeros(self.max_memory_size)
        self.next_state_memory =  np.zeros((self.max_memory_size, self.n_obs_space))
        self.terminal_memory =  np.zeros(self.max_memory_size)
        #self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

    # 11.トランジション保存のためstore_transitionメソドを作成する
    def store_transition(self, obs, action, reward, next_state, done):
        print('store_transition is working.')
        index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック
        print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())
        self.state_memory[index] = obs.detach().numpy().flatten()
        self.action_memory[index] = action.flatten()
        self.reward_memory[index] = reward.flatten()
        self.next_state_memory[index] = next_state.flatten()
        self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように
        print('state_memory :', self.state_memory)
        print('action_memory :', self.action_memory)
        print('reward_memory :', self.reward_memory)
        print('next_state_memory :', self.next_state_memory)
        print('memory.state_memory :', self.terminal_memory)

        print('type of state_memory :', type(self.state_memory[0][0]))
        print('type of action_memory :', type(self.action_memory[0][0]))
        print('type of reward_memory :', type(self.reward_memory[0]))
        print('type of next_state_memory :', type(self.next_state_memory[0][0]))
        print('type of memory.state_memory :', type(self.terminal_memory[0]))

        self.memory_count += 1
        print('memory_count :', agent.memory.memory_count)

    # 16 バッファメモリーからランダムに抽出する
    def sample_buffer(self, batch_size):
        # indexが最大メモリに到達していない場合を想定する。
        max_index = min(self.max_memory_size, self.memory_count)
        choosed_index = np.random.choice(max_index, batch_size)
        
        observations = self.state_memory[choosed_index]
        actions = self.action_memory[choosed_index]
        rewards = self.reward_memory[choosed_index]
        next_states = self.next_state_memory[choosed_index]
        terminals = self.terminal_memory[choosed_index]

        return observations, actions, rewards, next_states, terminals


# 6.ActorNNクラスを新規作成する
class ActorNN(nn.Module):
    def __init__(self, alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64):
        print('ActorNN.__init__ is working.')
        super(ActorNN, self).__init__()
        self.fc1 = nn.Linear(n_obs_space, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, n_action_space)

    def forward(self, obs):
        print('AgetDDPG.ActorNN.forward is working')
        print('====ここまではOK1====')
        x = self.fc1(obs)
        x = F.relu(x)
        x = self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        mu = F.tanh(x)
        print('action μ:', mu)
        print('====ここまではOK2====')

        # 必要であればあとでノイズを入れる：action = mu + noize
        action = mu
        return action
        

# 22.CriticNNクラスを新規作成する
class CriticNN(nn.Module):
    def __init__(self, input_dim, output_dim):
        print('CriticNN.__init__ is working.')
        super(CriticNN, self).__init__()

        # ActorNNの部分:ActorNnと同じ構造
        self.fc1 = nn.Linear(input_dim, 64)
        self.fc2 = nn.Linear(64, output_dim)
        
        #　CriticNNの部分
        self.fc11 = nn.Linear(n_action)
    def forward(self, obs, action):
        # ActorNNの部分:ActorNnと同じ構造
        x = self.fc1(obs)
        x = F.relu(x)
        mu = self.fc2(x)

# 3.エージェントクラスを定義する
class AgentDDPG:

    def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64):
        print('AgentDDPG.__init__ is working.')
        # 5.ActorNNクラスのインスタンスを生成する
        self.alpha = alpha
        self.beta = beta
        self.gamma = gamma
        self.tau = tau
        
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.n_state_action_value = n_state_action_value

        self.layer1_size = layer1_size
        self.layer2_size = layer2_size

        # 13.バッチサイズを決めておく
        self.batch_size = batch_size 

        self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                            layer1_size=64, layer2_size=64, batch_size=64)        
        
        # 9.memoryインスタンスを追加
        self.MAX_MEMORY_SIZE = 1000
        self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,
                                   n_obs_space=self.n_obs_space,
                                   n_action_space=self.n_action_space)
        
        # 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する
        # actorとtarget_actorのネットワークは同じActorNNで良い
        self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)

        
        # 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する
        #self.target_critic = CriticNN(input_dim=self.input_dim, output_dim=self.output_dim)

        #self.critic = CriticNN(input_dim=self.input_dim, output_dim=self.output_dim)

    def choose_action(self, obs):
        print('AgentDDPG.choose_action is working.')
        # 4.方策（アクター）はニューラルネットワークで表現する。
        #   ActorNNクラスを新規作成し、インスタンスactorとして使用する。
        action = self.actor.forward(obs)
        action = action.detach().numpy()
        print('====ここまではOK3====')
        return action
    
    # 8.remenberメソドを追加
    def remember(self, obs, action, reward, next_state, done):
        self.memory.store_transition(obs, action, reward, next_state, done)

    # 13.learnメソドを追加
    def learn(self):
        # 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。
        if self.memory.memory_count < self.batch_size:
            return
        
        # 15.メモリバッファからデータを抜き出す sample_buffer()
        # バッチ化されているので変数名を複数形にする
        observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)
        print('s:', observations)
        print(observations.shape)
        print('a :', actions)
        print('r :', rewards)
        print('s_ :', next_states)
        print('terminal :', terminals)

        # 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する
        observations = T.tensor(observations, dtype=T.float32)
        actions = T.tensor(actions, dtype=T.float32)
        rewards = T.tensor(rewards, dtype=T.float32)
        next_states = T.tensor(next_states, dtype=T.float32)
        terminals = T.tensor(terminals, dtype=T.float32)
       
        # 18.ターゲットアクターネットワークインスタンスtarget_actorに
        # 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。
        # このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。
        print('next_states :', next_states)
        target_actions = self.target_actor.forward(next_states)
        """

        # 20.ターゲットクリティックネットワークインスタンスtareget_criticに
        # 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して
        # 価値関数の推定値ターゲットバリューを出力する。
        # TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。
        # ターゲットクリティックバリューはターゲットアクターネットワークを使う
        target_critic_values = self.target_critic.forward(next_states, target_actions)

        # 22.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に
        # 現在の状態observationsと行動actionsを入力して
        # クリティックバリューを算出する
        critic_values = self.critic.forward(observations, actions)

        # 333.TDターゲットを算出する：r + γ*V(w)[s_t+1]
        GAMMA = 0.01
        self.gamma = GAMMA

        td_targets = []
        for i in range(self.batch_size):
            td_target = rewards[i] + GAMMA * target_critic_values[i] * terminals[i]
            td_targets.append(td_target)

        # TDターゲットの形をバッチに整える
        td_targets = T.tensor(td_targets)
        td_target = td_target.view(self.bathc_size, 1)
        """

# 2.エージェントクラスのインスタンスを生成する
agent = AgentDDPG(alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64)

env = gym.make("HalfCheetah-v4", render_mode= 'human')

EPISODES = 2
DELAY_TIME = 0.00 # sec
total_rewards = []
for eposode in range(EPISODES):
    obs = env.reset()
    obs = T.tensor(obs[0], dtype=T.float)
    # tensor([ 0.0040,  0.0199, -0.0622,  0.0594, -0.0605,  0.0577, -0.0056,  0.0333,        -0.0072,  0.0532, -0.0512,  0.0173, -0.0529, -0.1104,  0.0946, -0.0559,         0.0824])
    print(type(obs))
    # observation_space :  Box(-inf, inf, (17,), float64)
    print('observation_space : ', env.observation_space)
    print('obs :', obs)

    reward: float = 0
    total_reward: float = 0
    done: bool = False
    for j in range(40):
        env.render()
        
        # ここをDDPGに置き換えていく
        action = agent.choose_action(obs) # 1.Agentクラスを定義していく
        #action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]
        print('====ここまではOK4====')
        print('action_space : ', env.action_space)
        print('action : ', action)

        next_state, reward, done, _, info = env.step(action)
        print('next_state, reward, done, _, info :', next_state, reward, done, _, info)
        """
        action :  [ 0.06660474 -0.11753064  0.02527559  0.06465236  0.1050786   0.05048539]

        next_state, reward, done, _, info : 
        [-0.00265179  0.0229547   0.00463243 -0.04729936 -0.00959038  0.04734605
        0.03672746  0.02857842  0.09980254 -0.32065693  0.04221647  1.58668951
        -2.31089174  1.30338924 -0.25465526  1.08250465 -0.14134398]
        0.07553858359316026
        False
        False
        {'x_position': -0.09233384215910741, 'x_velocity': 0.07920445513883267, 'reward_run': 0.07920445513883267, 'reward_ctrl': -0.0036658715456724168}
        """
        print('====ここまではOK5====')

        #7. トラジェクトを保存する。経験再生(ReplayBuffer)
        agent.remember(obs, action, reward, next_state, int(done))

        # 12. ニューラルネットワークを学習する
        agent.learn()

        print('next_state:', next_state)
        obs = next_state
        obs = T.tensor(obs, dtype=T.float)

        total_reward += reward
        
        time.sleep(DELAY_TIME)

    print('total_reward : ', total_reward)
    total_rewards.append(total_reward)

print('total_rewards : ', total_rewards)

env.close() # 空なんですけど・・・
print('script is done.')
# https://gymnasium.farama.org/

100

101

102

103

104

105

106

107

108

109

110

111

112

113

114

115

116

117

118

119

120

121

122

123

124

125

126

127

128

129

130

131

132

133

134

135

136

137

138

139

140

141

142

143

144

145

146

147

148

149

150

151

152

153

154

155

156

157

158

159

160

161

162

163

164

165

166

167

168

169

170

171

172

173

174

175

176

177

178

179

180

181

182

183

184

185

186

187

188

189

190

191

192

193

194

195

196

197

198

199

200

201

202

203

204

205

206

207

208

209

210

211

212

213

214

215

216

217

218

219

220

221

222

223

224

225

226

227

228

229

230

231

232

233

234

235

236

237

238

239

240

241

242

243

244

245

246

247

248

249

250

251

252

253

254

255

256

257

258

259

260

261

262

263

264

265

266

267

268

269

270

271

272

273

274

275

276

277

278

279

280

281

282

283

284

285

286

287

288

289

290

291

292

import gymnasium as gym

import time

import torch as T

import torch.nn as nn

import torch.nn.functional as F

import torch.optim as opitm

import numpy as np

# 10. ReplayBufferクラスを新規作成する

class ReplayBuffer:

def __init__(self, max_memory_size, n_obs_space, n_action_space):

self.max_memory_size = max_memory_size

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.memory_count = 0

self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.action_memory = np.zeros((self.max_memory_size, self.n_action_space))

self.reward_memory = np.zeros(self.max_memory_size)

self.next_state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.terminal_memory = np.zeros(self.max_memory_size)

#self.terminal_memory = np.zeros(self.max_memory_size, dtype=np.bool)

# 11.トランジション保存のためstore_transitionメソドを作成する

def store_transition(self, obs, action, reward, next_state, done):

print('store_transition is working.')

index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック

print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())

self.state_memory[index] = obs.detach().numpy().flatten()

self.action_memory[index] = action.flatten()

self.reward_memory[index] = reward.flatten()

self.next_state_memory[index] = next_state.flatten()

self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように

print('state_memory :', self.state_memory)

print('action_memory :', self.action_memory)

print('reward_memory :', self.reward_memory)

print('next_state_memory :', self.next_state_memory)

print('memory.state_memory :', self.terminal_memory)

print('type of state_memory :', type(self.state_memory[0][0]))

print('type of action_memory :', type(self.action_memory[0][0]))

print('type of reward_memory :', type(self.reward_memory[0]))

print('type of next_state_memory :', type(self.next_state_memory[0][0]))

print('type of memory.state_memory :', type(self.terminal_memory[0]))

self.memory_count += 1

print('memory_count :', agent.memory.memory_count)

# 16 バッファメモリーからランダムに抽出する

def sample_buffer(self, batch_size):

# indexが最大メモリに到達していない場合を想定する。

max_index = min(self.max_memory_size, self.memory_count)

choosed_index = np.random.choice(max_index, batch_size)

observations = self.state_memory[choosed_index]

actions = self.action_memory[choosed_index]

rewards = self.reward_memory[choosed_index]

next_states = self.next_state_memory[choosed_index]

terminals = self.terminal_memory[choosed_index]

return observations, actions, rewards, next_states, terminals

# 6.ActorNNクラスを新規作成する

class ActorNN(nn.Module):

def __init__(self, alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64):

print('ActorNN.__init__ is working.')

super(ActorNN, self).__init__()

self.fc1 = nn.Linear(n_obs_space, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, n_action_space)

def forward(self, obs):

print('AgetDDPG.ActorNN.forward is working')

print('====ここまではOK1====')

x = self.fc1(obs)

x = F.relu(x)

x = self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

mu = F.tanh(x)

print('action μ:', mu)

print('====ここまではOK2====')

# 必要であればあとでノイズを入れる：action = mu + noize

action = mu

return action

# 22.CriticNNクラスを新規作成する

class CriticNN(nn.Module):

def __init__(self, input_dim, output_dim):

print('CriticNN.__init__ is working.')

super(CriticNN, self).__init__()

# ActorNNの部分:ActorNnと同じ構造

self.fc1 = nn.Linear(input_dim, 64)

self.fc2 = nn.Linear(64, output_dim)

#　CriticNNの部分

self.fc11 = nn.Linear(n_action)

def forward(self, obs, action):

# ActorNNの部分:ActorNnと同じ構造

x = self.fc1(obs)

x = F.relu(x)

mu = self.fc2(x)

# 3.エージェントクラスを定義する

class AgentDDPG:

def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64):

print('AgentDDPG.__init__ is working.')

# 5.ActorNNクラスのインスタンスを生成する

self.alpha = alpha

self.beta = beta

self.gamma = gamma

self.tau = tau

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.n_state_action_value = n_state_action_value

self.layer1_size = layer1_size

self.layer2_size = layer2_size

# 13.バッチサイズを決めておく

self.batch_size = batch_size

self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 9.memoryインスタンスを追加

self.MAX_MEMORY_SIZE = 1000

self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,

n_obs_space=self.n_obs_space,

n_action_space=self.n_action_space)

# 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する

# actorとtarget_actorのネットワークは同じActorNNで良い

self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する

#self.target_critic = CriticNN(input_dim=self.input_dim, output_dim=self.output_dim)

#self.critic = CriticNN(input_dim=self.input_dim, output_dim=self.output_dim)

def choose_action(self, obs):

print('AgentDDPG.choose_action is working.')

# 4.方策（アクター）はニューラルネットワークで表現する。

# ActorNNクラスを新規作成し、インスタンスactorとして使用する。

action = self.actor.forward(obs)

action = action.detach().numpy()

print('====ここまではOK3====')

return action

# 8.remenberメソドを追加

def remember(self, obs, action, reward, next_state, done):

self.memory.store_transition(obs, action, reward, next_state, done)

# 13.learnメソドを追加

def learn(self):

# 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。

if self.memory.memory_count < self.batch_size:

return

# 15.メモリバッファからデータを抜き出す sample_buffer()

# バッチ化されているので変数名を複数形にする

observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)

print('s:', observations)

print(observations.shape)

print('a :', actions)

print('r :', rewards)

print('s_ :', next_states)

print('terminal :', terminals)

# 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する

observations = T.tensor(observations, dtype=T.float32)

actions = T.tensor(actions, dtype=T.float32)

rewards = T.tensor(rewards, dtype=T.float32)

next_states = T.tensor(next_states, dtype=T.float32)

terminals = T.tensor(terminals, dtype=T.float32)

# 18.ターゲットアクターネットワークインスタンスtarget_actorに

# 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。

# このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。

print('next_states :', next_states)

target_actions = self.target_actor.forward(next_states)

"""

# 20.ターゲットクリティックネットワークインスタンスtareget_criticに

# 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して

# 価値関数の推定値ターゲットバリューを出力する。

# TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。

# ターゲットクリティックバリューはターゲットアクターネットワークを使う

target_critic_values = self.target_critic.forward(next_states, target_actions)

# 22.ベースラインとして機能するクリティックネットワーク（価値関数V(w)[s_t]ネットワーク）に

# 現在の状態observationsと行動actionsを入力して

# クリティックバリューを算出する

critic_values = self.critic.forward(observations, actions)

# 333.TDターゲットを算出する：r + γ*V(w)[s_t+1]

GAMMA = 0.01

self.gamma = GAMMA

td_targets = []

for i in range(self.batch_size):

td_target = rewards[i] + GAMMA * target_critic_values[i] * terminals[i]

td_targets.append(td_target)

# TDターゲットの形をバッチに整える

td_targets = T.tensor(td_targets)

td_target = td_target.view(self.bathc_size, 1)

"""

# 2.エージェントクラスのインスタンスを生成する

agent = AgentDDPG(alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64)

env = gym.make("HalfCheetah-v4", render_mode= 'human')

EPISODES = 2

DELAY_TIME = 0.00 # sec

total_rewards = []

for eposode in range(EPISODES):

obs = env.reset()

obs = T.tensor(obs[0], dtype=T.float)

# tensor([ 0.0040, 0.0199, -0.0622, 0.0594, -0.0605, 0.0577, -0.0056, 0.0333, -0.0072, 0.0532, -0.0512, 0.0173, -0.0529, -0.1104, 0.0946, -0.0559, 0.0824])

print(type(obs))

# observation_space : Box(-inf, inf, (17,), float64)

print('observation_space : ', env.observation_space)

print('obs :', obs)

reward: float = 0

total_reward: float = 0

done: bool = False

for j in range(40):

env.render()

# ここをDDPGに置き換えていく

action = agent.choose_action(obs) # 1.Agentクラスを定義していく

#action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

print('====ここまではOK4====')

print('action_space : ', env.action_space)

print('action : ', action)

next_state, reward, done, _, info = env.step(action)

print('next_state, reward, done, _, info :', next_state, reward, done, _, info)

"""

action : [ 0.06660474 -0.11753064 0.02527559 0.06465236 0.1050786 0.05048539]

next_state, reward, done, _, info :

[-0.00265179 0.0229547 0.00463243 -0.04729936 -0.00959038 0.04734605

0.03672746 0.02857842 0.09980254 -0.32065693 0.04221647 1.58668951

-2.31089174 1.30338924 -0.25465526 1.08250465 -0.14134398]

0.07553858359316026

False

{'x_position': -0.09233384215910741, 'x_velocity': 0.07920445513883267, 'reward_run': 0.07920445513883267, 'reward_ctrl': -0.0036658715456724168}

"""

print('====ここまではOK5====')

#7. トラジェクトを保存する。経験再生(ReplayBuffer)

agent.remember(obs, action, reward, next_state, int(done))

# 12. ニューラルネットワークを学習する

agent.learn()

print('next_state:', next_state)

obs = next_state

obs = T.tensor(obs, dtype=T.float)

total_reward += reward

time.sleep(DELAY_TIME)

print('total_reward : ', total_reward)

total_rewards.append(total_reward)

print('total_rewards : ', total_rewards)

env.close() # 空なんですけど・・・

print('script is done.')

# https://gymnasium.farama.org/

DDPG by gymnasium ４日目

前回はリプレイバッファを作りました。

今回はいよいよ、ニューラルネットワークの核心、学習部分を作っていきます。

12.メインスクリプトのagent.remember()直下にagent.learn()を作ります。

13.AgentDDPGクラス内にlearn()メソドを新規作成します。

14.バッチサイズ分のトランジションが集まるまでは何も実行しない。

if self.memory.memory_count< self.batch_size:

return

15.メモリバッファからデータを抜き出す sample_buffer()

obs, action, reward, new_state, done = self.memory.sample_buffer(self.batch_size)

16.ReplayBufferのメソドとしてsample_bufferメソドを追加する。

def sample_buffer(self, batch_size):

# indexが最大メモリに到達していない場合を想定する。

max_index = min(self.max_memory_size, self.memory_count)

choosed_index = np.random.choice(max.index, batch_size)

observations = self.state_memory[choosed_index]

actions = self.action_memory[choosed_index]

rewards = self.reward_memory[choosed_index]

next_states = self.next_state_memory[choosed_index]

terminals = self.terminal[choosed_index]

return observations, actions, rewards, next_states, terminals

17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する。torch.tensor( obs, dtype=float)

obs = T.tensor(obs, dtype=float)

action = T.tensor(action, dtype=float)

reward = T.tensor(reward, dtype=float)

new_state = T.tensor(new_state, dtype=float)

done = T.tensor(done, dtype=float)

18.ターゲットアクターネットワークインスタンスtarget_actorに

次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。

target_actions = self.target_actor.forward(next_states)

19.AgentDDPGクラスにターゲットアクターネットワークインスタンスtarget_actorを作成する。actorとtarget_actorのネットワークは同じActorNN構造で良い

self.target_actor = ActorNN(input_dim=self.input_dim, output_dim=self.output_dim)

20.ターゲットクリティックネットワークインスタンスtareget_criticに

# 次の状態next_statesと上記より算出したターゲットアクションの２つを入力して

# 価値関数の推定値ターゲットバリューを出力する。

# TDターゲット：r + γ*V(w)[s_t+1] の部分のこと。

# ターゲットクリティックバリューはターゲットアクターネットワークを使う

target_critic_values = self.target_critic.forward(next_states, target_actions)

21.AgentDDPGクラスにターゲットクリティックネットワークインスタンスtareget_criticを作成する

self.target_critic = CriticNN(input_dim=self.input_dim, output_dim=self.output_dim)

問題発生

2.エージェントクラスのインスタンスを生成するところで

agent = AgentDDPG(input_dim=17, output_dim=6)

としていますが、これだけの引数ではDDPGを表現できないことに気が付きました。

ActorNNは入力obsと出力actionだけなので17と6だけの情報で良かったのですが、CriticNNは入力がactionと obsの２つ、また出力が状態価値state_valueの１つあるので、ニューラルネットワークに必要な入出力の数が異なります。

よって、入力の形、学習率、ニューラルネットワークの各層とノード数など、アクターとクリティックで異なるであろう部分はエージェントクラスから含めるように修正していきます。

以下のようにエージェントクラスの引数をたくさん増やしました。

agent = AgentDDPG( alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001, n_obs_space=17 , n_action_space=6, layer1_size=64, layer2_size=64, batch_size=64)

またアクターニューラルネットワーククラスへ受け渡す引数を修正しました。

self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6, layer1_size=64, layer2_size=64, batch_size=64)

さらなる問題

“””エラーメッセージ

このエラーメッセージは、行列の乗算に問題があることを示しています。具体的には、1×64の行列と17×64の行列を乗算しようとしていますが、この操作は許容されません。なぜなら、最初の行列の列数(64)が2番目の行列の行数(17)と異なるためです。

この問題を解決するには、乗算しようとしている行列の次元を確認し、行列乗算に対して互換性のある次元になるように調整する必要があります。あるいは、行列の次元に合わせて、適切な演算や変換を使用することも検討してみてください。

“””

バッファメモリーへ保存した観測情報obsを

observations = self.state_memory[choosed_index]

によって64個のバッチサイズで抜き出して、ニューラルネットワークへ入力した時点でエラーが発生しました。

流れを今一度おさらいします。

ひとつの観測情報obsをActorNNへ入力することによって、１つ行動actionが生成されます。

このobsの形は1×17なので、バッチ64個分をバッファから抜き出すと 64×17のはずです。しかし、エラーでは1×64となっているので根本的に間違っています。

患部リプレイバッファーの初期化部分でした。

self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

これで、観測データ1000000 x 観測空間17　を確保するつもりが、

self.state_memory = np.zeros(self.max_memory_size)

となっており、観測空間17のメモリーしか確保されていませんでした。

さらに悪いことに保存データが

self.state_memory[index] = obs.detach().numpy().flatten()[0]

となっており、観測空間17個のうち先頭の１個しか保存されないという間違いがありました。

self.state_memory[index] = obs.detach().numpy().flatten()に修正しました。

ほかにも

self.new_state_memory[index]がありますので同様に修正が必要です。

ここまでの修正スクリプト

リプレイバッファが６４データ蓄積されるまでは動きます。

学習learn()メソドが始まるとエラーが出る状態です。

次回はここを解消していきます。

import gymnasium as gymimport time
import torch as T
import torch.nn as nn
import torch.nn.functional as F
import torch.optim as opitm
import numpy as np

# 10. ReplayBufferクラスを新規作成する
class ReplayBuffer:
    def __init__(self, max_memory_size, n_obs_space, n_action_space):
        self.max_memory_size = max_memory_size
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.memory_count = 0

        self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))
        self.action_memory =  np.zeros((self.max_memory_size, self.n_action_space))
        self.reward_memory =  np.zeros(self.max_memory_size)
        self.next_state_memory =  np.zeros((self.max_memory_size, self.n_obs_space))
        self.terminal_memory =  np.zeros(self.max_memory_size)     

    # 11.トランジション保存のためstore_transitionメソドを作成する
    def store_transition(self, obs, action, reward, next_state, done):
        print('store_transition is working.')
        index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック
        print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())
        self.state_memory[index] = obs.detach().numpy().flatten()
        self.action_memory[index] = action.flatten()
        self.reward_memory[index] = reward.flatten()
        self.next_state_memory[index] = next_state.flatten()
        self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように
        print('state_memory :', self.state_memory)
        print('action_memory :', self.action_memory)
        print('reward_memory :', self.reward_memory)
        print('next_state_memory :', self.next_state_memory)
        print('memory.state_memory :', self.terminal_memory)

        self.memory_count += 1
        print('memory_count :', agent.memory.memory_count)

    # 16 バッファメモリーからランダムに抽出する
    def sample_buffer(self, batch_size):
        # indexが最大メモリに到達していない場合を想定する。
        max_index = min(self.max_memory_size, self.memory_count)
        choosed_index = np.random.choice(max_index, batch_size)
        
        observations = self.state_memory[choosed_index]
        actions = self.action_memory[choosed_index]
        rewards = self.reward_memory[choosed_index]
        next_states = self.next_state_memory[choosed_index]
        terminals = self.terminal_memory[choosed_index]

        return observations, actions, rewards, next_states, terminals


# 6.ActorNNクラスを新規作成する
class ActorNN(nn.Module):
    def __init__(self, alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64):
        print('ActorNN.__init__ is working.')
        super(ActorNN, self).__init__()
        self.fc1 = nn.Linear(n_obs_space, layer1_size)
        self.fc2 = nn.Linear(layer1_size, layer2_size)
        self.fc3 = nn.Linear(layer2_size, n_action_space)

    def forward(self, obs):
        print('AgetDDPG.ActorNN.forward is working')
        print('====ここまではOK1====')
        x = self.fc1(obs)
        x = F.relu(x)
        x = self.fc2(x)
        x = F.relu(x)
        x = self.fc3(x)
        mu = F.tanh(x)
        print('action μ:', mu)
        print('====ここまではOK2====')

        # 必要であればあとでノイズを入れる：action = mu + noize
        action = mu
        return action

# 3.エージェントクラスを定義する
class AgentDDPG:

    def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64):
        print('AgentDDPG.__init__ is working.')
        # 5.ActorNNクラスのインスタンスを生成する
        self.alpha = alpha
        self.beta = beta
        self.gamma = gamma
        self.tau = tau
        
        self.n_obs_space = n_obs_space
        self.n_action_space = n_action_space

        self.n_state_action_value = n_state_action_value

        self.layer1_size = layer1_size
        self.layer2_size = layer2_size

        # 13.バッチサイズを決めておく
        self.batch_size = batch_size 

        self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                            layer1_size=64, layer2_size=64, batch_size=64)        
        
        # 9.memoryインスタンスを追加
        self.MAX_MEMORY_SIZE = 1000
        self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,
                                   n_obs_space=self.n_obs_space,
                                   n_action_space=self.n_action_space)
        
        # 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する
        # actorとtarget_actorのネットワークは同じActorNNで良い
        self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,
                                    layer1_size=64, layer2_size=64, batch_size=64)

        
        # 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する
        #self.target_critic = CriticNN(input_dim=self.input_dim, output_dim=self.output_dim)

        #self.critic = CriticNN(input_dim=self.input_dim, output_dim=self.output_dim)

    def choose_action(self, obs):
        print('AgentDDPG.choose_action is working.')
        # 4.方策（アクター）はニューラルネットワークで表現する。
        #   ActorNNクラスを新規作成し、インスタンスactorとして使用する。
        action = self.actor.forward(obs)
        action = action.detach().numpy()
        print('====ここまではOK3====')
        return action
    
    # 8.remenberメソドを追加
    def remember(self, obs, action, reward, next_state, done):
        self.memory.store_transition(obs, action, reward, next_state, done)

    # 13.learnメソドを追加
    def learn(self):
        # 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。
        if self.memory.memory_count< self.batch_size:
            return
        
        # 15.メモリバッファからデータを抜き出す sample_buffer()
        # バッチ化されているので変数名を複数形にする
        observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)
        print('s:', observations)
        print(observations.shape)
        print('a :', actions)
        print('r :', rewards)
        print('s_ :', next_states)
        print('terminal :', terminals)

        # 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する
        observations = T.tensor(observations, dtype=float)
        actions = T.tensor(actions, dtype=float)
        rewards = T.tensor(rewards, dtype=float)
        next_states = T.tensor(next_states, dtype=float)
        terminals = T.tensor(terminals, dtype=float)
       
        # 18.ターゲットアクターネットワークインスタンスtarget_actorに
        # 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。
        # このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。
        print('next_states :', next_states)
        target_actions = self.target_actor.forward(next_states)


# 2.エージェントクラスのインスタンスを生成する
agent = AgentDDPG(alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,
                  n_obs_space=17 , n_action_space=6, n_state_action_value=1,
                  layer1_size=64, layer2_size=64, batch_size=64)

env = gym.make("HalfCheetah-v4", render_mode= 'human')

EPISODES = 2
DELAY_TIME = 0.00 # sec
total_rewards = []
for eposode in range(EPISODES):
    obs = env.reset()
    obs = T.tensor(obs[0], dtype=T.float)

    # observation_space :  Box(-inf, inf, (17,), float64)
    print('observation_space : ', env.observation_space)
    print('obs :', obs)

    reward: float = 0
    total_reward: float = 0
    done: bool = False
    for j in range(40):
        env.render()

        action = agent.choose_action(obs) # 1.Agentクラスを定義していく

        print('====ここまではOK4====')
        print('action_space : ', env.action_space)
        print('action : ', action)

        next_state, reward, done, _, info = env.step(action)
        print('next_state, reward, done, _, info :', next_state, reward, done, _, info)

        print('====ここまではOK5====')

        #7. トラジェクトを保存する。経験再生(ReplayBuffer)
        agent.remember(obs, action, reward, next_state, int(done))

        # 12. ニューラルネットワークを学習する
        agent.learn()

        print('next_state:', next_state)
        obs = next_state
        obs = T.tensor(obs, dtype=T.float)

        total_reward += reward
        
        time.sleep(DELAY_TIME)

    print('total_reward : ', total_reward)
    total_rewards.append(total_reward)

print('total_rewards : ', total_rewards)

env.close() # 空なんですけど・・・
print('script is done.')
# https://gymnasium.farama.org/

print(len(agent.memory.reward_memory))
print(agent.memory.reward_memory)

100

101

102

103

104

105

106

107

108

109

110

111

112

113

114

115

116

117

118

119

120

121

122

123

124

125

126

127

128

129

130

131

132

133

134

135

136

137

138

139

140

141

142

143

144

145

146

147

148

149

150

151

152

153

154

155

156

157

158

159

160

161

162

163

164

165

166

167

168

169

170

171

172

173

174

175

176

177

178

179

180

181

182

183

184

185

186

187

188

189

190

191

192

193

194

195

196

197

198

199

200

201

202

203

204

205

206

207

208

209

210

211

212

213

214

215

216

217

218

219

220

221

222

223

224

225

226

227

228

229

import gymnasium as gymimport time

import torch as T

import torch.nn as nn

import torch.nn.functional as F

import torch.optim as opitm

import numpy as np

# 10. ReplayBufferクラスを新規作成する

class ReplayBuffer:

def __init__(self, max_memory_size, n_obs_space, n_action_space):

self.max_memory_size = max_memory_size

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.memory_count = 0

self.state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.action_memory = np.zeros((self.max_memory_size, self.n_action_space))

self.reward_memory = np.zeros(self.max_memory_size)

self.next_state_memory = np.zeros((self.max_memory_size, self.n_obs_space))

self.terminal_memory = np.zeros(self.max_memory_size)

# 11.トランジション保存のためstore_transitionメソドを作成する

def store_transition(self, obs, action, reward, next_state, done):

print('store_transition is working.')

index = self.memory_count % self.max_memory_size # 最大メモリー数に到達したら、古いデータから上書きされていくギミック

print('obs.detach().numpy().flatten():',obs.detach().numpy().flatten())

self.state_memory[index] = obs.detach().numpy().flatten()

self.action_memory[index] = action.flatten()

self.reward_memory[index] = reward.flatten()

self.next_state_memory[index] = next_state.flatten()

self.terminal_memory[index] = 1 - int(done) # ゴールならterminal = 0 となるように

print('state_memory :', self.state_memory)

print('action_memory :', self.action_memory)

print('reward_memory :', self.reward_memory)

print('next_state_memory :', self.next_state_memory)

print('memory.state_memory :', self.terminal_memory)

self.memory_count += 1

print('memory_count :', agent.memory.memory_count)

# 16 バッファメモリーからランダムに抽出する

def sample_buffer(self, batch_size):

# indexが最大メモリに到達していない場合を想定する。

max_index = min(self.max_memory_size, self.memory_count)

choosed_index = np.random.choice(max_index, batch_size)

observations = self.state_memory[choosed_index]

actions = self.action_memory[choosed_index]

rewards = self.reward_memory[choosed_index]

next_states = self.next_state_memory[choosed_index]

terminals = self.terminal_memory[choosed_index]

return observations, actions, rewards, next_states, terminals

# 6.ActorNNクラスを新規作成する

class ActorNN(nn.Module):

def __init__(self, alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64):

print('ActorNN.__init__ is working.')

super(ActorNN, self).__init__()

self.fc1 = nn.Linear(n_obs_space, layer1_size)

self.fc2 = nn.Linear(layer1_size, layer2_size)

self.fc3 = nn.Linear(layer2_size, n_action_space)

def forward(self, obs):

print('AgetDDPG.ActorNN.forward is working')

print('====ここまではOK1====')

x = self.fc1(obs)

x = F.relu(x)

x = self.fc2(x)

x = F.relu(x)

x = self.fc3(x)

mu = F.tanh(x)

print('action μ:', mu)

print('====ここまではOK2====')

# 必要であればあとでノイズを入れる：action = mu + noize

action = mu

return action

# 3.エージェントクラスを定義する

class AgentDDPG:

def __init__(self, alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64):

print('AgentDDPG.__init__ is working.')

# 5.ActorNNクラスのインスタンスを生成する

self.alpha = alpha

self.beta = beta

self.gamma = gamma

self.tau = tau

self.n_obs_space = n_obs_space

self.n_action_space = n_action_space

self.n_state_action_value = n_state_action_value

self.layer1_size = layer1_size

self.layer2_size = layer2_size

# 13.バッチサイズを決めておく

self.batch_size = batch_size

self.actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 9.memoryインスタンスを追加

self.MAX_MEMORY_SIZE = 1000

self.memory = ReplayBuffer(max_memory_size=self.MAX_MEMORY_SIZE,

n_obs_space=self.n_obs_space,

n_action_space=self.n_action_space)

# 19.ターゲットアクターネットワークインスタンスtarget_actorを作成する

# actorとtarget_actorのネットワークは同じActorNNで良い

self.target_actor = ActorNN(alpha=0.000025, n_obs_space=17, n_action_space=6,

layer1_size=64, layer2_size=64, batch_size=64)

# 21.ターゲットクリティックネットワークインスタンスtareget_criticを作成する

#self.target_critic = CriticNN(input_dim=self.input_dim, output_dim=self.output_dim)

#self.critic = CriticNN(input_dim=self.input_dim, output_dim=self.output_dim)

def choose_action(self, obs):

print('AgentDDPG.choose_action is working.')

# 4.方策（アクター）はニューラルネットワークで表現する。

# ActorNNクラスを新規作成し、インスタンスactorとして使用する。

action = self.actor.forward(obs)

action = action.detach().numpy()

print('====ここまではOK3====')

return action

# 8.remenberメソドを追加

def remember(self, obs, action, reward, next_state, done):

self.memory.store_transition(obs, action, reward, next_state, done)

# 13.learnメソドを追加

def learn(self):

# 14.バッチサイズ分のトランジションが集まるまでは何も実行しない。

if self.memory.memory_count< self.batch_size:

return

# 15.メモリバッファからデータを抜き出す sample_buffer()

# バッチ化されているので変数名を複数形にする

observations, actions, rewards, next_states, terminals = self.memory.sample_buffer(self.batch_size)

print('s:', observations)

print(observations.shape)

print('a :', actions)

print('r :', rewards)

print('s_ :', next_states)

print('terminal :', terminals)

# 17.抜き出したデータをpytorchで微分可能なようにtorch.tensor化する

observations = T.tensor(observations, dtype=float)

actions = T.tensor(actions, dtype=float)

rewards = T.tensor(rewards, dtype=float)

next_states = T.tensor(next_states, dtype=float)

terminals = T.tensor(terminals, dtype=float)

# 18.ターゲットアクターネットワークインスタンスtarget_actorに

# 次の状態next_satesを入れて、ターゲットアクションtarget_actionsとして取り出す。

# このターゲットネットワークはターゲットでないネットワークとNNパラメータを共有させる。

print('next_states :', next_states)

target_actions = self.target_actor.forward(next_states)

# 2.エージェントクラスのインスタンスを生成する

agent = AgentDDPG(alpha=0.000025, beta=0.00025, gamma=0.99, tau=0.001,

n_obs_space=17 , n_action_space=6, n_state_action_value=1,

layer1_size=64, layer2_size=64, batch_size=64)

env = gym.make("HalfCheetah-v4", render_mode= 'human')

EPISODES = 2

DELAY_TIME = 0.00 # sec

total_rewards = []

for eposode in range(EPISODES):

obs = env.reset()

obs = T.tensor(obs[0], dtype=T.float)

# observation_space : Box(-inf, inf, (17,), float64)

print('observation_space : ', env.observation_space)

print('obs :', obs)

reward: float = 0

total_reward: float = 0

done: bool = False

for j in range(40):

env.render()

action = agent.choose_action(obs) # 1.Agentクラスを定義していく

print('====ここまではOK4====')

print('action_space : ', env.action_space)

print('action : ', action)

next_state, reward, done, _, info = env.step(action)

print('next_state, reward, done, _, info :', next_state, reward, done, _, info)

print('====ここまではOK5====')

#7. トラジェクトを保存する。経験再生(ReplayBuffer)

agent.remember(obs, action, reward, next_state, int(done))

# 12. ニューラルネットワークを学習する

agent.learn()

print('next_state:', next_state)

obs = next_state

obs = T.tensor(obs, dtype=T.float)

total_reward += reward

time.sleep(DELAY_TIME)

print('total_reward : ', total_reward)

total_rewards.append(total_reward)

print('total_rewards : ', total_rewards)

env.close() # 空なんですけど・・・

print('script is done.')

# https://gymnasium.farama.org/

print(len(agent.memory.reward_memory))

print(agent.memory.reward_memory)

pytorchで謎の部分は下記のサイトを参考にさせていただきました。

https://qiita.com/tatsuya11bbs/items/86141fe3ca35bdae7338

2023年5月
月	火	水	木	金	土	日
1	2	3	4	5	6	7
8	9	10	11	12	13	14
15	16	17	18	19	20	21
22	23	24	25	26	27	28
29	30	31