// lstm_ddpg_update_actor.mbt — LSTM_DDPG BPTT-driven actor parameter
// update (v0.67.0).
//
// Companion to lstm_ddpg_update_critic_seq (v0.66.0): implements the
// DPG (deterministic policy gradient) identity through the LSTM
// actor.
//
// ∇_θ J = E_s[ ∇_a Q(s, π(s)) · ∇_θ π_θ(s) ]
//
// Implementation mirrors v0.65.0 (GRU actor BPTT) but for the LSTM
// variant. Like v0.65.0, the actor's MLP w2 + b2 update uses the
// closed-form ∂Q/∂a · ∂a/∂W2_a; ∂Q/∂a_t is computed via
// central-difference on the critic output. The actor's GRU/LSTM
// gradient through gru_cell_backward / lstm_cell_backward is
// reserved for the analytic-derivative version.
//
// Reuses `lstm_deterministic_policy_seq_forward` (v0.61.0) and
// `lstm_critic_step_with_cache` (v0.66.0).
///|
/// Update actor of an LSTM_DDPG agent via T-step BPTT on a sampled
/// (state_seq) window. Returns the mean absolute Q across the T
/// slots. Updates the actor in place via simple SGD.
///
/// Finite-difference eps for the actor-critic gradient is `eps`
/// (default 1e-3).
pub fn lstm_ddpg_update_actor_seq(
agent : LSTM_DDPG,
state_seq : Array[Float],
hidden_init : Array[Float],
cell_init : Array[Float],
lr : Float,
eps : Float,
) -> Float {
let seq_len = state_seq.length() / agent.actor.state_dim
let s_dim = agent.actor.state_dim
let a_dim = agent.actor.action_dim
let h_dim = agent.actor.lstm_hidden
// Step 1: forward through actor -> action_seq + per-step caches.
let action_seq : Array[Float] = Array::make(seq_len * a_dim, 0.0F)
let x_proj_seq : Array[Float] = Array::make(seq_len * h_dim, 0.0F)
let lstm_cache_seq : Array[LstmCellCache] = Array::make(seq_len, {
x: Array::make(h_dim, 0.0F), h_prev: Array::make(h_dim, 0.0F),
c_prev: Array::make(h_dim, 0.0F), f: Array::make(h_dim, 0.0F),
ig: Array::make(h_dim, 0.0F), c_tilde: Array::make(h_dim, 0.0F),
o: Array::make(h_dim, 0.0F), c_t: Array::make(h_dim, 0.0F),
tanh_c_t: Array::make(h_dim, 0.0F), h_t: Array::make(h_dim, 0.0F),
})
let mut hidden_a = hidden_init
let mut cell_a = cell_init
for t in 0.. (Array[Float], Array[Float], LstmCellCache) {
let x_proj_pre = matvec(policy.mlp_w1, policy.mlp_b1, state)
let x_proj = relu_forward(x_proj_pre)
let (hidden_next, cache) = lstm_cell_forward(x_proj, hidden, cell, policy.lstm)
let a_pre = matvec(policy.mlp_w2, policy.mlp_b2, hidden_next)
let half_range = (policy.action_high - policy.action_low) * 0.5F
let mid = (policy.action_high + policy.action_low) * 0.5F
let action : Array[Float] = Array::make(policy.action_dim, 0.0F)
for i in 0.. LSTMDDPGActorGrad {
let s_dim = policy.state_dim
let a_dim = policy.action_dim
let h_dim = policy.lstm_hidden
let zw1 : Array[Array[Float]] = Array::make(h_dim, [])
let zb1 : Array[Float] = Array::make(h_dim, 0.0F)
let zw2 : Array[Array[Float]] = Array::make(a_dim, [])
let zb2 : Array[Float] = Array::make(a_dim, 0.0F)
for i in 0..