// ddpg.mbt �?Deep Deterministic Policy Gradient (Lillicrap et al. 2015).
//
// DDPG learns a deterministic policy μ(s; θ^μ) that maps states to a
// continuous action, plus a critic Q(s, a; θ^Q) for off-policy TD
// learning. Like SAC/DSAC, it uses twin critics to mitigate Q-value
// overestimation and Polyak-averaged target networks for stability.
//
// Differences vs SAC/DSAC:
// - The actor is deterministic (no reparameterisation trick needed).
// Exploration comes from additive Gaussian noise during training.
// - Critic takes [state, action] as input and outputs a single scalar.
// - No entropy bonus in the actor loss.
//
// Differences vs TD3 (next version):
// - DDPG uses a single critic (we use twin for stability).
// - DDPG does NOT smooth the target policy or delay actor updates.
//
// References:
// - Lillicrap et al. 2015, "Continuous Control with Deep Reinforcement
// Learning" (ICLR).
// - Plappert et al. 2018 (twin-critic variant).
// =========================================================================
// DeterministicPolicy: state -> continuous action
// =========================================================================
///|
/// Deterministic policy network. One hidden layer with ReLU activation;
/// output is squashed via tanh to `[action_low, action_high]` per dim.
pub struct DeterministicPolicy {
state_dim : Int
action_dim : Int
hidden : Int
// W1[hidden, state_dim], b1[hidden]
w1 : Array[Array[Float]]
b1 : Array[Float]
// W2[action_dim, hidden], b2[action_dim]
w2 : Array[Array[Float]]
b2 : Array[Float]
action_low : Float
action_high : Float
}
///|
/// Construct a DeterministicPolicy. Hidden size defaults to 64.
pub fn DeterministicPolicy::new(
state_dim : Int,
action_dim : Int,
hidden : Int,
action_low : Float,
action_high : Float,
seed : UInt64,
) -> DeterministicPolicy {
let rng = Xoshiro::from_state(seed, seed + 7UL, seed + 13UL, seed + 17UL)
let bound_w1 = Float::from_double(@math.pow(1.0 / Float::from_int(state_dim).to_double(), 0.5))
let bound_w2 = Float::from_double(@math.pow(1.0 / Float::from_int(hidden).to_double(), 0.5))
let w1 : Array[Array[Float]] = Array::make(hidden, [])
let b1 : Array[Float] = Array::make(hidden, 0.0F)
for h in 0.. continuous action (in [action_low, action_high]).
/// `hidden_state_out` (optional) is filled with the post-ReLU hidden layer
/// for use in the critic's gradient computation.
pub fn deterministic_policy_forward(
net : DeterministicPolicy,
state : Array[Float],
hidden_state_out : Array[Float],
) -> Array[Float] {
// h = relu(W1 · state + b1)
for h in 0.. 0.0F { s } else { 0.0F }
ignore(hidden_state_out.set(h, relu))
}
// a = tanh(W2 · h + b2), then rescale to [action_low, action_high]
let action : Array[Float] = Array::make(net.action_dim, 0.0F)
let range = (net.action_high - net.action_low) * 0.5F
let mid = (net.action_high + net.action_low) * 0.5F
for a in 0.. Unit {
for h in 0.. scalar Q
// =========================================================================
///|
/// Continuous Q-network: takes state + action as input, outputs scalar Q.
/// One hidden layer with ReLU; linear output.
pub struct QNetworkContinuous {
state_dim : Int
action_dim : Int
hidden : Int
// W1[hidden, state_dim + action_dim], b1[hidden]
w1 : Array[Array[Float]]
b1 : Array[Float]
// W2[1, hidden], b2[1]
w2 : Array[Float]
mut b2 : Float
}
///|
/// Construct a continuous Q-network.
pub fn QNetworkContinuous::new(
state_dim : Int,
action_dim : Int,
hidden : Int,
seed : UInt64,
) -> QNetworkContinuous {
let rng = Xoshiro::from_state(seed, seed + 11UL, seed + 23UL, seed + 37UL)
let in_dim = state_dim + action_dim
let bound_w1 = Float::from_double(@math.pow(1.0 / Float::from_int(in_dim).to_double(), 0.5))
let bound_w2 = Float::from_double(@math.pow(1.0 / Float::from_int(hidden).to_double(), 0.5))
let w1 : Array[Array[Float]] = Array::make(hidden, [])
let b1 : Array[Float] = Array::make(hidden, 0.0F)
for h in 0.. scalar Q. `hidden_out` is filled with
/// the post-ReLU hidden state for use in critic gradient computation.
pub fn qnet_continuous_forward(
net : QNetworkContinuous,
state : Array[Float],
action : Array[Float],
hidden_out : Array[Float],
) -> Float {
let in_dim = net.state_dim + net.action_dim
// h = relu(W1 · [state; action] + b1)
for h in 0.. 0.0F { s } else { 0.0F }
ignore(hidden_out.set(h, relu))
}
// q = W2 · h + b2
let mut q = net.b2
for h in 0.. Unit {
for h in 0.. DDPG {
let actor = DeterministicPolicy::new(
state_dim, action_dim, hidden, action_low, action_high, seed,
)
let critic1 = QNetworkContinuous::new(state_dim, action_dim, hidden, seed + 1UL)
let critic2 = QNetworkContinuous::new(state_dim, action_dim, hidden, seed + 2UL)
let actor_target = DeterministicPolicy::new(
state_dim, action_dim, hidden, action_low, action_high, seed + 3UL,
)
let critic1_target = QNetworkContinuous::new(
state_dim, action_dim, hidden, seed + 4UL,
)
let critic2_target = QNetworkContinuous::new(
state_dim, action_dim, hidden, seed + 5UL,
)
policy_copy(actor_target, actor)
qnet_continuous_copy(critic1_target, critic1)
qnet_continuous_copy(critic2_target, critic2)
{
actor,
critic1,
critic2,
actor_target,
critic1_target,
critic2_target,
gamma,
tau,
exploration_noise,
}
}
///|
/// Deterministic policy action. Returns `actor(state)` for evaluation.
pub fn ddpg_select_action_eval(
ddpg : DDPG,
state : Array[Float],
) -> Array[Float] {
let hidden : Array[Float] = Array::make(ddpg.actor.hidden, 0.0F)
deterministic_policy_forward(ddpg.actor, state, hidden)
}
///|
/// Action with Gaussian exploration noise (clipped to action range).
/// Used during training.
pub fn ddpg_select_action(
ddpg : DDPG,
state : Array[Float],
rng : Xoshiro,
) -> Array[Float] {
let action = ddpg_select_action_eval(ddpg, state)
let noisy : Array[Float] = Array::make(ddpg.actor.action_dim, 0.0F)
for a in 0.. ddpg.actor.action_high {
v = ddpg.actor.action_high
}
ignore(noisy.set(a, v))
}
noisy
}
///|
/// Soft (Polyak) target update: target �?(1-τ)·target + τ·online.
pub fn ddpg_soft_update(ddpg : DDPG) -> Unit {
let tau = ddpg.tau
let one_minus_tau = 1.0F - tau
// Actor target.
for h in 0.. (Array[Float], Float, Bool),
ddpg : DDPG,
n_episodes : Int,
buffer_capacity : Int,
warmup_episodes : Int,
batch_size : Int,
sync_every : Int,
max_steps : Int,
seed : UInt64,
) -> Float {
let rng = Xoshiro::from_state(seed, seed + 7UL, seed + 13UL, seed + 17UL)
let buffer = ContinuousReplayBuffer::new(buffer_capacity, env_state_dim, env_action_dim)
let mut total_return = 0.0F
let mut return_count = 0
for ep in 0..= warmup_episodes && buffer.len() >= batch_size {
let batch = buffer.sample(batch_size, rng)
let _ = ddpg_update_critic(ddpg, batch, 0.001F)
let _ = ddpg_update_actor(ddpg, batch, 0.0001F)
}
// Polyak sync.
if (ep + 1) % sync_every == 0 {
ddpg_soft_update(ddpg)
}
let _ = env_action_dim
let _ = env_state_dim
}
if return_count > 0 {
total_return / Float::from_int(return_count)
} else {
0.0F
}
}