// utf.mbt
//
// UTF-8 与 UTF-16BE/LE 的编解码,逐条实现 encoding.bs(whatwg/encoding @ 2c3853e)的
// §utf-8-decoder、§utf-8-encoder、§shared-utf-16-decoder。
//
// 关键规范事实(全部读原文后落代码,不可凭记忆):
// - UTF-8 解码:0xC0/0xC1(过长两字节头)与 0xF5..0xFF 在**新鲜字节**分支直接报错;
//   0xE0 → 下界 0xA0、0xED → 上界 0x9F(挡住过长与 UTF-16 代理区)、
//   0xF0 → 下界 0x90、0xF4 → 上界 0x8F(挡住 >U+10FFFF);
//   **非法延续字节 → 全状态复位 + 恢复该字节 + error**(该字节随后按新鲜字节重处理);
//   序列在 end-of-queue 截断 → 恰好一个 U+FFFD。
//   规范注明这些约束等价于 Unicode 的 "Best Practices for Using U+FFFD"。
// - UTF-16 解码:共享状态机 + 字节序标志;未配对**低**代理 → error(两字节已消费);
//   高代理后跟非低代理 → 恢复**当前码元**的两字节 + error(高代理本身被消化为
//   一个 U+FFFD,当前码元原样重处理不丢字符);end-of-queue 有挂起 → 一个 U+FFFD。
// - **规范不定义 UTF-16 编码器**(§get an encoder 断言 encoding 不是
//   replacement 或 UTF-16BE/LE)——编解码入口对 utf-16 返回 UnsupportedEncoding。
// - BOM 不属于解码器算法(属于 decode 钩子)。本库不做 BOM 嗅探(已知限制),
//   因此 UTF-16 输入里的 BOM 会作为 U+FEFF 字符解出——与 CPython 的
//   utf-16-be / utf-16-le 行为一致。

///|
/// UTF-8 解码状态:规范的 code point / bytes seen / bytes needed /
/// lower boundary / upper boundary(初始 0 / 0 / 0 / 0x80 / 0xBF)。
priv struct Utf8State {
  mut code_point : Int
  mut bytes_seen : Int
  mut bytes_needed : Int
  mut lower_boundary : Int
  mut upper_boundary : Int
}

///|
fn Utf8State::new() -> Utf8State {
  {
    code_point: 0,
    bytes_seen: 0,
    bytes_needed: 0,
    lower_boundary: 0x80,
    upper_boundary: 0xBF,
  }
}

///|
fn Utf8State::has_pending(self : Utf8State) -> Bool {
  self.bytes_needed != 0
}

///|
fn Utf8State::reset(self : Utf8State) -> Unit {
  self.code_point = 0
  self.bytes_seen = 0
  self.bytes_needed = 0
  self.lower_boundary = 0x80
  self.upper_boundary = 0xBF
}

///|
/// UTF-8 解码一个字节块(§utf-8-decoder 逐步翻译)。
fn utf8_consume(state : Utf8State, input : Bytes) -> String {
  let bytes = input.to_array()
  let out : Array[Char] = []
  let mut replay : Array[Byte] = []
  let mut r = 0
  let mut i = 0
  while i < bytes.length() || r < replay.length() {
    let mut b = -1
    if r < replay.length() {
      b = replay[r].to_int()
      r += 1
    } else {
      b = bytes[i].to_int()
      i += 1
    }
    if state.bytes_needed == 0 {
      // 新鲜字节:ASCII 直通;合法序列头设置状态;其余(0x80..0xC1、0xF5..0xFF)报错
      if b < 0x80 {
        out.push(int_to_char(b))
      } else if b >= 0xC2 && b <= 0xDF {
        state.bytes_needed = 1
        state.code_point = b & 0x1F
      } else if b >= 0xE0 && b <= 0xEF {
        if b == 0xE0 {
          state.lower_boundary = 0xA0 // 挡住过长(U+0000..U+07FF 用三字节)
        }
        if b == 0xED {
          state.upper_boundary = 0x9F // 挡住 UTF-16 代理区
        }
        state.bytes_needed = 2
        state.code_point = b & 0xF
      } else if b >= 0xF0 && b <= 0xF4 {
        if b == 0xF0 {
          state.lower_boundary = 0x90 // 挡住过长四字节
        }
        if b == 0xF4 {
          state.upper_boundary = 0x8F // 挡住 >U+10FFFF
        }
        state.bytes_needed = 3
        state.code_point = b & 0x7
      } else {
        out.push(replacement_char()) // 0x80..0xC1(含过长头 C0/C1)、0xF5..0xFF
      }
    } else if b < state.lower_boundary || b > state.upper_boundary {
      // 非法延续字节:全状态复位、恢复该字节(它将按新鲜字节重处理)、报错
      state.reset()
      let restored : Array[Int] = [b]
      replay = replay_prepend(replay, r, restored)
      r = 0
      out.push(replacement_char())
    } else {
      state.lower_boundary = 0x80
      state.upper_boundary = 0xBF
      state.code_point = (state.code_point << 6) | (b & 0x3F)
      state.bytes_seen += 1
      if state.bytes_seen == state.bytes_needed {
        out.push(int_to_char(state.code_point))
        state.code_point = 0
        state.bytes_seen = 0
        state.bytes_needed = 0
      }
    }
  }
  String::from_iter(out.iter())
}

///|
/// UTF-8 流结束:序列截断 → 恰好一个 U+FFFD(§utf-8-decoder end-of-queue)。
fn utf8_finish(state : Utf8State) -> String {
  if state.has_pending() {
    state.reset()
    let out : Array[Char] = [replacement_char()]
    return String::from_iter(out.iter())
  }
  ""
}

///|
/// UTF-8 编码(§utf-8-encoder:按码点区间定 count/offset,逐 6 位发延续字节)。
/// 所有输入字符都是合法标量,三个区间必然命中,规范上不会失败。
fn utf8_encode(text : String) -> Result[Bytes, EncodingError] {
  let out : Array[Byte] = []
  for c in text {
    let cp = c.to_int()
    if cp < 0x80 {
      out.push(int_to_byte(cp))
    } else {
      let mut count = 0
      let mut offset = 0
      if cp <= 0x07FF {
        count = 1
        offset = 0xC0
      } else if cp <= 0xFFFF {
        count = 2
        offset = 0xE0
      } else {
        count = 3
        offset = 0xF0
      }
      out.push(int_to_byte((cp >> (6 * count)) + offset))
      let mut left = count
      while left > 0 {
        let temp = cp >> (6 * (left - 1))
        out.push(int_to_byte(0x80 | (temp & 0x3F)))
        left -= 1
      }
    }
  }
  Ok(Bytes::from_array(out.exact_view()))
}

///|
/// 共享 UTF-16 解码状态:lead byte / lead surrogate(-1 = null)+ 字节序标志。
priv struct Utf16State {
  mut lead_byte : Int
  mut lead_surrogate : Int
  is_be : Bool
}

///|
fn Utf16State::new(is_be : Bool) -> Utf16State {
  { lead_byte: -1, lead_surrogate: -1, is_be, }
}

///|
fn Utf16State::has_pending(self : Utf16State) -> Bool {
  self.lead_byte >= 0 || self.lead_surrogate >= 0
}

///|
fn Utf16State::reset(self : Utf16State) -> Unit {
  self.lead_byte = -1
  self.lead_surrogate = -1
}

///|
/// 共享 UTF-16 解码一个字节块(§shared-utf-16-decoder 逐步翻译)。
fn utf16_consume(state : Utf16State, input : Bytes) -> String {
  let bytes = input.to_array()
  let out : Array[Char] = []
  let mut replay : Array[Byte] = []
  let mut r = 0
  let mut i = 0
  while i < bytes.length() || r < replay.length() {
    let mut b = -1
    if r < replay.length() {
      b = replay[r].to_int()
      r += 1
    } else {
      b = bytes[i].to_int()
      i += 1
    }
    if state.lead_byte < 0 {
      state.lead_byte = b
      continue
    }
    let unit = if state.is_be {
      (state.lead_byte << 8) | b
    } else {
      (b << 8) | state.lead_byte
    }
    state.lead_byte = -1
    if state.lead_surrogate >= 0 {
      let lead = state.lead_surrogate
      state.lead_surrogate = -1
      if unit >= 0xDC00 && unit <= 0xDFFF {
        out.push(
          int_to_char(0x10000 + ((lead - 0xD800) << 10) + (unit - 0xDC00)),
        )
        continue
      }
      // 高代理未配对:恢复**当前码元**的两字节并报错——
      // 高代理被消化为一个 U+FFFD,当前码元随后按新鲜流程原样重处理
      let hi = unit >> 8
      let lo = unit & 0xFF
      let restored : Array[Int] = if state.is_be { [hi, lo] } else { [lo, hi] }
      replay = replay_prepend(replay, r, restored)
      r = 0
      out.push(replacement_char())
      continue
    }
    if unit >= 0xD800 && unit <= 0xDBFF {
      state.lead_surrogate = unit
      continue
    }
    if unit >= 0xDC00 && unit <= 0xDFFF {
      out.push(replacement_char()) // 孤立低代理(规范:不输出代理本身)
      continue
    }
    out.push(int_to_char(unit))
  }
  String::from_iter(out.iter())
}

///|
/// 共享 UTF-16 流结束:有挂起的首字节或高代理 → 一个 U+FFFD。
fn utf16_finish(state : Utf16State) -> String {
  if state.has_pending() {
    state.reset()
    let out : Array[Char] = [replacement_char()]
    return String::from_iter(out.iter())
  }
  ""
}