// codec.mbt
//
// 公共编解码接口:错误类型、流式 Decoder、一次性 decode / encode。
//
// 错误语义对齐 WHATWG Encoding Standard(encoding.bs @ 2c3853e):
// - 解码为 replacement 模式:错误字节输出 U+FFFD(规范 "process a queue" 的 replacement 分支);
// - 编码为 fatal 模式:不可映射字符直接报错(规范导出的 "encode or fail" 操作)。
//   规范另定义 html 模式(输出 &#码点; 形式,供 HTML 表单用),本库暴露 fatal 原语,
//   替换策略交还调用方——规范本身也鼓励非浏览器实现采用适合自己的 API。

///|
/// 编解码过程中可能的错误。
pub enum EncodingError {
  /// label 不在 WHATWG Encoding Standard 定义的 228 个 label 之内。
  UnknownLabel(String)
  /// label 合法,但该编码尚未实现——本项目按里程碑逐步补齐,见 README 的覆盖范围。
  UnsupportedEncoding(String)
  /// 编码时遇到目标编码无法表示的字符(fatal 模式)。
  /// 第一个参数是码点值,第二个是它在输入中的字符位置。
  Unmappable(Int, Int)
} derive(Eq, @debug.Debug)

///|
/// 显式声明 trait 方法提升(新编译器弃用隐式提升,见 implicit_impl_as_method)。
pub extend EncodingError with Eq::{not_equal, equal}

///|
pub extend EncodingError with @moonbitlang/core/debug.Debug::{to_repr}

///|
/// 解码器内部状态:单字节表驱动,或中文多字节状态机(GBK/gb18030 共用——
/// 规范规定 GBK 的解码器就是 gb18030 的解码器)。
enum DecoderState {
  SingleByte(Array[Int])
  Chinese(ChineseState)
  Big5(LeadState)
  ShiftJis(LeadState)
  EucJp(EucJpState)
  EucKr(LeadState)
  Utf8(Utf8State)
  Utf16(Utf16State)
}

///|
/// 流式解码器:`consume` 逐块喂入字节,`finish` 冲刷结尾。
/// 字段私有:可解码的编码在构造时已解析成状态,放开写会让外部破坏状态。
pub struct Decoder {
  priv state : DecoderState
}

///|
/// 由 WHATWG label 构造解码器(ASCII 大小写不敏感)。
/// label 合法但尚未实现的编码返回 `UnsupportedEncoding`。
pub fn Decoder::new(label : String) -> Result[Decoder, EncodingError] {
  match label_to_encoding(label) {
    Err(e) => Err(e)
    Ok(encoding) =>
      match single_byte_tables(encoding) {
        Some((decode_table, _)) => {
          let decoder : Decoder = {
            state: DecoderState::SingleByte(decode_table),
          }
          Ok(decoder)
        }
        None =>
          match encoding {
            "GBK" | "gb18030" => {
              let decoder : Decoder = {
                state: DecoderState::Chinese(ChineseState::new()),
              }
              Ok(decoder)
            }
            "Big5" => {
              let decoder : Decoder = {
                state: DecoderState::Big5(LeadState::new()),
              }
              Ok(decoder)
            }
            "Shift_JIS" => {
              let decoder : Decoder = {
                state: DecoderState::ShiftJis(LeadState::new()),
              }
              Ok(decoder)
            }
            "EUC-JP" => {
              let decoder : Decoder = {
                state: DecoderState::EucJp(EucJpState::new()),
              }
              Ok(decoder)
            }
            "EUC-KR" => {
              let decoder : Decoder = {
                state: DecoderState::EucKr(LeadState::new()),
              }
              Ok(decoder)
            }
            "UTF-8" => {
              let decoder : Decoder = {
                state: DecoderState::Utf8(Utf8State::new()),
              }
              Ok(decoder)
            }
            "UTF-16BE" => {
              let decoder : Decoder = {
                state: DecoderState::Utf16(Utf16State::new(true)),
              }
              Ok(decoder)
            }
            "UTF-16LE" => {
              let decoder : Decoder = {
                state: DecoderState::Utf16(Utf16State::new(false)),
              }
              Ok(decoder)
            }
            _ => Err(EncodingError::UnsupportedEncoding(encoding))
          }
      }
  }
}

///|
/// 喂入一个字节块,返回新解出的文本。
/// 单字节编码无跨块状态:任意切分的结果与整段解码一致。
pub fn Decoder::consume(self : Decoder, chunk : Bytes) -> String {
  match self.state {
    DecoderState::SingleByte(table) => single_byte_decode(table, chunk)
    DecoderState::Chinese(state) => chinese_consume(state, chunk)
    DecoderState::Big5(state) => big5_consume(state, chunk)
    DecoderState::ShiftJis(state) => shift_jis_consume(state, chunk)
    DecoderState::EucJp(state) => euc_jp_consume(state, chunk)
    DecoderState::EucKr(state) => euc_kr_consume(state, chunk)
    DecoderState::Utf8(state) => utf8_consume(state, chunk)
    DecoderState::Utf16(state) => utf16_consume(state, chunk)
  }
}

///|
/// 结束流,冲刷未完成的输入。单字节编码没有滞留字节,返回空串;
/// 中文编码在这里按规范 end-of-queue 把截断序列补成单个 U+FFFD。
pub fn Decoder::finish(self : Decoder) -> String {
  match self.state {
    DecoderState::SingleByte(_) => ""
    DecoderState::Chinese(state) => chinese_finish(state)
    DecoderState::Big5(state) => big5_finish(state)
    DecoderState::ShiftJis(state) => shift_jis_finish(state)
    DecoderState::EucJp(state) => euc_jp_finish(state)
    DecoderState::EucKr(state) => euc_kr_finish(state)
    DecoderState::Utf8(state) => utf8_finish(state)
    DecoderState::Utf16(state) => utf16_finish(state)
  }
}

///|
/// 本构建已接线的编码(WHATWG 规范名)——label 解析覆盖全部 40 种编码,
/// 其中这些可实际编解码;返回新数组,改它不影响内部状态。
pub fn supported_encodings() -> Array[String] {
  let out : Array[String] = []
  for name in single_byte_names {
    out.push(name)
  }
  for name in multi_byte_names {
    out.push(name)
  }
  out
}

///|
/// 一次性解码整段字节(等价于 new + consume + finish)。
pub fn decode(input : Bytes, label : String) -> Result[String, EncodingError] {
  match Decoder::new(label) {
    Err(e) => Err(e)
    Ok(decoder) => {
      let head = decoder.consume(input)
      let tail = decoder.finish()
      if tail.is_empty() {
        Ok(head)
      } else {
        Ok(head + tail)
      }
    }
  }
}

///|
/// 一次性编码:文本 -> 字节(fatal 模式,对应规范 `encode or fail`)。
/// 目标编码无法表示的字符返回 `Unmappable(码点, 字符位置)`。
pub fn encode(text : String, label : String) -> Result[Bytes, EncodingError] {
  match label_to_encoding(label) {
    Err(e) => Err(e)
    Ok(encoding) =>
      match single_byte_tables(encoding) {
        Some((_, encode_table)) => single_byte_encode(encode_table, text)
        None =>
          match encoding {
            "GBK" => chinese_encode(text, true)
            "gb18030" => chinese_encode(text, false)
            "Big5" => big5_encode(text)
            "Shift_JIS" => shift_jis_encode(text)
            "EUC-JP" => euc_jp_encode(text)
            "EUC-KR" => euc_kr_encode(text)
            "UTF-8" => utf8_encode(text)
            // 规范 §get an encoder 断言 encoding 不是 replacement 或
            // UTF-16BE/LE——WHATWG 不定义 UTF-16 编码器,此处如实报未支持。
            "UTF-16BE" | "UTF-16LE" =>
              Err(EncodingError::UnsupportedEncoding(encoding))
            _ => Err(EncodingError::UnsupportedEncoding(encoding))
          }
      }
  }
}