// Generated by tools/gen_charset.py from CPython's `encodings.aliases`.
// Charset handling for `HtmlDiff::make_file`, emulating Python's
// `str.encode(charset, "xmlcharrefreplace")` for the codecs listed below.

///|
/// Python's `encodings.normalize_encoding` followed by lowercasing: runs of
/// characters other than ASCII letters, digits and `.` become a single `_`,
/// and leading/trailing ones are dropped.
fn normalize_encoding(name : String) -> String {
  let buf = StringBuilder()
  let mut punct = false
  for c in name.to_lower() {
    if c is ('a'..='z' | '0'..='9' | '.') {
      if punct && buf.to_string() != "" {
        buf.write_char('_')
      }
      buf.write_char(c)
      punct = false
    } else {
      punct = true
    }
  }
  buf.to_string()
}

///|
/// Python's `encodings.aliases.aliases` restricted to the supported codecs.
fn codec_alias(name : String) -> String? {
  match name {
    "646"
    | "ansi_x3.4_1968"
    | "ansi_x3.4_1986"
    | "ansi_x3_4_1968"
    | "cp367"
    | "csascii"
    | "ibm367"
    | "iso646_us"
    | "iso_646.irv_1991"
    | "iso_ir_6"
    | "us"
    | "us_ascii" => Some("ascii")
    "8859"
    | "cp819"
    | "csisolatin1"
    | "ibm819"
    | "iso8859"
    | "iso8859_1"
    | "iso_8859_1"
    | "iso_8859_1_1987"
    | "iso_ir_100"
    | "l1"
    | "latin"
    | "latin1" => Some("latin_1")
    "cp65001" | "u8" | "utf" | "utf8" | "utf8_ucs2" | "utf8_ucs4" =>
      Some("utf_8")
    "u16" | "utf16" => Some("utf_16")
    "unicodelittleunmarked" | "utf_16le" => Some("utf_16_le")
    "unicodebigunmarked" | "utf_16be" => Some("utf_16_be")
    "u32" | "utf32" => Some("utf_32")
    "utf_32le" => Some("utf_32_le")
    "utf_32be" => Some("utf_32_be")
    _ => None
  }
}

///|
/// Resolves a charset name to a Python codec name like `encodings.search_function`:
/// the alias table is consulted for the normalized name and for the name
/// with `.` replaced by `_`, while a canonical codec module name must match
/// the normalized name exactly. Returns `None` for codecs this package does
/// not know about.
fn resolve_codec(charset : String) -> String? {
  let norm = normalize_encoding(charset)
  match codec_alias(norm) {
    Some(codec) => return Some(codec)
    None => ()
  }
  match codec_alias(norm.replace_all(old=".", new="_")) {
    Some(codec) => return Some(codec)
    None => ()
  }
  match norm {
    "ascii" => Some("ascii")
    "latin_1" => Some("latin_1")
    "utf_8" => Some("utf_8")
    "utf_8_sig" => Some("utf_8_sig")
    "utf_16" => Some("utf_16")
    "utf_16_le" => Some("utf_16_le")
    "utf_16_be" => Some("utf_16_be")
    "utf_32" => Some("utf_32")
    "utf_32_le" => Some("utf_32_le")
    "utf_32_be" => Some("utf_32_be")
    _ => None
  }
}

///|
/// Replaces characters that `charset` cannot encode by `&#NNN;` references,
/// emulating Python's `encode(charset, 'xmlcharrefreplace')` for ASCII,
/// Latin-1 and the UTF-8/16/32 codecs. Unknown charsets are assumed to encode
/// all of Unicode.
fn xml_charref_replace(s : String, charset : String) -> String {
  let max_code = match resolve_codec(charset) {
    Some("ascii") => 0x7f
    Some("latin_1") => 0xff
    Some(_) => 0x10ffff
    None => return s
  }
  let buf = StringBuilder()
  for c in s {
    let code = c.to_int()
    // lone surrogates cannot be encoded by any of these codecs
    if code > max_code || (code >= 0xd800 && code <= 0xdfff) {
      buf.write_string("&#\{code};")
    } else {
      buf.write_char(c)
    }
  }
  buf.to_string()
}