// Generated by tools/gen_charset.py from CPython's `encodings.aliases`.
// Charset handling for `HtmlDiff::make_file`, emulating Python's
// `str.encode(charset, "xmlcharrefreplace")` for the codecs listed below.
///|
/// Python's `encodings.normalize_encoding` followed by lowercasing: runs of
/// characters other than ASCII letters, digits and `.` become a single `_`,
/// and leading/trailing ones are dropped.
fn normalize_encoding(name : String) -> String {
let buf = StringBuilder()
let mut punct = false
for c in name.to_lower() {
if c is ('a'..='z' | '0'..='9' | '.') {
if punct && buf.to_string() != "" {
buf.write_char('_')
}
buf.write_char(c)
punct = false
} else {
punct = true
}
}
buf.to_string()
}
///|
/// Python's `encodings.aliases.aliases` restricted to the supported codecs.
fn codec_alias(name : String) -> String? {
match name {
"646"
| "ansi_x3.4_1968"
| "ansi_x3.4_1986"
| "ansi_x3_4_1968"
| "cp367"
| "csascii"
| "ibm367"
| "iso646_us"
| "iso_646.irv_1991"
| "iso_ir_6"
| "us"
| "us_ascii" => Some("ascii")
"8859"
| "cp819"
| "csisolatin1"
| "ibm819"
| "iso8859"
| "iso8859_1"
| "iso_8859_1"
| "iso_8859_1_1987"
| "iso_ir_100"
| "l1"
| "latin"
| "latin1" => Some("latin_1")
"cp65001" | "u8" | "utf" | "utf8" | "utf8_ucs2" | "utf8_ucs4" =>
Some("utf_8")
"u16" | "utf16" => Some("utf_16")
"unicodelittleunmarked" | "utf_16le" => Some("utf_16_le")
"unicodebigunmarked" | "utf_16be" => Some("utf_16_be")
"u32" | "utf32" => Some("utf_32")
"utf_32le" => Some("utf_32_le")
"utf_32be" => Some("utf_32_be")
_ => None
}
}
///|
/// Resolves a charset name to a Python codec name like `encodings.search_function`:
/// the alias table is consulted for the normalized name and for the name
/// with `.` replaced by `_`, while a canonical codec module name must match
/// the normalized name exactly. Returns `None` for codecs this package does
/// not know about.
fn resolve_codec(charset : String) -> String? {
let norm = normalize_encoding(charset)
match codec_alias(norm) {
Some(codec) => return Some(codec)
None => ()
}
match codec_alias(norm.replace_all(old=".", new="_")) {
Some(codec) => return Some(codec)
None => ()
}
match norm {
"ascii" => Some("ascii")
"latin_1" => Some("latin_1")
"utf_8" => Some("utf_8")
"utf_8_sig" => Some("utf_8_sig")
"utf_16" => Some("utf_16")
"utf_16_le" => Some("utf_16_le")
"utf_16_be" => Some("utf_16_be")
"utf_32" => Some("utf_32")
"utf_32_le" => Some("utf_32_le")
"utf_32_be" => Some("utf_32_be")
_ => None
}
}
///|
/// Replaces characters that `charset` cannot encode by `NNN;` references,
/// emulating Python's `encode(charset, 'xmlcharrefreplace')` for ASCII,
/// Latin-1 and the UTF-8/16/32 codecs. Unknown charsets are assumed to encode
/// all of Unicode.
fn xml_charref_replace(s : String, charset : String) -> String {
let max_code = match resolve_codec(charset) {
Some("ascii") => 0x7f
Some("latin_1") => 0xff
Some(_) => 0x10ffff
None => return s
}
let buf = StringBuilder()
for c in s {
let code = c.to_int()
// lone surrogates cannot be encoded by any of these codecs
if code > max_code || (code >= 0xd800 && code <= 0xdfff) {
buf.write_string("\{code};")
} else {
buf.write_char(c)
}
}
buf.to_string()
}