///|
fn pdf_text_tounicode_preamble() -> String {
  "/CIDInit /ProcSet findresource begin\n" +
  "12 dict begin\n" +
  "begincmap\n" +
  "/CIDSystemInfo <<\n" +
  "  /Registry (Adobe)\n" +
  "  /Ordering (UCS)\n" +
  "  /Supplement 0\n" +
  ">> def\n" +
  "/CMapName /Adobe-Identity-UCS def\n" +
  "/CMapType 2 def\n" +
  "1 begincodespacerange\n" +
  "<00>\n" +
  "endcodespacerange\n"
}

///|
/// A composite font addresses glyphs with two-byte codes, so its ToUnicode
/// CMap has to declare a two-byte codespace. A reader that finds `<00>`
/// here reads the high byte of every CID as a whole character and extracts
/// nothing usable.
fn pdf_text_tounicode_cid_preamble() -> String {
  "/CIDInit /ProcSet findresource begin\n" +
  "12 dict begin\n" +
  "begincmap\n" +
  "/CIDSystemInfo <<\n" +
  "  /Registry (Adobe)\n" +
  "  /Ordering (UCS)\n" +
  "  /Supplement 0\n" +
  ">> def\n" +
  "/CMapName /Adobe-Identity-UCS def\n" +
  "/CMapType 2 def\n" +
  "1 begincodespacerange\n" +
  "<0000>\n" +
  "endcodespacerange\n"
}

///|
fn pdf_text_tounicode_postamble() -> String {
  "endbfrange\n" +
  "endcmap\n" +
  "CMapName currentdict /CMap defineresource pop\n" +
  "end\n" +
  "end\n"
}

///|
fn pdf_text_tounicode_sorted(
  entries : ArrayView[(Int, @core.PdfBytes)],
) -> Array[(Int, @core.PdfBytes)] {
  let sorted = [ for entry in entries => entry ]
  sorted.sort_by(fn(left, right) { left.0.compare(right.0) })
  sorted
}

///|
fn pdf_text_hex_int_min2(value : Int) -> String {
  let text = value.to_string(radix=16)
  if text.length() == 1 {
    "0" + text
  } else {
    text
  }
}

///|
/// Four hex digits, the width a two-byte codespace requires. Every code in
/// a CMap must be written at its codespace's width, so a CID below 0x1000
/// still needs its leading zeros.
fn pdf_text_hex_int4(value : Int) -> String {
  let text = value.to_string(radix=16)
  match text.length() {
    1 => "000" + text
    2 => "00" + text
    3 => "0" + text
    _ => text
  }
}

///|
fn pdf_text_ascii_length(text : String) -> Int {
  @ascii.encode(text).length()
}

///|
fn pdf_text_write_ascii(
  output : Array[Byte],
  position : Int,
  text : String,
) -> Int {
  let mut current = position
  for byte in @ascii.encode(text) {
    output[current] = byte
    current += 1
  }
  current
}

///|
fn pdf_text_hex_bytes_length(bytes : BytesView) -> Int {
  bytes.length() * 2
}

///|
fn pdf_text_write_hex_bytes(
  output : Array[Byte],
  position : Int,
  bytes : BytesView,
) -> Int {
  let mut current = position
  for byte in bytes {
    let value = byte.to_int()
    output[current] = pdf_writer_hex_digit(value / 16)
    output[current + 1] = pdf_writer_hex_digit(value % 16)
    current += 2
  }
  current
}

///|
fn pdf_text_tounicode_bfrange_length(
  charcode_text : String,
  bytes : BytesView,
) -> Int {
  pdf_text_ascii_length("<") +
  pdf_text_ascii_length(charcode_text) +
  pdf_text_ascii_length("><") +
  pdf_text_ascii_length(charcode_text) +
  pdf_text_ascii_length("><") +
  pdf_text_hex_bytes_length(bytes) +
  pdf_text_ascii_length(">\n")
}

///|
fn pdf_text_write_tounicode_bfrange(
  output : Array[Byte],
  position : Int,
  charcode_text : String,
  bytes : BytesView,
) -> Int {
  let mut current = pdf_text_write_ascii(output, position, "<")
  current = pdf_text_write_ascii(output, current, charcode_text)
  current = pdf_text_write_ascii(output, current, "><")
  current = pdf_text_write_ascii(output, current, charcode_text)
  current = pdf_text_write_ascii(output, current, "><")
  current = pdf_text_write_hex_bytes(output, current, bytes)
  pdf_text_write_ascii(output, current, ">\n")
}

///|
/// Assemble a ToUnicode CMap from a preamble and one already-formatted
/// charcode per entry. The two callers differ only in those: a simple font
/// numbers its codes sequentially in a one-byte codespace, a composite font
/// uses the CID itself in a two-byte one.
fn pdf_text_tounicode_bytes_with(
  sorted : ArrayView[(Int, @core.PdfBytes)],
  preamble : String,
  charcode_texts : ArrayView[String],
) -> @core.PdfBytes {
  let count_text = sorted.length().to_string()
  let mut length = pdf_text_ascii_length(preamble) +
    pdf_text_ascii_length(count_text) +
    pdf_text_ascii_length(" beginbfrange\n") +
    pdf_text_ascii_length(pdf_text_tounicode_postamble())
  for index, entry in sorted {
    length += pdf_text_tounicode_bfrange_length(charcode_texts[index], entry.1)
  }
  let output = Array::make(length, b'\x00')
  let mut position = pdf_text_write_ascii(output, 0, preamble)
  position = pdf_text_write_ascii(output, position, count_text)
  position = pdf_text_write_ascii(output, position, " beginbfrange\n")
  for index, entry in sorted {
    position = pdf_text_write_tounicode_bfrange(
      output,
      position,
      charcode_texts[index],
      entry.1,
    )
  }
  let _ = pdf_text_write_ascii(output, position, pdf_text_tounicode_postamble())
  Bytes::from_array(output)
}

///|
fn pdf_text_tounicode_bytes(
  sorted : ArrayView[(Int, @core.PdfBytes)],
) -> @core.PdfBytes {
  // the simple-font embedder assigns charcodes in subset order from 33,
  // so the entry keys name codepoints rather than codes
  let charcode_texts : Array[String] = Array(capacity=sorted.length())
  let mut charcode = 33
  for _ in sorted {
    charcode_texts.push(pdf_text_hex_int_min2(charcode))
    charcode += 1
  }
  pdf_text_tounicode_bytes_with(
    sorted,
    pdf_text_tounicode_preamble(),
    charcode_texts,
  )
}

///|
fn pdf_text_tounicode_cid_bytes(
  sorted : ArrayView[(Int, @core.PdfBytes)],
) -> @core.PdfBytes {
  // a composite font's entry key *is* the code the content stream writes
  let charcode_texts : Array[String] = Array(capacity=sorted.length())
  for entry in sorted {
    charcode_texts.push(pdf_text_hex_int4(entry.0))
  }
  pdf_text_tounicode_bytes_with(
    sorted,
    pdf_text_tounicode_cid_preamble(),
    charcode_texts,
  )
}

///|
fn PdfDocument::pdf_text_write_tounicode(
  self : PdfDocument,
  entries : ArrayView[(Int, @core.PdfBytes)],
) -> Int {
  self.pdf_text_add_tounicode_stream(
    pdf_text_tounicode_bytes(pdf_text_tounicode_sorted(entries)),
  )
}

///|
/// Write a ToUnicode CMap for a composite font, keyed by the CIDs the
/// content stream actually writes.
fn PdfDocument::pdf_text_write_tounicode_cid(
  self : PdfDocument,
  entries : ArrayView[(Int, @core.PdfBytes)],
) -> Int {
  self.pdf_text_add_tounicode_stream(
    pdf_text_tounicode_cid_bytes(pdf_text_tounicode_sorted(entries)),
  )
}

///|
fn PdfDocument::pdf_text_add_tounicode_stream(
  self : PdfDocument,
  bytes : @core.PdfBytes,
) -> Int {
  self.add_object(
    @syntax.pdf_stream(
      PdfDictionary([(pdf_text_length_key_name, PdfInteger(bytes.length()))]),
      StreamGot(bytes),
    ),
  )
}