///|
/// The deepest nesting of doc comments (a doc comment in Elm code in a doc
/// comment) whose code is formatted. Code in deeper doc comments stays as
/// written, so formatting does not recurse without limit.
let max_doc_depth : Int = 3

///|
/// The doc comment `text` (`{-|` to `-}`) as elm-format writes it: its
/// Markdown through `@markdown.format_doc`, with Elm code formatted by
/// `format_code` (elm-format `formatDocComment`). `depth` is the number of
/// doc comments around it. Text that is not a doc comment, or whose
/// formatted text would not end the comment at its end, stays as it is.
fn format_documentation(
  text : String,
  dialect : @dialect.Dialect,
  depth : Int,
) -> String {
  guard text.length() >= 5 && text.has_prefix("{-|") && text.has_suffix("-}") else {
    return text
  }
  let inner = text.view(start_offset=3, end_offset=text.length() - 2).to_owned()
  let formatted = "{-|" +
    @markdown.format_doc(inner, format_code=code => {
      format_code(code, dialect, depth)
    }) +
    "-}"
  if closes_at_end(formatted) {
    formatted
  } else {
    text
  }
}

///|
/// Whether the block comment `text` ends at its last character and not
/// before: Elm block comments nest, so the `{-` and `-}` in it must balance.
fn closes_at_end(text : String) -> Bool {
  let n = text.length()
  let mut depth = 0
  let mut i = 0
  while i < n {
    let c = text.code_unit_at(i).to_int()
    let next = if i + 1 < n { text.code_unit_at(i + 1).to_int() } else { 0 }
    if c == 0x7B && next == 0x2D {
      // `{-`
      depth += 1
      i += 2
    } else if c == 0x2D && next == 0x7D {
      // `-}`
      depth -= 1
      i += 2
      if depth == 0 {
        return i == n
      }
    } else {
      if depth == 0 {
        return false
      }
      i += 1
    }
  }
  false
}

///|
/// Elm code from a doc comment, formatted as elm-format formats it there
/// (`formatDocComment`): as declarations, else as expressions, else as a
/// module. `None` when the code is none of these; the code then stays as
/// written. `depth` is the number of doc comments around the code.
///
/// krueger has no parser for parts of a module, so each form wraps the
/// code in a module, formats it with the elm-format layout and removes the
/// wrapper. In a doc comment, elm-format puts one blank line between
/// top-level items, and none after a comment.
fn format_code(
  code : String,
  dialect : @dialect.Dialect,
  depth : Int,
) -> String? {
  guard depth < max_doc_depth else { return None }
  let found = code_as_declarations(code, dialect, depth)
  guard found is None else { return found }
  let found = code_as_expressions(code, dialect, depth)
  guard found is None else { return found }
  code_as_module(code, dialect, depth)
}

///|
/// The header of the modules that wrap code: a port module, because
/// elm-format reads `port` declarations in code.
let code_header : String = "port module X__ exposing (..)\n\n\n"

///|
/// The kind of a top-level item of formatted code, for the blank lines
/// between items.
priv enum CodeItem {
  Header
  Imports
  Definition
  Fixity
  Comment
}

///|
/// A top-level item of formatted code: its rows (1-based, inclusive), and
/// for a definition the name it defines (`v:` for values, `t:` for types).
priv struct CodeRows {
  start : Int
  mut end : Int
  kind : CodeItem
  name : String?
}

///|
/// `source` formatted with the elm-format layout and parsed again, or
/// `None` when it does not format.
fn format_wrapped(
  source : String,
  dialect : @dialect.Dialect,
  depth : Int,
) -> (String, @ast.File)? {
  let text = format_source(source, ElmFormat, dialect, depth + 1) catch {
    _ => return None
  }
  let result = @parser.parse_module(
    @scanner.SourceText::new(text),
    @scanner.DefaultScanner::new(dialect~),
    dialect~,
  )
  guard result.ast is Some(file) &&
    result.diagnostics.iter().all(d => !(d.severity is Error)) else {
    return None
  }
  Some((text, file))
}

///|
/// The top-level items of `file` in order. Items that share a row are one
/// item (a declaration and the comments in it or after it on its last row).
fn code_items(file : @ast.File) -> Array[CodeRows] {
  let items : Array[CodeRows] = []
  let r = file.module_definition.range
  items.push({ start: r.start.row, end: r.end.row, kind: Header, name: None, })
  if file.imports is [first, ..] && file.imports.last() is Some(last) {
    items.push({
      start: first.range.start.row,
      end: last.range.end.row,
      kind: Imports,
      name: None,
    })
  }
  for d in file.declarations {
    let (kind, name) = match d.value {
      InfixDeclaration(_) => (Fixity, None)
      FunctionDeclaration(f) =>
        (Definition, Some("v:" + f.declaration.value.name.value))
      AliasDeclaration(a) => (Definition, Some("t:" + a.name.value))
      CustomTypeDeclaration(t) => (Definition, Some("t:" + t.name.value))
      PortDeclaration(p) => (Definition, Some("v:" + p.name.value))
      Destructuring(_, _) => (Definition, None)
    }
    items.push({ start: d.range.start.row, end: d.range.end.row, kind, name, })
  }
  for c in file.comments {
    items.push({
      start: c.range.start.row,
      end: c.range.end.row,
      kind: Comment,
      name: None,
    })
  }
  items.sort_by((a, b) => a.start.compare(b.start))
  let merged : Array[CodeRows] = []
  for item in items {
    match merged.last() {
      Some(last) if item.start <= last.end =>
        if item.end > last.end {
          last.end = item.end
        }
      _ => merged.push(item)
    }
  }
  merged
}

///|
/// The blank lines between two top-level items in a doc comment (elm-format
/// `formatTopLevelBody` with one line between items): none after a
/// comment, between infix declarations, and between two definitions of
/// one name.
fn code_spacing(a : CodeRows, b : CodeRows) -> Int {
  match (a.kind, b.kind) {
    (Comment, Comment | Definition | Fixity) => 0
    (Fixity, Fixity) => 0
    (Definition, Definition) if a.name is Some(_) && a.name == b.name => 0
    _ => 1
  }
}

///|
/// The text of `items` from the formatted `text`, with doc-comment
/// spacing. `body` gives the lines of an item, or `None` when the code
/// cannot be written.
fn code_text(
  text : String,
  items : Array[CodeRows],
  body : (CodeRows, Array[String]) -> Array[String]?,
) -> String? {
  let lines = text.split("\n").map(l => l.to_owned()).collect()
  let sb = StringBuilder()
  let mut previous : CodeRows? = None
  for item in items {
    if previous is Some(p) {
      sb.write_string("\n".repeat(code_spacing(p, item)))
    }
    let rows = lines[item.start - 1:item.end].to_owned()
    guard body(item, rows) is Some(out) else { return None }
    for line in out {
      sb.write_string(line)
      sb.write_char('\n')
    }
    previous = Some(item)
  }
  Some(sb.to_string())
}

///|
/// The code as declarations after a module header: the formatted module
/// without its header. elm-format reads no imports here, and no comment
/// after the last declaration.
fn code_as_declarations(
  code : String,
  dialect : @dialect.Dialect,
  depth : Int,
) -> String? {
  guard format_wrapped(code_header + code + "\n", dialect, depth)
    is Some((text, file)) &&
    file.imports.is_empty() else {
    return None
  }
  let items = code_items(file).filter(i => !(i.kind is Header))
  guard !(items.last() is Some({ kind: Comment, .. })) else { return None }
  code_text(text, items, (_, rows) => Some(rows))
}

///|
/// The code as expressions: each line in column 1 starts one (a comment
/// line is a comment). Each expression is the body of a definition
/// `x__N =`; the formatted bodies lose that line and 4 columns. A line
/// comment after an expression on one line stays after it (elm-format
/// `withEol`, `formatEolCommented`): on the same line, or on the next line
/// when the formatted expression takes more lines.
fn code_as_expressions(
  code : String,
  dialect : @dialect.Dialect,
  depth : Int,
) -> String? {
  guard expression_groups(code, dialect) is Some(groups) else { return None }
  for g in groups {
    while g.last() is Some(l) && l.trim(chars=" ") == "" {
      ignore(g.pop())
    }
  }
  // The line comments after expressions on one line, by group.
  let eol : Map[Int, String] = Map([])
  let first = @parser.parse_module(
    @scanner.SourceText::new(wrap_expressions(groups)),
    @scanner.DefaultScanner::new(dialect~),
    dialect~,
  )
  guard first.ast is Some(parsed) else { return None }
  // elm-format reads no comment after the last expression.
  let last_end = parsed.declarations
    .iter()
    .fold(init=0, (m, d) => {
      if d.range.end.row > m {
        d.range.end.row
      } else {
        m
      }
    })
  guard parsed.comments.iter().all(c => c.range.start.row <= last_end) else {
    return None
  }
  for i, g in groups {
    guard g is [line] && !is_comment_line(line) else { continue }
    // The definition of group `i` starts on row `row`; its body is the
    // row after it.
    let row = expression_row(parsed, i)
    for c in parsed.comments {
      if c.value.has_prefix("--") && c.range.start.row == row + 1 {
        eol[i] = c.value
        let chars = line.to_array()
        let cut = c.range.start.column - 1 - 4
        if cut >= 0 && cut <= chars.length() {
          g[0] = String::from_array(chars[:cut]).trim_end(chars=" ").to_owned()
        }
      }
    }
  }
  guard format_wrapped(wrap_expressions(groups), dialect, depth)
    is Some((text, file)) &&
    file.imports.is_empty() else {
    return None
  }
  let items = code_items(file).filter(i => !(i.kind is Header))
  // Each expression is still one definition (none went into a comment).
  let expressions = groups.filter(g => {
    !(g is [line, ..] && is_comment_line(line))
  })
  guard file.declarations.length() == expressions.length() else { return None }
  code_text(text, items, (item, rows) => {
    guard item.kind is Definition else { return Some(rows) }
    guard rows is [first, .. rest] &&
      first.strip_prefix("x__") is Some(after) &&
      after.split(" ").next() is Some(number) else {
      return None
    }
    let out = []
    for line in rest {
      if line == "" {
        out.push("")
      } else if line.has_prefix("    ") {
        out.push(line.view(start_offset=4).to_owned())
      } else {
        return None
      }
    }
    let mut index = 0
    for c in number {
      guard c.is_ascii_digit() else { return None }
      index = index * 10 + (c.to_int() - '0'.to_int())
    }
    match (eol.get(index), out) {
      (Some(comment), [single]) => out[0] = single + " " + comment
      (Some(comment), _) => out.push(comment)
      (None, _) => ()
    }
    Some(out)
  })
}

///|
/// The lines of `code` in groups: a line in column 1 starts a group,
/// except in a multi-line comment or string. Groups lose their blank lines
/// at the end. `None` when the code does not scan or does not start in
/// column 1.
fn expression_groups(
  code : String,
  dialect : @dialect.Dialect,
) -> Array[Array[String]]? {
  guard @scanner.DefaultScanner::new(dialect~).tokenize(
      @scanner.SourceText::new(code),
    )
    is Ok(stream) else {
    return None
  }
  // The lines (1-based) after the first line of a multi-line token or
  // comment.
  let inside : @hashset.HashSet[Int] = @hashset.HashSet([])
  let mark = (span : @scanner.Span) => {
    for l = span.start.line + 1; l <= span.end.line; l = l + 1 {
      inside.add(l)
    }
  }
  let marks_trivia = (trivia : Array[@scanner.Trivia]) => {
    for t in trivia {
      if t is Comment(c) {
        mark(c.span)
      }
    }
  }
  marks_trivia(stream.trivia)
  for t in stream.tokens {
    mark(t.span)
    marks_trivia(t.trivia_before)
    marks_trivia(t.trivia_after)
  }
  let groups : Array[Array[String]] = []
  for i, line in code.split("\n").collect() {
    let l = line.to_owned()
    if l != "" && !l.has_prefix(" ") && !inside.contains(i + 1) {
      groups.push([l])
    } else {
      match groups.last() {
        Some(g) => g.push(l)
        None => if l.trim(chars=" ") != "" { return None }
      }
    }
  }
  guard !groups.is_empty() else { return None }
  for g in groups {
    while g.last() is Some(l) && l.trim(chars=" ") == "" {
      ignore(g.pop())
    }
  }
  Some(groups)
}

///|
fn is_comment_line(line : String) -> Bool {
  line.has_prefix("--") || line.has_prefix("{-")
}

///|
/// A module with each expression group as the body of `x__N =` (`N` is the
/// group's index), and each comment group as it is.
fn wrap_expressions(groups : Array[Array[String]]) -> String {
  let wrapped = StringBuilder()
  wrapped.write_string(code_header)
  for i, g in groups {
    if g is [line, ..] && is_comment_line(line) {
      wrapped.write_string(g.join("\n"))
    } else {
      wrapped.write_string("x__\{i} =\n")
      wrapped.write_string(
        g
        .map(l => if l.trim(chars=" ") == "" { "" } else { "    " + l })
        .join("\n"),
      )
    }
    wrapped.write_string("\n\n\n")
  }
  wrapped.to_string()
}

///|
/// The row of the definition `x__N` (`N` is `index`) in `file`, or 0.
fn expression_row(file : @ast.File, index : Int) -> Int {
  let name = "x__\{index}"
  for d in file.declarations {
    if d.value is FunctionDeclaration(f) &&
      f.declaration.value.name.value == name {
      return d.range.start.row
    }
  }
  0
}

///|
/// The code as a module, laid out as elm-format's `formatModule` lays out
/// a module in a doc comment: the comments before the header, then two
/// blank lines; the header, the module documentation and the imports, with
/// one blank line between them; then one blank line before the body, or two when the body starts
/// with a comment (also when there is nothing before it). Without a module
/// header (elm-format reads one as optional), the code gets one, and the
/// formatted text loses it again.
fn code_as_module(
  code : String,
  dialect : @dialect.Dialect,
  depth : Int,
) -> String? {
  let (text, file, has_header) = match format_wrapped(code, dialect, depth) {
    Some((text, file)) => (text, file, true)
    None =>
      match
        format_wrapped(
          "port module X__ exposing (..)\n\n" + code + "\n",
          dialect,
          depth,
        ) {
        Some((text, file)) => (text, file, false)
        None => return None
      }
  }
  let all = code_items(file)
  let lines = text.split("\n").map(l => l.to_owned()).collect()
  let rows = (item : CodeRows) => lines[item.start - 1:item.end].to_owned()
  let initial = []
  let docs = []
  let mut header : CodeRows? = None
  let mut imports : CodeRows? = None
  let body = []
  let mut leading = leading_comments(code, dialect)
  for item in all {
    match item.kind {
      Header => header = Some(item)
      Imports => imports = Some(item)
      Comment if leading > 0 && body.is_empty() && imports is None => {
        initial.push(item)
        leading -= 1
      }
      Comment if has_header &&
        header is Some(_) &&
        imports is None &&
        body.is_empty() &&
        lines[item.start - 1].has_prefix("{-|") => docs.push(item)
      _ => body.push(item)
    }
  }
  let out = []
  for item in initial {
    out.append(rows(item))
  }
  if !initial.is_empty() {
    out.append(["", ""])
  }
  if has_header && header is Some(h) {
    out.append(rows(h))
    for d in docs {
      out.push("")
      out.append(rows(d))
    }
    if imports is Some(i) {
      out.push("")
      out.append(rows(i))
    }
  } else if imports is Some(i) {
    out.append(rows(i))
  }
  match body {
    [] => ()
    [{ kind: Comment, .. }, ..] => out.append(["", ""])
    _ => out.push("")
  }
  let sb = StringBuilder()
  for line in out {
    sb.write_string(line)
    sb.write_char('\n')
  }
  guard code_text(text, body, (_, r) => Some(r)) is Some(rest) else {
    return None
  }
  Some(sb.to_string() + rest)
}

///|
/// The number of comments before the first token of `code`. A block
/// comment that ends code with only comments is not one: elm-format reads
/// it after the module header.
fn leading_comments(code : String, dialect : @dialect.Dialect) -> Int {
  guard @scanner.DefaultScanner::new(dialect~).tokenize(
      @scanner.SourceText::new(code),
    )
    is Ok(stream) else {
    return 0
  }
  let trivia = match stream.tokens {
    [first, ..] => first.trivia_before
    [] => stream.trivia
  }
  let comments = trivia.filter_map(t => {
    if t is Comment(c) {
      Some(c)
    } else {
      None
    }
  })
  match (stream.tokens, comments.last()) {
    ([], Some({ kind: Block, .. })) => comments.length() - 1
    _ => comments.length()
  }
}