// Documentation in ATD's own "text" format, used in ``.
// A port of `doc.ml` and `doc_lexer.mll`.
///|
/// Inline element of a paragraph.
pub(all) enum DocInline {
Text(String)
Code(String)
} derive(Eq, Debug)
///|
/// Block of documentation.
pub(all) enum DocBlock {
Paragraph(Array[DocInline])
Pre(Array[String])
} derive(Eq, Debug)
///|
/// Documentation: a list of blocks.
pub type Doc = Array[DocBlock]
///|
priv suberror DocFailure {
DocFailure(String)
}
///|
fn is_space(c : Char) -> Bool {
c == ' ' || c == '\t' || c == '\r' || c == '\n'
}
///|
fn is_space_no_nl(c : Char) -> Bool {
c == ' ' || c == '\t' || c == '\r'
}
///|
fn is_par_not_special(c : Char) -> Bool {
!(c == '\\' || c == '{' || c == '}' || is_space(c))
}
///|
fn is_verb_not_special(c : Char) -> Bool {
!(c == '\\' || c == '}' || is_space(c))
}
///|
priv struct DocLexer {
s : Array[Char]
mut pos : Int
}
///|
fn DocLexer::at(self : DocLexer, i : Int) -> Char? {
if i < self.s.length() {
Some(self.s[i])
} else {
None
}
}
///|
fn DocLexer::has_prefix(self : DocLexer, i : Int, p : String) -> Bool {
let mut j = i
for c in p {
if self.at(j) != Some(c) {
return false
}
j += 1
}
true
}
///|
fn DocLexer::count(self : DocLexer, i : Int, pred : (Char) -> Bool) -> Int {
let mut j = i
while self.at(j) is Some(c) && pred(c) {
j += 1
}
j - i
}
///|
fn DocLexer::sub(self : DocLexer, i : Int, n : Int) -> String {
String::from_array(self.s[i:i + n].to_owned())
}
///|
/// Length of a match of `line_break` ('\r'? '\n') at position i, or -1.
fn DocLexer::line_break(self : DocLexer, i : Int) -> Int {
if self.at(i) == Some('\n') {
1
} else if self.at(i) == Some('\r') && self.at(i + 1) == Some('\n') {
2
} else {
-1
}
}
///|
/// Pick the longest match; ties are resolved in favor of the first rule.
/// `candidates` holds the match length of each rule, -1 for no match.
fn longest(candidates : Array[Int]) -> (Int, Int) {
let mut best = -1
let mut best_len = -1
for i, len in candidates {
if len > best_len {
best = i
best_len = len
}
}
(best, best_len)
}
///|
fn close_paragraph(
a1 : Array[DocBlock],
a2 : Array[DocInline],
a3 : Array[String],
) -> Unit {
let s = a3.join("")
if s != "" {
a2.push(Text(s))
}
if !a2.is_empty() {
a1.push(Paragraph(a2.copy()))
}
a2.clear()
a3.clear()
}
///|
fn DocLexer::paragraph(self : DocLexer) -> Array[DocBlock] raise DocFailure {
let a1 : Array[DocBlock] = []
let a2 : Array[DocInline] = []
let a3 : Array[String] = []
for ;; {
let i = self.pos
let at_eof = i >= self.s.length()
// R1: '\\' ('\\' | "{{" | "{{{")
let r1 = if self.at(i) == Some('\\') {
if self.has_prefix(i + 1, "{{{") {
4
} else if self.has_prefix(i + 1, "{{") {
3
} else if self.at(i + 1) == Some('\\') {
2
} else {
-1
}
} else {
-1
}
// R2: "{{"
let r2 = if self.has_prefix(i, "{{") { 2 } else { -1 }
// R3: space* "{{{" line_break?
let sp = self.count(i, is_space)
let r3 = if self.has_prefix(i + sp, "{{{") {
let lb = self.line_break(i + sp + 3)
sp + 3 + (if lb > 0 { lb } else { 0 })
} else {
-1
}
// R4: par_not_special+
let n4 = self.count(i, is_par_not_special)
let r4 = if n4 > 0 { n4 } else { -1 }
// R5: space'* "\n"? space'*
let s1 = self.count(i, is_space_no_nl)
let r5 = if self.at(i + s1) == Some('\n') {
s1 + 1 + self.count(i + s1 + 1, is_space_no_nl)
} else {
s1
}
// R6: space'* "\n" (space'* "\n")+ space'*
let r6 = {
let mut j = i + self.count(i, is_space_no_nl)
let mut newlines = 0
let mut last_end = -1
while self.at(j) == Some('\n') {
j += 1
newlines += 1
last_end = j
j += self.count(j, is_space_no_nl)
}
if newlines >= 2 {
last_end - i + self.count(last_end, is_space_no_nl)
} else {
-1
}
}
// R7: space* eof (the eof pseudo-character counts as one more char)
let r7 = if i + sp >= self.s.length() { sp + 1 } else { -1 }
// R8: any char
let r8 = if at_eof { -1 } else { 1 }
let (rule, len) = longest([r1, r2, r3, r4, r5, r6, r7, r8])
match rule {
0 => {
a3.push(self.sub(i + 1, len - 1))
self.pos += len
}
1 => {
self.pos += len
let code = self.inline_verbatim()
let s = a3.join("")
if s != "" {
a2.push(Text(s))
}
a3.clear()
a2.push(Code(code))
}
2 => {
self.pos += len
let pre = self.verbatim()
close_paragraph(a1, a2, a3)
a1.push(Pre(pre))
}
3 => {
a3.push(self.sub(i, len))
self.pos += len
}
4 => {
a3.push(" ")
self.pos += len
}
5 => {
close_paragraph(a1, a2, a3)
self.pos += len
}
6 => {
close_paragraph(a1, a2, a3)
return a1
}
_ => {
a3.push(self.sub(i, 1))
self.pos += 1
}
}
}
}
///|
fn DocLexer::inline_verbatim(self : DocLexer) -> String raise DocFailure {
let accu = []
for ;; {
let i = self.pos
let v1 = if self.has_prefix(i, "\\\\") { 2 } else { -1 }
let v2 = if self.has_prefix(i, "\\}}") { 3 } else { -1 }
let n3 = self.count(i, is_space)
let v3 = if n3 > 0 { n3 } else { -1 }
let n4 = self.count(i, is_verb_not_special)
let v4 = if n4 > 0 { n4 } else { -1 }
let v5 = if i < self.s.length() { 1 } else { -1 }
let v6 = if self.has_prefix(i + n3, "}}") { n3 + 2 } else { -1 }
let v7 = if i >= self.s.length() { 1 } else { -1 }
let (rule, len) = longest([v1, v2, v3, v4, v5, v6, v7])
match rule {
0 => accu.push("\\")
1 => accu.push("}}")
2 => accu.push(" ")
3 => accu.push(self.sub(i, len))
4 => accu.push(self.sub(i, 1))
5 => {
self.pos += len
return accu.join("")
}
_ => raise DocFailure("Missing `}}'")
}
self.pos += len
}
}
///|
fn count_leading_spaces(line : String) -> Int {
let mut n = 0
for c in line {
if c == ' ' {
n += 1
} else {
return n
}
}
// blank line = infinite indentation
2147483647
}
///|
/// Remove as many leading spaces as possible while maintaining the
/// relative visual positioning of the text.
fn trim_indentation(lines : Array[String]) -> Array[String] {
match lines {
[] => []
_ => {
let mut removable = 2147483647
for line in lines {
let n = count_leading_spaces(line)
if n < removable {
removable = n
}
}
lines.map(str => {
let chars = str.to_array()
if removable <= chars.length() {
String::from_array(chars[removable:].to_owned())
} else {
""
}
})
}
}
}
///|
fn DocLexer::verbatim(self : DocLexer) -> Array[String] raise DocFailure {
let lines = []
let line = []
for ;; {
let i = self.pos
let w1 = if self.has_prefix(i, "\\\\") { 2 } else { -1 }
let w2 = if self.has_prefix(i, "\\}}}") { 4 } else { -1 }
let w3 = if self.at(i) == Some('\t') { 1 } else { -1 }
let w4 = self.line_break(i)
let n5 = self.count(i, is_verb_not_special)
let w5 = if n5 > 0 { n5 } else { -1 }
let w6 = if i < self.s.length() { 1 } else { -1 }
let lb = self.line_break(i)
let lb = if lb > 0 { lb } else { 0 }
let w7 = if self.has_prefix(i + lb, "}}}") { lb + 3 } else { -1 }
let w8 = if i >= self.s.length() { 1 } else { -1 }
let (rule, len) = longest([w1, w2, w3, w4, w5, w6, w7, w8])
match rule {
0 => line.push("\\")
1 => line.push("}}}")
2 => line.push(" ")
3 => {
lines.push(line.join(""))
line.clear()
}
4 => line.push(self.sub(i, len))
5 => line.push(self.sub(i, 1))
6 => {
self.pos += len
lines.push(line.join(""))
return trim_indentation(lines)
}
_ => raise DocFailure("Missing `}}}'")
}
self.pos += len
}
}
///|
/// Parse documentation in ATD's text format.
pub fn parse_doc_text(loc : Loc, s : String) -> Doc raise AtdError {
let lexer : DocLexer = { s: s.to_array(), pos: 0, }
lexer.paragraph() catch {
DocFailure(msg) =>
error(
"\{string_of_loc(loc)}:\nInvalid format for doc.text \{ocaml_quote(s)}:\nFailure(\{ocaml_quote(msg)})",
)
}
}
///|
/// Replace each match of one of the patterns (tried in order at each
/// position) by its replacement.
fn substitute(s : String, rules : Array[(String, String)]) -> String {
let buf = StringBuilder()
let chars = s.to_array()
let mut i = 0
while i < chars.length() {
let mut matched = false
for rule in rules {
let (pat, repl) = rule
let pc = pat.to_array()
let mut ok = i + pc.length() <= chars.length()
if ok {
for k, c in pc {
if chars[i + k] != c {
ok = false
break
}
}
}
if ok {
buf.write_string(repl)
i += pc.length()
matched = true
break
}
}
if !matched {
buf.write_char(chars[i])
i += 1
}
}
buf.to_string()
}
///|
fn escape_text(s : String) -> String {
substitute(s, [("{{", "\\{\\{"), ("\\", "\\\\")])
}
///|
fn escape_code(s : String) -> String {
substitute(s, [("}}", "\\}\\}"), ("\\", "\\\\")])
}
///|
fn escape_pre_line(s : String) -> String {
substitute(s, [("}}}", "\\}\\}\\}"), ("\\", "\\\\")])
}
///|
/// OCaml's `String.trim`.
fn ocaml_trim(s : String) -> String {
let is_ws = (c : Char) => {
c == ' ' || c == '\u{0C}' || c == '\n' || c == '\r' || c == '\t'
}
let chars = s.to_array()
let mut i = 0
let mut j = chars.length()
while i < j && is_ws(chars[i]) {
i += 1
}
while j > i && is_ws(chars[j - 1]) {
j -= 1
}
String::from_array(chars[i:j].to_owned())
}
///|
/// Replicates the upstream regexp `(?: \t\r\n)+`, which matches repeated
/// occurrences of the exact 4-character sequence " \t\r\n".
fn compact_whitespace(s : String) -> String {
let chars = s.to_array()
let buf = StringBuilder()
let mut i = 0
let is_seq = (i : Int) => {
i + 4 <= chars.length() &&
chars[i] == ' ' &&
chars[i + 1] == '\t' &&
chars[i + 2] == '\r' &&
chars[i + 3] == '\n'
}
while i < chars.length() {
if is_seq(i) {
while is_seq(i) {
i += 4
}
buf.write_char(' ')
} else {
buf.write_char(chars[i])
i += 1
}
}
buf.to_string()
}
///|
fn normalize_inline(s : String) -> String {
compact_whitespace(ocaml_trim(s))
}
///|
fn print_doc_inline(x : DocInline) -> String {
match x {
Text(s) => escape_text(normalize_inline(s))
Code(s) =>
match escape_code(normalize_inline(s)) {
"" => ""
s => {
let first_space = if s.has_prefix("{") { " " } else { "" }
let last_space = if s.has_suffix("}") { " " } else { "" }
"{{" + first_space + s + last_space + "}}"
}
}
}
}
///|
fn print_doc_block(x : DocBlock) -> String {
match x {
Paragraph(xs) => xs.map(print_doc_inline).filter(s => s != "").join(" ")
Pre(lines) => {
let content = lines.map(escape_pre_line).join("\n")
match content {
"" => ""
s => {
let first_newline = if s.has_prefix("\n") { "" } else { "\n" }
let last_newline = if s.has_suffix("\n") { "" } else { "\n" }
"{{{" + first_newline + s + last_newline + "}}}"
}
}
}
}
}
///|
/// Print documentation in ATD's text format.
pub fn print_doc_text(blocks : Doc) -> String {
blocks.map(print_doc_block).join("\n\n")
}
///|
/// All the valid annotations of the form ``.
pub let doc_annot_schema : Schema = [
{
section: "doc",
fields: [
(ModuleHead, "text"),
(TypeDef, "text"),
(Variant, "text"),
(Field, "text"),
],
},
]
///|
/// Extract and parse the documentation from an annotation.
pub fn get_doc(loc : Loc, an : Annot) -> Doc? raise AtdError {
let mut err : AtdError? = None
let res = annot_get_opt_field(
an,
parse=s => {
Some(
parse_doc_text(loc, s) catch {
e => {
err = Some(e)
[]
}
},
)
},
sections=["doc"],
field="text",
)
match err {
Some(e) => raise e
None => res
}
}
///|
fn html_escape(buf : StringBuilder, s : String) -> Unit {
for c in s {
match c {
'<' => buf.write_string("<")
'>' => buf.write_string(">")
'&' => buf.write_string("&")
'"' => buf.write_string(""")
c => buf.write_char(c)
}
}
}
///|
/// Convert documentation to HTML.
pub fn html_of_doc(blocks : Doc) -> String {
let buf = StringBuilder()
buf.write_string("\n\n")
for block in blocks {
match block {
Paragraph(l) => {
buf.write_string("\n")
for x in l {
match x {
Text(s) => html_escape(buf, s)
Code(s) => {
buf.write_string("")
html_escape(buf, s)
buf.write_string("")
}
}
}
buf.write_string("\n
\n")
}
Pre(lines) => {
buf.write_string("\n")
for line in lines {
html_escape(buf, line)
buf.write_char('\n')
}
buf.write_string("\n")
}
}
}
buf.write_string("\n\n")
buf.to_string()
}
///|
/// Split a string on sequences of blanks, ignoring leading and trailing
/// blanks.
pub fn split_on_blank(str : String) -> Array[String] {
let res = []
let cur = StringBuilder()
for c in str {
if is_space(c) {
if cur.to_string() != "" {
res.push(cur.to_string())
cur.reset()
}
} else {
cur.write_char(c)
}
}
if cur.to_string() != "" {
res.push(cur.to_string())
}
res
}
///|
/// Concatenate words into lines of at most `max_length` bytes, if
/// possible.
pub fn concatenate_into_lines(
words : Array[String],
max_length~ : Int,
) -> Array[String] {
let max_length = if max_length < 0 { 0 } else { max_length }
let lines = []
let buf = StringBuilder()
let mut len = 0
for word in words {
let word_len = @format.utf8_length(word)
if len == 0 {
buf.write_string(word)
len = word_len
} else if len + 1 + word_len <= max_length {
buf.write_char(' ')
buf.write_string(word)
len = len + 1 + word_len
} else {
lines.push(buf.to_string())
buf.reset()
buf.write_string(word)
len = word_len
}
}
lines.push(buf.to_string())
lines
}
///|
/// Rewrap a paragraph into lines of at most `max_length` bytes.
pub fn rewrap_paragraph(str : String, max_length~ : Int) -> Array[String] {
concatenate_into_lines(split_on_blank(str), max_length~)
}