///|
/// One line of a doc comment, with the file position of its first character.
priv struct DocLine {
text : String
line : Int
column : Int
offset : Int
}
///|
/// The lines of a doc comment's text (without `{-|` and `-}`).
fn doc_lines(comment : @scanner.Comment) -> Array[DocLine] {
let text = comment.text
let start = if text.has_prefix("{-|") { 3 } else { 0 }
let end = if text.has_suffix("-}") {
text.length() - 2
} else {
text.length()
}
let lines = []
let mut from = start
let mut row = comment.span.start.line
let mut column = comment.span.start.column + start
for i = start; i <= end; i = i + 1 {
if i == end || text.code_unit_at(i) == '\n' {
let raw = text.unsafe_substring(start=from, end=i)
let line = if raw.has_suffix("\r") {
raw.unsafe_substring(start=0, end=raw.length() - 1)
} else {
raw
}
lines.push({
text: line,
line: row,
column,
offset: comment.span.start.offset + from,
})
from = i + 1
row += 1
column = 1
}
}
lines
}
///|
/// An attribute found in the line walk: its `@` line and continuation lines.
priv struct RawAttribute {
lines : Array[DocLine]
/// Open brackets after the lines so far (plain attributes only).
mut open : Int
/// Whether the lines so far hold a value (plain attributes only).
mut has_value : Bool
/// From an attributes block, where continuation lines are verbatim.
block : Bool
}
///|
fn is_blank(text : String) -> Bool {
text.trim().is_empty()
}
///|
/// The fence info string of an attributes block: ```` ```attributes ````.
let block_info : String = "attributes"
///|
/// Brackets opened minus closed in `text`, skipping string and char
/// literals.
fn bracket_balance(text : String) -> Int {
let mut balance = 0
let mut quote : Char? = None
let mut escaped = false
for c in text {
match quote {
Some(q) =>
if escaped {
escaped = false
} else if c == '\\' {
escaped = true
} else if c == q {
quote = None
}
None =>
match c {
'"' | '\'' => quote = Some(c)
'(' | '[' | '{' => balance += 1
')' | ']' | '}' => balance -= 1
_ => ()
}
}
}
balance
}
///|
/// The fence of a fenced code block line (```` ``` ```` or `~~~`, after up to
/// three spaces) and its info string.
fn fence_of(text : String) -> (String, String)? {
let trimmed = text.trim_start().to_owned()
if text.length() - trimmed.length() > 3 {
return None
}
for fence in ["```", "~~~"] {
if trimmed.has_prefix(fence) {
return Some(
(
fence,
trimmed
.unsafe_substring(start=3, end=trimmed.length())
.trim()
.to_owned(),
),
)
}
}
None
}
///|
/// The code-block state of the line walk.
priv enum Mode {
Prose
/// Inside a fenced block: its fence, and whether it is an attributes block.
Fenced(String, Bool)
/// Inside an indented code block.
Indented
}
///|
/// Attributes of a doc comment, in three forms:
/// - an attributes block: a fenced code block with the info string
/// `attributes` (read verbatim; an attribute runs until the next `@` line
/// or a blank line);
/// - `@name` with one code span holding its values;
/// - `@name values` in prose, continuing only while brackets are open or
/// while the `@` line has no value yet.
///
/// Other code blocks, prose and the comment's first line hold no attributes.
fn raw_attributes(lines : Array[DocLine]) -> Array[RawAttribute] {
let found : Array[RawAttribute] = []
let mut current : RawAttribute? = None
let mut mode : Mode = Prose
let mut previous_blank = false
for i, line in lines {
let text = line.text
match mode {
Fenced(fence, attributes) => {
if fence_of(text) is Some((f, _)) && f == fence {
mode = Prose
current = None
previous_blank = false
} else if !attributes {
()
} else if is_blank(text) {
current = None
} else if text.has_prefix("@") {
let attribute = {
lines: [line],
open: 0,
has_value: true,
block: true,
}
found.push(attribute)
current = Some(attribute)
} else if current is Some(attribute) {
attribute.lines.push(line)
}
continue
}
Indented =>
if is_blank(text) || text.has_prefix(" ") {
continue
} else {
mode = Prose
}
Prose => ()
}
if fence_of(text) is Some((fence, info)) {
mode = Fenced(fence, info == block_info)
current = None
continue
}
if is_blank(text) {
current = None
previous_blank = true
continue
}
if text.has_prefix(" ") && previous_blank && current is None {
mode = Indented
previous_blank = false
continue
}
previous_blank = false
if i > 0 && text.has_prefix("@") {
let value = text.unsafe_substring(start=1, end=text.length())
let after_name = match value.find(" ") {
Some(k) => value.unsafe_substring(start=k, end=value.length()).trim()
None => ""
}
let span = after_name.has_prefix("`")
let attribute = {
lines: [line],
open: if span {
0
} else {
bracket_balance(after_name.to_owned())
},
has_value: after_name != "",
block: false,
}
found.push(attribute)
current = Some(attribute)
} else if current is Some(attribute) &&
(attribute.open > 0 || !attribute.has_value) {
attribute.lines.push(line)
attribute.open += bracket_balance(text)
attribute.has_value = true
} else {
current = None
}
}
found
}
///|
fn is_name_start(c : Char) -> Bool {
c >= 'a' && c <= 'z'
}
///|
fn is_name_char(c : Char) -> Bool {
(c >= 'a' && c <= 'z') ||
(c >= 'A' && c <= 'Z') ||
(c >= '0' && c <= '9') ||
c == '_'
}
///|
/// Whether `name` is `lower ("." lower)*` with `lower = [a-z][A-Za-z0-9_]*`.
fn valid_attribute_name(name : String) -> Bool {
if name == "" {
return false
}
for part in name.split(".") {
let part = part.to_owned()
guard part.get_char(0) is Some(first) && is_name_start(first) else {
return false
}
for c in part {
if !is_name_char(c) {
return false
}
}
}
true
}
///|
/// Whether `e` is Elm data: literals, a negated number, unit, tuples, lists,
/// records, names, or a parenthesized constructor application of data.
fn is_data(e : @ast.Expression) -> Bool {
match e {
UnitExpr
| Literal(_)
| CharLiteral(_)
| Integer(_)
| Hex(_)
| Floatable(_)
| FunctionOrValue(_, _) => true
Negation(n) => n.value is (Integer(_) | Hex(_) | Floatable(_))
TupledExpression(items) | ListExpr(items) =>
items.all(i => is_data(i.value))
RecordExpr(setters) => setters.all(s => is_data(s.value.expression.value))
ParenthesizedExpression(inner) =>
match inner.value {
Application([{ value: FunctionOrValue(_, name), .. }, .. args]) =>
name.get_char(0) is Some(c) &&
@scanner.is_upper_start(c) &&
args.all(a => is_data(a.value))
other => is_data(other)
}
_ => false
}
}
///|
fn range_at(line : Int, column : Int, length : Int) -> @ast.Range {
{
start: { row: line, column, },
end: { row: line, column: column + length, },
}
}
///|
/// A token moved from attribute-text coordinates to file coordinates. `body`
/// is the values' text: its first line starts `shift` code units into the
/// attribute's first line, and its other lines are the attribute's following
/// lines, joined with `\n` (a CRLF in the file is one unit longer, so each
/// line is moved by its own start).
fn shift_token(
t : @scanner.Token,
raw : RawAttribute,
shift : Int,
) -> @scanner.Token {
let body_starts = []
let mut at = 0
for k, line in raw.lines {
body_starts.push(at)
at += line.text.length() + 1 - (if k == 0 { shift } else { 0 })
}
let first = raw.lines[0]
let relocate = fn(p : @scanner.Position) -> @scanner.Position {
let k = p.line - 1
let line = raw.lines[k]
let (start, column) = if k == 0 {
(
line.offset + shift,
line.column +
code_points(first.text.unsafe_substring(start=0, end=shift)),
)
} else {
(line.offset, line.column)
}
{
offset: start + p.offset - body_starts[k],
line: line.line,
column: column + p.column - 1,
}
}
{ ..t, span: { start: relocate(t.span.start), end: relocate(t.span.end), }, }
}
///|
fn malformed(raw : RawAttribute, reason : String) -> @scanner.Diagnostic {
let first = raw.lines[0]
let last = raw.lines[raw.lines.length() - 1]
let start : @scanner.Position = {
offset: first.offset,
line: first.line,
column: first.column,
}
let end : @scanner.Position = {
offset: last.offset + last.text.length(),
line: last.line,
column: last.column + code_points(last.text),
}
let span : @scanner.Span = { start, end, }
{
code: "KR-ATTR-001",
severity: Warning,
message: "Malformed doc attribute: \{reason}",
span,
title: "MALFORMED ATTRIBUTE",
report: [
plain("I could not read this attribute:"),
Excerpt(context=span, highlight=point(start)),
plain(reason),
Hint([
Plain("Attribute names are lower case, like "),
@scanner.Chunk::code("@deprecated"),
Plain(
", and values are Elm data: strings, numbers, lists, records, tuples and names, like ",
),
@scanner.Chunk::code("@derive [ Json.encoder ]"),
Plain("."),
]),
],
}
}
///|
/// Parse one attribute; `Err` is a `KR-ATTR-001` warning.
fn parse_attribute(
raw : RawAttribute,
dialect : @dialect.Dialect,
warnings : Array[@scanner.Diagnostic],
) -> Result[DocAttribute, @scanner.Diagnostic] {
let first = raw.lines[0]
let text = first.text
let mut name_end = 1
while name_end < text.length() &&
text.code_unit_at(name_end) != ' ' &&
text.code_unit_at(name_end) != '\t' {
name_end += 1
}
let name = text.unsafe_substring(start=1, end=name_end)
if !valid_attribute_name(name) {
return Err(malformed(raw, "`@\{name}` is not a valid attribute name."))
}
let name_node : @ast.Node[String] = {
range: range_at(first.line, first.column + 1, name.length()),
value: name,
}
let last = raw.lines[raw.lines.length() - 1]
let range : @ast.Range = {
start: { row: first.line, column: first.column, },
end: { row: last.line, column: last.column + code_points(last.text), },
}
let after_name = text.unsafe_substring(start=name_end, end=text.length())
let trimmed = after_name.trim_start().to_owned()
if !raw.block && trimmed.has_prefix("`") {
// `@name `values``: the values are the code span's content.
let lead = after_name.length() - trimmed.length()
guard trimmed.unsafe_substring(start=1, end=trimmed.length()).find("`")
is Some(close) else {
return Err(
malformed(raw, "The code span after `@\{name}` is not closed."),
)
}
let content = trimmed.unsafe_substring(start=1, end=close + 1)
return parse_values(
raw,
name,
name_node,
range,
content,
name_end + lead + 1,
dialect,
warnings,
)
}
let rest = [after_name]
for line in raw.lines[1:] {
rest.push(line.text)
}
let body = rest.join("\n")
if name == "docs" {
let names : Array[@ast.Node[String]] = []
for li, line_text in rest {
let line = raw.lines[li]
// Columns count code points; `index` counts UTF-16 units.
let base = if li == 0 {
first.column +
code_points(first.text.unsafe_substring(start=0, end=name_end))
} else {
line.column
}
let mut index = 0
for piece in line_text.split(",") {
let piece = piece.to_owned()
let trimmed = piece.trim().to_owned()
if trimmed != "" {
let lead = piece.length() - piece.trim_start().length()
let column = base +
code_points(line_text.unsafe_substring(start=0, end=index + lead))
names.push({
range: range_at(line.line, column, code_points(trimmed)),
value: trimmed,
})
}
index += piece.length() + 1
}
}
return Ok(Docs(names~, range~))
}
parse_values(raw, name, name_node, range, body, name_end, dialect, warnings)
}
///|
fn code_points(text : String) -> Int {
let mut n = 0
for _ in text {
n += 1
}
n
}
///|
/// Parse `body`, the values of an attribute, which start `shift` code units
/// into its first line.
fn parse_values(
raw : RawAttribute,
name : String,
name_node : @ast.Node[String],
range : @ast.Range,
body : String,
shift : Int,
dialect : @dialect.Dialect,
warnings : Array[@scanner.Diagnostic],
) -> Result[DocAttribute, @scanner.Diagnostic] {
let source = @scanner.SourceText::{ module_name: None, text: body, }
guard @scanner.lex_raw(source, dialect~) is Ok(raw_tokens) else {
return Err(malformed(raw, "I could not read the values of `@\{name}`."))
}
let tokens = @scanner.normalize_tokens(raw_tokens).tokens.map(t => {
shift_token(t, raw, shift)
})
let c = Cursor::new(tokens, dialect)
// Attribute values are not laid out like Elm code: elm-format may move
// continuation lines to column 1.
c.indent = 0
let arguments = []
while !c.at_end() {
let argument = sub_expression(c) catch {
SyntaxError(_, _, _) =>
return Err(malformed(raw, "I could not read the values of `@\{name}`."))
}
if !is_data(argument.value) {
return Err(
malformed(raw, "The values of `@\{name}` must be Elm data, not code."),
)
}
arguments.push(argument)
}
warnings.append(c.warnings)
Ok(Attribute(name=name_node, arguments~, range~))
}
///|
/// The attributes of a doc comment, and warnings for malformed ones.
fn doc_attributes(
comment : @scanner.Comment,
dialect : @dialect.Dialect,
) -> (Array[DocAttribute], Array[@scanner.Diagnostic]) {
let attributes = []
let warnings = []
for raw in raw_attributes(doc_lines(comment)) {
match parse_attribute(raw, dialect, warnings) {
Ok(a) => attributes.push(a)
Err(w) => warnings.push(w)
}
}
(attributes, warnings)
}
///|
fn encode_attribute(a : DocAttribute, exact_ints~ : Bool) -> Json {
match a {
Attribute(name~, arguments~, range~) =>
{
"name": @ast.encode_node(name, n => n.to_json()),
"arguments": Json::array(
arguments.map(arg => {
@ast.encode_node(arg, e => {
@ast.encode_expression_with(e, exact_ints~)
})
}),
),
"range": @ast.encode_range(range),
}
Docs(names~, range~) =>
{
"docs": Json::array(
names.map(n => @ast.encode_node(n, v => v.to_json())),
),
"range": @ast.encode_range(range),
}
}
}
///|
/// Attribute groups as JSON, for tools.
pub fn encode_attributes(groups : Array[AttributeGroup]) -> Json {
encode_attributes_with(groups, exact_ints=false)
}
///|
/// Attribute groups as JSON. With `exact_ints`, Int arguments are written
/// with their exact digits (see `@ast.encode_file_with`).
pub fn encode_attributes_with(
groups : Array[AttributeGroup],
exact_ints~ : Bool,
) -> Json {
Json::array(
groups.map(g => {
"target": match g.target {
Module => "module".to_json()
Declaration(name~, range~) =>
{ "declaration": name.to_json(), "range": @ast.encode_range(range) }
},
"attributes": Json::array(
g.attributes.map(a => encode_attribute(a, exact_ints~)),
),
}),
)
}