///|
/// Decode the escapes elm-syntax accepts: `\n` `\r` `\t` `\"` `\'` `\\` and
/// `\u{HEX}`. `body` is the literal without its quotes.
fn decode_escapes(
body : String,
code_point_limit? : Bool = false,
) -> String raise SyntaxError {
let sb = StringBuilder()
let chars = body.to_array()
let mut i = 0
while i < chars.length() {
let c = chars[i]
if c != '\\' {
sb.write_char(c)
i += 1
continue
}
guard chars.get(i + 1) is Some(e) else {
raise literal_error("UNKNOWN ESCAPE", "unfinished escape")
}
match e {
'n' => sb.write_char('\n')
'r' => sb.write_char('\r')
't' => sb.write_char('\t')
'"' => sb.write_char('"')
'\'' => sb.write_char('\'')
'\\' => sb.write_char('\\')
'u' => {
guard chars.get(i + 2) is Some('{') else {
raise literal_error(
"BAD UNICODE ESCAPE", "expected `{` in unicode escape",
)
}
let mut j = i + 3
let mut code = 0
while j < chars.length() && chars[j] != '}' {
guard hex_value(chars[j]) is Some(digit) else {
raise literal_error(
"BAD UNICODE ESCAPE", "expected a hex digit in unicode escape",
)
}
// Stop growing past the largest code point, so long runs cannot overflow.
if code <= 0x10FFFF {
code = code * 16 + digit
}
j += 1
}
if j >= chars.length() || j == i + 3 {
raise literal_error(
"BAD UNICODE ESCAPE", "expected hex digits and `}` in unicode escape",
)
}
if code_point_limit && code > 0x10FFFF {
raise literal_error("BAD UNICODE ESCAPE", "code point above 10FFFF")
}
// A value that is not a Unicode scalar value (above U+10FFFF, or a
// surrogate) becomes U+FFFD.
let scalar = code <= 0x10FFFF && !(code >= 0xD800 && code <= 0xDFFF)
sb.write_char(
if scalar {
Int::unsafe_to_char(code)
} else {
'\u{FFFD}'
},
)
i = j + 1
continue
}
_ => raise literal_error("UNKNOWN ESCAPE", "unknown escape `\\\{e}`")
}
i += 2
}
sb.to_string()
}
///|
let dummy_span : @scanner.Span = {
start: { offset: 0, line: 1, column: 1, },
end: { offset: 0, line: 1, column: 1, },
}
///|
fn hex_value(c : Char) -> Int? {
match c {
'0'..='9' => Some(c.to_int() - '0'.to_int())
'a'..='f' => Some(c.to_int() - 'a'.to_int() + 10)
'A'..='F' => Some(c.to_int() - 'A'.to_int() + 10)
_ => None
}
}
///|
/// A literal error before its token is known; `with_span` places it.
fn literal_error(title : String, message : String) -> SyntaxError {
SyntaxError(message, dummy_span, { title, report: [], })
}
///|
/// A literal error at `span`, with the report for its title.
fn literal_problem(
title : String,
message : String,
span : @scanner.Span,
) -> SyntaxError {
let highlight = point(span.start)
let problem = match title {
"UNKNOWN ESCAPE" =>
span_problem(
title,
"Backslashes always start escaped characters, but I do not recognize this one:",
highlight,
[
Text([
Plain("Valid escape characters include "),
@scanner.Chunk::code("\\n"),
Plain(", "),
@scanner.Chunk::code("\\t"),
Plain(", "),
@scanner.Chunk::code("\\\""),
Plain(", "),
@scanner.Chunk::code("\\'"),
Plain(", "),
@scanner.Chunk::code("\\\\"),
Plain(" and "),
@scanner.Chunk::code("\\u{1F600}"),
Plain("."),
]),
],
)
"BAD UNICODE ESCAPE" if message.has_prefix("code point") =>
span_problem(title, "This is not a valid code point:", highlight, [
plain("The valid code points are between 0 and 10FFFF inclusive."),
])
"BAD UNICODE ESCAPE" =>
span_problem(title, "I ran into an invalid Unicode escape:", highlight, [
Text([
Plain("A Unicode escape looks like "),
@scanner.Chunk::code("\\u{1F600}"),
Plain(", with 4 to 6 hex digits between the curly braces."),
]),
])
"NEEDS DOUBLE QUOTES" =>
span_problem(
title,
"The following string uses single quotes:",
highlight,
[
Text([
Plain("Please switch to double quotes instead: "),
@scanner.Chunk::code("'this'"),
Plain(" becomes "),
@scanner.Chunk::code("\"this\""),
Plain(". Elm uses single quotes only for one character, like "),
@scanner.Chunk::code("'a'"),
Plain("."),
]),
],
)
_ =>
span_problem(title, "I ran into a problem with this number:", highlight, [])
}
SyntaxError(message, span, problem)
}
///|
/// Value of a single- or triple-quoted string token.
fn string_value(
token : @scanner.Token,
dialect : @dialect.Dialect,
) -> String raise SyntaxError {
let text = token.lexeme
let quote = if text.has_prefix("\"\"\"") { 3 } else { 1 }
let body = text.unsafe_substring(start=quote, end=text.length() - quote)
decode_escapes(body, code_point_limit=dialect.has(BadUnicodeEscape)) catch {
SyntaxError(message, _, problem) =>
raise literal_problem(problem.title, message, token.span)
}
}
///|
/// Value of a char token.
fn char_value(
token : @scanner.Token,
dialect : @dialect.Dialect,
) -> Char raise SyntaxError {
let text = token.lexeme
let value = decode_escapes(
text.unsafe_substring(start=1, end=text.length() - 1),
code_point_limit=dialect.has(BadUnicodeEscape),
) catch {
SyntaxError(message, _, problem) =>
raise literal_problem(problem.title, message, token.span)
}
// `iter` takes a lone surrogate as one character (`char_length` aborts).
if dialect.has(CharLength) && value.iter().take(2).count() > 1 {
raise literal_problem(
"NEEDS DOUBLE QUOTES",
"a char literal holds one character [rule: char-length]",
token.span,
)
}
match value.get_char(0) {
Some(c) => c
None =>
raise literal_problem(
"NEEDS DOUBLE QUOTES",
"empty char literal",
token.span,
)
}
}
///|
/// A warning for a number too big to store exactly: krueger keeps the
/// nearest value it can (`Int64` max, or infinity), as elm make accepts it.
fn out_of_range(c : Cursor, token : @scanner.Token, kept : String) -> Unit {
c.warnings.push({
code: "KR-PARSE-009",
severity: Warning,
message: "Number literal out of range",
span: token.span,
title: "NUMBER OUT OF RANGE",
report: [
plain("This number is too big for me to store exactly:"),
Excerpt(
context=point(token.span.start),
highlight=point(token.span.start),
),
plain(
"I will use \{kept} instead. Elm compiles it, but the value may not be what you expect.",
),
],
})
}
///|
/// An integer token: `Ok(n)` for decimal, `Err(n)` for hexadecimal. A value
/// above the `Int64` range becomes `Int64` max, with a warning.
fn int_value(c : Cursor, token : @scanner.Token) -> Result[Int64, Int64] {
let text = token.lexeme
let max = 9223372036854775807L
if text.has_prefix("0x") {
let digits = text.unsafe_substring(start=2, end=text.length())
Err(
@string.parse_int64(digits, base=16) catch {
_ => {
out_of_range(c, token, "the largest 64-bit integer")
max
}
},
)
} else {
Ok(
@string.parse_int64(text) catch {
_ => {
out_of_range(c, token, "the largest 64-bit integer")
max
}
},
)
}
}
///|
/// A float token. A value beyond the `Double` range becomes infinity, with a
/// warning.
fn float_value(c : Cursor, token : @scanner.Token) -> Double {
@string.parse_double(token.lexeme) catch {
_ => {
out_of_range(c, token, "infinity")
1.0 / 0.0
}
}
}