///|
/// Lexical token categories exposed for diagnostics and editor integrations.
pub enum JsonTokenKind {
LeftBrace
RightBrace
LeftBracket
RightBracket
Colon
Comma
StringLiteral
NumberLiteral
TrueLiteral
FalseLiteral
NullLiteral
End
}
///|
pub(all) struct JsonToken {
kind : JsonTokenKind
lexeme : String
start : Int
end : Int
}
///|
pub fn JsonTokenKind::to_string(self : JsonTokenKind) -> String {
match self {
LeftBrace => "left-brace"
RightBrace => "right-brace"
LeftBracket => "left-bracket"
RightBracket => "right-bracket"
Colon => "colon"
Comma => "comma"
StringLiteral => "string"
NumberLiteral => "number"
TrueLiteral => "true"
FalseLiteral => "false"
NullLiteral => "null"
End => "end"
}
}
///|
pub fn JsonToken::to_json(self : JsonToken) -> String {
"{\"kind\":" +
canonical_string(self.kind.to_string()) +
",\"lexeme\":" +
canonical_string(self.lexeme) +
",\"start\":" +
self.start.to_string() +
",\"end\":" +
self.end.to_string() +
"}"
}
///|
/// Tokenize a strict JSON document. Parsing first guarantees that string
/// escapes and number boundaries have already passed the same RFC checks as
/// the value parser; the scanner then preserves source spans and lexemes.
pub fn tokenize(input : String) -> Result[Array[JsonToken], ParseError] {
match parse(input) {
Ok(_) => ()
Err(error) => return Err(error)
}
let chars : Array[Char] = []
for char in input {
chars.push(char)
}
let tokens : Array[JsonToken] = []
let mut position = 0
while position < chars.length() {
while position < chars.length() && is_space(chars[position]) {
position = position + 1
}
if position >= chars.length() {
break
}
let start = position
let char = chars[position]
match char {
'{' => {
tokens.push({ kind: LeftBrace, lexeme: "{", start, end: start + 1 })
position = position + 1
}
'}' => {
tokens.push({ kind: RightBrace, lexeme: "}", start, end: start + 1 })
position = position + 1
}
'[' => {
tokens.push({ kind: LeftBracket, lexeme: "[", start, end: start + 1 })
position = position + 1
}
']' => {
tokens.push({ kind: RightBracket, lexeme: "]", start, end: start + 1 })
position = position + 1
}
':' => {
tokens.push({ kind: Colon, lexeme: ":", start, end: start + 1 })
position = position + 1
}
',' => {
tokens.push({ kind: Comma, lexeme: ",", start, end: start + 1 })
position = position + 1
}
'"' => {
position = scan_string(chars, position)
tokens.push({
kind: StringLiteral,
lexeme: chars_between(chars, start, position),
start,
end: position,
})
}
'-' | '0' | '1' | '2' | '3' | '4' | '5' | '6' | '7' | '8' | '9' => {
position = scan_number(chars, position)
tokens.push({
kind: NumberLiteral,
lexeme: chars_between(chars, start, position),
start,
end: position,
})
}
't' => {
position = position + 4
tokens.push({ kind: TrueLiteral, lexeme: "true", start, end: position })
}
'f' => {
position = position + 5
tokens.push({
kind: FalseLiteral,
lexeme: "false",
start,
end: position,
})
}
'n' => {
position = position + 4
tokens.push({ kind: NullLiteral, lexeme: "null", start, end: position })
}
_ => return Err(UnexpectedCharacter(position, char))
}
}
tokens.push({ kind: End, lexeme: "", start: position, end: position })
Ok(tokens)
}
///|
pub fn token_summary(tokens : Array[JsonToken]) -> String {
let mut output = ""
for token in tokens {
if output != "" {
output = output + "\n"
}
output = output +
token.start.to_string() +
".." +
token.end.to_string() +
" " +
token.kind.to_string() +
" " +
token.lexeme
}
output
}
///|
fn scan_string(chars : Array[Char], start : Int) -> Int {
let mut position = start + 1
while position < chars.length() {
if chars[position] == '\\' {
position = position + 2
} else if chars[position] == '"' {
return position + 1
} else {
position = position + 1
}
}
position
}
///|
fn scan_number(chars : Array[Char], start : Int) -> Int {
let mut position = start
while position < chars.length() && is_number_char(chars[position]) {
position = position + 1
}
position
}
///|
fn is_number_char(char : Char) -> Bool {
char == '-' ||
char == '+' ||
char == '.' ||
char == 'e' ||
char == 'E' ||
token_is_digit(char)
}
///|
fn token_is_digit(char : Char) -> Bool {
char >= '0' && char <= '9'
}
///|
fn is_space(char : Char) -> Bool {
char == ' ' || char == '\n' || char == '\r' || char == '\t'
}
///|
fn chars_between(chars : Array[Char], start : Int, end : Int) -> String {
let mut output = ""
let mut index = start
while index < end && index < chars.length() {
output = output + chars[index].to_string()
index = index + 1
}
output
}