///|
/// Source-level measurements collected alongside lexical tokens.
pub(all) struct SourceStats {
character_count : Int
line_count : Int
whitespace_count : Int
token_count : Int
max_token_length : Int
source_bytes : Int
}
///|
pub fn inspect_source(input : String) -> Result[SourceStats, String] {
let tokens = match tokenize(input) {
Ok(tokens) => tokens
Err(error) => return Err(error.message())
}
let mut character_count = 0
let mut line_count = 1
let mut whitespace_count = 0
for char in input {
character_count = character_count + 1
if char == '\n' {
line_count = line_count + 1
}
if char == ' ' || char == '\n' || char == '\r' || char == '\t' {
whitespace_count = whitespace_count + 1
}
}
let mut max_token_length = 0
for token in tokens {
if token.lexeme.length() > max_token_length {
max_token_length = token.lexeme.length()
}
}
Ok({
character_count,
line_count,
whitespace_count,
token_count: tokens.length(),
max_token_length,
source_bytes: @utf8.encode(input).length(),
})
}
///|
pub fn SourceStats::to_json(self : SourceStats) -> String {
"{\"character_count\":" +
self.character_count.to_string() +
",\"line_count\":" +
self.line_count.to_string() +
",\"whitespace_count\":" +
self.whitespace_count.to_string() +
",\"token_count\":" +
self.token_count.to_string() +
",\"max_token_length\":" +
self.max_token_length.to_string() +
",\"source_bytes\":" +
self.source_bytes.to_string() +
"}"
}
///|
pub fn SourceStats::to_text(self : SourceStats) -> String {
"characters=" +
self.character_count.to_string() +
",lines=" +
self.line_count.to_string() +
",whitespace=" +
self.whitespace_count.to_string() +
",tokens=" +
self.token_count.to_string() +
",bytes=" +
self.source_bytes.to_string()
}