///|
/// Parse `text` as JSON, reporting the raw library error.
///
/// This is `parse_with_diagnostic` with no diagnostic built around the failure,
/// reading documents the same way and to the same limits.
pub fn parse(text : String) -> Result[@pjson.Json, @pjson.JsonError] {
parse_raw(text, None)
}
///|
/// The nesting depth the parser allows when the caller does not ask for a
/// different one.
///
/// This is the value `Nanaloveyuki/parsec/json` uses for `JsonLimits::standard()`,
/// restated because the struct keeps its fields private and exposes no
/// accessors. `parser_wbtest.mbt` measures the limit the library actually
/// applies and fails if the two ever disagree.
pub let standard_max_depth : Int = 128
///|
/// The largest document the parser accepts, in characters.
///
/// `Nanaloveyuki/parsec/json` caps input at 1 MiB, which is small enough that a
/// 10 MB export is refused outright rather than read. The scanner below is what
/// reads a document, so the cap is ours to set: 64 MiB leaves room for the file
/// sizes this tool is meant to handle while still bounding what one run will
/// hold in memory.
pub let standard_max_input_chars : Int = 67_108_864
///|
/// The string length `JsonLimits::standard()` allows, restated for the scanner.
let standard_max_string_chars : Int = 262_144
///|
/// The array length `JsonLimits::standard()` allows, restated for the scanner.
let standard_max_array_items : Int = 65_536
///|
/// The object member count `JsonLimits::standard()` allows, for the scanner.
let standard_max_object_members : Int = 65_536
///|
/// Parse `text`, turning any failure into a diagnostic that carries the line,
/// the column and a human-readable reason.
///
/// `max_depth` caps how many collections may nest, counting one for each array
/// or object entered. `Some(limit)` refuses anything deeper, reporting the
/// position where the limit was passed; `None` leaves the parser's own default
/// in place. Either way the other four size limits keep their standard values,
/// so this changes only how deep a document may go.
pub fn parse_with_diagnostic(
text : String,
max_depth : Int?,
) -> Result[@pjson.Json, JsonDiagnostic] {
match parse_raw(text, max_depth) {
Ok(json) => Ok(json)
Err(error) => Err(JsonDiagnostic::from_error(text, error))
}
}
// ---------------------------------------------------------------------------
// The fast path
//
// `Nanaloveyuki/parsec/json` is a backtracking combinator parser, and it reads a
// document at about a megabyte a second: a 10 MB file takes it ten seconds. The
// scanner below follows the same grammar directly, over the document's own code
// units, and reads the same file in a fraction of the time.
//
// It is not a second implementation of JSON in the sense that matters: it never
// says what is wrong with a document. Everything it will not accept is handed to
// the library parser, which has been describing bad documents all along, so the
// two cannot disagree about a diagnostic. What the scanner must therefore get
// exactly right is *which documents are good*: anything it accepts, the library
// parser would have accepted too. `parser_differential_wbtest.mbt` measures that
// over a corpus of documents and mutations of them.
// ---------------------------------------------------------------------------
///|
/// Structural characters as UTF-16 code units, which is what `code_units` hands
/// out. The scanner walks those units directly, so they are named by number,
/// with the character in a comment.
const CODE_TAB : UInt16 = 9
///|
const CODE_LF : UInt16 = 10
///|
const CODE_CR : UInt16 = 13
///|
const CODE_SPACE : UInt16 = 32
///|
const CODE_QUOTE : UInt16 = 34 // '"'
///|
const CODE_PLUS : UInt16 = 43 // '+'
///|
const CODE_COMMA : UInt16 = 44 // ','
///|
const CODE_MINUS : UInt16 = 45 // '-'
///|
const CODE_DOT : UInt16 = 46 // '.'
///|
const CODE_SLASH : UInt16 = 47 // '/'
///|
const CODE_ZERO : UInt16 = 48 // '0'
///|
const CODE_ONE : UInt16 = 49 // '1'
///|
const CODE_NINE : UInt16 = 57 // '9'
///|
const CODE_COLON : UInt16 = 58 // ':'
///|
const CODE_UPPER_E : UInt16 = 69 // 'E'
///|
const CODE_LBRACKET : UInt16 = 91 // '['
///|
const CODE_BACKSLASH : UInt16 = 92 // '\\'
///|
const CODE_RBRACKET : UInt16 = 93 // ']'
///|
const CODE_LOWER_B : UInt16 = 98 // 'b'
///|
const CODE_LOWER_E : UInt16 = 101 // 'e'
///|
const CODE_LOWER_F : UInt16 = 102 // 'f'
///|
const CODE_LOWER_N : UInt16 = 110 // 'n'
///|
const CODE_LOWER_R : UInt16 = 114 // 'r'
///|
const CODE_LOWER_T : UInt16 = 116 // 't'
///|
const CODE_LOWER_U : UInt16 = 117 // 'u'
///|
const CODE_LBRACE : UInt16 = 123 // '{'
///|
const CODE_RBRACE : UInt16 = 125 // '}'
///|
/// The size limits the scanner refuses a document for, which the library
/// restates in `JsonLimits`, privately. A refusal here only sends the document
/// to the library, which is where the limit is described, so these have to
/// agree with `ScanLimits::of` rather than with any message.
priv struct ScanLimits {
depth : Int
string_chars : Int
array_items : Int
object_members : Int
}
///|
fn ScanLimits::of(max_depth : Int) -> ScanLimits {
{
depth: max_depth,
string_chars: standard_max_string_chars,
array_items: standard_max_array_items,
object_members: standard_max_object_members,
}
}
///|
/// A scanner over one document's code units.
///
/// `pos` always points at the next unit to read. `view` is the document those
/// units came from, kept so that a stretch of text can be sliced back out of it
/// instead of being rebuilt a character at a time.
priv struct Scanner {
view : StringView
units : ArrayView[UInt16]
limits : ScanLimits
mut pos : Int
}
///|
fn Scanner::new(text : String, limits : ScanLimits) -> Scanner {
{ view: text, units: text.code_units(), limits, pos: 0, }
}
///|
fn Scanner::at_end(self : Scanner) -> Bool {
self.pos >= self.units.length()
}
///|
fn Scanner::skip_whitespace(self : Scanner) -> Unit {
while self.pos < self.units.length() {
let code = self.units[self.pos]
if code == CODE_SPACE ||
code == CODE_LF ||
code == CODE_TAB ||
code == CODE_CR {
self.pos = self.pos + 1
} else {
break
}
}
}
///|
/// Consume `code` if it is next, and say whether it was.
fn Scanner::take(self : Scanner, code : UInt16) -> Bool {
if self.pos < self.units.length() && self.units[self.pos] == code {
self.pos = self.pos + 1
true
} else {
false
}
}
///|
/// Consume `word` if it is next, letter by letter.
fn Scanner::take_word(self : Scanner, word : String) -> Bool {
let letters = word.code_units()
if self.pos + letters.length() > self.units.length() {
return false
}
for index in 0.. String {
self.view.exact_view(start~, end=stop).to_owned()
}
///|
/// Read one value, or answer `None` for anything the library should look at.
///
/// `depth` is the number of collections already entered, so a container is only
/// read when there is room for it: scalars are always allowed, which is what
/// makes a limit of zero still accept `5`.
fn Scanner::value(self : Scanner, depth : Int) -> @pjson.Json? {
self.skip_whitespace()
if self.at_end() {
return None
}
let code = self.units[self.pos]
if code == CODE_LBRACE {
self.object_value(depth)
} else if code == CODE_LBRACKET {
self.array_value(depth)
} else if code == CODE_QUOTE {
match self.text_value() {
Some(text) => Some(@pjson.Text(value=text))
None => None
}
} else if code == CODE_LOWER_T {
if self.take_word("true") {
Some(@pjson.Bool(value=true))
} else {
None
}
} else if code == CODE_LOWER_F {
if self.take_word("false") {
Some(@pjson.Bool(value=false))
} else {
None
}
} else if code == CODE_LOWER_N {
if self.take_word("null") {
Some(@pjson.Null)
} else {
None
}
} else if code == CODE_MINUS || (code >= CODE_ZERO && code <= CODE_NINE) {
self.number_value()
} else {
None
}
}
///|
/// Read one or more digits, saying whether there were any.
fn Scanner::digits(self : Scanner) -> Bool {
let start = self.pos
while self.pos < self.units.length() {
let code = self.units[self.pos]
if code >= CODE_ZERO && code <= CODE_NINE {
self.pos = self.pos + 1
} else {
break
}
}
self.pos > start
}
///|
/// Read a number, keeping its text exactly as it was written.
///
/// The shape is JSON's own, not the widest thing that could be read as one: no
/// leading zeroes, something after a `.`, something after an `e`. A number that
/// is only nearly right is left to the library to complain about.
fn Scanner::number_value(self : Scanner) -> @pjson.Json? {
let start = self.pos
ignore(self.take(CODE_MINUS))
if self.at_end() {
return None
}
let first = self.units[self.pos]
if first == CODE_ZERO {
self.pos = self.pos + 1
} else if first >= CODE_ONE && first <= CODE_NINE {
self.pos = self.pos + 1
ignore(self.digits())
} else {
return None
}
if self.take(CODE_DOT) && !self.digits() {
return None
}
if !self.at_end() &&
(
self.units[self.pos] == CODE_LOWER_E ||
self.units[self.pos] == CODE_UPPER_E
) {
self.pos = self.pos + 1
if !self.at_end() &&
(self.units[self.pos] == CODE_PLUS || self.units[self.pos] == CODE_MINUS) {
self.pos = self.pos + 1
}
if !self.digits() {
return None
}
}
Some(@pjson.Number(raw=self.slice(start, self.pos)))
}
///|
/// Read a string value, escapes and all.
///
/// A string with no escapes in it is sliced straight out of the document. One
/// with escapes is rebuilt, because the escapes are not what the value is.
fn Scanner::text_value(self : Scanner) -> String? {
self.pos = self.pos + 1
let start = self.pos
let units = self.units
let length = units.length()
let mut characters = 0
let mut escaped = false
while self.pos < length {
let code = units[self.pos]
if code == CODE_QUOTE {
let stop = self.pos
self.pos = self.pos + 1
if escaped {
return self.unescaped(start, stop)
}
return Some(self.slice(start, stop))
} else if code == CODE_BACKSLASH {
escaped = true
// The escape and its payload are two units, except for `\uXXXX`, which
// is six. Validity is settled by `unescaped`, which reads the same text
// again; here the only job is to find the closing quote.
if self.pos + 1 >= length || units[self.pos + 1] == CODE_LOWER_U {
self.pos = self.pos + 6
characters = characters + 1
} else {
self.pos = self.pos + 2
characters = characters + 1
}
} else {
// A control character has to be written as an escape, and the library holds
// documents to that.
if code < 0x20 {
return None
}
// A low surrogate is half of one character, not a character.
if code < 0xDC00 || code > 0xDFFF {
characters = characters + 1
}
self.pos = self.pos + 1
}
if characters > self.limits.string_chars {
return None
}
}
None
}
///|
/// Rebuild the string between `start` and `stop`, where the escapes are.
///
/// Answers `None` for an escape that is not one of JSON's, or for a `\u` escape
/// that is not a character on its own and not the first half of a pair, so that
/// the library gets to name the problem.
fn Scanner::unescaped(self : Scanner, start : Int, stop : Int) -> String? {
let buf = StringBuilder()
let units = self.units
let mut index = start
while index < stop {
let code = units[index]
if code != CODE_BACKSLASH {
buf.write_char(Int::unsafe_to_char(code.to_int()))
index = index + 1
continue
}
if index + 1 >= stop {
return None
}
let escaped = units[index + 1]
if escaped == CODE_LOWER_U {
let mut value = match self.hex4(index + 2, stop) {
Some(value) => value
None => return None
}
index = index + 6
if value >= 0xD800 && value <= 0xDBFF {
// A high surrogate is only a character with its low half behind it.
if index + 5 >= stop ||
units[index] != CODE_BACKSLASH ||
units[index + 1] != CODE_LOWER_U {
return None
}
let low = match self.hex4(index + 2, stop) {
Some(low) => low
None => return None
}
if low < 0xDC00 || low > 0xDFFF {
return None
}
value = 0x10000 + (value - 0xD800) * 1024 + (low - 0xDC00)
index = index + 6
} else if value >= 0xDC00 && value <= 0xDFFF {
// A low surrogate on its own is half of a character, which is not one.
return None
}
buf.write_char(Int::unsafe_to_char(value))
continue
}
let character = if escaped == CODE_QUOTE {
'"'
} else if escaped == CODE_BACKSLASH {
'\\'
} else if escaped == CODE_SLASH {
'/'
} else if escaped == CODE_LOWER_B {
'\u{8}'
} else if escaped == CODE_LOWER_F {
'\u{c}'
} else if escaped == CODE_LOWER_N {
'\n'
} else if escaped == CODE_LOWER_R {
'\r'
} else if escaped == CODE_LOWER_T {
'\t'
} else {
return None
}
buf.write_char(character)
index = index + 2
}
Some(buf.to_string())
}
///|
/// Four hex digits at `at`, or `None` if they are not four hex digits.
fn Scanner::hex4(self : Scanner, at : Int, stop : Int) -> Int? {
if at + 4 > stop {
return None
}
let mut value = 0
for offset in 0..<4 {
let digit = match hex_digit(self.units[at + offset]) {
Some(digit) => digit
None => return None
}
value = value * 16 + digit
}
Some(value)
}
///|
fn hex_digit(code : UInt16) -> Int? {
if code >= CODE_ZERO && code <= CODE_NINE {
Some(code.to_int() - 48)
} else if code >= 97 && code <= 102 {
Some(code.to_int() - 87)
} else if code >= 65 && code <= 70 {
Some(code.to_int() - 55)
} else {
None
}
}
///|
/// Read an array, up to the item limit.
fn Scanner::array_value(self : Scanner, depth : Int) -> @pjson.Json? {
if depth >= self.limits.depth {
return None
}
self.pos = self.pos + 1
let items : Array[@pjson.Json] = []
self.skip_whitespace()
if self.take(CODE_RBRACKET) {
return Some(@pjson.Array(items~))
}
while true {
let item = match self.value(depth + 1) {
Some(item) => item
None => return None
}
items.push(item)
if items.length() > self.limits.array_items {
return None
}
self.skip_whitespace()
if self.take(CODE_COMMA) {
continue
}
if self.take(CODE_RBRACKET) {
return Some(@pjson.Array(items~))
}
return None
} nobreak {
abort("unreachable: the array loop leaves only by returning")
}
}
///|
/// Read an object, up to the member limit.
///
/// A repeated key is refused here, not because the scanner has anything to say
/// about it but because the library refuses it: accepting one would be accepting
/// a document the tool has always turned down.
fn Scanner::object_value(self : Scanner, depth : Int) -> @pjson.Json? {
if depth >= self.limits.depth {
return None
}
self.pos = self.pos + 1
let members : Array[(String, @pjson.Json)] = []
// A hash map per object costs more than a walk down a handful of keys, so the
// map is only built once an object is big enough to need it.
let seen : Map[String, Unit] = Map([])
let mut indexed = false
self.skip_whitespace()
if self.take(CODE_RBRACE) {
return Some(@pjson.Object(members~))
}
while true {
self.skip_whitespace()
if self.at_end() || self.units[self.pos] != CODE_QUOTE {
return None
}
let key = match self.text_value() {
Some(key) => key
None => return None
}
if indexed {
if seen.contains(key) {
return None
}
seen.set(key, ())
} else {
for entry in members {
let (name, _) = entry
if name == key {
return None
}
}
if members.length() >= 8 {
for entry in members {
let (name, _) = entry
seen.set(name, ())
}
indexed = true
}
}
self.skip_whitespace()
if !self.take(CODE_COLON) {
return None
}
let value = match self.value(depth + 1) {
Some(value) => value
None => return None
}
members.push((key, value))
if members.length() > self.limits.object_members {
return None
}
self.skip_whitespace()
if self.take(CODE_COMMA) {
continue
}
if self.take(CODE_RBRACE) {
return Some(@pjson.Object(members~))
}
return None
} nobreak {
abort("unreachable: the object loop leaves only by returning")
}
}
///|
/// Scan `text` as JSON within `max_depth`, or answer `None`.
///
/// `None` means "not the scanner's to read": either the document is bad, or it
/// is good but past a limit the scanner does not describe. Both go to the
/// library parser.
fn scan(text : String, max_depth : Int) -> @pjson.Json? {
let scanner = Scanner::new(text, ScanLimits::of(max_depth))
match scanner.value(0) {
None => None
Some(json) => {
scanner.skip_whitespace()
if scanner.at_end() {
Some(json)
} else {
None
}
}
}
}
///|
/// The parser call itself, with no diagnostic built around it.
///
/// Split out because JSON Lines parses a record at a time and needs the raw
/// error to place it in the file as a whole rather than in the record.
///
/// The scanner in `scan` reads what it can. Anything it will not accept — a
/// syntax error, a limit passed, a repeated key — goes to the library parser,
/// which is the authority on what a bad document means and where it went wrong.
/// So the fast path can only ever make a good document quicker; it cannot
/// change what is said about a bad one.
fn parse_raw(
text : String,
max_depth : Int?,
) -> Result[@pjson.Json, @pjson.JsonError] {
let limit = match max_depth {
None => standard_max_depth
Some(limit) => limit
}
// A document over the cap is refused by the library, which is where the
// message for it is written. Asking first saves scanning a document that is
// only going to be turned away.
if over_input_limit(text) {
return @pjson.parse_with_limits(text, depth_limits(limit))
}
match scan(text, limit) {
Some(json) => Ok(json)
None => @pjson.parse_with_limits(text, depth_limits(limit))
}
}
///|
/// Whether `text` is too long to parse, counted in characters as the library
/// counts them.
///
/// The quick test is in code units, which are never fewer than characters, so a
/// document that passes it is under the cap without counting anything; only a
/// document that might be over is counted properly.
fn over_input_limit(text : String) -> Bool {
if text.length() <= standard_max_input_chars {
false
} else {
text.char_length() > standard_max_input_chars
}
}
///|
/// The standard limits, with `max_depth` replaced by `limit`.
///
/// Four of the five fields are the sizes `JsonLimits::standard()` picks, spelled
/// out again because only the depth is meant to change and the struct offers no
/// way to read the others back. The input cap is the one field that is not the
/// library's: see `standard_max_input_chars`.
fn depth_limits(limit : Int) -> @pjson.JsonLimits {
@pjson.JsonLimits::new(
max_input_chars=standard_max_input_chars,
max_depth=limit,
max_string_chars=standard_max_string_chars,
max_array_items=standard_max_array_items,
max_object_members=standard_max_object_members,
)
}
///|
/// Render a caught error as a concise message fit for one line of output.
///
/// Operating-system failures are reduced to a plain-language reason. The debug
/// form of these errors carries internal call context that only clutters a CLI
/// message, so the errno is translated through the public predicates instead.
fn describe_io_error(error : Error) -> String {
match error {
@os_error.OSError(errno, ..) as failure =>
if failure.is_ENOENT() {
"no such file or directory"
} else if failure.is_EACCES() {
"permission denied"
} else if failure.is_ENOTDIR() {
"not a directory"
} else {
"operating system error " + errno.to_string()
}
_ => @debug.to_string(error)
}
}
///|
/// Read `path` as UTF-8 text.
///
/// The failure is reported as a message rather than an error value so that the
/// caller can print it and exit with code 1 without unwrapping anything.
pub async fn read_file(path : String) -> Result[String, String] {
Ok(@fs.read_file(path).text()) catch {
error => Err(describe_io_error(error))
}
}
///|
/// Read all of standard input as UTF-8 text.
pub async fn read_stdin() -> Result[String, String] {
Ok(@stdio.stdin.read_all().text()) catch {
error => Err(describe_io_error(error))
}
}