Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 18 additions & 2 deletions src/ast.rs
Original file line number Diff line number Diff line change
Expand Up @@ -87,8 +87,8 @@ impl<'a> From<Value<'a>> for serde_json::Value {
}
} else {
// standard decimal number
let num_for_parsing = num.value.trim_start_matches('+');
match serde_json::Number::from_str(num_for_parsing) {
let num_for_parsing = crate::common::fill_bare_decimal_point(num.value.trim_start_matches('+'));
match serde_json::Number::from_str(&num_for_parsing) {
Ok(number) => serde_json::Value::Number(number),
Err(_) => serde_json::Value::String(num.value.to_string()),
}
Expand Down Expand Up @@ -656,6 +656,22 @@ mod test {
);
}

#[cfg(feature = "serde_json")]
#[test]
fn it_should_coerce_json5_numbers_to_serde_value() {
let ast = parse_to_ast(
r#"[.5, -.5, +5., 5.e3, Infinity]"#,
&Default::default(),
&ParseOptions::default(),
)
.unwrap();
let serde_value: serde_json::Value = ast.value.unwrap().into();

// serde_json cannot hold Infinity, so it keeps the existing fallback to a string
assert_eq!(serde_value, serde_json::json!([0.5, -0.5, 5.0, 5000.0, "Infinity"]));
assert!(serde_value[2].is_f64());
}

#[cfg(feature = "serde_json")]
#[test]
fn handle_weird_data() {
Expand Down
18 changes: 18 additions & 0 deletions src/common.rs
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,24 @@ impl Ranged for Range {
}
}

/// Adds the digit serde_json needs beside a leading or trailing decimal point (ex. `.5` to `0.5`).
#[cfg(feature = "serde_json")]
pub(crate) fn fill_bare_decimal_point(num: &str) -> std::borrow::Cow<'_, str> {
let Some(dot) = num.find('.') else {
return std::borrow::Cow::Borrowed(num);
};
let before = num[..dot].ends_with(|c: char| c.is_ascii_digit());
let after = num[dot + 1..].starts_with(|c: char| c.is_ascii_digit());
if before && after {
return std::borrow::Cow::Borrowed(num);
}
let mut filled = String::with_capacity(num.len() + 1);
filled.push_str(&num[..dot]);
filled.push_str(if before { ".0" } else { "0." });
filled.push_str(&num[dot + 1..]);
std::borrow::Cow::Owned(filled)
}

/// Represents an object that has a range in the text.
pub trait Ranged {
/// Gets the range.
Expand Down
63 changes: 59 additions & 4 deletions src/cst/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1518,8 +1518,8 @@ impl CstNumberLit {
}
} else {
// standard decimal number - strip leading + if present (serde_json doesn't accept it)
let num_for_parsing = raw.trim_start_matches('+');
match serde_json::Number::from_str(num_for_parsing) {
let parsed = serde_json::Number::from_str(&crate::common::fill_bare_decimal_point(raw.trim_start_matches('+')));
match parsed {
Ok(number) => Some(serde_json::Value::Number(number)),
// if the number is invalid, return it as a string (same behavior as AST conversion)
Err(_) => Some(serde_json::Value::String(raw)),
Expand Down Expand Up @@ -2372,9 +2372,10 @@ pub enum CstNewlineKind {
#[default]
LineFeed,
CarriageReturnLineFeed,
CarriageReturn,
}

/// Newline character (Lf or crlf).
/// Newline character (Lf, crlf, or cr).
#[derive(Debug, Clone)]
pub struct CstNewline(Rc<RefCell<CstValueInner<CstNewlineKind>>>);

Expand All @@ -2385,7 +2386,7 @@ impl CstNewline {
Self(CstValueInner::new(kind))
}

/// Whether this is a line feed (LF) or carriage return line feed (CRLF).
/// Whether this is a line feed (LF), carriage return line feed (CRLF), or carriage return (CR).
pub fn kind(&self) -> CstNewlineKind {
self.0.borrow().value
}
Expand All @@ -2407,6 +2408,7 @@ impl Display for CstNewline {
#[allow(clippy::write_with_newline)] // better to be explicit
CstNewlineKind::LineFeed => write!(f, "\n"),
CstNewlineKind::CarriageReturnLineFeed => write!(f, "\r\n"),
CstNewlineKind::CarriageReturn => write!(f, "\r"),
}
}
}
Expand Down Expand Up @@ -2614,6 +2616,10 @@ impl<'a> CstBuilder<'a> {
container.raw_append_child(CstNewline::new(CstNewlineKind::CarriageReturnLineFeed).into());
last_found_index = i + 2;
chars.next(); // move past the \n
} else if c == '\r' {
maybe_add_previous_text(last_found_index, i);
container.raw_append_child(CstNewline::new(CstNewlineKind::CarriageReturn).into());
last_found_index = i + 1;
} else if c == '\n' {
maybe_add_previous_text(last_found_index, i);
container.raw_append_child(CstNewline::new(CstNewlineKind::LineFeed).into());
Expand Down Expand Up @@ -3874,6 +3880,7 @@ mod test {
use pretty_assertions::assert_eq;

use crate::cst::CstInputValue;
use crate::cst::CstNewlineKind;
use crate::cst::TrailingCommaMode;
use crate::json;

Expand Down Expand Up @@ -5203,6 +5210,26 @@ value3: true
assert_eq!(value, SerdeValue::Null);
}

#[test]
fn test_cst_to_serde_value_json5_numbers() {
let root = build_cst(r#"[.5, -.5, +5., 5.e3, NaN]"#);
let value = root.to_serde_value().unwrap();
assert_eq!(value, serde_json::json!([0.5, -0.5, 5.0, 5000.0, "NaN"]));
assert!(value[2].is_f64());
assert_eq!(root.to_string(), "[.5, -.5, +5., 5.e3, NaN]");
}

#[test]
fn test_cst_json5_strings_and_line_terminators() {
let text = "{\r // a\r 'k': 'line\\\ncontinued \\x41\\v',\u{2028} // b\r \"it\\'s\": 1,\r}";
let root = build_cst(text);
assert_eq!(root.to_string(), text);
assert_eq!(
root.to_serde_value().unwrap(),
serde_json::json!({ "k": "linecontinued A\u{0B}", "it's": 1 })
);
}

#[test]
fn test_cst_to_serde_value_array() {
let root = build_cst(r#"[1, 2, 3]"#);
Expand Down Expand Up @@ -5513,4 +5540,32 @@ value3: true
.unwrap();
assert_eq!(decoded, "key\\with\\backslash");
}

#[test]
fn lone_carriage_return_ends_line_comment() {
// a lone \r ends a line comment, so edits should behave the same as with \n
#[track_caller]
fn run_test(json: &str, edit: impl Fn(&CstRootNode)) {
let lf_cst = build_cst(json);
edit(&lf_cst);
let cr_cst = build_cst(&json.replace('\n', "\r"));
assert_eq!(cr_cst.newline_kind(), CstNewlineKind::CarriageReturn);
edit(&cr_cst);
assert_eq!(cr_cst.to_string(), lf_cst.to_string().replace('\n', "\r"));
}

run_test("[ // note\n]", |cst| {
cst.array_value().unwrap().append(json!(3));
});
run_test("[\n 1 // one\n]", |cst| {
cst.array_value().unwrap().append(json!(3));
});
run_test("{\n \"b\": 1, // bee\n \"a\": 2\n}", |cst| {
cst
.object_value()
.unwrap()
.sort_properties()
.by_key(|prop| prop.decoded_name());
});
}
}
13 changes: 11 additions & 2 deletions src/errors.rs
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ use super::common::Range;

#[derive(Debug)]
pub enum ParseErrorKind {
BareDecimalPointNumbersNotAllowed,
CommentsNotAllowed,
ExpectedColonAfterObjectKey,
ExpectedObjectValue,
Expand All @@ -16,6 +17,7 @@ pub enum ParseErrorKind {
HexadecimalNumbersNotAllowed,
ExpectedComma,
MultipleRootJsonValues,
NonFiniteNumbersNotAllowed,
SingleQuotedStringsNotAllowed,
String(ParseStringErrorKind),
TrailingCommasNotAllowed,
Expand All @@ -40,6 +42,9 @@ impl std::fmt::Display for ParseErrorKind {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
use ParseErrorKind::*;
match self {
BareDecimalPointNumbersNotAllowed => {
write!(f, "Leading or trailing decimal points on numbers are not allowed")
}
CommentsNotAllowed => {
write!(f, "Comments are not allowed")
}
Expand Down Expand Up @@ -70,6 +75,9 @@ impl std::fmt::Display for ParseErrorKind {
MultipleRootJsonValues => {
write!(f, "Text cannot contain more than one JSON value")
}
NonFiniteNumbersNotAllowed => {
write!(f, "Infinity and NaN are not allowed")
}
SingleQuotedStringsNotAllowed => {
write!(f, "Single-quoted strings are not allowed")
}
Expand Down Expand Up @@ -206,8 +214,9 @@ impl fmt::Display for ParseError {
fn get_line_and_column_display(range: Range, file_text: &str) -> (usize, usize) {
let mut line_index = 0;
let mut column_index = 0;
for c in file_text[..range.start].chars() {
if c == '\n' {
for (i, c) in file_text[..range.start].char_indices() {
// a lone \r ends a line comment, so it also counts as a line break here
if c == '\n' || c == '\r' && file_text.as_bytes().get(i + 1) != Some(&b'\n') {
line_index += 1;
column_index = 0;
} else {
Expand Down
3 changes: 3 additions & 0 deletions src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -115,6 +115,9 @@
//! allow_single_quoted_strings: false,
//! allow_hexadecimal_numbers: false,
//! allow_unary_plus_numbers: false,
//! allow_bare_decimal_point_numbers: false,
//! allow_non_finite_numbers: false,
//! allow_extended_string_escapes: false,
//! })?;
//! # Ok(())
//! # }
Expand Down
94 changes: 89 additions & 5 deletions src/parse_to_ast.rs
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,12 @@ pub struct ParseOptions {
pub allow_hexadecimal_numbers: bool,
/// Allow unary plus sign on numbers like +42 (defaults to `true`).
pub allow_unary_plus_numbers: bool,
/// Allow a leading or trailing decimal point on numbers like .5 or 5. (defaults to `true`).
pub allow_bare_decimal_point_numbers: bool,
/// Allow the numbers Infinity, -Infinity and NaN (defaults to `true`).
pub allow_non_finite_numbers: bool,
/// Allow JSON5 string escapes like \x41, \v and line continuations (defaults to `true`).
pub allow_extended_string_escapes: bool,
}

impl Default for ParseOptions {
Expand All @@ -75,6 +81,9 @@ impl Default for ParseOptions {
allow_single_quoted_strings: true,
allow_hexadecimal_numbers: true,
allow_unary_plus_numbers: true,
allow_bare_decimal_point_numbers: true,
allow_non_finite_numbers: true,
allow_extended_string_escapes: true,
}
}
}
Expand Down Expand Up @@ -253,6 +262,9 @@ pub fn parse_to_ast<'a>(
allow_single_quoted_strings: parse_options.allow_single_quoted_strings,
allow_hexadecimal_numbers: parse_options.allow_hexadecimal_numbers,
allow_unary_plus_numbers: parse_options.allow_unary_plus_numbers,
allow_bare_decimal_point_numbers: parse_options.allow_bare_decimal_point_numbers,
allow_non_finite_numbers: parse_options.allow_non_finite_numbers,
allow_extended_string_escapes: parse_options.allow_extended_string_escapes,
},
),
comments: match collect_options.comments {
Expand Down Expand Up @@ -324,11 +336,11 @@ fn parse_object<'a>(context: &mut Context<'a>) -> Result<Object<'a>, ParseError>
Some(Token::String(prop_name)) => {
properties.push(parse_object_property(context, PropName::String(prop_name))?);
}
Some(Token::Word(prop_name)) | Some(Token::Number(prop_name)) => {
properties.push(parse_object_property(context, PropName::Word(prop_name))?);
}
None => return Err(context.create_error_for_current_range(ParseErrorKind::UnterminatedObject)),
_ => return Err(context.create_error(ParseErrorKind::UnexpectedTokenInObject)),
Some(token) => match token.as_loose_property_name() {
Some(prop_name) => properties.push(parse_object_property(context, PropName::Word(prop_name))?),
None => return Err(context.create_error(ParseErrorKind::UnexpectedTokenInObject)),
},
}

// skip the comma
Expand All @@ -342,7 +354,9 @@ fn parse_object<'a>(context: &mut Context<'a>) -> Result<Object<'a>, ParseError>
return Err(context.create_error_for_range(comma_range, ParseErrorKind::TrailingCommasNotAllowed));
}
}
Some(Token::String(_) | Token::Word(_) | Token::Number(_)) if !context.allow_missing_commas => {
Some(Token::String(_) | Token::Word(_) | Token::Number(_) | Token::Boolean(_) | Token::Null)
if !context.allow_missing_commas =>
{
let range = Range {
start: after_value_end,
end: after_value_end,
Expand Down Expand Up @@ -565,6 +579,11 @@ mod tests {
r#"{ word: 5 }"#,
"Expected string for object property on line 1 column 3",
);
assert_has_strict_error(
r#"{ true: 5 }"#,
"Expected string for object property on line 1 column 3",
);
assert_has_strict_error(r#"{ "a": 1 null: 2 }"#, "Expected comma on line 1 column 9");
}

#[test]
Expand All @@ -591,6 +610,27 @@ mod tests {
);
}

#[test]
fn strict_should_error_bare_decimal_point_number() {
assert_has_strict_error(
r#"{ "key": .5 }"#,
"Leading or trailing decimal points on numbers are not allowed on line 1 column 10",
);
assert_has_strict_error(
r#"{ "key": 5. }"#,
"Leading or trailing decimal points on numbers are not allowed on line 1 column 10",
);
}

#[test]
fn strict_should_error_non_finite_number() {
assert_has_strict_error(
r#"{ "key": -Infinity }"#,
"Infinity and NaN are not allowed on line 1 column 10",
);
assert_has_strict_error(r#"{ "key": NaN }"#, "Unexpected word on line 1 column 10");
}

#[track_caller]
fn assert_has_strict_error(text: &str, message: &str) {
let result = parse_to_ast(text, &Default::default(), &strict_options());
Expand All @@ -609,6 +649,9 @@ mod tests {
allow_single_quoted_strings: false,
allow_hexadecimal_numbers: false,
allow_unary_plus_numbers: false,
allow_bare_decimal_point_numbers: false,
allow_non_finite_numbers: false,
allow_extended_string_escapes: false,
}
}

Expand Down Expand Up @@ -708,6 +751,47 @@ mod tests {
assert_eq!(number_value.value, "+42");
}

#[test]
fn it_should_parse_keywords_as_loose_property_names() {
let result = parse_to_ast(
r#"{ true: 1, false: null null: true }"#,
&Default::default(),
&Default::default(),
)
.unwrap();
let value = result.value.unwrap();
let names = value
.as_object()
.unwrap()
.properties
.iter()
.map(|p| p.name.as_str())
.collect::<Vec<_>>();
assert_eq!(names, vec!["true", "false", "null"]);
}

#[test]
fn it_should_parse_json5_numbers_and_keep_non_finite_words_as_keys() {
let result = parse_to_ast(
r#"{ "a": .5, "b": 5., "c": -Infinity, Infinity: NaN }"#,
&Default::default(),
&Default::default(),
)
.unwrap();

let value = result.value.unwrap();
let obj = value.as_object().unwrap();
let values = obj
.properties
.iter()
.map(|p| (p.name.as_str(), p.value.as_number_lit().unwrap().value))
.collect::<Vec<_>>();
assert_eq!(
values,
vec![("a", ".5"), ("b", "5."), ("c", "-Infinity"), ("Infinity", "NaN")]
);
}

#[test]
fn missing_comma_between_properties() {
let text = r#"{
Expand Down
Loading
Loading